diff --git a/README.md b/README.md index e1491d4b..55662350 100644 --- a/README.md +++ b/README.md @@ -18,7 +18,7 @@ This repository contains reusable infrastructure modules designed for enterprise | `cache/` | `elasticache` | AWS ElastiCache clusters (Redis, Valkey, Memcached) | v1.0.0 | | `cdn/` | `cloudfront` | AWS CloudFront distributions with origins, cache behaviors, and edge redirects (includes `rvn-cloudfront` module definition) | v1.0.0 | | `compute/` | `autoscaling` | AWS Auto Scaling groups | v1.0.0 | -| `compute/` | `ec2` | AWS EC2 instances | Planned | +| `compute/` | `ec2_service` | Supervised EC2 workloads with configurable rolling deploys, standalone or ECS-cluster ALB routing, target tuning, and deployment-scoped CloudWatch logs | v1.0.0 | | `compute/` | `ecs_cluster` | AWS ECS clusters with Fargate/EC2 capacity providers and optional ALBs/NLBs | v1.0.0 | | `compute/` | `ecs_service` | AWS ECS services with task definitions, task IAM policies, load balancing, and auto scaling | v1.0.0 | | `compute/` | `lambda` | AWS Lambda functions | v1.0.0 | @@ -33,7 +33,7 @@ This repository contains reusable infrastructure modules designed for enterprise | `messaging/` | `sns` | AWS SNS topics and subscriptions | Planned | | `messaging/` | `sqs` | AWS SQS queues | Planned | | `monitoring/` | `cloudwatch` | AWS CloudWatch alarms and dashboards | Planned | -| `networking/` | `alb` | AWS Application Load Balancers | v1.0.0 | +| `networking/` | `alb` | Standalone AWS Application Load Balancer with shared HTTP/HTTPS listeners | v1.0.0 | | `networking/` | `eips` | AWS Elastic IP pool with deterministic Name tags and `/32` CIDR outputs | v1.0.0 | | `networking/` | `nlb` | AWS Network Load Balancers | v1.0.0 | | `networking/` | `route53` | AWS Route53 hosted zones and records | v1.0.0 | diff --git a/compute/ec2_service/.terraform.lock.hcl b/compute/ec2_service/.terraform.lock.hcl new file mode 100644 index 00000000..a079284f --- /dev/null +++ b/compute/ec2_service/.terraform.lock.hcl @@ -0,0 +1,39 @@ +# This file is maintained automatically by "tofu init". +# Manual edits may be lost in future updates. + +provider "registry.opentofu.org/hashicorp/aws" { + version = "6.53.0" + constraints = ">= 6.0.0" + hashes = [ + "h1:5P+SDwfY9xc3oNyFCi7T9ec2WKs0Kw6h+gBbLGp7xYA=", + "h1:7eaGyQaHqU0h48zwJg786kxdR1ox56WBl/Nfda/uU6U=", + "h1:MeKnCIwuNwdk10RB7T4y46fAaavFBQSoh0DP4Jd+cMQ=", + "h1:QIZhIYRX+FjNK1MM2j7d+SS+2sRuhGsStxGYdeuNuoI=", + "h1:RinUymyjUeWaHwS5hg6AkFgMmpn0+ZIMf80rx5BlgoY=", + "h1:WsPCHWzaWNYyqX4JvfDzmab/P+0NxHiqvHp/ClUOWqs=", + "h1:YXBlULX+maf70HlHnqqRCRV2yG/oLLzb/TbygeLONrM=", + "h1:cOEyj479FZE7cTxnDGM4vj/u0n0ILVRxyPl5oIAzfiY=", + "h1:dSdQS0Ue9Ob26BCt2Fv740LcoRzGmpu4W/L2I5K9YzE=", + "h1:k4vYcdMr0yU8bknkp6E4dfD4RjXzFFcJ/6G5oS6TiSY=", + "h1:n9aGPY4fOgjbvXX3TYrP4Amw2EiEfSqaz55lSuFS2Co=", + "h1:npiJXumWsg0dkJBn9yPVm+eqi+tXdie3ttD8P298+6Y=", + "h1:tTOBbDfzy8a4YVqIq/uoghTJL5kw32leOR2aQUloALI=", + "h1:uivUtKqX7AHRnBWkVrxlfWlauv/s0C4nMxQ9JeohHHs=", + "h1:xFtJcieVZfsv5xDW6F0ZgavumYLI2o7Fk1qZQHmQmFo=", + "zh:03fb02e200242a11252912d04be8da8eb80a72c06bfc9f4b73a8e97ad2bea21c", + "zh:19411bbcb38cf2644d0a426b52b8f28a29464a1749f5db713b80b443e706d8b8", + "zh:3ad53edba021e4a02415e079de846d2c385964e540b401801c7fd309f88b6b69", + "zh:4661891cb13b70df47f4a5913336c6b4ee81e0a72e22abba5561c0eb9e535f87", + "zh:4ea6ca42462e0377ce4ca50faef4b28a7059142eb88199fa966280b9307b525f", + "zh:6ce7d8598c2664cd3fa765ecebb564897910c7f32fb1139e8213b7d0fc5b86fc", + "zh:6e651398e2fe03b60a1cad41f45060838d47e26463317b34f644943e6e9ce760", + "zh:745d1c6b9c49cec684003fddc8aee4a99b8595cb9a7f1898dac5ee26d369b147", + "zh:7f85f9f0f523c2d220d93b892bae825ef4bca4a187c26a1207d40e3eb7a3693b", + "zh:a9a6c4f35d75b4f7511742d5ba3f02d1ad4dd720c5208c98fa85b47e5e37372b", + "zh:b306267308de2d1ef094702002417baf17359a2f8b09f3e1c9d557fa153506be", + "zh:d1ba9d27b28bb6b356b141b7d5015a37a78d92fb0ce715e13df6e8d98533ef46", + "zh:e78be305a8e0550a09ced9eaa6f5f98060c53079cb1d36cd904eb7afccf09138", + "zh:e9357d850c476ac35f3358ff102df7bce23bc303f87a77e0ecff0a6c308039bc", + "zh:f0918349619590f9f4213a86b74ecf8fa55f44971991c7ff2b460a76a6ae20c6", + ] +} diff --git a/compute/ec2_service/README.md b/compute/ec2_service/README.md new file mode 100644 index 00000000..9252f53a --- /dev/null +++ b/compute/ec2_service/README.md @@ -0,0 +1,196 @@ +# EC2 Service + +Runs an app on a stable group of EC2 instances behind an optional Application Load Balancer, with in-place deploys pushed through SSM Run Command. Instances are managed by an Auto Scaling Group but are never replaced by deploys, so per-instance state (root volume, optional data volume) survives every release. + +Two runtimes are supported: + +- **container** — each deploy pulls an image and runs the attached Docker process under supervisord. +- **manual** — deploys can check out an authenticated Git source, run release preparation commands, then start the configured long-lived command under supervisord. + +Instances install the prerequisites for both runtimes at launch, so `runtime` can switch between `container` and `manual` without replacing the instance group. Supervisord owns the app in both modes and restarts it after an unexpected exit. + +## How deploys work + +For the **container** runtime the module creates an SSM Command document (`-deploy` — the name is a platform contract derived from the group name) that encodes the whole in-place deploy for one instance: + +1. Rebuild the app env file: Terraform-rendered plain values, plus secret values fetched on the instance from Secrets Manager / SSM Parameter Store. +2. Drain: deregister the instance from the target group and wait (skipped in worker mode, and when the instance is the only registered target — with nothing to shift traffic to, draining only lengthens the outage). +3. Swap the release: `docker pull`, then update the supervisord-managed container process. +4. Health gate: poll `http://localhost:` until healthy, or fail the command. +5. Re-register the instance with the target group and wait until in service. + +An orchestrator (the Ravion deploy manager) runs this document against the Auto Scaling Group's instances with its own batching and failure policy, passing: + +| Parameter | Value | +|-----------|-------| +| `imageUri` | Full image URI including tag or digest | +| `deployId` | Optional release identifier | + +The module definition exposes the rolling deployment batch and failure limits. It defaults to one instance at a time and stops after the first failure. The SSM script timeout applies per instance; the module deployment has a separate 24-hour overall safety limit. + +For the **manual** runtime the document (same `-deploy` name) takes a `commands` parameter and an optional Git source. When source is present, the instance fetches a temporary credential from SSM Parameter Store, performs a clean checkout under `/srv/ravion//source`, and runs both the preparation commands and `manual_start_command` from the selected base path. When source is absent, commands keep their existing working-directory behavior. Any failure stops the deploy. The start command must remain in the foreground rather than daemonizing. Draining and health checking are up to the preparation commands. + +App stdout and stderr are shipped to `/ravion/ec2/`. Streams use `deployment//instance/`, which keeps every deployment and EC2 instance separate. The SSM deploy script tees its stdout and stderr into the same instance stream while preserving the native SSM command output. + +New instances launched by the Auto Scaling Group boot from the launch template but hold no release until the orchestrator repeats the deploy against them. + +## Usage + +### Web service running a container + +```hcl +module "web" { + source = "git::https://github.com/flightcontrolhq/modules.git//compute/ec2_service?ref=v1.0.0" + + name = "my-web" + vpc_id = "vpc-0123456789abcdef0" + subnet_ids = ["subnet-aaa", "subnet-bbb"] + + runtime = "container" + instance_type = "t3.small" + min_size = 2 + max_size = 4 + + app_port = 3000 + deploy_health_check_path = "/health" + + ecr_repository_creation_enabled = true + + load_balancer_attachment = { + target_group = { + port = 3000 + health_check = { + path = "/health" + } + } + listener_rules = [ + { + listener_arn = "arn:aws:elasticloadbalancing:us-east-1:123456789012:listener/app/shared/abc/def" + conditions = [ + { type = "host-header", values = ["app.example.com"] } + ] + } + ] + } + load_balancer_security_group_id = "sg-0123456789abcdef0" + + environment_variables = [ + { name = "NODE_ENV", value = "production" } + ] + secrets = [ + { name = "DATABASE_URL", value_from = "arn:aws:ssm:us-east-1:123456789012:parameter/my-app/database-url" } + ] +} +``` + +### Worker with manual command deploys + +```hcl +module "worker" { + source = "git::https://github.com/flightcontrolhq/modules.git//compute/ec2_service?ref=v1.0.0" + + name = "my-worker" + vpc_id = "vpc-0123456789abcdef0" + subnet_ids = ["subnet-aaa", "subnet-bbb"] + + runtime = "manual" + instance_type = "t3.small" + + manual_start_command = "cd /srv/app && ./bin/worker" + + data_volume_creation_enabled = true + data_volume_size = 50 +} +``` + +Deploys then run your release preparation commands on every instance, with the app env file refreshed and loaded first: + +```sh +aws ssm send-command \ + --document-name my-worker-deploy \ + --targets Key=tag:aws:autoscaling:groupName,Values=my-worker \ + --parameters commands='["cd /srv/app && git pull","cd /srv/app && ./bin/migrate"]' +``` + +## Stable storage + +- The root volume and optional data volume are per-instance EBS. They survive every in-place deploy because instances are not replaced, but they are lost when the Auto Scaling Group terminates the instance. +- For storage that must survive instance termination, mount an EFS file system (`efs_*` variables); it is mounted on every instance and, for the container runtime, bind-mounted into the app container. +- Launch template changes (AMI, user data, volumes) intentionally apply only to newly launched instances; there is no instance refresh. + +## Requirements + +| Name | Version | +|------|---------| +| opentofu/terraform | >= 1.10.0 | +| aws | >= 6.0 | + +Instances need outbound access to SSM, ECR/S3, CloudWatch Logs, PyPI for the pinned Supervisor installation, and (when secrets are configured) Secrets Manager (NAT gateway, public IPs, or VPC endpoints). The default AMI is Amazon Linux 2023; custom AMIs must run cloud-init and include the SSM agent. + +## Inputs + +| Name | Description | Type | Default | Required | +|------|-------------|------|---------|----------| +| name | Name prefix for all resources (1-28 chars) | `string` | n/a | yes | +| tags | Tags for all resources | `map(string)` | `{}` | no | +| region | AWS region (null = provider region) | `string` | `null` | no | +| vpc_id | VPC for the instances | `string` | n/a | yes | +| subnet_ids | Subnets for the Auto Scaling Group | `list(string)` | n/a | yes | +| public_ip_assignment_enabled | Assign public IPs to instances | `bool` | `false` | no | +| additional_security_group_ids | Extra security groups on the instances | `list(string)` | `[]` | no | +| direct_access_cidr_blocks | IPv4 CIDRs allowed to reach the app port directly | `list(string)` | `[]` | no | +| runtime | `container` or `manual` | `string` | n/a | yes | +| app_port | Port the app listens on | `number` | `null` | no | +| container_start_command | Optional container start command overriding the image CMD | `string` | `null` | no | +| manual_start_command | Long-running foreground manual app command managed by supervisord | `string` | `null` | yes for manual | +| environment_variables | Plain env vars written to the app env file | `list(object)` | `[]` | no | +| secrets | Secret env vars fetched on-instance from Secrets Manager / SSM Parameter Store (`{name, value_from}`) | `list(object)` | `[]` | no | +| deploy_health_check_path | Local HTTP path gating deploy success | `string` | `null` | no | +| deploy_timeout_seconds | Per-instance deploy script timeout | `number` | `1200` | no | +| instance_type | EC2 instance type | `string` | n/a | yes | +| ami_id | Custom AMI (null = latest AL2023) | `string` | `null` | no | +| key_name | SSH key pair name | `string` | `null` | no | +| root_volume_size | Root EBS volume size (GB) | `number` | `30` | no | +| root_volume_type | Root EBS volume type | `string` | `"gp3"` | no | +| data_volume_creation_enabled | Attach a formatted per-instance data volume | `bool` | `false` | no | +| data_volume_size | Data volume size (GB) | `number` | `20` | no | +| data_volume_type | Data volume type | `string` | `"gp3"` | no | +| data_volume_mount_path | Host mount path for the data volume | `string` | `"/data"` | no | +| additional_user_data | Extra shell script appended to bootstrap | `string` | `""` | no | +| min_size | Minimum instances | `number` | `1` | no | +| max_size | Maximum instances | `number` | `3` | no | +| desired_capacity | Desired instances (null = group-managed) | `number` | `null` | no | +| health_check_type | ASG health check: `EC2` or `ELB` | `string` | `"EC2"` | no | +| health_check_grace_period | Seconds before ASG health checks apply | `number` | `300` | no | +| cpu_autoscaling_enabled | Scale on average CPU utilization | `bool` | `false` | no | +| cpu_target_value | CPU utilization target (%) | `number` | `70` | no | +| load_balancer_attachment | Target group + listener rules (null = worker mode) | `object` | `null` | no | +| load_balancer_security_group_id | ALB security group allowed to reach the app port | `string` | `null` | no | +| efs_enabled | Mount an EFS file system on every instance | `bool` | `false` | no | +| efs_file_system_id | EFS file system ID | `string` | `null` | no | +| efs_access_point_id | EFS access point to mount through | `string` | `null` | no | +| efs_client_security_group_id | EFS client security group attached to instances | `string` | `null` | no | +| efs_mount_path | Host mount path for EFS | `string` | `"/mnt/efs"` | no | +| ecr_repository_creation_enabled | Create an ECR repository for built images | `bool` | `false` | no | +| ecr_force_deletion_enabled | Delete the ECR repository even with images | `bool` | `false` | no | +| ecr_scan_on_push_enabled | Scan images for vulnerabilities after push | `bool` | `true` | no | +| log_retention_in_days | CloudWatch app log retention | `number` | `30` | no | + +## Outputs + +| Name | Description | +|------|-------------| +| autoscaling_group_name | Name of the Auto Scaling Group deploys target | +| autoscaling_group_arn | ARN of the Auto Scaling Group | +| ssm_document_name | Name of the SSM deploy document | +| ssm_document_arn | ARN of the SSM deploy document | +| ecr_repository_arn | ARN of the service ECR repository (when created) | +| ecr_repository_url | URL of the service ECR repository (when created) | +| target_group_arn | ARN of the service target group (when attached) | +| target_group_arn_suffix | Target group ARN suffix for CloudWatch dimensions | +| security_group_id | ID of the instance security group | +| instance_role_arn | ARN of the instance IAM role | +| log_group_name | CloudWatch log group receiving app logs | +| log_stream_prefix | Prefix of deployment- and instance-scoped app log streams | +| aws_account_id | AWS account ID | +| region | AWS region | diff --git a/compute/ec2_service/autoscaling.tf b/compute/ec2_service/autoscaling.tf new file mode 100644 index 00000000..035166ff --- /dev/null +++ b/compute/ec2_service/autoscaling.tf @@ -0,0 +1,72 @@ +################################################################################ +# Auto Scaling Group +################################################################################ + +module "autoscaling" { + source = "../autoscaling" + + name = var.name + vpc_zone_identifier = var.subnet_ids + + # Capacity + min_size = var.min_size + max_size = var.max_size + desired_capacity = var.desired_capacity + + # Health checks. EC2 by default: in-place deploys briefly deregister + # instances from the target group, which ELB health checks would treat + # as unhealthy and replace mid-deploy. + health_check_type = var.health_check_type + health_check_grace_period = var.health_check_grace_period + + # Use existing launch template (don't create new one) + launch_template_creation_enabled = false + launch_template_id = aws_launch_template.app.id + launch_template_version = "$Latest" + + # Register instances with the service target group + target_group_arns = local.load_balancer_creation_enabled ? [aws_lb_target_group.app[0].arn] : [] + + enabled_metrics = [ + "GroupDesiredCapacity", + "GroupInServiceInstances", + ] + + # Visibility-only lifecycle hooks: they emit "EC2 Instance-launch/-terminate + # Lifecycle Action" EventBridge events that Ravion ingests to show instances + # while they are still Pending/Terminating. Nothing completes the action, so + # the minimum 30s heartbeat with CONTINUE keeps the added launch/terminate + # delay as small as possible. + lifecycle_hooks = [ + { + name = "ravion-launch-visibility" + lifecycle_transition = "autoscaling:EC2_INSTANCE_LAUNCHING" + default_result = "CONTINUE" + heartbeat_timeout = 30 + }, + { + name = "ravion-terminate-visibility" + lifecycle_transition = "autoscaling:EC2_INSTANCE_TERMINATING" + default_result = "CONTINUE" + heartbeat_timeout = 30 + }, + ] + + scaling_policies = var.cpu_autoscaling_enabled ? [ + { + name = "${var.name}-cpu-target-tracking" + policy_type = "TargetTrackingScaling" + + target_tracking_configuration = { + target_value = var.cpu_target_value + predefined_metric_specification = { + predefined_metric_type = "ASGAverageCPUUtilization" + } + } + } + ] : [] + + tags = merge(local.tags, { + Name = var.name + }) +} diff --git a/compute/ec2_service/cloudwatch.tf b/compute/ec2_service/cloudwatch.tf new file mode 100644 index 00000000..6cf55f59 --- /dev/null +++ b/compute/ec2_service/cloudwatch.tf @@ -0,0 +1,14 @@ +################################################################################ +# CloudWatch Log Group +# +# App stdout and stderr from every instance land here. Supervisord writes +# each release to its own file, and the CloudWatch agent publishes it to a +# stream scoped by deployment ID and instance ID. +################################################################################ + +resource "aws_cloudwatch_log_group" "app" { + name = local.log_group_name + retention_in_days = var.log_retention_in_days + + tags = local.tags +} diff --git a/compute/ec2_service/data.tf b/compute/ec2_service/data.tf new file mode 100644 index 00000000..1d1da38d --- /dev/null +++ b/compute/ec2_service/data.tf @@ -0,0 +1,27 @@ +################################################################################ +# Data Sources +################################################################################ + +# Architecture of the selected instance type, used to pick the matching +# default AMI. No user input needed: the instance type determines it. +data "aws_ec2_instance_type" "selected" { + count = var.ami_id == null ? 1 : 0 + + instance_type = var.instance_type +} + +# Latest Amazon Linux 2023 AMI for the instance type's CPU architecture +data "aws_ssm_parameter" "al2023_ami" { + count = var.ami_id == null ? 1 : 0 + + name = "/aws/service/ami-amazon-linux-latest/al2023-ami-kernel-default-${local.cpu_architecture}" +} + +# Get current AWS region +data "aws_region" "current" {} + +# Get current AWS account ID +data "aws_caller_identity" "current" {} + +# Get current AWS partition +data "aws_partition" "current" {} diff --git a/compute/ec2_service/ecr.tf b/compute/ec2_service/ecr.tf new file mode 100644 index 00000000..2ae6c663 --- /dev/null +++ b/compute/ec2_service/ecr.tf @@ -0,0 +1,20 @@ +################################################################################ +# ECR Repository +# +# Optional. Holds images built for this service when the container runtime +# builds from source. The instance role gets pull permissions on it. +################################################################################ + +module "ecr" { + count = var.ecr_repository_creation_enabled ? 1 : 0 + + source = "../../containers/ecr" + + name = var.name + tags = var.tags + + image_scan_on_push_enabled = var.ecr_scan_on_push_enabled + force_delete_enabled = var.ecr_force_deletion_enabled + + default_lifecycle_policy_enabled = true +} diff --git a/compute/ec2_service/iam.tf b/compute/ec2_service/iam.tf new file mode 100644 index 00000000..e247ee0f --- /dev/null +++ b/compute/ec2_service/iam.tf @@ -0,0 +1,174 @@ +################################################################################ +# IAM Role for Service Instances +################################################################################ + +resource "aws_iam_role" "instance" { + name = "${var.name}-instance" + + assume_role_policy = jsonencode({ + Version = "2012-10-17" + Statement = [ + { + Action = "sts:AssumeRole" + Effect = "Allow" + Principal = { + Service = "ec2.amazonaws.com" + } + } + ] + }) + + tags = local.tags +} + +# SSM agent access: required for the SSM-based deploy documents and +# Session Manager shell access. +resource "aws_iam_role_policy_attachment" "instance_ssm" { + role = aws_iam_role.instance.name + policy_arn = "arn:aws:iam::aws:policy/AmazonSSMManagedInstanceCore" +} + +resource "aws_iam_instance_profile" "instance" { + name = "${var.name}-instance" + role = aws_iam_role.instance.name + + tags = local.tags +} + +################################################################################ +# Instance Permissions +################################################################################ + +data "aws_iam_policy_document" "instance" { + # SSM Run Command resolves the configured CloudWatch destination before + # creating its per-command stdout/stderr streams. DescribeLogGroups does + # not support resource-level permissions. + statement { + sid = "DescribeAppLogGroups" + actions = ["logs:DescribeLogGroups"] + resources = ["*"] + } + + # App log shipping (Docker awslogs driver and CloudWatch agent) + statement { + sid = "AppLogs" + actions = [ + "logs:CreateLogStream", + "logs:PutLogEvents", + "logs:DescribeLogStreams", + ] + resources = [ + aws_cloudwatch_log_group.app.arn, + "${aws_cloudwatch_log_group.app.arn}:*", + ] + } + + # Container image pulls from the service ECR repository + dynamic "statement" { + for_each = local.container_runtime ? [1] : [] + content { + sid = "EcrAuth" + actions = ["ecr:GetAuthorizationToken"] + resources = ["*"] + } + } + + # Manual deploy source checkout receives only the name of a temporary + # SecureString. The credential itself is fetched on the instance and is + # deleted by the deploy workflow after the command finishes. + statement { + sid = "GitDeployTokenRead" + actions = [ + "ssm:GetParameter", + ] + resources = [ + "arn:${data.aws_partition.current.partition}:ssm:*:${data.aws_caller_identity.current.account_id}:parameter/ravion/git-tokens/ec2/*" + ] + } + + dynamic "statement" { + for_each = local.container_runtime && var.ecr_repository_creation_enabled ? [1] : [] + content { + sid = "EcrPull" + actions = [ + "ecr:BatchGetImage", + "ecr:GetDownloadUrlForLayer", + "ecr:BatchCheckLayerAvailability", + ] + resources = [module.ecr[0].repository_arn] + } + } + + # Connection draining around in-place deploys + dynamic "statement" { + for_each = local.load_balancer_creation_enabled ? [1] : [] + content { + sid = "TargetGroupDrain" + actions = [ + "elasticloadbalancing:RegisterTargets", + "elasticloadbalancing:DeregisterTargets", + ] + resources = [aws_lb_target_group.app[0].arn] + } + } + + dynamic "statement" { + for_each = local.load_balancer_creation_enabled ? [1] : [] + content { + sid = "TargetGroupDescribe" + actions = ["elasticloadbalancing:DescribeTargetHealth"] + resources = ["*"] + } + } + + # Secret env vars: the env-file builder fetches values on the instance. + # Same-account grants matching the ECS execution role's secrets access + # (compute/ecs_service/task_definition.tf). + dynamic "statement" { + for_each = length(var.secrets) > 0 ? [1] : [] + content { + sid = "SecretsManagerRead" + actions = ["secretsmanager:GetSecretValue"] + resources = ["*"] + condition { + test = "StringEquals" + variable = "aws:ResourceAccount" + values = [data.aws_caller_identity.current.account_id] + } + } + } + + dynamic "statement" { + for_each = length(var.secrets) > 0 ? [1] : [] + content { + sid = "SsmParameterRead" + actions = [ + "ssm:GetParameter", + "ssm:GetParameters", + ] + resources = [ + "arn:${data.aws_partition.current.partition}:ssm:*:${data.aws_caller_identity.current.account_id}:parameter/*" + ] + } + } + + dynamic "statement" { + for_each = length(var.secrets) > 0 ? [1] : [] + content { + sid = "SecretsKmsDecrypt" + actions = ["kms:Decrypt"] + resources = ["*"] + condition { + test = "StringEquals" + variable = "aws:ResourceAccount" + values = [data.aws_caller_identity.current.account_id] + } + } + } +} + +resource "aws_iam_role_policy" "instance" { + name = "${var.name}-instance" + role = aws_iam_role.instance.id + policy = data.aws_iam_policy_document.instance.json +} diff --git a/compute/ec2_service/launch_template.tf b/compute/ec2_service/launch_template.tf new file mode 100644 index 00000000..17742cbf --- /dev/null +++ b/compute/ec2_service/launch_template.tf @@ -0,0 +1,113 @@ +################################################################################ +# Launch Template +# +# Launch template changes (AMI, user data, volumes) apply to newly +# launched instances only. Existing instances are intentionally never +# replaced by this module so in-place state on them is preserved; recycle +# instances manually when a bootstrap change must roll out. +################################################################################ + +resource "aws_launch_template" "app" { + name = var.name + + image_id = local.ami_id + instance_type = var.instance_type + key_name = var.key_name + + user_data = local.user_data + + iam_instance_profile { + arn = aws_iam_instance_profile.instance.arn + } + + network_interfaces { + associate_public_ip_address = var.public_ip_assignment_enabled + security_groups = concat( + [module.instance_security_group.security_group_id], + var.efs_client_security_group_id != null ? [var.efs_client_security_group_id] : [], + var.additional_security_group_ids + ) + } + + block_device_mappings { + device_name = "/dev/xvda" + + ebs { + volume_size = var.root_volume_size + volume_type = var.root_volume_type + encrypted = true + delete_on_termination = true + } + } + + dynamic "block_device_mappings" { + for_each = var.data_volume_creation_enabled ? [1] : [] + content { + device_name = "/dev/xvdf" + + ebs { + volume_size = var.data_volume_size + volume_type = var.data_volume_type + encrypted = true + delete_on_termination = true + } + } + } + + metadata_options { + http_endpoint = "enabled" + http_tokens = "required" + http_put_response_hop_limit = 2 + } + + monitoring { + enabled = true + } + + tag_specifications { + resource_type = "instance" + + tags = merge(local.tags, { + Name = var.name + }) + } + + tag_specifications { + resource_type = "volume" + + tags = merge(local.tags, { + Name = var.name + }) + } + + tags = local.tags + + lifecycle { + create_before_destroy = true + + precondition { + condition = !local.load_balancer_creation_enabled || var.app_port != null + error_message = "The app_port is required when a load balancer is attached." + } + + precondition { + condition = var.deploy_health_check_path == null || var.app_port != null + error_message = "The app_port is required when deploy_health_check_path is set." + } + + precondition { + condition = !var.efs_enabled || var.efs_file_system_id != null + error_message = "The efs_file_system_id is required when efs_enabled is true." + } + + precondition { + condition = var.min_size <= var.max_size + error_message = "The min_size must not be greater than max_size." + } + + precondition { + condition = var.desired_capacity == null || (var.desired_capacity >= var.min_size && var.desired_capacity <= var.max_size) + error_message = "The desired_capacity must be between min_size and max_size." + } + } +} diff --git a/compute/ec2_service/locals.tf b/compute/ec2_service/locals.tf new file mode 100644 index 00000000..7b8061dc --- /dev/null +++ b/compute/ec2_service/locals.tf @@ -0,0 +1,122 @@ +################################################################################ +# Local Values +################################################################################ + +locals { + # Default tags for all resources + default_tags = { + ManagedBy = "terraform" + Module = "compute/ec2_service" + } + + tags = merge(local.default_tags, var.tags) + + region = coalesce(var.region, data.aws_region.current.region) + + container_runtime = var.runtime == "container" + + load_balancer_creation_enabled = var.load_balancer_attachment != null ? var.load_balancer_attachment.creation_enabled : false + + cpu_architecture = var.ami_id == null ? ( + contains(data.aws_ec2_instance_type.selected[0].supported_architectures, "arm64") ? "arm64" : "x86_64" + ) : null + + ami_id = var.ami_id != null ? var.ami_id : data.aws_ssm_parameter.al2023_ami[0].value + + log_group_name = "/ravion/ec2/${var.name}" + + # App layout shared by both deploy modes. Each deployment gets its own + # app log file and CloudWatch stream under the service log group. + env_file_path = "/etc/ravion/${var.name}.env" + log_directory = "/var/log/ravion/${var.name}" + supervisor_program = "ravion-${var.name}" + supervisor_conf = "/etc/supervisord.d/${var.name}.ini" + app_runner_path = "/usr/local/bin/ravion-${var.name}-run" + image_ref_path = "/etc/ravion/${var.name}.image" + start_command_path = "/etc/ravion/${var.name}.start-command" + source_working_directory_path = "/etc/ravion/${var.name}.source-working-directory" + + supervisor_install_script = templatefile("${path.module}/templates/install_supervisor.sh.tpl", {}) + + deployment_log_script = templatefile("${path.module}/templates/configure_deployment_logs.sh.tpl", { + log_directory = local.log_directory + log_group_name = local.log_group_name + }) + + supervisor_program_script = templatefile("${path.module}/templates/configure_supervisor_program.sh.tpl", { + app_runner_path = local.app_runner_path + supervisor_conf = local.supervisor_conf + supervisor_program = local.supervisor_program + }) + + # Env-file builder script shared by instance boot and both deploy modes. + # Plain values are Terraform-rendered (like + # an ECS task definition's environment); secrets are fetched on the + # instance so their values stay out of state and the SSM document. + env_file_script = templatefile("${path.module}/templates/env_file.sh.tpl", { + environment_variables = var.environment_variables + secrets = var.secrets + app_port = var.app_port + env_file_path = local.env_file_path + region = local.region + }) + + manual_deploy_prelude = templatefile("${path.module}/templates/deploy_manual_before.sh.tpl", { + deployment_log_script = local.deployment_log_script + env_file_path = local.env_file_path + env_file_script = local.env_file_script + git_source_checkout_script = templatefile("${path.module}/templates/checkout_git_source.sh.tpl", { + name = var.name + region = local.region + source_working_directory_path = local.source_working_directory_path + }) + name = var.name + supervisor_install_script = local.supervisor_install_script + supervisor_program = local.supervisor_program + }) + + manual_deploy_postlude = templatefile("${path.module}/templates/deploy_manual_after.sh.tpl", { + app_runner_path = local.app_runner_path + env_file_path = local.env_file_path + manual_start_command_base64 = base64encode(var.manual_start_command != null ? var.manual_start_command : "") + start_command_path = local.start_command_path + source_working_directory_path = local.source_working_directory_path + supervisor_program = local.supervisor_program + supervisor_program_script = local.supervisor_program_script + }) + + deploy_script = local.container_runtime ? templatefile("${path.module}/templates/deploy_container.sh.tpl", { + name = var.name + region = local.region + app_port = var.app_port + start_command = var.container_start_command != null ? var.container_start_command : "" + deploy_health_check_path = var.deploy_health_check_path != null ? var.deploy_health_check_path : "" + env_file_script = local.env_file_script + env_file_path = local.env_file_path + app_runner_path = local.app_runner_path + deployment_log_script = local.deployment_log_script + image_ref_path = local.image_ref_path + supervisor_conf = local.supervisor_conf + supervisor_install_script = local.supervisor_install_script + supervisor_program = local.supervisor_program + supervisor_program_script = local.supervisor_program_script + target_group_arn = local.load_balancer_creation_enabled ? aws_lb_target_group.app[0].arn : "" + data_volume_mount_path = var.data_volume_creation_enabled ? var.data_volume_mount_path : "" + efs_mount_path = var.efs_enabled ? var.efs_mount_path : "" + }) : null + + user_data = base64encode(templatefile("${path.module}/templates/user_data.sh.tpl", { + name = var.name + region = local.region + env_file_script = local.env_file_script + env_file_path = local.env_file_path + supervisor_install_script = local.supervisor_install_script + data_volume_creation_enabled = var.data_volume_creation_enabled + data_volume_mount_path = var.data_volume_mount_path + efs_enabled = var.efs_enabled + efs_file_system_id = var.efs_file_system_id != null ? var.efs_file_system_id : "" + efs_access_point_id = var.efs_access_point_id != null ? var.efs_access_point_id : "" + efs_mount_path = var.efs_mount_path + additional_user_data = var.additional_user_data + })) +} diff --git a/compute/ec2_service/outputs.tf b/compute/ec2_service/outputs.tf new file mode 100644 index 00000000..65b33e19 --- /dev/null +++ b/compute/ec2_service/outputs.tf @@ -0,0 +1,97 @@ +################################################################################ +# Auto Scaling Group +################################################################################ + +output "autoscaling_group_name" { + description = "The name of the Auto Scaling Group. Deploys target this group's instances." + value = module.autoscaling.autoscaling_group_name +} + +output "autoscaling_group_arn" { + description = "The ARN of the Auto Scaling Group." + value = module.autoscaling.autoscaling_group_arn +} + +################################################################################ +# Deploy Contract +################################################################################ + +output "ssm_document_name" { + description = "The name of the SSM deploy document run on each instance by the deploy manager." + value = aws_ssm_document.deploy.name +} + +output "ssm_document_arn" { + description = "The ARN of the SSM deploy document." + value = aws_ssm_document.deploy.arn +} + +################################################################################ +# Artifact Stores +################################################################################ + +output "ecr_repository_arn" { + description = "The ARN of the service ECR repository, when created." + value = var.ecr_repository_creation_enabled ? module.ecr[0].repository_arn : null +} + +output "ecr_repository_url" { + description = "The URL of the service ECR repository, when created." + value = var.ecr_repository_creation_enabled ? module.ecr[0].repository_url : null +} + +################################################################################ +# Load Balancer +################################################################################ + +output "target_group_arn" { + description = "The ARN of the service target group, when a load balancer is attached." + value = local.load_balancer_creation_enabled ? aws_lb_target_group.app[0].arn : null +} + +output "target_group_arn_suffix" { + description = "The ARN suffix of the service target group for CloudWatch dimensions." + value = local.load_balancer_creation_enabled ? aws_lb_target_group.app[0].arn_suffix : null +} + +################################################################################ +# Networking and IAM +################################################################################ + +output "security_group_id" { + description = "The ID of the instance security group." + value = module.instance_security_group.security_group_id +} + +output "instance_role_arn" { + description = "The ARN of the instance IAM role." + value = aws_iam_role.instance.arn +} + +################################################################################ +# Logging +################################################################################ + +output "log_group_name" { + description = "The name of the CloudWatch log group receiving app logs." + value = aws_cloudwatch_log_group.app.name +} + +output "log_stream_prefix" { + description = "Prefix of deployment- and instance-scoped app log streams inside the log group." + value = "deployment" +} + +################################################################################ +# Context +################################################################################ + +output "aws_account_id" { + description = "The AWS account ID where resources are created." + value = data.aws_caller_identity.current.account_id +} + +output "region" { + description = "The AWS region where resources are created." + value = local.region +} diff --git a/compute/ec2_service/provider.tf b/compute/ec2_service/provider.tf new file mode 100644 index 00000000..dc58d9a2 --- /dev/null +++ b/compute/ec2_service/provider.tf @@ -0,0 +1,3 @@ +provider "aws" { + region = var.region +} diff --git a/compute/ec2_service/rvn-ec2-service-definition.yml b/compute/ec2_service/rvn-ec2-service-definition.yml new file mode 100644 index 00000000..3eb5528f --- /dev/null +++ b/compute/ec2_service/rvn-ec2-service-definition.yml @@ -0,0 +1,1067 @@ +definition: + type: rvn-ec2-service + name: EC2 Service + description: Runs supervised workloads on a stable EC2 Auto Scaling Group, with optional shared ALB routing and switchable container or manual in-place deploys. +release: + version: 0.1.1 + description: Reorganize EC2 service inputs and clarify web and storage configuration. +module: + inputs: + - id: network + immutable: true + label: VPC network + type: $ref:rvn-aws-network + description: Existing Ravion network that supplies the AWS account, region, VPC, and public and private subnets for the instances. + mapped_inputs: + - $template: ../../partials/templates/network-ref-mapped-inputs.yml + with: + aws_account_id_default: << ref.input.aws_account_id >> + aws_region_default: << ref.input.aws_region >> + execution_environment_default: << ref.input.execution_environment_id >> + private_subnet_ids_default: <> + public_subnet_ids_default: <> + vpc_id_default: <> + required: true + - $include: ../../partials/inputs/private-subnet-placement.yml + - id: section_service + label: EC2 service + type: section + - id: name + immutable: true + label: Service name + type: string + description: Name for the instance group and related resources. + required: true + default: <>-<>-<> + patterns: + - message: 1-28 lowercase letters, numbers, and hyphens. Start and end with a letter or number. + pattern: ^[a-z0-9]([a-z0-9-]{0,26}[a-z0-9])?$ + - id: instance_type + label: Instance type + type: string + required: true + no_options_message: Select a VPC Network, or enter an AWS account and region, to load available EC2 instance types. + values: $values:aws/ec2/instances?awsAccountId=<>®ion=<> + - id: ami_id + label: Custom AMI ID + type: string + description: Custom AMI for new instances. Leave blank for the latest architecture-matched Amazon Linux 2023 AMI. + collapsible: true + placeholder: ami-... + required: false + - id: key_name + label: SSH key pair name + type: string + description: Optional EC2 key pair for SSH access. + collapsible: true + required: false + - id: additional_user_data + label: Additional user data + type: text + description: Shell script run as root after Ravion's bootstrap on every newly launched instance. + collapsible: true + required: false + - id: http_traffic_enabled + label: Serve HTTP traffic + type: boolean + description: Off for a worker. On to serve HTTP traffic through a load balancer. + default: true + - id: deploy_type + label: Deploy type + type: string + description: Choose in-place container image deploys or host-level shell commands. + required: true + default: container + values: + - label: Container + description: Pull an image and replace the supervised Docker container on each existing instance. + value: container + - label: Manual + description: Run release preparation commands on each instance, then start a supervised foreground command on the host. + value: manual + - id: section_build + label: Build config + show_when: + deploy_type: container + type: section + - id: build_source + label: Build source + type: string + description: Build a container image from source with Dockerfile or Railpack, or deploy an existing image from a configured registry repository. + required: true + show_when: + deploy_type: container + default: dockerfile + values: + - description: Build an image from a Dockerfile in the selected Git repository. + label: Dockerfile + value: dockerfile + - description: Let Railpack detect the application and build a container image from source. + label: Railpack + value: railpack + - description: Deploy an image from a registry configured on this module. Provide only the tag or digest at deploy time. + label: Pull from image registry + value: image_registry + - $template: ../../partials/inputs/build-git-source.yml + with: + show_when: + deploy_type: container + build_source: + - dockerfile + - railpack + - id: image_repository + label: Image repository + type: string + description: Repository without a tag or digest, such as `nginx` or `123456789012.dkr.ecr.us-east-1.amazonaws.com/app`. Registries requiring Docker credentials are unsupported. Same-region private ECR also needs a repository policy that lets the service instance role pull images. + placeholder: nginx + required: true + show_when: + deploy_type: container + build_source: image_registry + - id: container_start_command + label: Start command override + type: string + description: Optional command string that overrides the image CMD. The image ENTRYPOINT is preserved. + collapsible: true + required: false + show_when: + deploy_type: container + - $template: ../../partials/inputs/dockerfile-build.yml + with: + show_when: + deploy_type: container + build_source: dockerfile + - $template: ../../partials/inputs/railpack.yml + with: + show_when: + deploy_type: container + build_source: railpack + - id: railpack_start_cmd + label: Start command + type: string + description: Optional start command embedded in the Railpack-built image. Leave blank to use Railpack detection. + placeholder: Railpack default + show_when: + deploy_type: container + build_source: railpack + - id: section_deployment + label: Deployment + type: section + - id: deploy_source_repo + label: Git repository + type: gitrepo + description: Optional repository to check out before manual deploy commands run. Leave blank to run the commands without a managed source checkout. + required: false + show_when: + deploy_type: manual + - id: deploy_source_base_path + label: Source base path + type: string + description: Repository-relative working directory for manual deploy and start commands when a Git repository is selected. + required: false + default: . + show_when: + deploy_type: manual + deploy_source_repo: + not: "" + - id: deploy_commands + label: Deploy commands + type: string_array + description: Release preparation commands run as root, in order, on every instance during each manual deploy. Keep them idempotent; any command that exits non-zero fails that instance's deploy. + add_button_label: Add command + placeholder: cd /srv/app && git pull + required: true + show_when: + deploy_type: manual + - id: start_command + label: Start command + type: string + description: Long-running foreground app command that supervisord runs as root after preparation and restarts if it exits. Use a wrapper to drop privileges when needed; the command must not daemonize. + placeholder: cd /srv/app && ./bin/server + required: true + show_when: + deploy_type: manual + - id: deployment_concurrency_max + label: Maximum concurrent instances + type: string + description: Maximum instances updated at once. Use an instance count such as 1 or a percentage such as 25%. + collapsible: true + required: true + default: "1" + patterns: + - message: Positive instance count or percentage from 1% to 100%. + pattern: ^([1-9][0-9]*|[1-9][0-9]?%|100%)$ + - id: deployment_errors_max + label: Maximum deployment errors + type: string + description: Number or percentage of failed instances tolerated before the deployment stops on remaining instances. + collapsible: true + required: true + default: "0" + patterns: + - message: Zero, a positive error count, or percentage from 1% to 100%. + pattern: ^(0|[1-9][0-9]*|[1-9][0-9]?%|100%)$ + - id: deploy_timeout_seconds + label: Per-instance deploy timeout (secs) + type: number + description: Maximum time allowed for the SSM deploy script on each instance, for both container and manual deploys. + collapsible: true + max: 14400 + min: 60 + default: 1200 + - id: section_web + label: HTTP routing + show_when: + http_traffic_enabled: true + type: section + - id: load_balancer_source + immutable: true + label: Load balancer source + type: string + description: Must use the same AWS account, region, and VPC as the VPC network. + required: true + default: standalone_alb + show_when: + http_traffic_enabled: true + values: + - label: Standalone ALB module + description: Attach to a standalone Ravion Application Load Balancer. + value: standalone_alb + - label: ALB from ECS Cluster module + description: Attach to the public or private ALB created by a Ravion ECS cluster. + value: ecs_cluster + - id: alb + immutable: true + label: Standalone ALB module + type: $ref:rvn-aws-alb + description: Attach listener rules to this. + mapped_inputs: + - default: <> + id: alb_http_listener_arn + immutable: true + label: ALB HTTP listener ARN + required: false + type: string + - default: <> + id: alb_https_listener_arn + immutable: true + label: ALB HTTPS listener ARN + required: false + type: string + - default: <> + id: alb_security_group_id + immutable: true + label: ALB security group ID + required: true + type: string + - default: <> + id: alb_arn_suffix + immutable: true + label: ALB ARN suffix + required: true + type: string + required: true + show_when: + http_traffic_enabled: true + load_balancer_source: standalone_alb + - id: ecs_cluster + immutable: true + label: ECS Cluster + type: $ref:rvn-ecs-cluster + description: Attach listener rules to the ALB in this cluster. + mapped_inputs: + - $include: ../../partials/inputs/ecs-service-cluster-alb-mapped-inputs.yml + - default: <> + id: public_alb_arn_suffix + immutable: true + label: Public ALB ARN suffix + type: string + - default: <> + id: private_alb_arn_suffix + immutable: true + label: Private ALB ARN suffix + type: string + required: true + show_when: + http_traffic_enabled: true + load_balancer_source: ecs_cluster + - id: ecs_cluster_alb_visibility + label: ALB to use from ECS Cluster module + type: string + description: The selected load balancer must be enabled on the cluster. + required: true + default: public + show_when: + http_traffic_enabled: true + load_balancer_source: ecs_cluster + values: + - label: Public + description: Route through the cluster's internet-facing Application Load Balancer. + value: public + - label: Private + description: Route through the cluster's internal Application Load Balancer. + value: private + - id: app_port + immutable: true + label: App port + type: number + description: Port the host application listens on. Container mode publishes the same container port, and PORT is set automatically. + max: 65535 + min: 1 + required: true + show_when: + http_traffic_enabled: true + default: 80 + - $template: ../../partials/templates/alb-listener-rule-inputs.yml + with: + field_overrides: + show_when: + http_traffic_enabled: true + - id: section_health + label: Health check + type: section + - id: health_check_path + label: Health check path + type: string + description: HTTP path used by the load balancer health check and by the Container-mode deploy gate on each instance. + required: true + show_when: + http_traffic_enabled: true + default: / + - id: health_check_matcher + label: Success codes + type: string + description: HTTP status codes the ALB treats as healthy, such as 200-399. + collapsible: true + required: true + show_when: + http_traffic_enabled: true + default: 200-399 + - id: health_check_interval + label: Interval (secs) + type: number + description: Seconds between Application Load Balancer health checks. + collapsible: true + max: 300 + min: 5 + show_when: + http_traffic_enabled: true + default: 10 + - id: health_check_timeout + label: Timeout (secs) + type: number + description: Seconds the Application Load Balancer waits for a health-check response. Must be lower than the interval. + collapsible: true + max: 120 + min: 2 + show_when: + http_traffic_enabled: true + default: 5 + - id: healthy_threshold + label: Healthy threshold + type: number + description: Consecutive successful ALB checks required before an instance is healthy. + collapsible: true + max: 10 + min: 2 + show_when: + http_traffic_enabled: true + default: 2 + - id: unhealthy_threshold + label: Unhealthy threshold + type: number + description: Consecutive failed ALB checks required before an instance is unhealthy. + collapsible: true + max: 10 + min: 2 + show_when: + http_traffic_enabled: true + default: 2 + - id: health_check_grace_period + label: Instance health check grace period (secs) + type: number + description: Seconds after launch before Auto Scaling health checks apply. + collapsible: true + min: 0 + default: 300 + - id: target_group_slow_start + label: Slow start duration (secs) + type: number + description: Seconds the load balancer gradually increases traffic to a newly healthy instance. Use 0 to disable. + collapsible: true + max: 900 + min: 0 + show_when: + http_traffic_enabled: true + default: 0 + - id: deregistration_delay + label: Deregistration delay (secs) + type: number + description: Seconds allowed for in-flight requests to drain before a Container-mode in-place swap. + collapsible: true + max: 3600 + min: 0 + show_when: + http_traffic_enabled: true + default: 30 + - id: target_group_stickiness_enabled + label: Sticky sessions + type: boolean + description: Keep repeat requests on the same instance using a load balancer or application cookie. + collapsible: true + show_when: + http_traffic_enabled: true + default: false + - id: target_group_stickiness_type + label: Stickiness type + type: string + description: Use a load-balancer-generated cookie or an application cookie. + required: true + show_when: + http_traffic_enabled: true + target_group_stickiness_enabled: true + default: lb_cookie + values: + - label: Load balancer cookie + value: lb_cookie + - label: Application cookie + value: app_cookie + - id: target_group_stickiness_cookie_name + label: Application cookie name + type: string + description: Application cookie name used for target stickiness. + required: true + show_when: + http_traffic_enabled: true + target_group_stickiness_enabled: true + target_group_stickiness_type: app_cookie + - id: target_group_stickiness_cookie_duration + label: Stickiness cookie duration (secs) + type: number + description: How long the cookie keeps a client routed to the same instance. + collapsible: true + max: 604800 + min: 1 + show_when: + http_traffic_enabled: true + target_group_stickiness_enabled: true + default: 86400 + - id: section_storage + label: Storage + type: section + description: The root volume is the required boot disk, the data volume adds per-instance storage, and EFS provides shared storage that survives instance replacement. + - id: root_volume_size + label: Root volume size (GB) + type: number + description: Required encrypted boot disk for the operating system, Docker, and local files. It is deleted when the instance is terminated. + collapsible: true + max: 16384 + min: 8 + default: 30 + - id: data_volume_creation_enabled + label: Data volume + type: boolean + description: Attach a second encrypted EBS disk for per-instance files such as caches or generated artifacts. Each instance gets its own disk, which is deleted with that instance. + default: false + - id: data_volume_size + label: Data volume size (GB) + type: number + description: Size of the dedicated EBS data volume attached to each instance. + max: 16384 + min: 1 + show_when: + data_volume_creation_enabled: true + default: 20 + - id: data_volume_mount_path + label: Data volume mount path + type: string + description: Absolute host path for the data volume. + required: true + show_when: + data_volume_creation_enabled: true + default: /data + patterns: + - message: Use an absolute path. + pattern: ^/ + - id: efs_enabled + label: EFS file system + type: boolean + description: Mount a shared network filesystem that every instance can access. Files survive instance replacement and remain available at the configured mount path. + default: false + - id: efs + label: EFS file system + type: $ref:rvn-efs + mapped_inputs: + - default: <> + id: efs_file_system_id + label: EFS file system ID + type: string + - default: <> + id: efs_access_point_id + label: EFS access point ID + required: false + type: string + - default: <> + id: efs_client_security_group_id + label: EFS client security group ID + type: string + required: true + show_when: + efs_enabled: true + - id: efs_mount_path + label: EFS mount path + type: string + description: Host path where the file system is mounted. + required: true + show_when: + efs_enabled: true + default: /mnt/efs + patterns: + - message: Use an absolute path. + pattern: ^/ + - id: section_scaling + label: Scaling + type: section + - $template: ../../partials/templates/autoscaling-capacity-inputs.yml + with: + min_capacity_overrides: + label: Minimum instances + required: true + max_capacity_overrides: + label: Maximum instances + required: true + - id: cpu_autoscaling_enabled + label: CPU autoscaling + type: boolean + default: false + - id: cpu_target_value + label: CPU target (%) + type: number + description: Average EC2 CPU utilization that target-tracking autoscaling tries to maintain. + max: 100 + min: 1 + show_when: + cpu_autoscaling_enabled: true + default: 70 + - id: desired_capacity + label: Desired instances + placeholder: Defaults to minimum instances + type: number + description: Initial group size when CPU autoscaling is off. Leave blank to start at the minimum. + min: 0 + required: false + show_when: + cpu_autoscaling_enabled: false + - id: section_app_config + label: Environment variables + type: section + - $merge: + - ../../partials/inputs/build-environment-variables.yml + show_when: + deploy_type: container + build_source: + - dockerfile + - railpack + - $merge: + - ../../partials/inputs/dockerfile-env-injection.yml + show_when: + deploy_type: container + build_source: dockerfile + - id: environment_variables + label: Runtime environment variables + type: array + description: Plain, single-line environment variables written to each instance at boot and refreshed on every deploy. + - id: secrets + label: Runtime secrets + type: array + description: Secret environment variables as {name, value_from} objects fetched on each instance. + - id: section_networking + label: Networking + type: section + - id: additional_security_group_ids + label: Additional security groups + type: string_array + description: Additional security groups attached to the instances. + add_button_label: Add security group ID + collapsible: true + required: false + - id: direct_access_cidr_blocks + label: Allowed CIDR blocks + type: string_array + description: IPv4 CIDR blocks allowed to reach the app port directly, bypassing the load balancer. + add_button_label: Add CIDR block + collapsible: true + required: false + show_when: + http_traffic_enabled: true + - $template: ../../partials/templates/builder-infrastructure-inputs.yml + with: + no_options_message: Select a VPC Network, or enter an AWS account and region, to load available EC2 instance types. + show_when: + deploy_type: container + build_source: + - dockerfile + - railpack + - id: section_ecr + label: Container registry + show_when: + deploy_type: container + build_source: + - dockerfile + - railpack + type: section + - id: ecr_scan_on_push_enabled + label: Scan images on push + type: boolean + description: Scan images for vulnerabilities after they are pushed to the Ravion-created ECR repository. + collapsible: true + show_when: + deploy_type: container + build_source: + - dockerfile + - railpack + default: true + - id: ecr_force_deletion_enabled + label: Force delete image repository + type: boolean + description: Allow the Ravion-created ECR repository to be deleted while it still contains images. + collapsible: true + show_when: + deploy_type: container + build_source: + - dockerfile + - railpack + default: false + - id: section_logging + label: Logging + type: section + - id: log_retention_in_days + label: Log retention (days) + type: number + description: CloudWatch retention for app logs. + collapsible: true + min: 1 + default: 30 + - $include: ../../partials/inputs/tags.yml + - $include: ../../partials/inputs/terraform-settings.yml + stack: + $template: ../../partials/templates/opentofu-stack.yml + with: + base_path: compute/ec2_service + terraform_variables: + ...overrides: << module.input.advanced_terraform_variables >> + additional_security_group_ids: << module.input.additional_security_group_ids || [] >> + additional_user_data: << module.input.additional_user_data || "" >> + direct_access_cidr_blocks: "<< module.input.http_traffic_enabled ? (module.input.direct_access_cidr_blocks || []) : [] >>" + ami_id: << module.input.ami_id || nil >> + app_port: "<< module.input.http_traffic_enabled ? module.input.app_port : nil >>" + public_ip_assignment_enabled: "<< module.input.private_subnet_placement_enabled ? false : true >>" + cpu_autoscaling_enabled: << module.input.cpu_autoscaling_enabled >> + cpu_target_value: << module.input.cpu_target_value >> + data_volume_creation_enabled: << module.input.data_volume_creation_enabled >> + data_volume_mount_path: << module.input.data_volume_mount_path >> + data_volume_size: << module.input.data_volume_size >> + deploy_timeout_seconds: << module.input.deploy_timeout_seconds >> + desired_capacity: "<< module.input.cpu_autoscaling_enabled ? nil : (module.input.desired_capacity || nil) >>" + ecr_force_deletion_enabled: << module.input.ecr_force_deletion_enabled >> + ecr_repository_creation_enabled: '<< module.input.deploy_type == "container" && (module.input.build_source == "dockerfile" || module.input.build_source == "railpack") >>' + ecr_scan_on_push_enabled: << module.input.ecr_scan_on_push_enabled >> + efs_access_point_id: "<< module.input.efs_enabled ? (module.input.efs_access_point_id || nil) : nil >>" + efs_client_security_group_id: "<< module.input.efs_enabled ? module.input.efs_client_security_group_id : nil >>" + efs_enabled: << module.input.efs_enabled >> + efs_file_system_id: "<< module.input.efs_enabled ? module.input.efs_file_system_id : nil >>" + efs_mount_path: << module.input.efs_mount_path >> + environment_variables: << module.input.environment_variables || [] >> + health_check_grace_period: << module.input.health_check_grace_period >> + deploy_health_check_path: "<< module.input.http_traffic_enabled ? module.input.health_check_path : nil >>" + instance_type: << module.input.instance_type >> + key_name: << module.input.key_name || nil >> + load_balancer_attachment: + $template: ../../partials/templates/alb-load-balancer-attachment.yml + with: + additional_fields: {} + attachment_control: + creation_enabled: << module.input.http_traffic_enabled >> + listener_arn: >- + << !module.input.http_traffic_enabled ? "" : + module.input.load_balancer_source == "standalone_alb" ? + (module.input.alb_https_listener_arn || module.input.alb_http_listener_arn) : + module.input.ecs_cluster_alb_visibility == "public" ? + (module.input.public_alb_https_listener_arn || module.input.public_alb_http_listener_arn) : + (module.input.private_alb_https_listener_arn || module.input.private_alb_http_listener_arn) >> + target_group: + port: << module.input.app_port >> + deregistration_delay: << module.input.deregistration_delay >> + slow_start: << module.input.target_group_slow_start >> + health_check: + enabled: true + path: << module.input.health_check_path >> + matcher: << module.input.health_check_matcher >> + interval: << module.input.health_check_interval >> + timeout: << module.input.health_check_timeout >> + healthy_threshold: << module.input.healthy_threshold >> + unhealthy_threshold: << module.input.unhealthy_threshold >> + stickiness: >- + << module.input.target_group_stickiness_enabled ? { + "enabled": true, + "type": module.input.target_group_stickiness_type, + "cookie_duration": module.input.target_group_stickiness_cookie_duration, + "cookie_name": module.input.target_group_stickiness_type == "app_cookie" ? module.input.target_group_stickiness_cookie_name : nil + } : nil >> + load_balancer_security_group_id: >- + << !module.input.http_traffic_enabled ? nil : + module.input.load_balancer_source == "standalone_alb" ? module.input.alb_security_group_id : + module.input.ecs_cluster_alb_visibility == "public" ? module.input.public_alb_security_group_id : + module.input.private_alb_security_group_id >> + log_retention_in_days: << module.input.log_retention_in_days >> + manual_start_command: '<< module.input.deploy_type == "manual" ? module.input.start_command : nil >>' + max_size: << module.input.max_capacity >> + min_size: << module.input.min_capacity >> + name: << module.input.name >> + region: << module.input.aws_region >> + root_volume_size: << module.input.root_volume_size >> + runtime: << module.input.deploy_type >> + secrets: << module.input.secrets || [] >> + container_start_command: '<< module.input.deploy_type == "container" ? (module.input.container_start_command || nil) : nil >>' + subnet_ids: "<< module.input.private_subnet_placement_enabled ? module.input.private_subnet_ids : module.input.public_subnet_ids >>" + tags: + $include: ../../partials/stack/ravion-tags.yml + vpc_id: << module.input.vpc_id >> + build: + $merge: ../../partials/build/ecs-service-image-build.yml + type: '<< module.input.deploy_type == "manual" || module.input.build_source == "image_registry" ? "disabled" : "image" >>' + deploy: + type: aws:ec2 + concurrency: + queue_overflow: oldest + queue_size: 1 + timeout: 86400 + strategy: + type: rolling + concurrency_max: << module.input.deployment_concurrency_max >> + errors_max: << module.input.deployment_errors_max >> + inputs: + - id: image_ref + label: Image tag or digest + type: string + description: Image tag or digest to deploy. + placeholder: sha256:... or latest or -railpack + patterns: + - message: Image tags and digests must not contain whitespace. + pattern: ^\S+$ + required: true + show_when: + deploy_type: container + - id: branch + label: Git branch + type: string + description: Branch to deploy. Defaults to the repository's default branch. + required: false + show_when: + deploy_type: manual + - id: ref + label: Git ref (commit or tag) + type: string + description: Optional commit SHA, tag, or ref to deploy. + required: false + show_when: + deploy_type: manual + source: >- + << module.input.deploy_type == "manual" && module.input.deploy_source_repo ? { + "type": "git", + "repo": module.input.deploy_source_repo, + "branch": deploy.input.branch, + "ref": deploy.input.ref, + "base_path": module.input.deploy_source_base_path || "." + } : nil >> + definition: >- + << module.input.deploy_type == "container" ? {"runtime": "container", "image_uri": + (module.input.build_source == "image_registry" ? (deploy.input.image_ref contains "sha256:" ? + module.input.image_repository + "@" + deploy.input.image_ref : module.input.image_repository + + ":" + deploy.input.image_ref) : (deploy.input.image_ref contains "sha256:" ? + stack.output.ecr_repository_url + "@" + deploy.input.image_ref : stack.output.ecr_repository_url + + ":" + deploy.input.image_ref))} : {"runtime": "manual", "commands": + module.input.deploy_commands} >> + infrastructure: + autoscaling_group_name: << stack.output.autoscaling_group_name >> + aws_account_id: << module.input.aws_account_id >> + log_group_name: << stack.output.log_group_name >> + region: << stack.output.region >> + target_group_arn: << stack.output.target_group_arn >> + ui: + logs: + - id: app_logs + name: App logs + source: + type: cloudwatch + aws_account_id: << module.input.aws_account_id >> + region: << stack.output.region >> + log_group: << stack.output.log_group_name >> + log_stream_prefix: << stack.output.log_stream_prefix >> + metrics: >- + << /* ALB metrics use LoadBalancer: and TargetGroup: dimensions. */ [{"id":"desired_instances","name":"Desired instances","type":"line","source":{"type":"cloudwatch","aws_account_id":module.input.aws_account_id,"dimensions":{"AutoScalingGroupName":stack.output.autoscaling_group_name},"name":"GroupDesiredCapacity","namespace":"AWS/AutoScaling","region":stack.output.region,"statistic":"Average"}},{"id":"in_service_instances","name":"In-service instances","type":"line","source":{"type":"cloudwatch","aws_account_id":module.input.aws_account_id,"dimensions":{"AutoScalingGroupName":stack.output.autoscaling_group_name},"name":"GroupInServiceInstances","namespace":"AWS/AutoScaling","region":stack.output.region,"statistic":"Average"}}] | + concat(module.input.http_traffic_enabled ? [{"id":"request_count","name":"Request count","type":"line","source":{"type":"cloudwatch","aws_account_id":module.input.aws_account_id,"dimensions":{"LoadBalancer":(module.input.load_balancer_source == "standalone_alb" ? module.input.alb_arn_suffix : module.input.ecs_cluster_alb_visibility == "public" ? module.input.public_alb_arn_suffix : module.input.private_alb_arn_suffix),"TargetGroup":stack.output.target_group_arn_suffix},"name":"RequestCount","namespace":"AWS/ApplicationELB","region":stack.output.region,"statistic":"Sum"}},{"id":"target_response_time","name":"Target response time","type":"line","source":{"type":"cloudwatch","aws_account_id":module.input.aws_account_id,"dimensions":{"LoadBalancer":(module.input.load_balancer_source == "standalone_alb" ? module.input.alb_arn_suffix : module.input.ecs_cluster_alb_visibility == "public" ? module.input.public_alb_arn_suffix : module.input.private_alb_arn_suffix),"TargetGroup":stack.output.target_group_arn_suffix},"name":"TargetResponseTime","namespace":"AWS/ApplicationELB","region":stack.output.region,"statistic":"Average"}},{"id":"target_4xx_errors","name":"4xx errors","type":"line","source":{"type":"cloudwatch","aws_account_id":module.input.aws_account_id,"dimensions":{"LoadBalancer":(module.input.load_balancer_source == "standalone_alb" ? module.input.alb_arn_suffix : module.input.ecs_cluster_alb_visibility == "public" ? module.input.public_alb_arn_suffix : module.input.private_alb_arn_suffix),"TargetGroup":stack.output.target_group_arn_suffix},"name":"HTTPCode_Target_4XX_Count","namespace":"AWS/ApplicationELB","region":stack.output.region,"statistic":"Sum"}},{"id":"target_5xx_errors","name":"5xx errors","type":"line","source":{"type":"cloudwatch","aws_account_id":module.input.aws_account_id,"dimensions":{"LoadBalancer":(module.input.load_balancer_source == "standalone_alb" ? module.input.alb_arn_suffix : module.input.ecs_cluster_alb_visibility == "public" ? module.input.public_alb_arn_suffix : module.input.private_alb_arn_suffix),"TargetGroup":stack.output.target_group_arn_suffix},"name":"HTTPCode_Target_5XX_Count","namespace":"AWS/ApplicationELB","region":stack.output.region,"statistic":"Sum"}},{"id":"healthy_hosts","name":"Healthy hosts","type":"line","source":{"type":"cloudwatch","aws_account_id":module.input.aws_account_id,"dimensions":{"LoadBalancer":(module.input.load_balancer_source == "standalone_alb" ? module.input.alb_arn_suffix : module.input.ecs_cluster_alb_visibility == "public" ? module.input.public_alb_arn_suffix : module.input.private_alb_arn_suffix),"TargetGroup":stack.output.target_group_arn_suffix},"name":"HealthyHostCount","namespace":"AWS/ApplicationELB","region":stack.output.region,"statistic":"Average"}},{"id":"unhealthy_hosts","name":"Unhealthy hosts","type":"line","source":{"type":"cloudwatch","aws_account_id":module.input.aws_account_id,"dimensions":{"LoadBalancer":(module.input.load_balancer_source == "standalone_alb" ? module.input.alb_arn_suffix : module.input.ecs_cluster_alb_visibility == "public" ? module.input.public_alb_arn_suffix : module.input.private_alb_arn_suffix),"TargetGroup":stack.output.target_group_arn_suffix},"name":"UnHealthyHostCount","namespace":"AWS/ApplicationELB","region":stack.output.region,"statistic":"Average"}}] : []) >> + readme: | + Runs supervised workloads on a stable EC2 Auto Scaling Group, with optional shared ALB routing and switchable container or manual in-place deploys. + + ## Overview + + The EC2 Service module deploys supervisord-managed workloads across an EC2 Auto Scaling Group. Deploys update the existing hosts in place instead of creating a new task or instance for every release. Container mode replaces a Docker container on each host; Manual mode runs host-level preparation commands and then starts a foreground command. You can switch modes without replacing the group. + + Ravion provisions the launch template, Auto Scaling Group, instance role and security group, SSM deploy document, app log group, and optional target group and listener rule. Web services can attach to a standalone Application Load Balancer or reuse the public or private ALB from an ECS cluster. The selected network and load balancer must use the same AWS account, region, and VPC. The Auto Scaling Group can still terminate instances because of scaling, health failures, or operator action. Root and data EBS volumes survive releases, but not instance termination. + + Terraform source: [flightcontrolhq/modules/compute/ec2_service](https://github.com/flightcontrolhq/modules/tree/$local.module_tag/compute/ec2_service) + + ## Use cases + + | Scenario | Why this module fits | Example | + | --- | --- | --- | + | Existing host-installed application | Manual deploys can update files and dependencies directly on every host | A legacy Node, Ruby, or JVM service that is not packaged as a container | + | Host customization is part of the workload | A custom AMI and Additional user data can install agents, libraries, or host services | A vendor runtime that needs an OS package or host monitoring agent | + | Disposable per-host state should survive releases | In-place deploys retain each instance's encrypted root and optional data volume | Model files, build artifacts, or a rebuildable local cache | + | A container workload needs stable hosts | Container mode replaces the app container without replacing the EC2 instance | A licensed service tied to host identity, when its license permits scaling | + | Background processing without HTTP routing | Web service can be disabled, leaving a supervised worker group | A queue consumer that specifically needs host access or local cache | + | Several services share one ALB | Each service owns a target group and host/path listener rule | `api.example.com` and `admin.example.com` on one shared ALB | + + Do not treat per-instance EBS as durable application storage. Use EFS for shared files that must survive instance replacement, and use a managed database for relational or transactional data. + + ## EC2 Service or ECS? + + For a normal stateless containerized web service or worker, prefer an ECS Web Service, ECS Worker Service, or ECS Network Service. ECS gives you immutable task revisions, scheduler-managed replacement, sidecars, Fargate or shared EC2 capacity, and rolling or traffic-shift deployment strategies. + + Choose EC2 Service only when a concrete host-level requirement outweighs those ECS benefits: + + | Decision | EC2 Service | ECS deployment | + | --- | --- | --- | + | Release model | Updates the app on existing hosts through SSM | Starts new task revisions and replaces old tasks | + | Packaging | Container image or host-level manual commands | Container image | + | Host access | Direct instance access, custom AMI, and bootstrap script | Tasks are the deployment unit; hosts are abstracted or cluster-managed | + | Process model | One supervised app per instance group | App container plus optional sidecars and release tasks | + | Capacity | Dedicated Auto Scaling Group for this app | Fargate, Fargate Spot, or shared/dedicated ECS EC2 capacity | + | Local storage | EBS survives deploys but is lost with an instance | Task-local storage is ephemeral; EFS can survive task replacement | + | Deployment safety | Sequential in-place updates; a one-instance service has an interruption | Rolling, blue/green, linear, or canary options, depending on the ECS module | + + Selecting EC2 capacity for an ECS service does not make it equivalent to this module. ECS on EC2 still deploys replaceable tasks through the ECS scheduler. This module deploys directly to stable hosts and gives the application the whole instance. + + ## Deploy types + + | Deploy type | Build behavior | On each instance | + | --- | --- | --- | + | Container | Dockerfile, Railpack, or image registry | Pull the selected image, replace the app container, and run it under supervisord | + | Manual | Build disabled | Run Deploy commands as root on every host, then run Start command under supervisord | + + Instances are prepared for both modes at launch. Supervisord owns the long-running process and restarts it after an unexpected exit. In Container mode, Start command override replaces the image CMD but preserves its ENTRYPOINT. In Manual mode, the start command must remain in the foreground and should explicitly drop root privileges when the app should run as another user. + + ## Build sources + + | Build source | Build output | Deploy input | + | --- | --- | --- | + | Dockerfile | Build from the selected Git repository and push to the service ECR repository | Image tag or digest | + | Railpack | Detect and build the app from the selected Git repository, then push to service ECR | Image tag or digest | + | Pull from image registry | Skip the build and use Image repository from the module | Tag or digest within that repository | + + Build settings apply only to Container mode. Manual mode disables the build pipeline and does not create an ECR repository. Built-image repositories scan images on push by default. Registry username/password credentials are not supported. Public images work directly; same-region private ECR requires repository permissions for the generated `-instance` role. + + ## Worked examples + + ### Host-installed application + + Use Manual mode when the release genuinely needs to update the host rather than replace a container. For example, bootstrap an `app` user and an initial checkout with Additional user data, then configure: + + ```text + Deploy type: Manual + Deploy commands: + runuser -u app -- git -C /srv/orders pull --ff-only + runuser -u app -- bash -c 'cd /srv/orders && ./bin/install-dependencies' + runuser -u app -- bash -c 'cd /srv/orders && ./bin/build-assets' + Start command: + exec runuser -u app -- /srv/orders/bin/server + ``` + + Every Deploy command runs on every instance, so keep commands repeatable and do not put a cluster-wide one-time migration in this list. Select Git repository to have Ravion check out a clean authenticated source before running the commands, or leave it blank when the commands manage their own source or artifacts. Deploy commands and the Start command run from Source base path when source is configured. Use Container mode for digest-addressed image releases; use ECS when you need scheduler-managed rollback or traffic shifting. + + ### Container that needs host customization + + Choose Container mode with a custom AMI or Additional user data when the app is containerized but also needs host software that ECS does not model for the service. A typical web configuration uses at least two instances, App port `3000`, Health check path `/health`, and host or path rules on a shared ALB. Use Dockerfile or Railpack for Ravion builds, or Pull from image registry when an external pipeline already publishes the image. + + If the requirement is only "run my container," use ECS instead. The EC2 choice should come from a real host dependency, stable-host requirement, or manual operational constraint. + + ### Worker with a rebuildable local cache + + Turn Web service off, keep Container mode, and enable a Data volume mounted at a path such as `/var/cache/app`. This is appropriate when the worker can recreate the cache after an instance is replaced. It is not appropriate for the only copy of jobs, uploads, or database data. + + ## Load balancing and worker mode + + With Web service enabled, select a shared Ravion Application Load Balancer. The service creates an HTTP instance target group and a listener rule on the ALB's HTTPS listener when available, otherwise its HTTP listener. Route traffic with Domain host rules, Path rules, or both; when both are empty, the rule matches `/*`. + + Several EC2 services (and other services) can attach to the same load balancer with different host or path rules. Slow start can ramp traffic to newly registered instances, and sticky sessions can use either an ALB-managed cookie or an application cookie. + + Allowed CIDR blocks permit direct access to the app port on the instances. This bypasses load balancer TLS termination, WAF, authentication, and access logging, so leave the list empty unless direct access is required. + + Turn Web service off for worker groups. No target group, listener rule, ALB health check, or local HTTP deploy gate is configured. + + ## Deploys + + Container deploys run through an SSM command document created by the module. On each instance, the deploy: + + 1. Rebuilds the app environment file and fetches configured secret values on the instance. + 2. Pulls the selected image before interrupting the running app. + 3. For a web service with another registered target, deregisters and drains this instance. + 4. Stops the prior process, replaces the app container, and starts it under supervisord. + 5. Polls the local health path for an HTTP response below 400, failing that instance's deploy otherwise. + 6. Re-registers a drained instance and waits for the ALB to mark it in service. + + A web service with only one registered target skips draining because there is nowhere to move traffic; replacing its process still causes an interruption. Use at least two instances when availability during container deploys matters. + + Manual deploys refresh and load the app environment file, stop the prior app, and run Deploy commands as root in order on every instance. Any non-zero exit fails that instance and leaves its prior app stopped. Supervisord starts the Start command only after all preparation commands succeed. Manual mode does not automatically drain the ALB or run a health gate; implement those behaviors in your commands if required. + + Maximum concurrent instances controls how many hosts update together, while Maximum deployment errors controls when the rollout stops sending commands to remaining hosts. Defaults are sequential and fail-fast. Per-instance deploy timeout limits each SSM script; the overall deployment has a 24-hour safety limit. + + The EC2 deploy manager also runs the current release document against instances launched by autoscaling or replacement so they catch up to the active release. + + ## Logs and metrics + + App stdout and stderr from the supervised process are sent to `/ravion/ec2/` in CloudWatch Logs. Every deployment and instance has a distinct stream named `deployment//instance/`. SSM preparation and deploy-script stdout and stderr are copied into the same instance stream while remaining available with the deployment command. + + The dashboard shows desired and in-service instance counts for every service. Web services also show request count, target response time, target 4xx and 5xx responses, and healthy and unhealthy host counts. Worker groups omit load-balancer metrics. + + ## Stable storage + + | Storage | Survives deploys | Survives instance termination | + | --- | --- | --- | + | Root volume | Yes | No | + | Data volume | Yes | No | + | EFS file system | Yes | Yes | + + Root and Data volumes are encrypted gp3 EBS volumes with delete-on-termination enabled. Use the Data volume only for independent, rebuildable per-host files such as caches, model downloads, or generated artifacts. Do not use it for the only copy of a database, upload, or job. + + Enable EFS and select a Ravion EFS module for shared files that must outlive instances. The service mounts the file system through its access point when present and attaches the EFS client security group to each instance. Container mode bind-mounts the Data and EFS host paths into the app container at the same paths. + + ## Scaling and instance health + + Minimum instances and Maximum instances bound the group. Enable CPU autoscaling to target average EC2 CPU utilization. When autoscaling is off, Desired instances sets the initial group size; later external scaling changes are preserved. Instance health check grace period controls how long new instances have before Auto Scaling begins health evaluation. + + The Auto Scaling Group replaces instances that fail EC2 health checks. Load-balancer-based instance replacement is intentionally off by default because in-place deploys briefly deregister instances; override with Advanced Terraform variables (`health_check_type`) if you prefer it. + + Launch template changes such as a new AMI, instance type, volume setting, or Additional user data apply to newly launched instances only. The module does not perform an instance refresh, so recycle existing instances deliberately when a host-level change must roll out. + + ## Application configuration + + Runtime environment variables must be single-line values. They are rendered into the app environment file at instance boot and on every deploy. Container mode passes the file to Docker; Manual mode loads it before Deploy commands and the Start command. A configuration change requires a stack update and then a deploy to restart the app with the new values. + + Runtime secrets use same-account Secrets Manager or SSM Parameter Store references. Each instance fetches values while rebuilding its environment file, so the values do not enter Terraform state or the SSM document. A rotated value takes effect on the next deploy without a stack change. Multi-line secret values are unsupported. + + Build environment variables apply to Railpack and Dockerfile builds; for Dockerfile builds, optionally inject them as build arguments. + + ## Configuration + + ### Service and instances + + | Field | Required | Default | Description | + | --- | --- | --- | --- | + | VPC network | Yes | - | Existing network supplying account, region, VPC, and subnets | + | Service name | Yes | {project}-{env}-{module} | Name for the instance group and related resources | + | Deploy type | Yes | container | Switchable Container or Manual in-place deployment | + | Deploy commands | Yes* | - | Root-level, per-instance preparation commands for Manual mode | + | Start command | Yes* | - | Foreground Manual-mode process managed by supervisord | + | Git repository | No | - | Optional authenticated source checkout for Manual mode | + | Source base path | No | . | Working directory within the selected repository | + | Maximum concurrent instances | No | 1 | Instance count or percentage updated together | + | Maximum deployment errors | No | 0 | Failures tolerated before the rollout stops | + | Per-instance deploy timeout (secs) | No | 1200 | SSM script timeout for either deploy mode | + | Run in private subnets | No | true | Use private subnets without public IPs; outbound access is still required | + | Instance type | Yes | - | EC2 type used for every instance; architecture must match the app | + | Minimum instances / Maximum instances | Yes | 1 / 3 | Auto Scaling Group capacity bounds | + | CPU autoscaling | No | false | Target-tracking scaling on average EC2 CPU utilization | + | CPU target (%) | No | 70 | CPU target used when autoscaling is enabled | + | Desired instances | No | - | Initial target when CPU autoscaling is off; blank starts at the minimum | + | Instance health check grace period (secs) | No | 300 | Startup window before Auto Scaling health checks apply | + | SSH key pair name | No | - | Optional SSH key; Session Manager does not require one | + | Custom AMI ID | No | Latest AL2023 | AMI for future instances; must support the module bootstrap | + | Additional security groups | No | [] | Extra security groups attached to every instance | + | Additional user data | No | - | Root shell script run after Ravion bootstrap on future instances | + + ### Container builds + + | Field | Required | Default | Description | + | --- | --- | --- | --- | + | Build source | Yes* | dockerfile | Dockerfile, Railpack, or Pull from image registry in Container mode | + | Git repository | Yes* | - | Source repository for Dockerfile and Railpack builds | + | Source base path | No | . | Repository-relative application and build root | + | Dockerfile path / build context | No | Dockerfile / . | Dockerfile build locations relative to the source root | + | Railpack version | No | Ravion default | Optional Railpack version override | + | Railpack install / build / start commands | No | Detected | Optional Railpack command overrides | + | Image repository | Yes* | - | Repository without tag or digest for registry deployments | + | Start command override | No | Image CMD | Container command override; preserves image ENTRYPOINT | + | Build environment variables | No | {} | Values and secret references available only during image builds | + | Inject environment variables in Dockerfile | No | false | Pass build environment values as Docker build arguments | + | Builder instance type / size | Yes* | ec2 / c7a.4xlarge | On-demand or Spot build runner and its EC2 size | + | Scan images on push | No | true | Run ECR basic vulnerability scanning for built images | + | Force delete image repository | No | false | Permit deletion of a non-empty Ravion-created ECR repository | + + Git branch and Git ref are optional inputs on each build run. They default to the repository's default branch and its current head. + + ### Web routing and health + + A web service can use a standalone Application Load Balancer or the public or private ALB from an ECS cluster. The selected load balancer must belong to the same AWS account, region, and VPC as the selected VPC network. When both HTTPS and HTTP listeners exist, the service attaches its listener rule to HTTPS. + + | Field | Required | Default | Description | + | --- | --- | --- | --- | + | Web service | No | true | Attach HTTP routing to a shared ALB; disable for workers | + | Load balancer source | Yes* | Standalone ALB module | Use a standalone ALB or an ALB from an ECS cluster | + | Standalone ALB module | Yes* | - | Standalone Ravion ALB receiving this service's listener rule | + | ALB from ECS Cluster module | Yes* | - | Cluster that supplies a public or private ALB | + | ALB to use from ECS Cluster module | Yes* | Public | Select the cluster's public or private ALB | + | App port | Yes* | 80 | Host app port; Container mode maps the same container port and sets PORT | + | Domain host rules / Path rules | No | empty / empty | Listener conditions; both empty produces `/*` | + | Listener rule priority | No | AWS assigned | Optional explicit ALB rule order | + | Health check path | Yes* | / | Path used by the ALB and local container deploy gate | + | Success codes | Yes* | 200-399 | ALB health matcher; local deploys separately require a status below 400 | + | Interval / Timeout (secs) | No | 10 / 5 | ALB health-check timing | + | Healthy / Unhealthy threshold | No | 2 / 2 | Consecutive ALB checks needed to change health state | + | Slow start duration (secs) | No | 0 | Gradually ramp traffic to new instances; 0 disables it | + | Deregistration delay (secs) | No | 30 | Drain time when another target can serve traffic | + | Sticky sessions | No | false | Keep repeat clients on one instance with an ALB or application cookie | + | Stickiness type / application cookie name | Yes* | Load balancer cookie / - | Select cookie ownership and name application cookies | + | Stickiness cookie duration (secs) | No | 86400 | Cookie lifetime when stickiness is enabled | + | Allowed CIDR blocks | No | [] | Direct app-port access that bypasses the load balancer | + + ### Application data and operations + + | Field | Required | Default | Description | + | --- | --- | --- | --- | + | Runtime environment variables | No | [] | Plain single-line values refreshed on every deploy | + | Runtime secrets | No | [] | Same-account Secrets Manager or SSM references fetched on each host | + | Root volume size (GB) | No | 30 | Encrypted per-instance gp3 volume deleted with the instance | + | Data volume / size / mount path | No | false / 20 / /data | Independent encrypted EBS volume for rebuildable per-host files | + | EFS file system | No | false | Enable shared storage that outlives instances | + | EFS file system reference / mount path | Yes* | - / /mnt/efs | Ravion EFS module and its host/container mount path | + | Log retention (days) | No | 30 | CloudWatch retention for supervised app logs | + | Tags | No | Standard Ravion tags | Additional tags applied to resources | + | Advanced Terraform variables | No | {} | Lower-level overrides for exceptional settings | + | OpenTofu version override | No | Ravion default | Stack OpenTofu version override | + | Ravion Terraform workspace name | No | {project}-{env}-{module}-{stack} | State backend workspace override | + + *Conditionally required or visible based on Deploy type, Build source, Web service, Load balancer source, CPU autoscaling, Data volume, or EFS file system. + + ## Design decisions + + - ECS is the default recommendation for stateless container services. Use this module when manual host deployment, host customization, or stable-host behavior is an actual requirement. + - Deploys are in place by design. They do not replace instances, but scaling, health recovery, and manual operations still can. + - Instances are prepared for both deploy modes so the Deploy type can change without replacing the group. + - Supervisord manages one app process per instance. A one-instance service is interrupted while that process is replaced. + - Supervisor 4.3.0 is installed from PyPI. Instances need outbound access to PyPI or an equivalent package source during bootstrap and when deploys ensure the pinned version. + - One app per instance group. Sharing instances between apps is intentionally out of scope; share the load balancer instead. + - Load balancing stays shared: use either a standalone ALB module or a public/private ALB from an ECS cluster while this service owns its target group and listener rule. + - The ASG health check type defaults to EC2 because in-place deploys briefly deregister instances from the target group. + - Launch template changes do not trigger an instance refresh, preserving per-host state until you choose to recycle instances. + - Worker groups omit load-balancer metrics instead of rendering charts with empty target-group dimensions. + + ## Learn more + + - [Amazon EC2 Auto Scaling](https://docs.aws.amazon.com/autoscaling/ec2/userguide/what-is-amazon-ec2-auto-scaling.html) + - [AWS Systems Manager Run Command](https://docs.aws.amazon.com/systems-manager/latest/userguide/run-command.html) + - [Application Load Balancer listener rules](https://docs.aws.amazon.com/elasticloadbalancing/latest/application/listener-update-rules.html) + - [Amazon ECS services](https://docs.aws.amazon.com/AmazonECS/latest/developerguide/ecs_services.html) + - [Supervisor documentation](https://supervisord.org/) + - [Railpack documentation](https://railpack.com/) diff --git a/compute/ec2_service/security_group.tf b/compute/ec2_service/security_group.tf new file mode 100644 index 00000000..8f9973e3 --- /dev/null +++ b/compute/ec2_service/security_group.tf @@ -0,0 +1,38 @@ +################################################################################ +# Security Group for Service Instances +################################################################################ + +module "instance_security_group" { + source = "../../networking/security-groups" + + name = var.name + name_suffix = "instance" + description = "Security group for ${var.name} EC2 service instances" + vpc_id = var.vpc_id + tags = var.tags + + all_egress_enabled = true + + ingress_rules = concat( + # Allow the load balancer to reach the app port + local.load_balancer_creation_enabled && var.load_balancer_security_group_id != null && var.app_port != null ? [ + { + description = "Allow inbound from load balancer" + from_port = var.app_port + to_port = var.app_port + ip_protocol = "tcp" + referenced_security_group_id = var.load_balancer_security_group_id + } + ] : [], + # Additional direct sources + var.app_port != null ? [ + for cidr in var.direct_access_cidr_blocks : { + description = "Allow inbound from ${cidr}" + from_port = var.app_port + to_port = var.app_port + ip_protocol = "tcp" + cidr_ipv4 = cidr + } + ] : [] + ) +} diff --git a/compute/ec2_service/ssm_document.tf b/compute/ec2_service/ssm_document.tf new file mode 100644 index 00000000..823c2428 --- /dev/null +++ b/compute/ec2_service/ssm_document.tf @@ -0,0 +1,121 @@ +################################################################################ +# Deploy SSM Document +# +# The aws:ec2 deploy contract. The deploy manager runs this document +# against the group's instances (batched, with per-instance status). The +# same document is run on scale-out instances to catch them up to the +# current release. +# +# Container runtime: the document takes imageUri (the full image URI to +# run), installs a supervisord program for it, and encodes the whole +# in-place deploy. +# +# Manual runtime: the document takes commands (the service's deploy +# command list). Before running them it stops the prior app and rebuilds +# the app env file. Afterward it starts the configured long-running app +# command under supervisord. +# +# NAMING CONTRACT: the document name is derived by the platform as +# "-deploy" — both names come from var.name in +# this module, so the convention holds by construction. Do not rename +# one without the other (and the platform's Ec2DeployDocumentName). +################################################################################ + +resource "aws_ssm_document" "deploy" { + name = "${var.name}-deploy" + document_type = "Command" + document_format = "YAML" + + content = local.container_runtime ? yamlencode({ + schemaVersion = "2.2" + description = "In-place supervised container deploy for the ${var.name} EC2 service." + + parameters = { + imageUri = { + type = "String" + description = "Full container image URI to deploy, including tag or digest." + allowedPattern = "^[^\\s]+$" + } + deployId = { + type = "String" + description = "Identifier for this deploy, used for release directories and logging." + default = "" + allowedPattern = "^[A-Za-z0-9._-]*$" + } + } + + mainSteps = [ + { + action = "aws:runShellScript" + name = "deploy" + inputs = { + timeoutSeconds = tostring(var.deploy_timeout_seconds) + runCommand = [local.deploy_script] + } + } + ] + }) : yamlencode({ + schemaVersion = "2.2" + description = "Manual deploy for the ${var.name} EC2 service: prepares the release, then runs the app under supervisord." + + parameters = { + commands = { + type = "String" + description = "Release preparation script to run on the instance with the app env file loaded." + } + deployId = { + type = "String" + description = "Identifier for this deploy, used for logging." + default = "" + allowedPattern = "^[A-Za-z0-9._-]*$" + } + sourceRepo = { + type = "String" + description = "Optional Git repository URL to check out before running the manual deploy commands." + default = "" + allowedPattern = "^[A-Za-z0-9:/@._+-]*$" + } + sourceBranch = { + type = "String" + description = "Optional Git branch to check out." + default = "" + allowedPattern = "^[A-Za-z0-9._/@:+-]*$" + } + sourceRef = { + type = "String" + description = "Optional immutable Git ref to check out." + default = "" + allowedPattern = "^[A-Za-z0-9._/@:+-]*$" + } + sourceBasePath = { + type = "String" + description = "Repository-relative working directory for the deploy and start commands." + default = "." + allowedPattern = "^[A-Za-z0-9._/-]*$" + } + gitTokenParameterName = { + type = "String" + description = "Optional SSM SecureString parameter containing the temporary Git credential." + default = "" + allowedPattern = "^(|/ravion/git-tokens/ec2/[A-Za-z0-9._/-]+)$" + } + } + + mainSteps = [ + { + action = "aws:runShellScript" + name = "deploy" + inputs = { + timeoutSeconds = tostring(var.deploy_timeout_seconds) + runCommand = [join("\n", [ + local.manual_deploy_prelude, + "{{ commands }}", + local.manual_deploy_postlude, + ])] + } + } + ] + }) + + tags = local.tags +} diff --git a/compute/ec2_service/target_group.tf b/compute/ec2_service/target_group.tf new file mode 100644 index 00000000..cc100d21 --- /dev/null +++ b/compute/ec2_service/target_group.tf @@ -0,0 +1,104 @@ +################################################################################ +# Target Group and Listener Rules +# +# One instance target group. In-place deploys keep serving from it: the +# deploy script drains each instance, swaps the app, and re-registers. +################################################################################ + +resource "aws_lb_target_group" "app" { + count = local.load_balancer_creation_enabled ? 1 : 0 + + name = "${substr(var.name, 0, min(length(var.name), 28))}-tg" + port = var.load_balancer_attachment.target_group.port + protocol = "HTTP" + vpc_id = var.vpc_id + target_type = "instance" + + deregistration_delay = var.load_balancer_attachment.target_group.deregistration_delay + slow_start = var.load_balancer_attachment.target_group.slow_start + + health_check { + enabled = var.load_balancer_attachment.target_group.health_check.enabled + path = var.load_balancer_attachment.target_group.health_check.path + port = var.load_balancer_attachment.target_group.health_check.port + protocol = "HTTP" + matcher = var.load_balancer_attachment.target_group.health_check.matcher + interval = var.load_balancer_attachment.target_group.health_check.interval + timeout = var.load_balancer_attachment.target_group.health_check.timeout + healthy_threshold = var.load_balancer_attachment.target_group.health_check.healthy_threshold + unhealthy_threshold = var.load_balancer_attachment.target_group.health_check.unhealthy_threshold + } + + dynamic "stickiness" { + for_each = var.load_balancer_attachment.target_group.stickiness != null ? [var.load_balancer_attachment.target_group.stickiness] : [] + content { + enabled = stickiness.value.enabled + type = stickiness.value.type + cookie_duration = stickiness.value.cookie_duration + cookie_name = stickiness.value.cookie_name + } + } + + tags = merge(local.tags, { + Name = "${var.name}-tg" + }) + + lifecycle { + create_before_destroy = true + } +} + +resource "aws_lb_listener_rule" "app" { + for_each = local.load_balancer_creation_enabled ? { + for idx, rule in var.load_balancer_attachment.listener_rules : idx => rule + } : {} + + listener_arn = each.value.listener_arn + priority = each.value.priority + + action { + type = "forward" + target_group_arn = aws_lb_target_group.app[0].arn + } + + dynamic "condition" { + for_each = [for c in each.value.conditions : c if c.type == "path-pattern"] + content { + path_pattern { + values = condition.value.values + } + } + } + + dynamic "condition" { + for_each = [for c in each.value.conditions : c if c.type == "host-header"] + content { + host_header { + values = condition.value.values + } + } + } + + dynamic "condition" { + for_each = [for c in each.value.conditions : c if c.type == "http-header"] + content { + http_header { + http_header_name = condition.value.values[0] + values = slice(condition.value.values, 1, length(condition.value.values)) + } + } + } + + dynamic "condition" { + for_each = [for c in each.value.conditions : c if c.type == "source-ip"] + content { + source_ip { + values = condition.value.values + } + } + } + + tags = merge(local.tags, { + Name = "${var.name}-rule-${each.key}" + }) +} diff --git a/compute/ec2_service/templates/checkout_git_source.sh.tpl b/compute/ec2_service/templates/checkout_git_source.sh.tpl new file mode 100644 index 00000000..75f4ccf2 --- /dev/null +++ b/compute/ec2_service/templates/checkout_git_source.sh.tpl @@ -0,0 +1,114 @@ +SOURCE_REPO="{{ sourceRepo }}" +SOURCE_BRANCH="{{ sourceBranch }}" +SOURCE_REF="{{ sourceRef }}" +SOURCE_BASE_PATH="{{ sourceBasePath }}" +GIT_TOKEN_PARAMETER_NAME="{{ gitTokenParameterName }}" +SOURCE_ROOT="/srv/ravion/${name}" +SOURCE_DIRECTORY="$SOURCE_ROOT/source" +SOURCE_STAGING_DIRECTORY="$SOURCE_ROOT/.source-$DEPLOY_ID" +SOURCE_WORKING_DIRECTORY_FILE="${source_working_directory_path}" + +if [ -z "$SOURCE_REPO" ]; then + rm -f "$SOURCE_WORKING_DIRECTORY_FILE" +else + if ! command -v git >/dev/null 2>&1; then + dnf install -y git + fi + + if [ -z "$GIT_TOKEN_PARAMETER_NAME" ]; then + echo "Git source requires a temporary credential parameter" >&2 + exit 1 + fi + + GIT_CREDENTIAL=$(aws ssm get-parameter \ + --name "$GIT_TOKEN_PARAMETER_NAME" \ + --with-decryption \ + --region "${region}" \ + --query 'Parameter.Value' \ + --output text) + if [ -z "$GIT_CREDENTIAL" ] || [ "$GIT_CREDENTIAL" = "None" ]; then + echo "Git credential parameter returned no value" >&2 + exit 1 + fi + + case "$GIT_CREDENTIAL" in + *:*) + RVN_GIT_USERNAME=$${GIT_CREDENTIAL%%:*} + RVN_GIT_PASSWORD=$${GIT_CREDENTIAL#*:} + ;; + *) + RVN_GIT_USERNAME="x-access-token" + RVN_GIT_PASSWORD="$GIT_CREDENTIAL" + ;; + esac + unset GIT_CREDENTIAL + export RVN_GIT_USERNAME RVN_GIT_PASSWORD GIT_TERMINAL_PROMPT=0 + + GIT_ASKPASS_SCRIPT=$(mktemp /tmp/ravion-git-askpass.XXXXXX) + cleanup_git_credentials() { + rm -f "$GIT_ASKPASS_SCRIPT" + rm -rf "$SOURCE_STAGING_DIRECTORY" + unset RVN_GIT_USERNAME RVN_GIT_PASSWORD GIT_ASKPASS + } + trap cleanup_git_credentials EXIT + cat > "$GIT_ASKPASS_SCRIPT" <<'GIT_ASKPASS' +#!/bin/sh +case "$1" in + *Username*) printf '%s\n' "$RVN_GIT_USERNAME" ;; + *) printf '%s\n' "$RVN_GIT_PASSWORD" ;; +esac +GIT_ASKPASS + chmod 700 "$GIT_ASKPASS_SCRIPT" + export GIT_ASKPASS="$GIT_ASKPASS_SCRIPT" + + case "$SOURCE_REPO" in + git@*:*) + SOURCE_REPO_HOST=$${SOURCE_REPO#git@} + SOURCE_REPO_HOST=$${SOURCE_REPO_HOST%%:*} + SOURCE_REPO_PATH=$${SOURCE_REPO#*:} + SOURCE_REPO="https://$SOURCE_REPO_HOST/$SOURCE_REPO_PATH" + ;; + esac + + mkdir -p "$SOURCE_ROOT" + rm -rf "$SOURCE_STAGING_DIRECTORY" + git clone --no-checkout "$SOURCE_REPO" "$SOURCE_STAGING_DIRECTORY" + + if [ -z "$SOURCE_BRANCH" ]; then + SOURCE_BRANCH=$(git -C "$SOURCE_STAGING_DIRECTORY" symbolic-ref --short refs/remotes/origin/HEAD) + SOURCE_BRANCH=$${SOURCE_BRANCH#origin/} + fi + git -C "$SOURCE_STAGING_DIRECTORY" fetch origin "$SOURCE_BRANCH" --tags + + if [ -n "$SOURCE_REF" ]; then + git -C "$SOURCE_STAGING_DIRECTORY" checkout --detach "$SOURCE_REF" + else + git -C "$SOURCE_STAGING_DIRECTORY" checkout -B "$SOURCE_BRANCH" "origin/$SOURCE_BRANCH" + fi + git -C "$SOURCE_STAGING_DIRECTORY" submodule sync --recursive + git -C "$SOURCE_STAGING_DIRECTORY" submodule update --init --recursive + + if [ -z "$SOURCE_BASE_PATH" ]; then SOURCE_BASE_PATH="."; fi + SOURCE_WORKING_DIRECTORY=$(realpath -m "$SOURCE_STAGING_DIRECTORY/$SOURCE_BASE_PATH") + case "$SOURCE_WORKING_DIRECTORY" in + "$SOURCE_STAGING_DIRECTORY"|"$SOURCE_STAGING_DIRECTORY"/*) ;; + *) + echo "Git source base path must stay inside the repository" >&2 + exit 1 + ;; + esac + if [ ! -d "$SOURCE_WORKING_DIRECTORY" ]; then + echo "Git source base path does not exist or is not a directory: $SOURCE_BASE_PATH" >&2 + exit 1 + fi + + rm -rf "$SOURCE_DIRECTORY" + mv "$SOURCE_STAGING_DIRECTORY" "$SOURCE_DIRECTORY" + SOURCE_WORKING_DIRECTORY=$(realpath -m "$SOURCE_DIRECTORY/$SOURCE_BASE_PATH") + printf '%s' "$SOURCE_WORKING_DIRECTORY" > "$SOURCE_WORKING_DIRECTORY_FILE" + chmod 600 "$SOURCE_WORKING_DIRECTORY_FILE" + + cleanup_git_credentials + trap - EXIT + cd "$SOURCE_WORKING_DIRECTORY" +fi diff --git a/compute/ec2_service/templates/configure_deployment_logs.sh.tpl b/compute/ec2_service/templates/configure_deployment_logs.sh.tpl new file mode 100644 index 00000000..89442fb0 --- /dev/null +++ b/compute/ec2_service/templates/configure_deployment_logs.sh.tpl @@ -0,0 +1,27 @@ +LOG_PATH="${log_directory}/deployment-$${DEPLOY_ID}.log" +mkdir -p "${log_directory}" +touch "$LOG_PATH" +chmod 640 "$LOG_PATH" + +dnf install -y amazon-cloudwatch-agent +cat > /opt/aws/amazon-cloudwatch-agent/etc/app-logs.json < "${supervisor_conf}" </dev/null 2>&1 || true + +SUPERVISOR_RUNNING=0 +for _ in $(seq 1 30); do + if /usr/local/bin/supervisorctl -c /etc/supervisord.conf status "${supervisor_program}" | grep -q RUNNING; then + SUPERVISOR_RUNNING=1 + break + fi + sleep 1 +done +if [ "$SUPERVISOR_RUNNING" -ne 1 ]; then + echo "${supervisor_program} did not reach the RUNNING state" >&2 + /usr/local/bin/supervisorctl -c /etc/supervisord.conf status "${supervisor_program}" >&2 || true + tail -n 100 "$LOG_PATH" >&2 || true + exit 1 +fi diff --git a/compute/ec2_service/templates/deploy_container.sh.tpl b/compute/ec2_service/templates/deploy_container.sh.tpl new file mode 100644 index 00000000..2ede6363 --- /dev/null +++ b/compute/ec2_service/templates/deploy_container.sh.tpl @@ -0,0 +1,114 @@ +#!/bin/bash +# In-place container deploy for ${name}. Runs on each instance via the +# SSM deploy document; {{ }} placeholders are SSM parameter substitutions. +set -euo pipefail + +IMAGE_URI="{{ imageUri }}" +DEPLOY_ID="{{ deployId }}" +if [ -z "$DEPLOY_ID" ]; then DEPLOY_ID=$(date +%s); fi + +# SSM runs this script as root but with the agent's bare environment (no +# HOME, TERM=dumb). Restore normal root-shell semantics so release tooling +# that resolves `~` or queries terminal capabilities does not abort. +export HOME="$${HOME:-/root}" +if [ "$${TERM:-dumb}" = "dumb" ]; then export TERM=xterm; fi + +${deployment_log_script} +exec > >(tee -a "$LOG_PATH") 2> >(tee -a "$LOG_PATH" >&2) + +echo "Deploying image $IMAGE_URI (deploy $DEPLOY_ID)" + +TOKEN=$(curl -sf -X PUT http://169.254.169.254/latest/api/token -H "X-aws-ec2-metadata-token-ttl-seconds: 300") +INSTANCE_ID=$(curl -sf -H "X-aws-ec2-metadata-token: $TOKEN" http://169.254.169.254/latest/meta-data/instance-id) + +# Make supervisord available on both newly launched and existing instances. +${supervisor_install_script} + +# Rebuild the app env file on every deploy. +${env_file_script} + +# Log in to ECR when pulling from an ECR registry +case "$IMAGE_URI" in + *.dkr.ecr.*.amazonaws.com/*) + ECR_REGISTRY="$${IMAGE_URI%%/*}" + aws ecr get-login-password --region ${region} | docker login --username AWS --password-stdin "$ECR_REGISTRY" + ;; +esac + +docker pull "$IMAGE_URI" + +%{ if target_group_arn != "" ~} +# Drain this instance from the target group before swapping the container. +# Skipped when this instance is the only registered target: there is +# nothing to shift traffic to, and skipping the deregistration delay and +# in-service wait shortens the outage the swap causes anyway. +REGISTERED_TARGETS=$(aws elbv2 describe-target-health --region ${region} --target-group-arn "${target_group_arn}" --query 'length(TargetHealthDescriptions)' --output text) +DRAIN=0 +if [ "$REGISTERED_TARGETS" -gt 1 ]; then + DRAIN=1 + echo "Deregistering $INSTANCE_ID from the target group" + aws elbv2 deregister-targets --region ${region} --target-group-arn "${target_group_arn}" --targets Id="$INSTANCE_ID" + aws elbv2 wait target-deregistered --region ${region} --target-group-arn "${target_group_arn}" --targets Id="$INSTANCE_ID" +else + echo "Only registered target in the target group; skipping drain" +fi +%{ endif ~} + +# Stop the old supervised process after draining. The runner removes any +# stale same-named container before starting the requested image. +/usr/local/bin/supervisorctl -c /etc/supervisord.conf stop "${supervisor_program}" >/dev/null 2>&1 || true +docker rm -f ${name} >/dev/null 2>&1 || true +printf '%s\n' "$IMAGE_URI" > "${image_ref_path}" + +cat > "${app_runner_path}" <<'APP_RUNNER' +#!/bin/bash +set -euo pipefail +# Supervisord starts the runner without a login environment; docker reads +# credential config from $HOME. +export HOME="$${HOME:-/root}" +IMAGE_URI=$(cat "${image_ref_path}") +docker rm -f ${name} >/dev/null 2>&1 || true + +RUN_ARGS=(--rm --name ${name} --env-file "${env_file_path}") +%{ if app_port != null ~} +RUN_ARGS+=(-p ${app_port}:${app_port}) +%{ endif ~} +%{ if data_volume_mount_path != "" ~} +RUN_ARGS+=(-v ${data_volume_mount_path}:${data_volume_mount_path}) +%{ endif ~} +%{ if efs_mount_path != "" ~} +RUN_ARGS+=(-v ${efs_mount_path}:${efs_mount_path}) +%{ endif ~} +exec docker run "$${RUN_ARGS[@]}" "$IMAGE_URI" ${start_command} +APP_RUNNER +chmod 755 "${app_runner_path}" + +${supervisor_program_script} + +%{ if deploy_health_check_path != "" && app_port != null ~} +# Gate deploy success on the local health check +HEALTHY=0 +for _ in $(seq 1 60); do + if curl -fsS -o /dev/null "http://localhost:${app_port}${deploy_health_check_path}"; then + HEALTHY=1 + break + fi + sleep 5 +done +if [ "$HEALTHY" -ne 1 ]; then + echo "App failed the local health check on port ${app_port}${deploy_health_check_path}" >&2 + tail -n 100 "$LOG_PATH" >&2 || true + exit 1 +fi +%{ endif ~} + +%{ if target_group_arn != "" ~} +if [ "$DRAIN" -eq 1 ]; then + echo "Re-registering $INSTANCE_ID with the target group" + aws elbv2 register-targets --region ${region} --target-group-arn "${target_group_arn}" --targets Id="$INSTANCE_ID" + aws elbv2 wait target-in-service --region ${region} --target-group-arn "${target_group_arn}" --targets Id="$INSTANCE_ID" +fi +%{ endif ~} + +docker image prune -f >/dev/null 2>&1 || true +echo "Deploy $DEPLOY_ID complete" diff --git a/compute/ec2_service/templates/deploy_manual_after.sh.tpl b/compute/ec2_service/templates/deploy_manual_after.sh.tpl new file mode 100644 index 00000000..5ac24001 --- /dev/null +++ b/compute/ec2_service/templates/deploy_manual_after.sh.tpl @@ -0,0 +1,23 @@ +printf '%s' '${manual_start_command_base64}' | base64 -d > "${start_command_path}" +chmod 600 "${start_command_path}" + +cat > "${app_runner_path}" <<'APP_RUNNER' +#!/bin/bash +set -euo pipefail +# Supervisord starts the app without a login environment; give the start +# command the same root-shell semantics as the deploy commands. +export HOME="$${HOME:-/root}" +set -a +. "${env_file_path}" +set +a +if [ -s "${source_working_directory_path}" ]; then + cd "$(cat "${source_working_directory_path}")" +fi +START_COMMAND=$(cat "${start_command_path}") +exec /bin/bash -lc "$START_COMMAND" +APP_RUNNER +chmod 755 "${app_runner_path}" + +${supervisor_program_script} + +echo "Manual deploy $DEPLOY_ID complete; supervisord is managing ${supervisor_program}" diff --git a/compute/ec2_service/templates/deploy_manual_before.sh.tpl b/compute/ec2_service/templates/deploy_manual_before.sh.tpl new file mode 100644 index 00000000..68a2c73b --- /dev/null +++ b/compute/ec2_service/templates/deploy_manual_before.sh.tpl @@ -0,0 +1,30 @@ +#!/bin/bash +set -euo pipefail + +DEPLOY_ID="{{ deployId }}" +if [ -z "$DEPLOY_ID" ]; then DEPLOY_ID=$(date +%s); fi + +# SSM runs this script as root but with the agent's bare environment (no +# HOME, TERM=dumb). Restore normal root-shell semantics so release tooling +# that resolves `~` or queries terminal capabilities does not abort. +export HOME="$${HOME:-/root}" +if [ "$${TERM:-dumb}" = "dumb" ]; then export TERM=xterm; fi + +${deployment_log_script} +exec > >(tee -a "$LOG_PATH") 2> >(tee -a "$LOG_PATH" >&2) + +echo "Preparing manual deploy $DEPLOY_ID" + +${supervisor_install_script} +${env_file_script} + +# Stop the previous supervised app before release preparation. Removing a +# same-named container also makes Container -> Manual switches deterministic. +/usr/local/bin/supervisorctl -c /etc/supervisord.conf stop "${supervisor_program}" >/dev/null 2>&1 || true +docker rm -f ${name} >/dev/null 2>&1 || true + +set -a +. "${env_file_path}" +set +a + +${git_source_checkout_script} diff --git a/compute/ec2_service/templates/env_file.sh.tpl b/compute/ec2_service/templates/env_file.sh.tpl new file mode 100644 index 00000000..9789ec93 --- /dev/null +++ b/compute/ec2_service/templates/env_file.sh.tpl @@ -0,0 +1,32 @@ +# Build the app environment file. Plain values are rendered by Terraform +# (like an ECS task definition's environment); secret values are fetched +# from Secrets Manager / SSM Parameter Store here, so they never land in +# Terraform state or the SSM document. A failed fetch aborts before the +# old env file is replaced. +umask 077 +ENV_TMP=$(mktemp) +cat > "$ENV_TMP" <<'RVNENV' +%{ for ev in environment_variables ~} +${ev.name}=${ev.value} +%{ endfor ~} +%{ if app_port != null ~} +PORT=${app_port} +%{ endif ~} +RVNENV +%{ for s in secrets ~} +SECRET_SOURCE='${s.value_from}' +case "$SECRET_SOURCE" in + arn:*:secretsmanager:*) + SECRET_VALUE=$(aws secretsmanager get-secret-value --region "$(echo "$SECRET_SOURCE" | cut -d: -f4)" --secret-id "$SECRET_SOURCE" --query SecretString --output text) + ;; + arn:*:ssm:*) + SECRET_VALUE=$(aws ssm get-parameter --region "$(echo "$SECRET_SOURCE" | cut -d: -f4)" --name "$SECRET_SOURCE" --with-decryption --query Parameter.Value --output text) + ;; + *) + SECRET_VALUE=$(aws ssm get-parameter --region ${region} --name "$SECRET_SOURCE" --with-decryption --query Parameter.Value --output text) + ;; +esac +printf '%s=%s\n' '${s.name}' "$SECRET_VALUE" >> "$ENV_TMP" +%{ endfor ~} +mkdir -p "$(dirname ${env_file_path})" +mv "$ENV_TMP" "${env_file_path}" diff --git a/compute/ec2_service/templates/install_supervisor.sh.tpl b/compute/ec2_service/templates/install_supervisor.sh.tpl new file mode 100644 index 00000000..fc1622d3 --- /dev/null +++ b/compute/ec2_service/templates/install_supervisor.sh.tpl @@ -0,0 +1,52 @@ +# Install Supervisor from PyPI because Amazon Linux 2023 does not ship a +# Supervisor RPM. Pin the version so instance bootstrap is reproducible. +if ! /usr/local/bin/supervisord --version 2>/dev/null | grep -qx '4.3.0'; then + dnf install -y python3-pip + python3 -m pip install --prefix /usr/local supervisor==4.3.0 +fi + +mkdir -p /etc/supervisord.d /var/log/supervisor +cat > /etc/supervisord.conf <<'SUPERVISOR_CONFIG' +[unix_http_server] +file=/run/supervisor/supervisor.sock +chmod=0700 + +[supervisord] +logfile=/var/log/supervisor/supervisord.log +pidfile=/run/supervisord.pid +childlogdir=/var/log/supervisor +nodaemon=false + +[rpcinterface:supervisor] +supervisor.rpcinterface_factory=supervisor.rpcinterface:make_main_rpcinterface + +[supervisorctl] +serverurl=unix:///run/supervisor/supervisor.sock + +[include] +files=/etc/supervisord.d/*.ini +SUPERVISOR_CONFIG + +cat > /etc/systemd/system/supervisord.service <<'SYSTEMD_UNIT' +[Unit] +Description=Supervisor process control system +After=network-online.target docker.service +Wants=network-online.target + +[Service] +Type=forking +ExecStart=/usr/local/bin/supervisord -c /etc/supervisord.conf +ExecStop=/usr/local/bin/supervisorctl -c /etc/supervisord.conf shutdown +ExecReload=/usr/local/bin/supervisorctl -c /etc/supervisord.conf reload +PIDFile=/run/supervisord.pid +RuntimeDirectory=supervisor +RuntimeDirectoryMode=0700 +Restart=on-failure +RestartSec=5s + +[Install] +WantedBy=multi-user.target +SYSTEMD_UNIT + +systemctl daemon-reload +systemctl enable --now supervisord diff --git a/compute/ec2_service/templates/user_data.sh.tpl b/compute/ec2_service/templates/user_data.sh.tpl new file mode 100644 index 00000000..a6e564e5 --- /dev/null +++ b/compute/ec2_service/templates/user_data.sh.tpl @@ -0,0 +1,57 @@ +#!/bin/bash +# Bootstrap for ${name} EC2 service instances. +# Runs once at launch. Deploys are pushed separately through SSM Run +# Command, so this script prepares the host for either deploy mode. +set -euo pipefail + +dnf install -y git jq unzip + +%{ if data_volume_creation_enabled ~} +# Format and mount the data volume on first boot. The volume is the only +# attached disk without a filesystem; on later boots fstab mounts it. +DATA_DEVICE="" +for dev in $(lsblk -dnpo NAME -e 7,11); do + if [ -z "$(lsblk -no FSTYPE "$dev" | tr -d '[:space:]')" ]; then + DATA_DEVICE="$dev" + break + fi +done +if [ -n "$DATA_DEVICE" ]; then + mkfs -t xfs "$DATA_DEVICE" + mkdir -p ${data_volume_mount_path} + DATA_UUID=$(blkid -s UUID -o value "$DATA_DEVICE") + echo "UUID=$DATA_UUID ${data_volume_mount_path} xfs defaults,nofail 0 2" >> /etc/fstab + mount -a +fi +%{ endif ~} + +%{ if efs_enabled ~} +# Mount the EFS file system +dnf install -y amazon-efs-utils +mkdir -p ${efs_mount_path} +%{ if efs_access_point_id != "" ~} +echo "${efs_file_system_id} ${efs_mount_path} efs _netdev,tls,accesspoint=${efs_access_point_id} 0 0" >> /etc/fstab +%{ else ~} +echo "${efs_file_system_id} ${efs_mount_path} efs _netdev,tls 0 0" >> /etc/fstab +%{ endif ~} +mount -a -t efs +%{ endif ~} + +mkdir -p "$(dirname ${env_file_path})" + +# Install both runtime prerequisites so deploy mode can change without +# replacing the instance group. The same idempotent supervisor bootstrap +# also runs during deploys to upgrade existing instances in place. +dnf install -y docker +systemctl enable --now docker +${supervisor_install_script} + +# Initialize the app env file. Deploys refresh it before running either mode. +${env_file_script} + +dnf install -y amazon-cloudwatch-agent + +%{ if additional_user_data != "" ~} +# Additional user data +${additional_user_data} +%{ endif ~} diff --git a/compute/ec2_service/tests/deploy_process_and_logs.tftest.hcl b/compute/ec2_service/tests/deploy_process_and_logs.tftest.hcl new file mode 100644 index 00000000..4be05520 --- /dev/null +++ b/compute/ec2_service/tests/deploy_process_and_logs.tftest.hcl @@ -0,0 +1,231 @@ +mock_provider "aws" { + override_data { + target = data.aws_caller_identity.current + values = { + account_id = "123456789012" + } + } + + override_data { + target = data.aws_region.current + values = { + id = "us-east-1" + name = "us-east-1" + } + } + + override_data { + target = data.aws_partition.current + values = { + partition = "aws" + } + } + + override_resource { + target = aws_iam_instance_profile.instance + values = { + arn = "arn:aws:iam::123456789012:instance-profile/supervised-app-instance" + } + } + + override_resource { + target = module.instance_security_group.aws_security_group.this + values = { + id = "sg-12345678" + } + } + + override_resource { + target = aws_launch_template.app + values = { + id = "lt-12345678" + } + } + + override_resource { + target = aws_lb_target_group.app + values = { + arn = "arn:aws:elasticloadbalancing:us-east-1:123456789012:targetgroup/supervised-app/1234567890abcdef" + } + } +} + +variables { + name = "supervised-app" + region = "us-east-1" + vpc_id = "vpc-12345678" + subnet_ids = ["subnet-12345678"] + instance_type = "t3.micro" + ami_id = "ami-12345678" +} + +run "container_is_supervised_and_logs_per_deployment" { + command = plan + + variables { + runtime = "container" + } + + assert { + condition = strcontains(aws_ssm_document.deploy.content, "autorestart=true") + error_message = "Container deploys must configure supervisord to restart the app." + } + + assert { + condition = strcontains(aws_ssm_document.deploy.content, "deployment/$${DEPLOY_ID}/instance/{instance_id}") + error_message = "Container logs must use deployment- and instance-scoped CloudWatch streams." + } + + assert { + condition = strcontains(yamldecode(aws_ssm_document.deploy.content).mainSteps[0].inputs.runCommand[0], "exec > >(tee -a \"$LOG_PATH\") 2> >(tee -a \"$LOG_PATH\" >&2)") + error_message = "Container SSM stdout and stderr must also be copied to the deployment instance log." + } + + assert { + condition = strcontains(base64decode(aws_launch_template.app.user_data), "supervisor==4.3.0") + error_message = "Instances must install the pinned Supervisor version at bootstrap." + } + + assert { + condition = strcontains(yamldecode(aws_ssm_document.deploy.content).mainSteps[0].inputs.runCommand[0], "export HOME=\"$${HOME:-/root}\"") + error_message = "Container deploys must restore root-shell HOME for the release and runner scripts." + } + + assert { + condition = output.log_stream_prefix == "deployment" + error_message = "The log stream output must select all deployment-scoped streams." + } + +} + +run "manual_start_command_is_supervised" { + command = plan + + variables { + runtime = "manual" + manual_start_command = "cd /srv/app && ./bin/server" + } + + assert { + condition = strcontains(aws_ssm_document.deploy.content, base64encode("cd /srv/app && ./bin/server")) + error_message = "The manual start command must be embedded safely in the deploy document." + } + + assert { + condition = strcontains(aws_ssm_document.deploy.content, "exec /bin/bash -lc") + error_message = "Manual deploys must run the long-lived start command through the supervisor runner." + } + + assert { + condition = strcontains(aws_ssm_document.deploy.content, "autorestart=true") + error_message = "Manual deploys must configure supervisord to restart the app." + } + + assert { + condition = strcontains(aws_ssm_document.deploy.content, "sourceRepo") && strcontains(aws_ssm_document.deploy.content, "gitTokenParameterName") + error_message = "Manual deploys must expose the optional nested source transport parameters." + } + + assert { + condition = strcontains(base64decode(aws_launch_template.app.user_data), "dnf install -y git jq unzip") + error_message = "Instances must install Git at bootstrap for source-backed manual deploys." + } + + assert { + condition = strcontains(aws_ssm_document.deploy.content, "GIT_ASKPASS") && strcontains(aws_ssm_document.deploy.content, "SOURCE_DIRECTORY=\"$SOURCE_ROOT/source\"") + error_message = "Manual deploys must authenticate transiently and check source out under the Ravion-managed directory." + } + + assert { + condition = strcontains(aws_ssm_document.deploy.content, "if ! command -v git") && strcontains(aws_ssm_document.deploy.content, "dnf install -y git") + error_message = "Source-backed manual deploys must install Git on existing instances before checkout." + } + + assert { + condition = strcontains(aws_ssm_document.deploy.content, "source-working-directory") + error_message = "Manual deploy and start commands must share the selected source working directory." + } + + assert { + condition = strcontains(yamldecode(aws_ssm_document.deploy.content).mainSteps[0].inputs.runCommand[0], "export HOME=\"$${HOME:-/root}\"") + error_message = "Manual deploy commands and the supervised start command must run with root-shell HOME." + } + + assert { + condition = strcontains(yamldecode(aws_ssm_document.deploy.content).mainSteps[0].inputs.runCommand[0], "then export TERM=xterm; fi") + error_message = "Manual deploy commands must run with a usable TERM instead of the SSM agent's dumb terminal." + } + + assert { + condition = strcontains(yamldecode(aws_ssm_document.deploy.content).mainSteps[0].inputs.runCommand[0], "exec > >(tee -a \"$LOG_PATH\") 2> >(tee -a \"$LOG_PATH\" >&2)") + error_message = "Manual SSM stdout and stderr must also be copied to the deployment instance log." + } + + assert { + condition = yamldecode(aws_ssm_document.deploy.content).parameters.commands.type == "String" + error_message = "Manual deploy commands must use a String parameter so the command script can be embedded between the prelude and postlude." + } + + assert { + condition = length(yamldecode(aws_ssm_document.deploy.content).mainSteps[0].inputs.runCommand) == 1 && strcontains(yamldecode(aws_ssm_document.deploy.content).mainSteps[0].inputs.runCommand[0], "{{ commands }}") + error_message = "Manual deploy setup, commands, and teardown must be one runCommand string so SSM does not create a nested command array during parameter substitution." + } +} + +run "web_target_group_supports_slow_start_stickiness_and_direct_access" { + command = plan + + variables { + runtime = "container" + app_port = 3000 + deploy_health_check_path = "/health" + health_check_grace_period = 450 + load_balancer_security_group_id = "sg-87654321" + direct_access_cidr_blocks = ["10.0.0.0/8"] + load_balancer_attachment = { + creation_enabled = true + target_group = { + port = 3000 + slow_start = 60 + stickiness = { + enabled = true + type = "app_cookie" + cookie_duration = 3600 + cookie_name = "SESSION_ID" + } + } + listener_rules = [] + } + } + + assert { + condition = aws_lb_target_group.app[0].slow_start == 60 + error_message = "The target group must receive the configured slow start duration." + } + + assert { + condition = aws_lb_target_group.app[0].stickiness[0].type == "app_cookie" && aws_lb_target_group.app[0].stickiness[0].cookie_name == "SESSION_ID" && aws_lb_target_group.app[0].stickiness[0].cookie_duration == 3600 + error_message = "The target group must receive application-cookie stickiness settings." + } + +} + +run "slow_start_rejects_values_below_30_seconds" { + command = plan + + variables { + runtime = "container" + app_port = 3000 + deploy_health_check_path = "/health" + load_balancer_attachment = { + creation_enabled = true + target_group = { + port = 3000 + slow_start = 29 + } + listener_rules = [] + } + } + + expect_failures = [var.load_balancer_attachment] +} diff --git a/compute/ec2_service/variables.tf b/compute/ec2_service/variables.tf new file mode 100644 index 00000000..022b2421 --- /dev/null +++ b/compute/ec2_service/variables.tf @@ -0,0 +1,513 @@ +################################################################################ +# General +################################################################################ + +variable "name" { + type = string + description = "Name prefix for all resources created by this module." + + validation { + condition = can(regex("^[a-z0-9]([a-z0-9-]{0,26}[a-z0-9])?$", var.name)) + error_message = "The name must be 1-28 characters, contain only lowercase letters, numbers, and hyphens, and start and end with a letter or number." + } +} + +variable "tags" { + type = map(string) + description = "A map of tags to assign to all resources." + default = {} +} + +variable "region" { + type = string + description = "AWS region. When null, the provider's configured region is used." + default = null +} + +################################################################################ +# Network +################################################################################ + +variable "vpc_id" { + type = string + description = "The ID of the VPC where the instances run." + + validation { + condition = can(regex("^vpc-", var.vpc_id)) + error_message = "The vpc_id must be a valid VPC ID starting with 'vpc-'." + } +} + +variable "subnet_ids" { + type = list(string) + description = "Subnets for the Auto Scaling Group instances." + + validation { + condition = length(var.subnet_ids) >= 1 + error_message = "At least 1 subnet ID is required." + } + + validation { + condition = alltrue([for s in var.subnet_ids : can(regex("^subnet-", s))]) + error_message = "All subnet_ids must be valid subnet IDs starting with 'subnet-'." + } +} + +variable "public_ip_assignment_enabled" { + type = bool + description = "Assign public IPs to instances. Use when instances run in public subnets without NAT egress." + default = false +} + +variable "additional_security_group_ids" { + type = list(string) + description = "Additional security group IDs attached to the instances." + default = [] + + validation { + condition = alltrue([for sg in var.additional_security_group_ids : can(regex("^sg-", sg))]) + error_message = "All additional_security_group_ids must be valid security group IDs starting with 'sg-'." + } +} + +variable "direct_access_cidr_blocks" { + type = list(string) + description = "IPv4 CIDR blocks allowed to reach the app port directly, in addition to the load balancer." + default = [] + + validation { + condition = alltrue([for cidr in var.direct_access_cidr_blocks : can(cidrhost(cidr, 0))]) + error_message = "All direct_access_cidr_blocks must be valid IPv4 CIDR blocks." + } +} + +################################################################################ +# Runtime +################################################################################ + +variable "runtime" { + type = string + description = "How the app is deployed on the instances. 'container' swaps a Docker container through the module's SSM deploy document; 'manual' means the deploy manager runs a user-provided list of shell commands on each instance." + + validation { + condition = contains(["container", "manual"], var.runtime) + error_message = "The runtime must be 'container' or 'manual'." + } +} + +variable "app_port" { + type = number + description = "Port the app listens on. Required when a load balancer is attached or a local health check path is set." + default = null + + validation { + condition = var.app_port == null || (var.app_port >= 1 && var.app_port <= 65535) + error_message = "The app_port must be between 1 and 65535." + } +} + +variable "container_start_command" { + type = string + description = "Optional command for the container runtime that overrides the image default command." + default = null +} + +variable "manual_start_command" { + type = string + description = "Long-running foreground application command managed and restarted by supervisord in the manual runtime. Deploy commands prepare each release; this command starts it." + default = null + + validation { + condition = var.runtime != "manual" || (var.manual_start_command != null && length(trimspace(var.manual_start_command)) > 0) + error_message = "The manual_start_command must be set when runtime is 'manual'." + } +} + +variable "environment_variables" { + type = list(object({ + name = string + value = string + })) + description = "Plain environment variables written to the app environment file at deploy time." + default = [] +} + +variable "secrets" { + type = list(object({ + name = string + value_from = string + })) + description = "Secret environment variables fetched on the instance and appended to the app environment file on every deploy (and at instance boot for manual). value_from is a Secrets Manager secret ARN, an SSM parameter ARN, or a bare SSM parameter name. Multi-line secret values are not supported (env-file format)." + default = [] + + validation { + condition = alltrue([for s in var.secrets : can(regex("^[A-Za-z_][A-Za-z0-9_]*$", s.name))]) + error_message = "Each secret name must be a valid environment variable name." + } +} + +variable "deploy_health_check_path" { + type = string + description = "Local HTTP path polled on the instance after each deploy to gate success, such as /health. Requires app_port. Null disables the local health gate." + default = null + + validation { + condition = var.deploy_health_check_path == null || can(regex("^/", var.deploy_health_check_path)) + error_message = "The deploy_health_check_path must be an absolute path starting with '/'." + } +} + +variable "deploy_timeout_seconds" { + type = number + description = "Timeout in seconds for the deploy document's script on each instance." + default = 1200 + + validation { + condition = var.deploy_timeout_seconds >= 60 && var.deploy_timeout_seconds <= 14400 + error_message = "The deploy_timeout_seconds must be between 60 and 14400." + } +} + +################################################################################ +# Instances +################################################################################ + +variable "instance_type" { + type = string + description = "EC2 instance type for the Auto Scaling Group." +} + +variable "ami_id" { + type = string + description = "Custom AMI for the instances. Leave null to use the latest Amazon Linux 2023 AMI. Custom AMIs must run cloud-init and include the SSM agent." + default = null + + validation { + condition = var.ami_id == null || can(regex("^ami-", var.ami_id)) + error_message = "The ami_id must be a valid AMI ID starting with 'ami-'." + } +} + +variable "key_name" { + type = string + description = "Key pair name for SSH access to the instances." + default = null +} + +variable "root_volume_size" { + type = number + description = "Root EBS volume size in GB." + default = 30 + + validation { + condition = var.root_volume_size >= 8 && var.root_volume_size <= 16384 + error_message = "The root_volume_size must be between 8 and 16384 GB." + } +} + +variable "root_volume_type" { + type = string + description = "Root EBS volume type." + default = "gp3" + + validation { + condition = contains(["gp3", "gp2", "io1", "io2"], var.root_volume_type) + error_message = "The root_volume_type must be 'gp3', 'gp2', 'io1', or 'io2'." + } +} + +variable "data_volume_creation_enabled" { + type = bool + description = "Attach a dedicated data EBS volume to each instance, formatted and mounted on first boot. Data survives in-place deploys but not instance termination." + default = false +} + +variable "data_volume_size" { + type = number + description = "Data EBS volume size in GB." + default = 20 + + validation { + condition = var.data_volume_size >= 1 && var.data_volume_size <= 16384 + error_message = "The data_volume_size must be between 1 and 16384 GB." + } +} + +variable "data_volume_type" { + type = string + description = "Data EBS volume type." + default = "gp3" + + validation { + condition = contains(["gp3", "gp2", "io1", "io2"], var.data_volume_type) + error_message = "The data_volume_type must be 'gp3', 'gp2', 'io1', or 'io2'." + } +} + +variable "data_volume_mount_path" { + type = string + description = "Host path where the data volume is mounted. Container apps see it at the same path." + default = "/data" + + validation { + condition = can(regex("^/", var.data_volume_mount_path)) + error_message = "The data_volume_mount_path must be an absolute path starting with '/'." + } +} + +variable "additional_user_data" { + type = string + description = "Additional shell script appended to the instance bootstrap user data." + default = "" +} + +################################################################################ +# Auto Scaling Group +################################################################################ + +variable "min_size" { + type = number + description = "Minimum instances in the Auto Scaling Group." + default = 1 + + validation { + condition = var.min_size >= 0 + error_message = "The min_size must be at least 0." + } +} + +variable "max_size" { + type = number + description = "Maximum instances in the Auto Scaling Group." + default = 3 + + validation { + condition = var.max_size >= 1 + error_message = "The max_size must be at least 1." + } +} + +variable "desired_capacity" { + type = number + description = "Desired instances in the Auto Scaling Group. Null lets the group manage it within min/max." + default = null +} + +variable "health_check_type" { + type = string + description = "ASG health check type. 'EC2' replaces instances only on instance failure. 'ELB' also replaces instances failing load balancer health checks; note that in-place deploys briefly deregister instances, so prefer 'EC2' unless deploys are infrequent." + default = "EC2" + + validation { + condition = contains(["EC2", "ELB"], var.health_check_type) + error_message = "The health_check_type must be 'EC2' or 'ELB'." + } +} + +variable "health_check_grace_period" { + type = number + description = "Seconds after launch before ASG health checks apply." + default = 300 + + validation { + condition = var.health_check_grace_period >= 0 + error_message = "The health_check_grace_period must be 0 or greater." + } +} + +variable "cpu_autoscaling_enabled" { + type = bool + description = "Scale the group to maintain the target average CPU utilization." + default = false +} + +variable "cpu_target_value" { + type = number + description = "Average CPU utilization target for target tracking scaling." + default = 70 + + validation { + condition = var.cpu_target_value >= 1 && var.cpu_target_value <= 100 + error_message = "The cpu_target_value must be between 1 and 100." + } +} + +################################################################################ +# Load Balancer Attachment +################################################################################ + +variable "load_balancer_attachment" { + type = object({ + creation_enabled = optional(bool, true) + + target_group = object({ + port = number + deregistration_delay = optional(number, 30) + slow_start = optional(number, 0) + + health_check = optional(object({ + enabled = optional(bool, true) + path = optional(string, "/") + port = optional(string, "traffic-port") + matcher = optional(string, "200-399") + interval = optional(number, 30) + timeout = optional(number, 5) + healthy_threshold = optional(number, 3) + unhealthy_threshold = optional(number, 3) + }), {}) + + stickiness = optional(object({ + enabled = optional(bool, false) + type = optional(string, "lb_cookie") + cookie_duration = optional(number, 86400) + cookie_name = optional(string, null) + }), null) + }) + + listener_rules = optional(list(object({ + listener_arn = string + priority = optional(number, null) + + conditions = list(object({ + type = string + values = list(string) + })) + })), []) + }) + description = "Application Load Balancer attachment: an instance target group plus listener rules on an existing ALB listener. Null runs the group without a load balancer (worker mode)." + default = null + + validation { + condition = try( + var.load_balancer_attachment.target_group.slow_start == 0 || ( + var.load_balancer_attachment.target_group.slow_start >= 30 && + var.load_balancer_attachment.target_group.slow_start <= 900 + ), + true + ) + error_message = "The target group slow_start must be 0 or between 30 and 900 seconds." + } + + validation { + condition = try( + var.load_balancer_attachment.target_group.stickiness == null || + contains(["lb_cookie", "app_cookie"], var.load_balancer_attachment.target_group.stickiness.type), + true + ) + error_message = "The target group stickiness type must be 'lb_cookie' or 'app_cookie'." + } + + validation { + condition = try( + var.load_balancer_attachment.target_group.stickiness == null || + !var.load_balancer_attachment.target_group.stickiness.enabled || + var.load_balancer_attachment.target_group.stickiness.type != "app_cookie" || + length(trimspace(coalesce(var.load_balancer_attachment.target_group.stickiness.cookie_name, ""))) > 0, + true + ) + error_message = "The target group stickiness cookie_name is required for app_cookie stickiness." + } + + validation { + condition = try( + var.load_balancer_attachment.target_group.stickiness == null || ( + var.load_balancer_attachment.target_group.stickiness.cookie_duration >= 1 && + var.load_balancer_attachment.target_group.stickiness.cookie_duration <= 604800 + ), + true + ) + error_message = "The target group stickiness cookie_duration must be between 1 and 604800 seconds." + } +} + +variable "load_balancer_security_group_id" { + type = string + description = "Security group of the load balancer, allowed to reach the app port on the instances." + default = null + + validation { + condition = var.load_balancer_security_group_id == null || can(regex("^sg-", var.load_balancer_security_group_id)) + error_message = "The load_balancer_security_group_id must be a valid security group ID starting with 'sg-'." + } +} + +################################################################################ +# EFS +################################################################################ + +variable "efs_enabled" { + type = bool + description = "Mount an EFS file system on every instance." + default = false +} + +variable "efs_file_system_id" { + type = string + description = "EFS file system ID to mount. Required when efs_enabled is true." + default = null +} + +variable "efs_access_point_id" { + type = string + description = "EFS access point to mount through, when set." + default = null +} + +variable "efs_client_security_group_id" { + type = string + description = "EFS client security group attached to the instances so NFS traffic is allowed." + default = null + + validation { + condition = var.efs_client_security_group_id == null || can(regex("^sg-", var.efs_client_security_group_id)) + error_message = "The efs_client_security_group_id must be a valid security group ID starting with 'sg-'." + } +} + +variable "efs_mount_path" { + type = string + description = "Host path where the EFS file system is mounted." + default = "/mnt/efs" + + validation { + condition = can(regex("^/", var.efs_mount_path)) + error_message = "The efs_mount_path must be an absolute path starting with '/'." + } +} + +################################################################################ +# Artifact Stores +################################################################################ + +variable "ecr_repository_creation_enabled" { + type = bool + description = "Create an ECR repository for images built for this service. Used by the container runtime." + default = false +} + +variable "ecr_force_deletion_enabled" { + type = bool + description = "Allow deleting the ECR repository even when it contains images." + default = false +} + +variable "ecr_scan_on_push_enabled" { + type = bool + description = "Scan images for vulnerabilities after they are pushed to the ECR repository." + default = true +} + +################################################################################ +# Logging +################################################################################ + +variable "log_retention_in_days" { + type = number + description = "CloudWatch log retention for app logs." + default = 30 + + validation { + condition = var.log_retention_in_days >= 1 + error_message = "The log_retention_in_days must be at least 1." + } +} diff --git a/compute/ec2_service/versions.tf b/compute/ec2_service/versions.tf new file mode 100644 index 00000000..985850ec --- /dev/null +++ b/compute/ec2_service/versions.tf @@ -0,0 +1,16 @@ +################################################################################ +# OpenTofu/Terraform and Provider Requirements +################################################################################ + +terraform { + required_version = ">= 1.10.0" + + cloud {} + + required_providers { + aws = { + source = "hashicorp/aws" + version = ">= 6.0" + } + } +} diff --git a/compute/ecs_cluster/rvn-ecs-cluster-definition.yml b/compute/ecs_cluster/rvn-ecs-cluster-definition.yml index e5fac57e..bb663e70 100644 --- a/compute/ecs_cluster/rvn-ecs-cluster-definition.yml +++ b/compute/ecs_cluster/rvn-ecs-cluster-definition.yml @@ -3,8 +3,8 @@ definition: name: ECS Cluster description: Production-ready AWS ECS cluster with Fargate, Fargate Spot, optional EC2 capacity, and shared load balancers. release: - version: 0.3.0 - description: "BREAKING: Replace the boolean Container insights input with a select supporting enhanced observability." + version: 0.3.1 + description: Add additional ALB certificate ARN fields for SNI. module: inputs: - id: network @@ -230,24 +230,17 @@ module: show_when: public_alb_enabled: true type: boolean - - description: ACM certificate module for public ALB HTTPS. - id: public_alb_certificate - label: Certificate - mapped_inputs: - - add_button_label: Add certificate ARN - default: - - <> - description: ACM certificate ARNs for public ALB HTTPS. The first ARN is the default certificate; additional ARNs are attached for SNI. At least one is required when HTTPS is enabled. - id: public_alb_certificate_arns - label: Certificate ARNs - placeholder: arn:aws:acm:... - required: true - type: string_array - required: true - show_when: - public_alb_enabled: true - public_alb_https_enabled: true - type: $ref:rvn-acm-certificate + - $template: ../../partials/templates/alb-certificate-input.yml + with: + additional_certificate_arns_description: Additional ACM certificate ARNs attached to the public ALB HTTPS listener for SNI. + additional_certificate_arns_input_id: public_alb_additional_certificate_arns + certificate_arns_description: ACM certificate ARNs resolved from the selected module. The first ARN is the default certificate. + certificate_arns_input_id: public_alb_certificate_arns + certificate_description: Primary ACM certificate module for public ALB HTTPS. + certificate_input_id: public_alb_certificate + show_when: + public_alb_enabled: true + public_alb_https_enabled: true - collapsible: true description: SSL policy for public ALB HTTPS. id: public_alb_ssl_policy @@ -307,24 +300,17 @@ module: show_when: private_alb_enabled: true type: boolean - - description: ACM certificate module for private ALB HTTPS. - id: private_alb_certificate - label: Certificate - mapped_inputs: - - add_button_label: Add certificate ARN - default: - - <> - description: ACM certificate ARNs for private ALB HTTPS. The first ARN is the default certificate; additional ARNs are attached for SNI. At least one is required when HTTPS is enabled. - id: private_alb_certificate_arns - label: Certificate ARNs - placeholder: arn:aws:acm:... - required: true - type: string_array - required: true - show_when: - private_alb_enabled: true - private_alb_https_enabled: true - type: $ref:rvn-acm-certificate + - $template: ../../partials/templates/alb-certificate-input.yml + with: + additional_certificate_arns_description: Additional ACM certificate ARNs attached to the private ALB HTTPS listener for SNI. + additional_certificate_arns_input_id: private_alb_additional_certificate_arns + certificate_arns_description: ACM certificate ARNs resolved from the selected module. The first ARN is the default certificate. + certificate_arns_input_id: private_alb_certificate_arns + certificate_description: Primary ACM certificate module for private ALB HTTPS. + certificate_input_id: private_alb_certificate + show_when: + private_alb_enabled: true + private_alb_https_enabled: true - collapsible: true description: SSL policy for private ALB HTTPS. id: private_alb_ssl_policy @@ -559,6 +545,8 @@ module: - Backend services - Private HTTP/HTTPS routing + For either ALB, the selected Certificate is the default HTTPS certificate. Use Additional certificate ARNs to attach more ACM certificates to the same listener through SNI. + ### Network load balancers Use Network Load Balancers for TCP/UDP workloads, static IP requirements, very high-throughput connections, or protocols that do not need HTTP routing. For normal web apps and HTTP APIs, prefer an Application Load Balancer. @@ -660,7 +648,9 @@ module: public_nlb_enabled: << module.input.public_nlb_enabled >> name: << module.input.name >> private_alb_access_logs_bucket_arn: << module.input.private_alb_access_logs_bucket_arn >> - private_alb_certificate_arns: << module.input.private_alb_certificate_arns >> + private_alb_certificate_arns: >- + << (module.input.private_alb_certificate_arns != nil ? module.input.private_alb_certificate_arns : []) | + concat(module.input.private_alb_additional_certificate_arns != nil ? module.input.private_alb_additional_certificate_arns : []) >> private_alb_access_logs_enabled: << module.input.private_alb_access_logs_enabled >> private_alb_https_enabled: << module.input.private_alb_https_enabled >> private_alb_idle_timeout: << module.input.private_alb_idle_timeout >> @@ -674,7 +664,9 @@ module: private_nlb_security_group_ids: << module.input.private_nlb_security_group_ids >> private_subnet_ids: << module.input.private_subnet_ids >> public_alb_access_logs_bucket_arn: << module.input.public_alb_access_logs_bucket_arn >> - public_alb_certificate_arns: << module.input.public_alb_certificate_arns >> + public_alb_certificate_arns: >- + << (module.input.public_alb_certificate_arns != nil ? module.input.public_alb_certificate_arns : []) | + concat(module.input.public_alb_additional_certificate_arns != nil ? module.input.public_alb_additional_certificate_arns : []) >> public_alb_access_logs_enabled: << module.input.public_alb_access_logs_enabled >> public_alb_https_enabled: << module.input.public_alb_https_enabled >> public_alb_idle_timeout: << module.input.public_alb_idle_timeout >> diff --git a/compute/ecs_service/rvn-ecs-nlb-definition.yml b/compute/ecs_service/rvn-ecs-nlb-definition.yml index f24be398..8a1b93f6 100644 --- a/compute/ecs_service/rvn-ecs-nlb-definition.yml +++ b/compute/ecs_service/rvn-ecs-nlb-definition.yml @@ -3,9 +3,8 @@ definition: name: ECS Network Service description: Network Load Balanced ECS service for running TCP, UDP, or TLS workloads behind an ECS cluster Network Load Balancer. release: - version: 0.3.0 - description: >- - BREAKING: To upgrade, replace `container_port`, `listener_port`, and `listener_protocol` with one or more `listeners` items; move each TLS certificate, SSL policy, ALPN policy, and target protocol into its TLS listener. All deployments are now rolling. + version: 0.3.1 + description: Improve build source, Railpack command, and builder instance guidance. module: inputs: - id: section_cluster diff --git a/compute/ecs_service/rvn-ecs-web-definition.yml b/compute/ecs_service/rvn-ecs-web-definition.yml index f5740e00..d41b7559 100644 --- a/compute/ecs_service/rvn-ecs-web-definition.yml +++ b/compute/ecs_service/rvn-ecs-web-definition.yml @@ -3,8 +3,8 @@ definition: name: ECS Web Service description: Web server ECS service for running an HTTP application behind an ECS cluster load balancer. release: - version: 0.8.2 - description: "Accept mixed IAM policy document shapes for task role inline policies." + version: 0.8.3 + description: Improve build source, Railpack command, and builder instance guidance. module: inputs: - id: section_cluster @@ -36,11 +36,7 @@ module: type: boolean description: Expose this service through the public ALB. Turn off to use the private ALB. default: true - - id: private_subnet_placement_enabled - label: Run in private subnets - type: boolean - description: Recommended. Requires a NAT gateway or equivalent for internet access and a static IP. - default: true + - $include: ../../partials/inputs/private-subnet-placement.yml - $include: ../../partials/inputs/ecs-service-build-inputs.yml - id: section_deployment label: Deployment @@ -283,33 +279,9 @@ module: required: false show_when: target_group_stickiness_type: app_cookie - - id: section_routing - label: HTTP listener rules - type: section - description: Add at least one domain host rule or path rule. - - id: host_header_values - label: Domain host rules - type: string_array - description: Hostnames that should route to this service, such as app.example.com or *.example.com. Leave empty to use path-based routing. - add_button_label: Add domain host - placeholder: app.example.com - required: false - default: null - - id: path_pattern_values - label: Path rules - type: string_array - description: Path patterns that should route to this service, such as /*, /api/*, or /app/*. If both domain host rules and path rules are empty, the service routes all paths with /*. - add_button_label: Add path pattern - collapsible: true - required: false - default: null - - id: listener_rule_priority - label: Listener rule priority - type: number - description: Optional ALB listener rule priority. Leave blank to let AWS assign the next available priority. - collapsible: true - max: 50000 - min: 1 + - $template: ../../partials/templates/alb-listener-rule-inputs.yml + with: + field_overrides: {} - $include: ../../partials/inputs/ecs-service-capacity-inputs.yml - $include: ../../partials/inputs/ecs-service-autoscaling-inputs.yml - $template: ../../partials/templates/ecs-service-runtime-env-inputs.yml @@ -341,42 +313,35 @@ module: - container_port: << module.input.container_port >> health_check_grace_period_seconds: << module.input.health_check_grace_period_seconds >> load_balancer_attachment: - container_port: << module.input.container_port >> - enabled: true - listener_rules: - - conditions: - - >- - ...<< module.input.host_header_values != nil && module.input.host_header_values != - [] ? [{type: "host-header", values: module.input.host_header_values}] : [] >> - - >- - ...<< module.input.path_pattern_values != nil && module.input.path_pattern_values != - [] ? [{type: "path-pattern", values: module.input.path_pattern_values}] : - module.input.host_header_values != nil && module.input.host_header_values != [] ? [] - : [{type: "path-pattern", values: ["/*"]}] >> - listener_arn: >- - << module.input.public_web_service_enabled ? (module.input.public_alb_https_listener_arn || - module.input.public_alb_http_listener_arn) : - (module.input.private_alb_https_listener_arn || - module.input.private_alb_http_listener_arn) >> - priority: << module.input.listener_rule_priority >> - target_group: - health_check: + $template: ../../partials/templates/alb-load-balancer-attachment.yml + with: + additional_fields: + container_port: << module.input.container_port >> + attachment_control: enabled: true - healthy_threshold: << module.input.healthy_threshold >> - interval: << module.input.health_check_interval >> - matcher: << module.input.health_check_matcher >> - path: << module.input.health_check_path >> - timeout: << module.input.health_check_timeout >> - unhealthy_threshold: << module.input.unhealthy_threshold >> - port: << module.input.container_port >> - protocol: HTTP - slow_start: << module.input.target_group_slow_start >> - stickiness: - cookie_duration: << module.input.target_group_stickiness_cookie_duration >> - cookie_name: '<< module.input.target_group_stickiness_type == "app_cookie" ? module.input.target_group_stickiness_cookie_name : nil >>' - enabled: << module.input.target_group_stickiness_enabled >> - type: << module.input.target_group_stickiness_type >> - target_type: ip + listener_arn: >- + << module.input.public_web_service_enabled ? (module.input.public_alb_https_listener_arn || + module.input.public_alb_http_listener_arn) : + (module.input.private_alb_https_listener_arn || + module.input.private_alb_http_listener_arn) >> + target_group: + health_check: + enabled: true + healthy_threshold: << module.input.healthy_threshold >> + interval: << module.input.health_check_interval >> + matcher: << module.input.health_check_matcher >> + path: << module.input.health_check_path >> + timeout: << module.input.health_check_timeout >> + unhealthy_threshold: << module.input.unhealthy_threshold >> + port: << module.input.container_port >> + protocol: HTTP + slow_start: << module.input.target_group_slow_start >> + stickiness: + cookie_duration: << module.input.target_group_stickiness_cookie_duration >> + cookie_name: '<< module.input.target_group_stickiness_type == "app_cookie" ? module.input.target_group_stickiness_cookie_name : nil >>' + enabled: << module.input.target_group_stickiness_enabled >> + type: << module.input.target_group_stickiness_type >> + target_type: ip load_balancer_security_group_id: >- << module.input.public_web_service_enabled ? module.input.public_alb_security_group_id : module.input.private_alb_security_group_id >> diff --git a/compute/ecs_service/rvn-ecs-worker-definition.yml b/compute/ecs_service/rvn-ecs-worker-definition.yml index d0b11f01..0ad85751 100644 --- a/compute/ecs_service/rvn-ecs-worker-definition.yml +++ b/compute/ecs_service/rvn-ecs-worker-definition.yml @@ -3,8 +3,8 @@ definition: name: ECS Worker description: Background worker ECS service for running private container workloads without exposed ports. release: - version: 0.3.2 - description: "Accept mixed IAM policy document shapes for task role inline policies." + version: 0.3.3 + description: Improve build source, Railpack command, and builder instance guidance. module: inputs: - id: section_cluster diff --git a/compute/lambda/rvn-lambda-definition.yml b/compute/lambda/rvn-lambda-definition.yml index c3608c66..eeab0cb0 100644 --- a/compute/lambda/rvn-lambda-definition.yml +++ b/compute/lambda/rvn-lambda-definition.yml @@ -3,8 +3,8 @@ definition: name: Lambda Function description: AWS Lambda function with zip or container image deployments, alias-based releases, IAM, logs, and optional function URLs. release: - version: 0.3.1 - description: "Add Railpack support for zip package and container image builds." + version: 0.3.2 + description: Improve build source and builder instance guidance. module: inputs: - $include: ../../partials/inputs/aws-account.yml @@ -884,6 +884,11 @@ module: - $template: ../../partials/templates/builder-infrastructure-inputs.yml with: no_options_message: Select an AWS account and region to load available EC2 instance types. + show_when: + build_source: + - dockerfile + - railpack + - nixpacks - id: section_ecr label: Image registry lifecycle type: section diff --git a/hosting/static_site/rvn-aws-static-definition.yml b/hosting/static_site/rvn-aws-static-definition.yml index 55bd00c0..25a8da68 100644 --- a/hosting/static_site/rvn-aws-static-definition.yml +++ b/hosting/static_site/rvn-aws-static-definition.yml @@ -3,8 +3,8 @@ definition: name: Static Hosting description: Static file hosting for S3-backed sites and assets delivered through CloudFront. release: - version: 0.3.3 - description: Replace legacy CloudFront rewrite functions without leaving unmanaged resources. + version: 0.3.4 + description: Improve build source, Railpack command, and builder instance guidance. module: build: builder: >- @@ -318,6 +318,11 @@ module: - $template: ../../partials/templates/builder-infrastructure-inputs.yml with: no_options_message: Select an AWS account and region to load available EC2 instance types. + show_when: + build_source: + - dockerfile + - railpack + - nixpacks - id: section_deploy label: Deploy config type: section diff --git a/networking/alb/rvn-aws-alb-definition.yml b/networking/alb/rvn-aws-alb-definition.yml new file mode 100644 index 00000000..65da6fcd --- /dev/null +++ b/networking/alb/rvn-aws-alb-definition.yml @@ -0,0 +1,345 @@ +definition: + type: rvn-aws-alb + name: AWS Application Load Balancer + description: Standalone Application Load Balancer (ALB) with HTTP/HTTPS listeners that multiple services can attach listener rules to. +release: + version: 0.1.0 + description: Initial module definition. +module: + inputs: + - id: network + immutable: true + label: VPC network + type: $ref:rvn-aws-network + mapped_inputs: + - $template: ../../partials/templates/network-ref-mapped-inputs.yml + with: + aws_account_id_default: << ref.input.aws_account_id >> + aws_region_default: << ref.input.aws_region >> + execution_environment_default: << ref.input.execution_environment_id >> + private_subnet_ids_default: <> + public_subnet_ids_default: <> + vpc_id_default: <> + required: true + - id: section_load_balancer + label: Load balancer + type: section + - id: name + immutable: true + label: Name slug + type: string + description: Name for the load balancer and related resources. + required: true + default: <>-<>-<> + patterns: + - message: 1-32 lowercase letters, numbers, and hyphens. Start and end with a letter or number. + pattern: ^[a-z0-9]([a-z0-9-]{0,30}[a-z0-9])?$ + - id: internal_load_balancer_enabled + label: Internal load balancer + type: boolean + description: Create an internal load balancer reachable only inside the VPC. Turn off for an internet-facing load balancer in the public subnets. + default: false + - id: deletion_protection_enabled + label: Deletion protection + type: boolean + description: Prevent accidental deletion of the load balancer via the AWS API. + default: true + - id: idle_timeout + label: Idle timeout (seconds) + type: number + description: Seconds a connection is allowed to be idle before the load balancer closes it. + collapsible: true + max: 4000 + min: 1 + default: 60 + - id: section_listeners + label: Listeners + type: section + - id: http_listener_enabled + label: HTTP + type: boolean + description: Create an HTTP listener. + default: true + - id: http_listener_port + label: HTTP port + type: number + collapsible: true + max: 65535 + min: 1 + show_when: + http_listener_enabled: true + default: 80 + - id: https_listener_enabled + label: HTTPS + type: boolean + description: Create an HTTPS listener. Select an ACM certificate when enabled. + default: false + - id: https_listener_port + label: HTTPS port + type: number + collapsible: true + max: 65535 + min: 1 + show_when: + https_listener_enabled: true + default: 443 + - $template: ../../partials/templates/alb-certificate-input.yml + with: + additional_certificate_arns_description: Additional ACM certificate ARNs attached to the HTTPS listener for SNI. + additional_certificate_arns_input_id: additional_certificate_arns + certificate_arns_description: ACM certificate ARNs resolved from the selected module. The first ARN is the default certificate. + certificate_arns_input_id: certificate_arns + certificate_description: Primary ACM certificate module for HTTPS. The certificate must use the same AWS account and region as the load balancer. + certificate_input_id: certificate + show_when: + https_listener_enabled: true + - id: ssl_policy + label: SSL policy + type: string + description: SSL policy for the HTTPS listener. + collapsible: true + placeholder: ELBSecurityPolicy-TLS13-1-2-2021-06 + show_when: + https_listener_enabled: true + - id: http_to_https_redirect_enabled + label: Redirect HTTP to HTTPS + type: boolean + description: Redirect HTTP traffic to the HTTPS listener. Applies only when both listeners are enabled. + show_when: + http_listener_enabled: true + https_listener_enabled: true + default: true + - id: section_security + label: Security + type: section + - id: ingress_cidr_blocks + label: Ingress IPv4 CIDRs + type: string_array + description: IPv4 CIDR blocks allowed to reach the load balancer. Defaults to public internet access for an internet-facing load balancer and RFC1918 private ranges for an internal load balancer. Use an empty list to allow no IPv4 ingress. + add_button_label: Add CIDR block + collapsible: true + placeholder: 0.0.0.0/0 + - id: ingress_ipv6_cidr_blocks + label: Ingress IPv6 CIDRs + type: string_array + description: IPv6 CIDR blocks allowed to reach the load balancer. Defaults to public internet access for an internet-facing load balancer and no IPv6 ingress for an internal load balancer. Use an empty list to allow no IPv6 ingress. + add_button_label: Add IPv6 CIDR block + collapsible: true + placeholder: "::/0" + - id: web_acl_arn + label: WAF web ACL ARN + type: string + description: WAFv2 Web ACL to associate with the load balancer. + collapsible: true + placeholder: arn:aws:wafv2:... + - id: section_access_logs + label: Access logs + type: section + - id: access_logs_enabled + label: Access logs + type: boolean + description: Write load balancer access logs to S3. + default: false + - id: access_logs_bucket_arn + label: Access logs bucket ARN + type: string + description: Existing S3 bucket for access logs. Leave blank to create a bucket. + collapsible: true + placeholder: arn:aws:s3:::my-bucket + show_when: + access_logs_enabled: true + - id: access_logs_retention_days + label: Access logs retention (days) + type: number + description: Days to retain access logs in the created bucket. + collapsible: true + min: 1 + show_when: + access_logs_enabled: true + default: 90 + - $include: ../../partials/inputs/misc-section.yml + - $include: ../../partials/inputs/tags.yml + - $include: ../../partials/inputs/terraform-settings.yml + stack: + $template: ../../partials/templates/opentofu-stack.yml + with: + base_path: networking/alb + terraform_variables: + ...overrides: << module.input.advanced_terraform_variables >> + access_logs_bucket_arn: << module.input.access_logs_bucket_arn || nil >> + access_logs_enabled: << module.input.access_logs_enabled >> + access_logs_retention_days: << module.input.access_logs_retention_days >> + certificate_arns: >- + << module.input.https_listener_enabled ? + ((module.input.certificate_arns != nil ? module.input.certificate_arns : []) | + concat(module.input.additional_certificate_arns != nil ? module.input.additional_certificate_arns : [])) : [] >> + deletion_protection_enabled: << module.input.deletion_protection_enabled >> + http_listener_enabled: << module.input.http_listener_enabled >> + http_listener_port: << module.input.http_listener_port >> + http_to_https_redirect_enabled: << module.input.http_to_https_redirect_enabled >> + https_listener_enabled: << module.input.https_listener_enabled >> + https_listener_port: << module.input.https_listener_port >> + idle_timeout: << module.input.idle_timeout >> + ingress_cidr_blocks: >- + << module.input.ingress_cidr_blocks != nil ? module.input.ingress_cidr_blocks : + (module.input.internal_load_balancer_enabled ? ["10.0.0.0/8", "172.16.0.0/12", "192.168.0.0/16"] : ["0.0.0.0/0"]) >> + ingress_ipv6_cidr_blocks: >- + << module.input.ingress_ipv6_cidr_blocks != nil ? module.input.ingress_ipv6_cidr_blocks : + (module.input.internal_load_balancer_enabled ? [] : ["::/0"]) >> + internal_load_balancer_enabled: << module.input.internal_load_balancer_enabled >> + name: << module.input.name >> + region: << module.input.aws_region >> + ssl_policy: << module.input.ssl_policy || nil >> + subnet_ids: "<< module.input.internal_load_balancer_enabled ? module.input.private_subnet_ids : module.input.public_subnet_ids >>" + tags: + $include: ../../partials/stack/ravion-tags.yml + vpc_id: << module.input.vpc_id >> + waf_association_enabled: '<< module.input.web_acl_arn != nil && module.input.web_acl_arn != "" >>' + web_acl_arn: << module.input.web_acl_arn || nil >> + ui: + links: |- + << + (module.input.https_listener_enabled || module.input.http_listener_enabled) && stack.output.alb_dns_name ? + [{"name":"Load balancer","href":(module.input.https_listener_enabled ? "https://" : "http://") + stack.output.alb_dns_name}] : + [] + >> + metrics: + - id: request_count + name: Request count + type: line + source: + type: cloudwatch + aws_account_id: << module.input.aws_account_id >> + dimensions: + LoadBalancer: << stack.output.alb_arn_suffix >> + name: RequestCount + namespace: AWS/ApplicationELB + region: << stack.output.region >> + statistic: Sum + - id: target_response_time + name: Target response time + type: line + source: + type: cloudwatch + aws_account_id: << module.input.aws_account_id >> + dimensions: + LoadBalancer: << stack.output.alb_arn_suffix >> + name: TargetResponseTime + namespace: AWS/ApplicationELB + region: << stack.output.region >> + statistic: Average + - id: target_5xx_errors + name: Target 5xx errors + type: line + source: + type: cloudwatch + aws_account_id: << module.input.aws_account_id >> + dimensions: + LoadBalancer: << stack.output.alb_arn_suffix >> + name: HTTPCode_Target_5XX_Count + namespace: AWS/ApplicationELB + region: << stack.output.region >> + statistic: Sum + - id: elb_5xx_errors + name: Load balancer 5xx errors + type: line + source: + type: cloudwatch + aws_account_id: << module.input.aws_account_id >> + dimensions: + LoadBalancer: << stack.output.alb_arn_suffix >> + name: HTTPCode_ELB_5XX_Count + namespace: AWS/ApplicationELB + region: << stack.output.region >> + statistic: Sum + - id: active_connections + name: Active connections + type: line + source: + type: cloudwatch + aws_account_id: << module.input.aws_account_id >> + dimensions: + LoadBalancer: << stack.output.alb_arn_suffix >> + name: ActiveConnectionCount + namespace: AWS/ApplicationELB + region: << stack.output.region >> + statistic: Sum + readme: | + Standalone Application Load Balancer (ALB) with HTTP/HTTPS listeners that multiple services can attach listener rules to. + + ## Overview + + Use this module to create an Application Load Balancer inside a VPC that is shared by one or more services. Select a VPC network and Ravion provisions the load balancer, its security group, HTTP and HTTPS listeners with a fixed default response, optional access logging, and optional WAF association. + + The load balancer intentionally serves no traffic by itself. Services that reference this module, such as EC2 services, create their own target groups and listener rules on the shared listeners. + + Terraform source: [flightcontrolhq/modules/networking/alb](https://github.com/flightcontrolhq/modules/tree/$local.module_tag/networking/alb) + + ## Use cases + + | Scenario | Benefit | + | --- | --- | + | Several services on one domain | Share one load balancer, route by host or path rules per service. | + | Public web apps and APIs | Internet-facing load balancer in the public subnets with HTTPS. | + | Internal services | Internal load balancer reachable only inside the VPC. | + | Cost control | One load balancer instead of one per service. | + + ## Placement and security + + The load balancer is internet-facing in the public subnets by default. Enable Internal load balancer to place it in the private subnets, reachable only inside the VPC. + + The module creates a dedicated security group. Internet-facing load balancers allow 0.0.0.0/0 and ::/0 on the listener ports by default. Internal load balancers allow the RFC1918 IPv4 ranges and no IPv6 ingress by default. Restrict or remove access with Ingress IPv4 CIDRs and Ingress IPv6 CIDRs; an explicitly empty list creates no rules for that IP version. Services that attach to the load balancer allow traffic from this security group on their app ports. + + ## Listeners and TLS + + HTTP is enabled on port 80 by default. Enable HTTPS and select a primary ACM Certificate module in the same AWS account and region. Add Additional certificate ARNs when the listener should serve more certificates through SNI. When both listeners are enabled, HTTP traffic redirects to HTTPS by default. The HTTP and HTTPS listener ports remain configurable for workloads that cannot use ports 80 and 443. + + Requests that match no service listener rule receive a fixed 503 response, so an unmatched host or path never reaches a service by accident. + + ## Access logs and WAF + + Enable Access logs to write load balancer access logs to S3. Leave the bucket ARN blank to create a bucket with the configured retention, or point at an existing bucket. Associate a WAFv2 Web ACL by providing its ARN. + + ## Configuration + + | Field | Required | Default | Description | + | --- | --- | --- | --- | + | VPC network | Yes | - | Existing rvn-aws-network module instance | + | Name slug | Yes | {project}-{env}-{module} | Name for the load balancer, 32 characters max | + | Internal load balancer | No | false | Place the load balancer in private subnets | + | Deletion protection | No | true | Prevent accidental deletion | + | Idle timeout (seconds) | No | 60 | Connection idle timeout | + | HTTP | No | true | Create an HTTP listener | + | HTTP port | No | 80 | HTTP listener port | + | HTTPS | No | false | Create an HTTPS listener | + | HTTPS port | No | 443 | HTTPS listener port | + | Certificate | Yes* | - | Primary ACM certificate module for HTTPS | + | Additional certificate ARNs | No | [] | Additional ACM certificates attached for SNI | + | SSL policy | No | TLS 1.3 policy | SSL policy for HTTPS | + | Redirect HTTP to HTTPS | No | true | Applies when both listeners are enabled | + | Ingress IPv4/IPv6 CIDRs | No | Visibility-dependent | Sources allowed to reach the load balancer; empty lists create no ingress rules | + | WAF web ACL ARN | No | - | WAFv2 Web ACL to associate | + | Access logs | No | false | Write access logs to S3 | + | Access logs bucket ARN | No | Created bucket | Existing bucket for access logs | + | Access logs retention (days) | No | 90 | Retention for the created bucket | + | Tags | No | Standard Ravion tags | Additional tags applied to resources | + | Advanced Terraform variables | No | {} | Raw lower-level overrides for exceptional cases | + | OpenTofu version override | No | Ravion default | Override the OpenTofu version for the stack | + | Ravion Terraform workspace name | No | {project}-{env}-{module} | Override the state backend workspace name | + + *Required when HTTPS is enabled. + + ## Design decisions + + - The default listener action is a fixed 503 response. Routing is owned by the services that attach listener rules, so the load balancer stays service-agnostic. + - Internal placement uses the network's private subnets and internet-facing placement uses the public subnets; subnet selection is not exposed separately. + - Internal ingress defaults to RFC1918 IPv4 ranges with no IPv6 ingress, while internet-facing ingress defaults to all IPv4 and IPv6 sources. + - HTTP-only is allowed for development, but production load balancers should enable HTTPS with an ACM certificate. + - Advanced ALB behaviors such as desync mitigation, header handling, and HTTP/2 keep their safe Terraform defaults and are available through Advanced Terraform variables. + + ## Learn more + + - [Application Load Balancers](https://docs.aws.amazon.com/elasticloadbalancing/latest/application/introduction.html) + - [Listener rules](https://docs.aws.amazon.com/elasticloadbalancing/latest/application/listener-update-rules.html) + - [ACM certificates](https://docs.aws.amazon.com/acm/latest/userguide/acm-overview.html) + - [AWS WAF](https://docs.aws.amazon.com/waf/latest/developerguide/waf-chapter.html) diff --git a/partials/inputs/build-git-source.yml b/partials/inputs/build-git-source.yml index 20d2bddb..2862b459 100644 --- a/partials/inputs/build-git-source.yml +++ b/partials/inputs/build-git-source.yml @@ -1,5 +1,6 @@ - id: source_repo label: Git repository + description: Repository containing the application source for Dockerfile or Railpack builds. required: true show_when: $with.show_when type: gitrepo diff --git a/partials/inputs/ecs-service-autoscaling-inputs.yml b/partials/inputs/ecs-service-autoscaling-inputs.yml index 8d500521..ec8dd665 100644 --- a/partials/inputs/ecs-service-autoscaling-inputs.yml +++ b/partials/inputs/ecs-service-autoscaling-inputs.yml @@ -5,21 +5,17 @@ label: Autoscaling type: boolean default: true -- id: min_capacity - label: Minimum tasks - type: number - description: Recommend at least 2 for production. - min: 0 - show_when: - auto_scaling_enabled: true - default: 1 -- id: max_capacity - label: Maximum tasks - type: number - min: 1 - show_when: - auto_scaling_enabled: true - default: 3 +- $template: ../templates/autoscaling-capacity-inputs.yml + with: + min_capacity_overrides: + label: Minimum tasks + description: Recommend at least 2 for production. + show_when: + auto_scaling_enabled: true + max_capacity_overrides: + label: Maximum tasks + show_when: + auto_scaling_enabled: true - id: desired_count label: Desired tasks type: number diff --git a/partials/inputs/ecs-service-builder-ecr-misc-inputs.yml b/partials/inputs/ecs-service-builder-ecr-misc-inputs.yml index b4353ced..d083927a 100644 --- a/partials/inputs/ecs-service-builder-ecr-misc-inputs.yml +++ b/partials/inputs/ecs-service-builder-ecr-misc-inputs.yml @@ -1,6 +1,11 @@ - $template: ../templates/builder-infrastructure-inputs.yml with: no_options_message: Select a VPC Network, or enter an AWS account and region, to load available EC2 instance types. + show_when: + build_source: + - dockerfile + - railpack + - nixpacks - id: section_ecr label: Image registry lifecycle type: section diff --git a/partials/inputs/private-subnet-placement.yml b/partials/inputs/private-subnet-placement.yml new file mode 100644 index 00000000..7b88b267 --- /dev/null +++ b/partials/inputs/private-subnet-placement.yml @@ -0,0 +1,5 @@ +id: private_subnet_placement_enabled +label: Run in private subnets +type: boolean +description: Recommended. Requires a NAT gateway or equivalent for internet access and a static IP. +default: true diff --git a/partials/inputs/railpack.yml b/partials/inputs/railpack.yml index 89c25201..bd96c867 100644 --- a/partials/inputs/railpack.yml +++ b/partials/inputs/railpack.yml @@ -14,11 +14,13 @@ type: string - id: railpack_install_cmd label: Install command + description: Optional dependency installation command. Leave blank to use Railpack detection. placeholder: Railpack default show_when: $with.show_when type: string - id: railpack_build_cmd label: Build command + description: Optional application build command. Leave blank to use Railpack detection. placeholder: Railpack default show_when: $with.show_when type: string diff --git a/partials/templates/alb-certificate-input.yml b/partials/templates/alb-certificate-input.yml new file mode 100644 index 00000000..fc29c1c6 --- /dev/null +++ b/partials/templates/alb-certificate-input.yml @@ -0,0 +1,25 @@ +- description: $with.certificate_description + id: $with.certificate_input_id + label: Certificate + mapped_inputs: + - add_button_label: Add certificate ARN + default: + - <> + description: $with.certificate_arns_description + id: $with.certificate_arns_input_id + label: Certificate ARNs + placeholder: arn:aws:acm:... + required: true + type: string_array + required: true + show_when: $with.show_when + type: $ref:rvn-acm-certificate +- add_button_label: Add certificate ARN + collapsible: true + default: [] + description: $with.additional_certificate_arns_description + id: $with.additional_certificate_arns_input_id + label: Additional certificate ARNs + placeholder: arn:aws:acm:... + show_when: $with.show_when + type: string_array diff --git a/partials/templates/alb-listener-rule-inputs.yml b/partials/templates/alb-listener-rule-inputs.yml new file mode 100644 index 00000000..561bd990 --- /dev/null +++ b/partials/templates/alb-listener-rule-inputs.yml @@ -0,0 +1,31 @@ +- $merge: $with.field_overrides + id: section_routing + label: HTTP listener rules + type: section + description: Add at least one domain host rule or path rule. +- $merge: $with.field_overrides + id: host_header_values + label: Domain host rules + type: string_array + description: Hostnames that should route to this service, such as app.example.com or *.example.com. Leave empty to use path-based routing. + add_button_label: Add domain host + placeholder: app.example.com + required: false + default: null +- $merge: $with.field_overrides + id: path_pattern_values + label: Path rules + type: string_array + description: Path patterns that should route to this service, such as /*, /api/*, or /app/*. If both domain host rules and path rules are empty, the service routes all paths with /*. + add_button_label: Add path pattern + collapsible: true + required: false + default: null +- $merge: $with.field_overrides + id: listener_rule_priority + label: Listener rule priority + type: number + description: Optional ALB listener rule priority. Leave blank to let AWS assign the next available priority. + collapsible: true + max: 50000 + min: 1 diff --git a/partials/templates/alb-load-balancer-attachment.yml b/partials/templates/alb-load-balancer-attachment.yml new file mode 100644 index 00000000..90ca4e92 --- /dev/null +++ b/partials/templates/alb-load-balancer-attachment.yml @@ -0,0 +1,16 @@ +$merge: + - $with.additional_fields + - $with.attachment_control +listener_rules: + - conditions: + - >- + ...<< module.input.host_header_values != nil && module.input.host_header_values != + [] ? [{type: "host-header", values: module.input.host_header_values}] : [] >> + - >- + ...<< module.input.path_pattern_values != nil && module.input.path_pattern_values != + [] ? [{type: "path-pattern", values: module.input.path_pattern_values}] : + module.input.host_header_values != nil && module.input.host_header_values != [] ? [] + : [{type: "path-pattern", values: ["/*"]}] >> + listener_arn: $with.listener_arn + priority: << module.input.listener_rule_priority >> +target_group: $with.target_group diff --git a/partials/templates/autoscaling-capacity-inputs.yml b/partials/templates/autoscaling-capacity-inputs.yml new file mode 100644 index 00000000..c78ae4be --- /dev/null +++ b/partials/templates/autoscaling-capacity-inputs.yml @@ -0,0 +1,14 @@ +- $merge: + - id: min_capacity + label: Minimum capacity + type: number + min: 0 + default: 1 + - $with.min_capacity_overrides +- $merge: + - id: max_capacity + label: Maximum capacity + type: number + min: 1 + default: 3 + - $with.max_capacity_overrides diff --git a/partials/templates/builder-infrastructure-inputs.yml b/partials/templates/builder-infrastructure-inputs.yml index b0c307bf..9482df33 100644 --- a/partials/templates/builder-infrastructure-inputs.yml +++ b/partials/templates/builder-infrastructure-inputs.yml @@ -1,26 +1,20 @@ - id: section_builder_config label: Builder config - show_when: - build_source: - - dockerfile - - railpack - - nixpacks + show_when: $with.show_when type: section - default: ec2 - description: Use EC2 for quick start or EC2 spot for cheaper, but potentially delayed builds. + description: Use on-demand EC2 for predictable availability or EC2 Spot for lower cost with possible capacity delays or interruption. id: build_infrastructure_type label: Builder instance type required: true - show_when: - build_source: - - dockerfile - - railpack - - nixpacks + show_when: $with.show_when type: string values: - label: EC2 + description: Use on-demand capacity for predictable availability without Spot interruption. value: ec2 - label: EC2 spot + description: Use lower-cost Spot capacity that can wait for capacity or be interrupted by AWS. value: ec2-spot - default: c7a.4xlarge description: EC2 instance type for builds. Start with the default value, then increase or decrease it based on the resource usage report at the end of builds. @@ -28,11 +22,7 @@ label: Builder instance size no_options_message: $with.no_options_message required: true - show_when: - build_source: - - dockerfile - - railpack - - nixpacks + show_when: $with.show_when type: string values: $values:aws/ec2/instances?awsAccountId=<>®ion=<>&costUnit=minute - collapsible: true @@ -40,11 +30,7 @@ id: build_execution_environment_id label: Builder execution environment required: false - show_when: - build_source: - - dockerfile - - railpack - - nixpacks + show_when: $with.show_when type: string values: $values:ravion/execution_environments - collapsible: true @@ -52,9 +38,5 @@ id: build_ami label: Builder AMI required: false - show_when: - build_source: - - dockerfile - - railpack - - nixpacks + show_when: $with.show_when type: string diff --git a/tools/ravion-modules/test/compiler.test.ts b/tools/ravion-modules/test/compiler.test.ts index 5b235264..e89094be 100644 --- a/tools/ravion-modules/test/compiler.test.ts +++ b/tools/ravion-modules/test/compiler.test.ts @@ -143,6 +143,8 @@ describe("compiler", () => { assert.deepEqual(getBuildSourceShowWhen(findInput(inputs, "section_builder_config")), ["dockerfile", "railpack", "nixpacks"]); assert.deepEqual(getBuildSourceShowWhen(findInput(inputs, "section_ecr")), ["dockerfile", "railpack", "nixpacks"]); + assert.equal(findInput(inputs, "min_capacity").label, "Minimum tasks"); + assert.equal(findInput(inputs, "max_capacity").label, "Maximum tasks"); const build = getModuleBuild(compiled.module); const builder = assertString(build.builder); @@ -191,6 +193,329 @@ describe("compiler", () => { assert.doesNotMatch(builder, /cache_from: \{tag: "railpack"\}/); }); + it("compiles primary ALB certificate references, additional SNI certificates, and visibility-aware ingress defaults", async () => { + const [alb, cluster] = await Promise.all([ + compileDefinitionFile(join(repoRoot, "networking", "alb", "rvn-aws-alb-definition.yml")), + compileDefinitionFile(join(repoRoot, "compute", "ecs_cluster", "rvn-ecs-cluster-definition.yml")), + ]); + const albInputs = getModuleInputs(alb.module); + + assert.equal(findInput(albInputs, "internal_load_balancer_enabled").label, "Internal load balancer"); + assert.equal(findInput(albInputs, "http_listener_enabled").label, "HTTP"); + assert.equal(findInput(albInputs, "https_listener_enabled").label, "HTTPS"); + assert.ok(findInput(albInputs, "http_listener_port")); + assert.ok(findInput(albInputs, "https_listener_port")); + + const certificate = findInput(albInputs, "certificate"); + assert.equal(certificate.type, "$ref:rvn-acm-certificate"); + assert.equal(certificate.required, true); + assert.deepEqual(certificate.show_when, { https_listener_enabled: true }); + const certificateMappedInputs = certificate.mapped_inputs; + assert.ok(Array.isArray(certificateMappedInputs), "certificate.mapped_inputs should be an array"); + const certificateArns = assertRecord(certificateMappedInputs[0], "certificate ARN mapped input"); + assert.equal(certificateArns.id, "certificate_arns"); + assert.equal(certificateArns.type, "string_array"); + assert.deepEqual(certificateArns.default, ["<>"]); + const additionalCertificateArns = findInput(albInputs, "additional_certificate_arns"); + assert.equal(additionalCertificateArns.type, "string_array"); + assert.deepEqual(additionalCertificateArns.default, []); + assert.deepEqual(additionalCertificateArns.show_when, { https_listener_enabled: true }); + + const clusterInputs = getModuleInputs(cluster.module); + for (const [inputId, mappedInputId, additionalInputId] of [ + ["public_alb_certificate", "public_alb_certificate_arns", "public_alb_additional_certificate_arns"], + ["private_alb_certificate", "private_alb_certificate_arns", "private_alb_additional_certificate_arns"], + ]) { + const clusterCertificate = findInput(clusterInputs, inputId); + assert.equal(clusterCertificate.type, "$ref:rvn-acm-certificate"); + const mappedInputs = clusterCertificate.mapped_inputs; + assert.ok(Array.isArray(mappedInputs), `${inputId}.mapped_inputs should be an array`); + const mappedInput = assertRecord(mappedInputs[0], `${inputId} ARN mapped input`); + assert.equal(mappedInput.id, mappedInputId); + assert.equal(mappedInput.type, "string_array"); + assert.deepEqual(findInput(clusterInputs, additionalInputId).default, []); + } + + assert.match( + assertString(getTerraformVariable(alb.module, "certificate_arns")), + /concat\(module\.input\.additional_certificate_arns/, + ); + assert.match( + assertString(getTerraformVariable(cluster.module, "public_alb_certificate_arns")), + /concat\(module\.input\.public_alb_additional_certificate_arns/, + ); + assert.match( + assertString(getTerraformVariable(cluster.module, "private_alb_certificate_arns")), + /concat\(module\.input\.private_alb_additional_certificate_arns/, + ); + + const ipv4Ingress = assertString(getTerraformVariable(alb.module, "ingress_cidr_blocks")); + assert.match(ipv4Ingress, /ingress_cidr_blocks != nil/); + assert.doesNotMatch(ipv4Ingress, /ingress_cidr_blocks != \[\]/); + assert.match(ipv4Ingress, /internal_load_balancer_enabled/); + assert.match(ipv4Ingress, /10\.0\.0\.0\/8/); + assert.match(ipv4Ingress, /0\.0\.0\.0\/0/); + + const ipv6Ingress = assertString(getTerraformVariable(alb.module, "ingress_ipv6_cidr_blocks")); + assert.match(ipv6Ingress, /internal_load_balancer_enabled \? \[\]/); + assert.match(ipv6Ingress, /::\/0/); + + const ui = assertRecord(alb.module.ui, "module.ui"); + const links = assertString(ui.links); + assert.match(links, /https_listener_enabled/); + assert.match(links, /http_listener_enabled/); + assert.match(links, /stack\.output\.alb_dns_name/); + }); + + it("compiles EC2 service load balancer source choices", async () => { + const compiled = await compileDefinitionFile( + join(repoRoot, "compute", "ec2_service", "rvn-ec2-service-definition.yml"), + ); + const inputs = getModuleInputs(compiled.module); + + assert.equal(compiled.type, "rvn-ec2-service"); + assert.deepEqual( + inputs.filter((input) => input.type === "section").map((input) => input.id), + [ + "section_service", + "section_build", + "section_dockerfile", + "section_railpack", + "section_deployment", + "section_web", + "section_routing", + "section_health", + "section_storage", + "section_scaling", + "section_app_config", + "section_networking", + "section_builder_config", + "section_ecr", + "section_logging", + "section_advanced", + ], + ); + + for (const inputId of [ + "deployment_concurrency_max", + "deployment_errors_max", + "target_group_slow_start", + "target_group_stickiness_type", + "target_group_stickiness_cookie_name", + "health_check_grace_period", + "direct_access_cidr_blocks", + "data_volume_creation_enabled", + "min_capacity", + "max_capacity", + "cpu_autoscaling_enabled", + "ecr_scan_on_push_enabled", + ]) { + assert.ok(findInput(inputs, inputId), `expected EC2 service input ${inputId}`); + } + + assert.deepEqual(getBuildSourceShowWhen(findInput(inputs, "section_builder_config")), ["dockerfile", "railpack"]); + assert.deepEqual(findInput(inputs, "deploy_source_base_path").show_when, { + deploy_type: "manual", + deploy_source_repo: { not: "" }, + }); + assert.equal(inputs.some((input) => input.id === "min_size" || input.id === "max_size"), false); + assert.equal(findInput(inputs, "min_capacity").label, "Minimum instances"); + assert.equal(findInput(inputs, "max_capacity").label, "Maximum instances"); + assert.equal(getTerraformVariable(compiled.module, "min_size"), "<< module.input.min_capacity >>"); + assert.equal(getTerraformVariable(compiled.module, "max_size"), "<< module.input.max_capacity >>"); + + const imageRef = findInput(getDeployInputs(compiled.module), "image_ref"); + assert.deepEqual(imageRef.patterns, [ + { + message: "Image tags and digests must not contain whitespace.", + pattern: "^\\S+$", + }, + ]); + + const loadBalancerSource = findInput(inputs, "load_balancer_source"); + assert.equal(loadBalancerSource.default, "standalone_alb"); + assert.equal(loadBalancerSource.immutable, true); + assert.deepEqual(getValueOptions(loadBalancerSource), ["standalone_alb", "ecs_cluster"]); + assert.deepEqual(loadBalancerSource.show_when, { http_traffic_enabled: true }); + + const standaloneAlb = findInput(inputs, "alb"); + assert.equal(standaloneAlb.required, true); + assert.deepEqual(standaloneAlb.show_when, { + http_traffic_enabled: true, + load_balancer_source: "standalone_alb", + }); + const standaloneMappedInputs = standaloneAlb.mapped_inputs; + assert.ok(Array.isArray(standaloneMappedInputs), "alb.mapped_inputs should be an array"); + assert.ok( + standaloneMappedInputs.some((input) => assertRecord(input, "ALB mapped input").id === "alb_arn_suffix"), + "expected standalone ALB ARN suffix mapping", + ); + + const ecsCluster = findInput(inputs, "ecs_cluster"); + assert.equal(ecsCluster.required, true); + assert.equal(ecsCluster.immutable, true); + assert.deepEqual(ecsCluster.show_when, { + http_traffic_enabled: true, + load_balancer_source: "ecs_cluster", + }); + const clusterMappedInputs = ecsCluster.mapped_inputs; + assert.ok(Array.isArray(clusterMappedInputs), "ecs_cluster.mapped_inputs should be an array"); + const clusterMappedInputIds = clusterMappedInputs.map((input) => assertRecord(input, "ECS cluster mapped input").id); + for (const inputId of [ + "public_alb_http_listener_arn", + "public_alb_https_listener_arn", + "public_alb_security_group_id", + "public_alb_arn_suffix", + "private_alb_http_listener_arn", + "private_alb_https_listener_arn", + "private_alb_security_group_id", + "private_alb_arn_suffix", + ]) { + assert.ok(clusterMappedInputIds.includes(inputId), `expected ECS cluster mapping ${inputId}`); + } + + const clusterAlbVisibility = findInput(inputs, "ecs_cluster_alb_visibility"); + assert.equal(clusterAlbVisibility.default, "public"); + assert.equal(clusterAlbVisibility.immutable, undefined); + assert.deepEqual(getValueOptions(clusterAlbVisibility), ["public", "private"]); + assert.deepEqual(clusterAlbVisibility.show_when, { + http_traffic_enabled: true, + load_balancer_source: "ecs_cluster", + }); + const loadBalancerAttachment = assertRecord( + getTerraformVariable(compiled.module, "load_balancer_attachment"), + "load_balancer_attachment", + ); + const listenerRules = loadBalancerAttachment.listener_rules; + assert.equal(loadBalancerAttachment.creation_enabled, "<< module.input.http_traffic_enabled >>"); + assert.ok(Array.isArray(listenerRules) && listenerRules.length === 1, "load balancer attachment should have one listener rule"); + const listenerArn = assertString(assertRecord(listenerRules[0], "listener rule").listener_arn); + assert.match(listenerArn, /load_balancer_source == "standalone_alb"/); + assert.match(listenerArn, /alb_https_listener_arn \|\| module\.input\.alb_http_listener_arn/); + assert.match(listenerArn, /ecs_cluster_alb_visibility == "public"/); + assert.match(listenerArn, /public_alb_https_listener_arn \|\| module\.input\.public_alb_http_listener_arn/); + assert.match(listenerArn, /private_alb_https_listener_arn \|\| module\.input\.private_alb_http_listener_arn/); + + const targetGroup = assertRecord(loadBalancerAttachment.target_group, "load_balancer_attachment.target_group"); + assert.equal(targetGroup.slow_start, "<< module.input.target_group_slow_start >>"); + const stickiness = assertString(targetGroup.stickiness); + assert.match(stickiness, /module\.input\.target_group_stickiness_type/); + assert.match(stickiness, /module\.input\.target_group_stickiness_cookie_name/); + + const loadBalancerSecurityGroupId = assertString( + getTerraformVariable(compiled.module, "load_balancer_security_group_id"), + ); + assert.match(loadBalancerSecurityGroupId, /load_balancer_source == "standalone_alb"/); + assert.match(loadBalancerSecurityGroupId, /alb_security_group_id/); + assert.match(loadBalancerSecurityGroupId, /public_alb_security_group_id/); + assert.match(loadBalancerSecurityGroupId, /private_alb_security_group_id/); + + const ecrRepositoryCreationEnabled = assertString( + getTerraformVariable(compiled.module, "ecr_repository_creation_enabled"), + ); + assert.match(ecrRepositoryCreationEnabled, /module\.input\.deploy_type == "container"/); + assert.equal( + getTerraformVariable(compiled.module, "ecr_scan_on_push_enabled"), + "<< module.input.ecr_scan_on_push_enabled >>", + ); + assert.equal( + getTerraformVariable(compiled.module, "health_check_grace_period"), + "<< module.input.health_check_grace_period >>", + ); + assert.match( + assertString(getTerraformVariable(compiled.module, "direct_access_cidr_blocks")), + /module\.input\.http_traffic_enabled/, + ); + assert.equal( + getTerraformVariable(compiled.module, "public_ip_assignment_enabled"), + "<< module.input.private_subnet_placement_enabled ? false : true >>", + ); + assert.equal( + getTerraformVariable(compiled.module, "data_volume_creation_enabled"), + "<< module.input.data_volume_creation_enabled >>", + ); + assert.match( + assertString(getTerraformVariable(compiled.module, "deploy_health_check_path")), + /module\.input\.health_check_path/, + ); + assert.match( + assertString(getTerraformVariable(compiled.module, "container_start_command")), + /module\.input\.container_start_command/, + ); + + const deploy = assertRecord(compiled.module.deploy, "module.deploy"); + assert.equal(deploy.timeout, 86400); + assert.deepEqual(deploy.concurrency, { queue_overflow: "oldest", queue_size: 1 }); + assert.deepEqual(deploy.strategy, { + type: "rolling", + concurrency_max: "<< module.input.deployment_concurrency_max >>", + errors_max: "<< module.input.deployment_errors_max >>", + }); + + const ui = assertRecord(compiled.module.ui, "module.ui"); + const metrics = assertString(ui.metrics); + assert.match(metrics, /module\.input\.http_traffic_enabled/); + assert.match(metrics, /GroupDesiredCapacity/); + assert.match(metrics, /GroupInServiceInstances/); + assert.match(metrics, /LoadBalancer:/); + assert.match(metrics, /TargetGroup:/); + assert.match(metrics, /HTTPCode_Target_4XX_Count/); + assert.match(metrics, /UnHealthyHostCount/); + assert.doesNotMatch(metrics, /namespace:"AWS\/EC2"/); + }); + + it("propagates shared build input guidance to every consumer", async () => { + const definitionPaths = [ + ["compute", "ec2_service", "rvn-ec2-service-definition.yml"], + ["compute", "ecs_service", "rvn-ecs-nlb-definition.yml"], + ["compute", "ecs_service", "rvn-ecs-web-definition.yml"], + ["compute", "ecs_service", "rvn-ecs-worker-definition.yml"], + ["compute", "lambda", "rvn-lambda-definition.yml"], + ["hosting", "static_site", "rvn-aws-static-definition.yml"], + ]; + const definitions = await Promise.all(definitionPaths.map((path) => compileDefinitionFile(join(repoRoot, ...path)))); + + for (const definition of definitions) { + const inputs = getModuleInputs(definition.module); + assert.equal( + findInput(inputs, "source_repo").description, + "Repository containing the application source for Dockerfile or Railpack builds.", + `${definition.type} should include shared Git source guidance`, + ); + + const builderType = findInput(inputs, "build_infrastructure_type"); + assert.equal( + builderType.description, + "Use on-demand EC2 for predictable availability or EC2 Spot for lower cost with possible capacity delays or interruption.", + `${definition.type} should include shared builder guidance`, + ); + const builderOptions = builderType.values; + assert.ok(Array.isArray(builderOptions), `${definition.type} builder type should have values`); + assert.deepEqual( + builderOptions.map((option) => { + const value = assertRecord(option, `${definition.type} builder option`); + return [value.value, value.description]; + }), + [ + ["ec2", "Use on-demand capacity for predictable availability without Spot interruption."], + ["ec2-spot", "Use lower-cost Spot capacity that can wait for capacity or be interrupted by AWS."], + ], + ); + } + + for (const definition of definitions.filter((candidate) => candidate.type !== "rvn-lambda")) { + const inputs = getModuleInputs(definition.module); + assert.equal( + findInput(inputs, "railpack_install_cmd").description, + "Optional dependency installation command. Leave blank to use Railpack detection.", + ); + assert.equal( + findInput(inputs, "railpack_build_cmd").description, + "Optional application build command. Leave blank to use Railpack detection.", + ); + } + }); + it("compiles rolling ECS NLB services with per-listener configuration", async () => { const compiled = await compileDefinitionFile(join(repoRoot, "compute", "ecs_service", "rvn-ecs-nlb-definition.yml")); const listeners = findInput(getModuleInputs(compiled.module), "listeners"); @@ -240,6 +565,13 @@ function getModuleBuild(module: Record): Record): Record[] { + const deploy = assertRecord(module.deploy, "module.deploy"); + const inputs = deploy.inputs; + assert.ok(Array.isArray(inputs), "module.deploy.inputs should be an array"); + return inputs.map((input) => assertRecord(input, "deploy input")); +} + function findInput(inputs: Record[], id: string): Record { const input = inputs.find((candidate) => candidate.id === id); assert.ok(input, `expected input ${id}`);