From ad2277da935bfc332264c6d5406f7b5154d5c496 Mon Sep 17 00:00:00 2001 From: Giacomo Marciani Date: Tue, 18 Aug 2026 16:32:04 -0400 Subject: [PATCH 1/8] [Test] test_patching_cluster: add crc32_generic to lazily-loaded kernel modules on every head node to prevent false alarms. (cherry picked from commit 7a20fe44fcc0b2be1e9239008bf1146191168521) --- .../tests/patching/test_patching.py | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/tests/integration-tests/tests/patching/test_patching.py b/tests/integration-tests/tests/patching/test_patching.py index 17df4dff93..63c9670d61 100644 --- a/tests/integration-tests/tests/patching/test_patching.py +++ b/tests/integration-tests/tests/patching/test_patching.py @@ -55,37 +55,37 @@ # Maintained per OS and per node type. Keep module names alphabetically sorted. LAZY_KERNEL_MODULES = { "alinux2023": { - HEAD_NODE: ["tls"], + HEAD_NODE: ["crc32_generic", "tls"], COMPUTE_NODE: ["tls"], LOGIN_NODE: ["tls"], }, "rhel8": { - HEAD_NODE: ["af_packet_diag", "inet_diag", "tcp_diag", "tls", "udp_diag"], + HEAD_NODE: ["af_packet_diag", "crc32_generic", "inet_diag", "tcp_diag", "tls", "udp_diag"], COMPUTE_NODE: ["tls"], LOGIN_NODE: ["tls"], }, "rhel9": { - HEAD_NODE: ["tls"], + HEAD_NODE: ["crc32_generic", "tls"], COMPUTE_NODE: ["tls"], LOGIN_NODE: ["tls"], }, "rocky8": { - HEAD_NODE: ["af_packet_diag", "inet_diag", "tcp_diag", "tls", "udp_diag"], + HEAD_NODE: ["af_packet_diag", "crc32_generic", "inet_diag", "tcp_diag", "tls", "udp_diag"], COMPUTE_NODE: ["tls"], LOGIN_NODE: ["tls"], }, "rocky9": { - HEAD_NODE: ["tls"], + HEAD_NODE: ["crc32_generic", "tls"], COMPUTE_NODE: ["tls"], LOGIN_NODE: ["tls"], }, "ubuntu2204": { - HEAD_NODE: ["tls"], + HEAD_NODE: ["crc32_generic", "tls"], COMPUTE_NODE: ["tls"], LOGIN_NODE: ["tls"], }, "ubuntu2404": { - HEAD_NODE: ["tls"], + HEAD_NODE: ["crc32_generic", "tls"], COMPUTE_NODE: ["tls"], LOGIN_NODE: ["tls"], }, From dc3bcdac1459f488a79aacd3a92f100b0f3f9eec Mon Sep 17 00:00:00 2001 From: Giacomo Marciani Date: Tue, 18 Aug 2026 16:32:24 -0400 Subject: [PATCH 2/8] [Test] test_patching_cluster: deduplicate head-node lazy kernel modules via COMMON_HEAD_NODE_LAZY_MODULES to avoid code duplication. (cherry picked from commit b3bc52704ae623710e8f9294b1609eb242e07af3) --- .../tests/patching/test_patching.py | 12 +++++++----- 1 file changed, 7 insertions(+), 5 deletions(-) diff --git a/tests/integration-tests/tests/patching/test_patching.py b/tests/integration-tests/tests/patching/test_patching.py index 63c9670d61..ace2f3beee 100644 --- a/tests/integration-tests/tests/patching/test_patching.py +++ b/tests/integration-tests/tests/patching/test_patching.py @@ -53,9 +53,11 @@ # may or may not have run by the time we snapshot). We force-load these after patching so that # the before/after comparison does not flag them as missing. # Maintained per OS and per node type. Keep module names alphabetically sorted. +# Modules tolerated on the head node across all OSes. +COMMON_HEAD_NODE_LAZY_MODULES = ["crc32_generic", "tls"] LAZY_KERNEL_MODULES = { "alinux2023": { - HEAD_NODE: ["crc32_generic", "tls"], + HEAD_NODE: COMMON_HEAD_NODE_LAZY_MODULES, COMPUTE_NODE: ["tls"], LOGIN_NODE: ["tls"], }, @@ -65,7 +67,7 @@ LOGIN_NODE: ["tls"], }, "rhel9": { - HEAD_NODE: ["crc32_generic", "tls"], + HEAD_NODE: COMMON_HEAD_NODE_LAZY_MODULES, COMPUTE_NODE: ["tls"], LOGIN_NODE: ["tls"], }, @@ -75,17 +77,17 @@ LOGIN_NODE: ["tls"], }, "rocky9": { - HEAD_NODE: ["crc32_generic", "tls"], + HEAD_NODE: COMMON_HEAD_NODE_LAZY_MODULES, COMPUTE_NODE: ["tls"], LOGIN_NODE: ["tls"], }, "ubuntu2204": { - HEAD_NODE: ["crc32_generic", "tls"], + HEAD_NODE: COMMON_HEAD_NODE_LAZY_MODULES, COMPUTE_NODE: ["tls"], LOGIN_NODE: ["tls"], }, "ubuntu2404": { - HEAD_NODE: ["crc32_generic", "tls"], + HEAD_NODE: COMMON_HEAD_NODE_LAZY_MODULES, COMPUTE_NODE: ["tls"], LOGIN_NODE: ["tls"], }, From 537ab622a6c5ce8999d40aa56e52f9ce3696106b Mon Sep 17 00:00:00 2001 From: Giacomo Marciani Date: Wed, 19 Aug 2026 12:54:38 -0400 Subject: [PATCH 3/8] [Test] test_patching_cluster: prevent patching failure on Ubuntu caused by dpkg user prompts. Such reconfiguration may occur when efs-utils get upgraded as part of the patching. (cherry picked from commit cf4407a519e21707d91d67f0d08878c99da022ff) --- .../test_patching/test_patching_cluster/patch_node.sh | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/tests/integration-tests/tests/patching/test_patching/test_patching_cluster/patch_node.sh b/tests/integration-tests/tests/patching/test_patching/test_patching_cluster/patch_node.sh index e520aa9d05..72de232fdc 100644 --- a/tests/integration-tests/tests/patching/test_patching/test_patching_cluster/patch_node.sh +++ b/tests/integration-tests/tests/patching/test_patching/test_patching_cluster/patch_node.sh @@ -183,6 +183,9 @@ elif command -v apt-get >/dev/null 2>&1; then # DEBIAN_FRONTEND is passed *through* sudo, which resets the environment by default, # so an exported var alone would not reach the root apt-get process. _envars=(DEBIAN_FRONTEND=noninteractive) + # dpkg conffile options: keep locally-modified config files (e.g. efs-utils.conf); without + # them dpkg prompts and aborts on EOF (DEBIAN_FRONTEND=noninteractive does not cover this). + _dpkg_opts=(-o Dpkg::Options::="--force-confold" -o Dpkg::Options::="--force-confdef") sudo "${_envars[@]}" apt-get update -y if [[ "${CAPPED}" == "true" ]]; then cap_kernel_debian @@ -194,7 +197,7 @@ elif command -v apt-get >/dev/null 2>&1; then sudo "${_envars[@]}" unattended-upgrade -v else # Apply all available package updates, including kernel packages. - sudo "${_envars[@]}" apt-get upgrade -y + sudo "${_envars[@]}" apt-get upgrade -y "${_dpkg_opts[@]}" fi else echo "ERROR: no supported package manager found (dnf/yum/apt-get)" >&2 From fd6668f629f0e3ef5cb4b1cb8847c06acf104bf6 Mon Sep 17 00:00:00 2001 From: Giacomo Marciani Date: Wed, 19 Aug 2026 13:36:15 -0400 Subject: [PATCH 4/8] [Test] test_patching_cluster: prevent ICE by removing placement group from cluster config, which is not required for this test. (cherry picked from commit 781dad2d0e09c105aecfec67aae9828048bd7955) --- .../test_patching/test_patching_cluster/pcluster.config.yaml | 5 ----- 1 file changed, 5 deletions(-) diff --git a/tests/integration-tests/tests/patching/test_patching/test_patching_cluster/pcluster.config.yaml b/tests/integration-tests/tests/patching/test_patching/test_patching_cluster/pcluster.config.yaml index 72887fe1d7..cf29878ea1 100644 --- a/tests/integration-tests/tests/patching/test_patching/test_patching_cluster/pcluster.config.yaml +++ b/tests/integration-tests/tests/patching/test_patching/test_patching_cluster/pcluster.config.yaml @@ -40,12 +40,7 @@ Scheduling: Enabled: true Networking: PlacementGroup: - {% if capacity_reservation_framework_placement_group %} - Enabled: true - Name: {{ capacity_reservation_framework_placement_group }} - {% else %} Enabled: false - {% endif %} SubnetIds: - {{ private_subnet_id }} ComputeResources: From 8c5abf7be83d6c5db1776609fbbc42fdc788771a Mon Sep 17 00:00:00 2001 From: Giacomo Marciani Date: Tue, 18 Aug 2026 16:33:00 -0400 Subject: [PATCH 5/8] [Test] test_build_image_no_internet: prevent Proxy verification timeout by decoupling ProxyVerificationWaitCondition from the Proxy instance. Also reduced the risk of proxy infra deployment failure by increasing the timeout of the proxy instance to not fail on slow user data. (cherry picked from commit 1fa1447ca3fe3360e89baf901c364881d63cf508) --- cloudformation/proxy/proxy.yaml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/cloudformation/proxy/proxy.yaml b/cloudformation/proxy/proxy.yaml index 4a3a54d796..cc908d95b2 100644 --- a/cloudformation/proxy/proxy.yaml +++ b/cloudformation/proxy/proxy.yaml @@ -661,11 +661,11 @@ Resources: ProxyVerificationWaitCondition: Type: AWS::CloudFormation::WaitCondition DependsOn: - - Proxy + - ProxyClient Properties: Count: 1 Handle: !Ref ProxyVerificationWaitConditionHandle - Timeout: 600 + Timeout: 900 Outputs: From 50864cfb9470e60906815a50cf99cc9fa4c0fb9b Mon Sep 17 00:00:00 2001 From: Giacomo Marciani Date: Tue, 18 Aug 2026 17:42:32 -0400 Subject: [PATCH 6/8] [Test] test_cluster_with_gpu_health_checks: prevent ICE by reserving all the instances required by the test and let the cluster use the specific subnet where they have been reserved. (cherry picked from commit 9e4db017589caf08d9e78def8148ad53ff108b87) --- tests/integration-tests/configs/develop.yaml | 2 +- .../test_cluster_with_gpu_health_checks/pcluster.config.yaml | 4 ---- 2 files changed, 1 insertion(+), 5 deletions(-) diff --git a/tests/integration-tests/configs/develop.yaml b/tests/integration-tests/configs/develop.yaml index c1a22c1e0c..db0a0d9984 100644 --- a/tests/integration-tests/configs/develop.yaml +++ b/tests/integration-tests/configs/develop.yaml @@ -322,7 +322,7 @@ test-suites: health_checks: test_gpu_health_checks.py::test_cluster_with_gpu_health_checks: dimensions: - - regions: [{{ US_WEST_2_GPU_INSTANCE_TYPE_0_CAPACITY_RESERVATION_4_INSTANCES_2_HOURS_NOPG_OS_X86_5 }}] + - regions: [{{ US_WEST_2_GPU_INSTANCE_TYPE_0_CAPACITY_RESERVATION_4_INSTANCES_2_HOURS_NOPG_OS_X86_5__c5_xlarge_CAPACITY_RESERVATION_5_INSTANCES_2_HOURS_NOPG_OS_X86_5 }}] instances: [{{ US_WEST_2_GPU_INSTANCE_TYPE_0 }}] oss: [{{ OS_X86_5 }}] schedulers: ["slurm"] diff --git a/tests/integration-tests/tests/health_checks/test_gpu_health_checks/test_cluster_with_gpu_health_checks/pcluster.config.yaml b/tests/integration-tests/tests/health_checks/test_gpu_health_checks/test_cluster_with_gpu_health_checks/pcluster.config.yaml index 828edf26af..09f36dacf6 100644 --- a/tests/integration-tests/tests/health_checks/test_gpu_health_checks/test_cluster_with_gpu_health_checks/pcluster.config.yaml +++ b/tests/integration-tests/tests/health_checks/test_gpu_health_checks/test_cluster_with_gpu_health_checks/pcluster.config.yaml @@ -93,9 +93,7 @@ Scheduling: - InstanceType: {{ non_gpu_instance }} Networking: SubnetIds: - {% for private_subnet_id in private_subnet_ids %} - {{ private_subnet_id }} - {% endfor %} - Name: queue-2 ComputeResources: - Name: compute-resource-1 @@ -112,7 +110,5 @@ Scheduling: Enabled: true Networking: SubnetIds: - {% for private_subnet_id in private_subnet_ids %} - {{ private_subnet_id }} - {% endfor %} {% endif %} From 26200ff6bb0518ada65f4f6edd65969e23472956 Mon Sep 17 00:00:00 2001 From: Giacomo Marciani Date: Fri, 14 Aug 2026 17:54:51 -0400 Subject: [PATCH 7/8] [Test] test_multiple_efs: retry EFS mount assertion for IAM-authorized EFS in test_multiple_efs. IAM policy evaluation can transiently delay the mount on head node reboot, causing flaky failures. Retry the mount check only for IAM-authorized EFS. (cherry picked from commit a57a04f203f18b32497177a3aeb1aa5562b559fd) --- .../tests/storage/test_efs.py | 24 ++++++++++++++----- 1 file changed, 18 insertions(+), 6 deletions(-) diff --git a/tests/integration-tests/tests/storage/test_efs.py b/tests/integration-tests/tests/storage/test_efs.py index 54ebda4ba3..c635efe8b3 100644 --- a/tests/integration-tests/tests/storage/test_efs.py +++ b/tests/integration-tests/tests/storage/test_efs.py @@ -17,6 +17,8 @@ from assertpy import assert_that from cfn_stacks_factory import CfnVpcStack from remote_command_executor import RemoteCommandExecutor +from retrying import retry +from time_utils import minutes, seconds from utils import get_arn_partition, get_compute_nodes_instance_ips from tests.common.utils import get_sts_endpoint, reboot_head_node @@ -350,12 +352,22 @@ def _check_efs_correctly_mounted_and_shared( all_mount_dirs, remote_command_executor, scheduler_commands, iam_authorizations, encryption_in_transits ): for i, mount_dir in enumerate(all_mount_dirs): - test_efs_correctly_mounted( - remote_command_executor, - mount_dir, - encryption_in_transits[i], - iam_authorizations[i], - ) + if iam_authorizations[i]: + # An EFS with IAM authorization can be transiently slower to mount because the mount depends on + # IAM policy evaluation. As such, we retry the assertion to prevent flaky test failures. + retry(wait_fixed=seconds(10), stop_max_delay=minutes(3))(test_efs_correctly_mounted)( + remote_command_executor, + mount_dir, + encryption_in_transits[i], + iam_authorizations[i], + ) + else: + test_efs_correctly_mounted( + remote_command_executor, + mount_dir, + encryption_in_transits[i], + iam_authorizations[i], + ) _test_efs_correctly_shared(remote_command_executor, mount_dir, scheduler_commands) From 26f3e054973e9005a3716545a4dbd1bc97c5aba6 Mon Sep 17 00:00:00 2001 From: Giacomo Marciani Date: Fri, 14 Aug 2026 17:55:51 -0400 Subject: [PATCH 8/8] [Test] test_queue_parameters_update: fix flakiness asserting TERMINATE requeue deterministically The TERMINATE branch snapshotted the transient CONFIGURING state via a single-shot assert_job_state after a fixed sleep(10). When replacement nodes boot quickly the whole node swap finishes inside the update-cluster --wait window, so the job is already RUNNING again by the time the test samples it, causing false failures. Assert the deterministic outcome instead: the job was requeued (wait_job_requeued) and is running again, and the original instance it ran on was terminated (wait_instance_replaced_or_terminating). Add SlurmCommands.get_job_instance_id to resolve the job's instance via Slurm, and wait_job_requeued helper. (cherry picked from commit 31f162adfd42e980e557562c89195a4ff5dce94e) --- .../tests/common/schedulers_common.py | 15 +++++++++++++ .../tests/update/test_update.py | 21 ++++++++++++++++--- 2 files changed, 33 insertions(+), 3 deletions(-) diff --git a/tests/integration-tests/tests/common/schedulers_common.py b/tests/integration-tests/tests/common/schedulers_common.py index 9b8a37af8d..3b454764de 100644 --- a/tests/integration-tests/tests/common/schedulers_common.py +++ b/tests/integration-tests/tests/common/schedulers_common.py @@ -496,6 +496,15 @@ def get_node_addr(self, node_name): node_addr = re.search(r"NodeAddr=(.*) NodeHostName", result).group(1) return node_addr + def get_job_instance_id(self, job_id): + """Return the id of the EC2 instance the job is running on, as reported by Slurm.""" + node_name = self._remote_command_executor.run_remote_command( + f'scontrol show jobs {job_id} --json | jq -r ".jobs[0].batch_host"' + ).stdout.strip() + return self._remote_command_executor.run_remote_command( + f'scontrol show nodes {node_name} --json | jq -r ".nodes[0].instance_id"' + ).stdout.strip() + def submit_command_and_assert_job_accepted(self, submit_command_args): """Submit a command and assert the job is accepted by scheduler.""" result = self.submit_command(**submit_command_args) @@ -520,6 +529,12 @@ def wait_job_running(self, job_id): result = self._remote_command_executor.run_remote_command("scontrol show jobs -o {0}".format(job_id)) assert_that(result.stdout).contains("JobState=RUNNING") + @retry(wait_fixed=seconds(10), stop_max_delay=minutes(13)) + def wait_job_requeued(self, job_id, times=1): + """Wait till the job has been requeued at least `times` times (Restarts>=times).""" + restarts = self.get_job_info(job_id, field="Restarts") + assert_that(int(restarts)).is_greater_than_or_equal_to(times) + def get_node_info(self, nodename): """Get node info.""" return self._remote_command_executor.run_remote_command("scontrol show nodes {0}".format(nodename)).stdout diff --git a/tests/integration-tests/tests/update/test_update.py b/tests/integration-tests/tests/update/test_update.py index 9aca180d36..0b5fc47b5d 100644 --- a/tests/integration-tests/tests/update/test_update.py +++ b/tests/integration-tests/tests/update/test_update.py @@ -42,7 +42,12 @@ wait_for_computefleet_changed, ) -from tests.common.assertions import assert_instance_config_version_on_ddb, assert_lines_in_logs, assert_no_msg_in_logs +from tests.common.assertions import ( + assert_instance_config_version_on_ddb, + assert_lines_in_logs, + assert_no_msg_in_logs, + wait_instance_replaced_or_terminating, +) from tests.common.hit_common import ( assert_compute_node_states, assert_initial_conditions, @@ -888,6 +893,10 @@ def _test_update_queue_strategy_with_running_job( scheduler_commands.wait_job_running(queue2_job_id) logging.info(f"Job {queue2_job_id} is running on queue2") + # Ask Slurm exactly which instance the queue2 job is running on, so that after a TERMINATE update + # we can assert that exact instance was terminated (the job's node was replaced). + queue2_instance_id = scheduler_commands.get_job_instance_id(queue2_job_id) + logging.info(f"Updating cluster with strategy {queue_update_strategy} with running jobs") updated_config_file = pcluster_config_reader( config_file="pcluster.config.update_with_running_job.yaml", @@ -913,8 +922,14 @@ def _test_update_queue_strategy_with_running_job( queue1_nodes = scheduler_commands.get_compute_nodes("queue1") assert_compute_node_states(scheduler_commands, queue1_nodes, expected_states=["mixed", "allocated"]) if queue_update_strategy == "TERMINATE": - time.sleep(10) - scheduler_commands.assert_job_state(queue2_job_id, "CONFIGURING") + # TERMINATE forcibly terminates queue2's compute nodes and requeues the running job onto + # freshly launched replacement nodes. Assert the deterministic outcome (the job was requeued + # and is running again on the replaced instances) rather than snapshotting the transient + # CONFIGURING state, which can close before we sample it when replacement nodes boot quickly + # (the whole swap can finish inside the update-cluster --wait window). + scheduler_commands.wait_job_requeued(queue2_job_id) + wait_instance_replaced_or_terminating(queue2_instance_id, region) + scheduler_commands.wait_job_running(queue2_job_id) # check queue1 AMIs are not replaced _check_queue_ami(cluster, ec2, pcluster_ami_id, "queue1")