diff --git a/vm_images/linux-amd64/linux-amd64.pkr.hcl b/vm_images/linux-amd64/linux-amd64.pkr.hcl index fd711d1..905096c 100644 --- a/vm_images/linux-amd64/linux-amd64.pkr.hcl +++ b/vm_images/linux-amd64/linux-amd64.pkr.hcl @@ -7,6 +7,15 @@ packer { } } +variable "iam_instance_profile" { + type = string + description = "EC2 instance profile with permission to pull xgb-ci.gpu from ECR." + validation { + condition = length(trimspace(var.iam_instance_profile)) > 0 + error_message = "An instance profile with ECR pull permissions is required." + } +} + locals { ami_name_prefix = "xgboost-ci" image_name = "RunsOn worker with Ubuntu 24.04 AMD64 + CUDA driver 580" @@ -34,20 +43,21 @@ source "amazon-ebs" "runs-on-linux-amd64" { associate_public_ip_address = true communicator = "ssh" instance_type = "g4dn.xlarge" + iam_instance_profile = var.iam_instance_profile region = "${local.region}" ssh_timeout = "10m" ssh_username = "ubuntu" ssh_file_transfer_method = "sftp" user_data_file = "setup_ssh.sh" launch_block_device_mappings { - device_name = "/dev/sda1" - volume_size = "${local.volume_size}" - volume_type = "gp3" + device_name = "/dev/sda1" + volume_size = "${local.volume_size}" + volume_type = "gp3" delete_on_termination = true } - aws_polling { # Wait up to 1 hour until the AMI is ready + aws_polling { # Wait up to 1 hour until the AMI is ready delay_seconds = 15 - max_attempts = 240 + max_attempts = 240 } snapshot_tags = { Name = "${local.image_name}" @@ -76,4 +86,11 @@ build { pause_before = "1m0s" script = "bootstrap.sh" } + + provisioner "shell" { + script = "preload_gpu_image.sh" + environment_vars = [ + "AWS_DEFAULT_REGION=${local.region}", + ] + } } diff --git a/vm_images/linux-amd64/preload_gpu_image.sh b/vm_images/linux-amd64/preload_gpu_image.sh new file mode 100644 index 0000000..3406312 --- /dev/null +++ b/vm_images/linux-amd64/preload_gpu_image.sh @@ -0,0 +1,23 @@ +#!/bin/bash +# Cache Docker layers in the AMI to avoid downloading/extracting them on every job. +# Remove this step if runners gain a persistent image cache or stop using xgb-ci.gpu. +set -euo pipefail + +registry=492475357299.dkr.ecr.us-west-2.amazonaws.com +image="${registry}/xgb-ci.gpu:main" + +# Use the builder's instance profile and an ephemeral Docker config so ECR tokens +# are not baked into the AMI. CI must still pull its requested tag to pick up updates. +docker_config=$(mktemp -d) +trap 'sudo rm -rf "$docker_config"' EXIT +sudo systemctl start docker +aws ecr get-login-password --region "$AWS_DEFAULT_REGION" | + sudo docker --config "$docker_config" login --username AWS --password-stdin "$registry" +sudo docker --config "$docker_config" pull "$image" +sudo docker run --rm --pull=never --gpus all --entrypoint nvidia-smi "$image" +# Record the digest and size in the build log to identify what the snapshot contains. +sudo docker image inspect --format '{{json .RepoDigests}} {{.Size}}' "$image" + +# Preserve the downloaded layers, but stop writes before Packer snapshots the disk. +sudo systemctl stop docker.service docker.socket containerd.service +sync