Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
40 changes: 5 additions & 35 deletions lib/asg-runner-stack.ts
Original file line number Diff line number Diff line change
Expand Up @@ -71,11 +71,7 @@ export class ASGRunnerStack extends cdk.Stack implements IASGRunnerStack {
case PlatformType.MAC: {
rootDeviceName = '/dev/sda1';
if (this.arch === 'arm') {
// Use mac-m4.metal (Apple M4) for arm macOS runners. The older mac2.metal (M2)
// has scarce dedicated-host capacity in our regions, causing ASG launch failures
// and wedged CloudFormation deploys. aws-cdk-lib 2.266.0 has no M4 InstanceClass
// enum, so construct the type from its string identifier.
instanceType = new ec2.InstanceType('mac-m4.metal');
instanceType = ec2.InstanceType.of(ec2.InstanceClass.MAC2, ec2.InstanceSize.METAL);
} else {
instanceType = ec2.InstanceType.of(ec2.InstanceClass.MAC1, ec2.InstanceSize.METAL);
}
Expand Down Expand Up @@ -242,23 +238,10 @@ export class ASGRunnerStack extends cdk.Stack implements IASGRunnerStack {
]
};

// Dedicated-host runners (macOS/Windows) run on scarce metal capacity that is
// allocated just-in-time. A rolling update that terminates an instance BEFORE its
// replacement is InService leaves a capacity gap: if metal capacity is momentarily
// unavailable, the replacement cannot launch, the batch never stabilizes, and the
// CloudFormation update fails (and can wedge in UPDATE_ROLLBACK_FAILED). To make
// deploys resilient we give these ASGs N+1 headroom (maxCapacity = desired + 1) and
// keep desired instances in service during a rolling update (launch-before-terminate),
// which requires a reserved spare host slot (see the host resource group below, which
// retains hosts rather than auto-releasing them). maxCapacity > desired also means a
// bad instance can always be replaced without deadlocking.
const usesDedicatedHosts = this.requiresDedicatedHosts();
const maxCapacity = usesDedicatedHosts ? props.type.desiredInstances + 1 : props.type.desiredInstances;

const asg = new autoscaling.AutoScalingGroup(this, asgName, {
vpc,
desiredCapacity: props.type.desiredInstances,
maxCapacity,
maxCapacity: props.type.desiredInstances,
minCapacity: 0,
healthCheck: autoscaling.HealthCheck.ec2({
grace: cdk.Duration.seconds(3600)
Expand All @@ -268,13 +251,7 @@ export class ASGRunnerStack extends cdk.Stack implements IASGRunnerStack {
// Defaults shown here explicitly except for pauseTime
// and minSuccesPercentage
maxBatchSize: 1,
// For dedicated-host (metal) runners, keep all desired instances in service while
// rolling so a replacement must reach InService before the old instance is
// terminated (launch-before-terminate). This avoids the "terminate first, then
// fail to launch on insufficient metal capacity" wedge. Requires N+1 capacity
// (maxCapacity above) and a reserved host slot. Non-metal ASGs keep the prior
// behavior (0), since they scale on abundant capacity.
minInstancesInService: usesDedicatedHosts ? props.type.desiredInstances : 0,
minInstancesInService: 0,
suspendProcesses: [
autoscaling.ScalingProcess.HEALTH_CHECK,
autoscaling.ScalingProcess.REPLACE_UNHEALTHY,
Expand Down Expand Up @@ -316,15 +293,8 @@ export class ASGRunnerStack extends cdk.Stack implements IASGRunnerStack {
values: ['true']
},
{
// Retain dedicated hosts instead of auto-releasing them after an instance
// terminates. macOS/Windows metal capacity is scarce and allocated
// just-in-time; auto-releasing forces the ASG to re-compete for metal
// capacity on every deploy/replacement, which causes "Insufficient
// capacity" launch failures and wedged CloudFormation updates. Retaining
// hosts keeps the (N+1) reserved slots so launch-before-terminate rolling
// updates always have somewhere to place the replacement. Note: macOS
// dedicated hosts have a ~24h minimum allocation regardless, so retaining
// them adds little effective cost while removing the capacity gamble.
// Retain the dedicated host so a replacement instance relaunches onto the same
// held host (avoids re-allocating a scarce Mac dedicated host).
name: 'auto-release-host',
values: ['false']
},
Expand Down
54 changes: 2 additions & 52 deletions test/asg-runner-stack.test.ts
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
import * as cdk from 'aws-cdk-lib';
import { Match, Template } from 'aws-cdk-lib/assertions';
import { Template } from 'aws-cdk-lib/assertions';
import { PlatformType, RunnerConfig, RunnerType } from '../config/runner-config';
import { ASGRunnerStack } from '../lib/asg-runner-stack';
import { ENVIRONMENT_STAGE } from '../lib/finch-pipeline-app-stage';
Expand Down Expand Up @@ -54,7 +54,7 @@ describe('ASGRunnerStack test', () => {
}
case PlatformType.MAC: {
if (type.arch === 'arm') {
instanceType = 'mac-m4.metal';
instanceType = 'mac2.metal';
} else {
instanceType = 'mac1.metal';
}
Expand Down Expand Up @@ -82,54 +82,4 @@ describe('ASGRunnerStack test', () => {
expect(stack.terminationProtection).toBeTruthy();
});
});

it('gives dedicated-host runners N+1 capacity and launch-before-terminate', () => {
// macOS/Windows runners run on scarce dedicated-host (metal) capacity. To avoid a
// rolling update terminating an instance before its replacement can be placed, they
// get maxCapacity = desired + 1 and keep desired instances in service during rollout.
runnerConfig.runnerTypes
.filter((type) => type.platform === PlatformType.MAC || type.platform === PlatformType.WINDOWS)
.forEach((type) => {
const stack = stacks.find((s) => s.stackName === generateASGStackName(type));
const template = Template.fromStack(stack!);
template.hasResourceProperties('AWS::AutoScaling::AutoScalingGroup', {
MinSize: '0',
MaxSize: `${type.desiredInstances + 1}`,
DesiredCapacity: `${type.desiredInstances}`
});
// Rolling update keeps desired instances in service (launch-before-terminate).
template.hasResource('AWS::AutoScaling::AutoScalingGroup', {
UpdatePolicy: {
AutoScalingRollingUpdate: {
MinInstancesInService: type.desiredInstances,
MaxBatchSize: 1
}
}
});
// Dedicated hosts are retained (not auto-released) so the reserved slot persists.
template.hasResourceProperties('AWS::ResourceGroups::Group', {
Configuration: Match.arrayWith([
Match.objectLike({
Type: 'AWS::EC2::HostManagement',
Parameters: Match.arrayWith([
Match.objectLike({ Name: 'auto-release-host', Values: ['false'] })
])
})
])
});
});
});

it('non-dedicated-host runners keep maxCapacity equal to desired', () => {
runnerConfig.runnerTypes
.filter((type) => type.platform === PlatformType.AMAZONLINUX)
.forEach((type) => {
const stack = stacks.find((s) => s.stackName === generateASGStackName(type));
const template = Template.fromStack(stack!);
template.hasResourceProperties('AWS::AutoScaling::AutoScalingGroup', {
MaxSize: `${type.desiredInstances}`,
DesiredCapacity: `${type.desiredInstances}`
});
});
});
});
Loading