Skip to content

Instantly share code, notes, and snippets.

@NeilMasters
Created August 5, 2026 09:52
Show Gist options
  • Select an option

  • Save NeilMasters/f20bb791108b6701f323acd9a2e6cde2 to your computer and use it in GitHub Desktop.

Select an option

Save NeilMasters/f20bb791108b6701f323acd9a2e6cde2 to your computer and use it in GitHub Desktop.
A simple script that will pull codedeployment history to help isolate if failures are clustered around specific instances.
#!/usr/bin/env bash
# Dump full CodeDeploy deployment + target history for an application (every target on every
# deployment, not just failures), to figure out whether failures cluster on specific instances
# or are randomly distributed.
#
# Usage: ./codedeploy-history.sh <application-name> [deployment-group-name] [max-deployments]
# Example: ./codedeploy-history.sh example_name "" 200
#
# Output will include each codedeploy and the reason it failed (based on first failure from the
# instances), for each instance percentage of failure(if your doing in place) along with first
# error, last error and last good deploy timestamps.
set -euo pipefail
APP_NAME="${1:?Usage: $0 <application-name> [deployment-group-name] [max-deployments]}"
GROUP_NAME="${2:-}"
MAX_DEPLOYMENTS="${3:-200}"
OUT_DIR="codedeploy-history/${APP_NAME}_$(date -u +%Y%m%dT%H%M%SZ)"
mkdir -p "$OUT_DIR"
LIST_FILTER=(--application-name "$APP_NAME")
[[ -n "$GROUP_NAME" ]] && LIST_FILTER+=(--deployment-group-name "$GROUP_NAME")
echo "==> Listing deployments for $APP_NAME ${GROUP_NAME:+(group: $GROUP_NAME)}" >&2
DEPLOYMENT_IDS=()
NEXT_TOKEN=""
while true; do
if [[ -n "$NEXT_TOKEN" ]]; then
RESP=$(aws deploy list-deployments "${LIST_FILTER[@]}" --next-token "$NEXT_TOKEN")
else
RESP=$(aws deploy list-deployments "${LIST_FILTER[@]}")
fi
while IFS= read -r line; do
[[ -n "$line" ]] && DEPLOYMENT_IDS+=("$line")
done < <(echo "$RESP" | jq -r '.deployments[]')
NEXT_TOKEN=$(echo "$RESP" | jq -r '.nextToken // empty')
[[ ${#DEPLOYMENT_IDS[@]} -ge $MAX_DEPLOYMENTS || -z "$NEXT_TOKEN" ]] && break
done
DEPLOYMENT_IDS=("${DEPLOYMENT_IDS[@]:0:$MAX_DEPLOYMENTS}")
echo "==> Found ${#DEPLOYMENT_IDS[@]} deployments, fetching details..." >&2
TMP_DIR=$(mktemp -d)
trap 'rm -rf "$TMP_DIR"' EXIT
chunk_num=0
for ((i = 0; i < ${#DEPLOYMENT_IDS[@]}; i += 25)); do
CHUNK=("${DEPLOYMENT_IDS[@]:i:25}")
aws deploy batch-get-deployments --deployment-ids "${CHUNK[@]}" > "$TMP_DIR/chunk_${chunk_num}.json"
((chunk_num++))
done
jq -s '[.[].deploymentsInfo[]]' "$TMP_DIR"/chunk_*.json > "$OUT_DIR/deployments.json"
echo "==> Fetching per-target history (every target, every outcome, not just failures)..." >&2
for id in "${DEPLOYMENT_IDS[@]}"; do
TARGET_IDS=$(aws deploy list-deployment-targets --deployment-id "$id" | jq -r '.targetIds[]?')
[[ -z "$TARGET_IDS" ]] && continue
aws deploy batch-get-deployment-targets --deployment-id "$id" --target-ids $TARGET_IDS \
> "$OUT_DIR/targets_${id}.json"
done
jq -s '[.[].deploymentTargets[]?.instanceTarget]' "$OUT_DIR"/targets_*.json > "$OUT_DIR/all_targets.json"
echo "==> Looking up 'Roles' tag for each instance seen..." >&2
INSTANCE_IDS=()
while IFS= read -r line; do
[[ -n "$line" ]] && INSTANCE_IDS+=("$line")
done < <(jq -r '[.[].targetId] | unique | .[]' "$OUT_DIR/all_targets.json")
echo '{}' > "$OUT_DIR/instance_roles.json"
if [[ ${#INSTANCE_IDS[@]} -gt 0 ]]; then
TAG_TMP_DIR=$(mktemp -d)
chunk_num=0
for ((i = 0; i < ${#INSTANCE_IDS[@]}; i += 200)); do
CHUNK=("${INSTANCE_IDS[@]:i:200}")
IDS_CSV=$(IFS=,; echo "${CHUNK[*]}")
aws ec2 describe-tags \
--filters "Name=resource-id,Values=$IDS_CSV" "Name=key,Values=Roles" \
--query 'Tags[].{ResourceId:ResourceId,Value:Value}' \
> "$TAG_TMP_DIR/tags_${chunk_num}.json"
((chunk_num++))
done
jq -s '[.[][]] | reduce .[] as $t ({}; .[$t.ResourceId] = $t.Value)' "$TAG_TMP_DIR"/tags_*.json > "$OUT_DIR/instance_roles.json"
rm -rf "$TAG_TMP_DIR"
fi
echo "==> Deployment summary" | tee "$OUT_DIR/summary.tsv" >&2
{
echo -e "DeploymentId\tStatus\tCreator\tStart\tComplete\tErrorCode\tErrorMessage"
jq -r '.[] | [.deploymentId, .status, (.creator // "-"), (.startTime // "-"), (.completeTime // "-"), (.errorInformation.code // "-"), (.errorInformation.message // "-")] | @tsv' "$OUT_DIR/deployments.json"
} | tee -a "$OUT_DIR/summary.tsv" | column -t -s $'\t' >&2
echo "==> Per-instance failure rate + first/last failure timing (within the ${#DEPLOYMENT_IDS[@]} deployments pulled) — this answers 'same instances vs random'" >&2
{
echo -e "InstanceId\tRoles\tAttempts\tSucceeded\tFailed\tFailureRate%\tFirstFailedAt\tLastSucceededAt\tLastAttemptAt\tLastAttemptStatus\tLastFailedEvent"
jq -r --slurpfile rolesFile "$OUT_DIR/instance_roles.json" '
def toepoch: if type=="number" then floor else (.[0:19] + "Z" | fromdateiso8601) end;
($rolesFile[0]) as $roles
| map(.lastUpdatedAt |= (if . == null then null else toepoch end))
| group_by(.targetId)
| map({
targetId: .[0].targetId,
roles: ($roles[.[0].targetId] // "-"),
attempts: length,
succeeded: (map(select(.status=="Succeeded")) | length),
failed: (map(select(.status=="Failed")) | length),
lastFailedEvent: ([.[] | select(.status=="Failed") | (.lifecycleEvents[]? | select(.status=="Failed") | .lifecycleEventName)] | last // "-"),
firstFailedAt: ([.[] | select(.status=="Failed" and .lastUpdatedAt != null) | .lastUpdatedAt] | min),
lastSucceededAt: ([.[] | select(.status=="Succeeded" and .lastUpdatedAt != null) | .lastUpdatedAt] | max),
lastAttempt: (sort_by(.lastUpdatedAt) | last)
})
| map(. + {lastAttemptAt: .lastAttempt.lastUpdatedAt, lastAttemptStatus: .lastAttempt.status} | del(.lastAttempt))
| sort_by(.lastSucceededAt // -1)
| .[] | [
.targetId,
.roles,
.attempts,
.succeeded,
.failed,
((.failed / .attempts * 100) | round),
(if .firstFailedAt then (.firstFailedAt | todate) else "-" end),
(if .lastSucceededAt then (.lastSucceededAt | todate) else "NEVER" end),
(if .lastAttemptAt then (.lastAttemptAt | todate) else "-" end),
.lastAttemptStatus,
.lastFailedEvent
]
| @tsv
' "$OUT_DIR/all_targets.json"
} | tee "$OUT_DIR/instance_failure_summary.tsv" | column -t -s $'\t' >&2
echo "==> Failed lifecycle event detail (per target/deployment) — the actual script/exit reason" >&2
jq -r '
.[] | select(.status=="Failed") as $t
| ($t.lifecycleEvents[]? | select(.status=="Failed")) as $ev
| [$t.targetId, $t.deploymentId, $ev.lifecycleEventName, ($ev.diagnostics.scriptName // "-"), ($ev.diagnostics.message // "-"), ($ev.diagnostics.logTail // "-")]
| @tsv
' "$OUT_DIR/all_targets.json" | tee "$OUT_DIR/failed_lifecycle_events.tsv" | column -t -s $'\t' >&2
echo "==> Raw data written to $OUT_DIR/" >&2
Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment