Created
August 5, 2026 09:52
-
-
Save NeilMasters/f20bb791108b6701f323acd9a2e6cde2 to your computer and use it in GitHub Desktop.
A simple script that will pull codedeployment history to help isolate if failures are clustered around specific instances.
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| #!/usr/bin/env bash | |
| # Dump full CodeDeploy deployment + target history for an application (every target on every | |
| # deployment, not just failures), to figure out whether failures cluster on specific instances | |
| # or are randomly distributed. | |
| # | |
| # Usage: ./codedeploy-history.sh <application-name> [deployment-group-name] [max-deployments] | |
| # Example: ./codedeploy-history.sh example_name "" 200 | |
| # | |
| # Output will include each codedeploy and the reason it failed (based on first failure from the | |
| # instances), for each instance percentage of failure(if your doing in place) along with first | |
| # error, last error and last good deploy timestamps. | |
| set -euo pipefail | |
| APP_NAME="${1:?Usage: $0 <application-name> [deployment-group-name] [max-deployments]}" | |
| GROUP_NAME="${2:-}" | |
| MAX_DEPLOYMENTS="${3:-200}" | |
| OUT_DIR="codedeploy-history/${APP_NAME}_$(date -u +%Y%m%dT%H%M%SZ)" | |
| mkdir -p "$OUT_DIR" | |
| LIST_FILTER=(--application-name "$APP_NAME") | |
| [[ -n "$GROUP_NAME" ]] && LIST_FILTER+=(--deployment-group-name "$GROUP_NAME") | |
| echo "==> Listing deployments for $APP_NAME ${GROUP_NAME:+(group: $GROUP_NAME)}" >&2 | |
| DEPLOYMENT_IDS=() | |
| NEXT_TOKEN="" | |
| while true; do | |
| if [[ -n "$NEXT_TOKEN" ]]; then | |
| RESP=$(aws deploy list-deployments "${LIST_FILTER[@]}" --next-token "$NEXT_TOKEN") | |
| else | |
| RESP=$(aws deploy list-deployments "${LIST_FILTER[@]}") | |
| fi | |
| while IFS= read -r line; do | |
| [[ -n "$line" ]] && DEPLOYMENT_IDS+=("$line") | |
| done < <(echo "$RESP" | jq -r '.deployments[]') | |
| NEXT_TOKEN=$(echo "$RESP" | jq -r '.nextToken // empty') | |
| [[ ${#DEPLOYMENT_IDS[@]} -ge $MAX_DEPLOYMENTS || -z "$NEXT_TOKEN" ]] && break | |
| done | |
| DEPLOYMENT_IDS=("${DEPLOYMENT_IDS[@]:0:$MAX_DEPLOYMENTS}") | |
| echo "==> Found ${#DEPLOYMENT_IDS[@]} deployments, fetching details..." >&2 | |
| TMP_DIR=$(mktemp -d) | |
| trap 'rm -rf "$TMP_DIR"' EXIT | |
| chunk_num=0 | |
| for ((i = 0; i < ${#DEPLOYMENT_IDS[@]}; i += 25)); do | |
| CHUNK=("${DEPLOYMENT_IDS[@]:i:25}") | |
| aws deploy batch-get-deployments --deployment-ids "${CHUNK[@]}" > "$TMP_DIR/chunk_${chunk_num}.json" | |
| ((chunk_num++)) | |
| done | |
| jq -s '[.[].deploymentsInfo[]]' "$TMP_DIR"/chunk_*.json > "$OUT_DIR/deployments.json" | |
| echo "==> Fetching per-target history (every target, every outcome, not just failures)..." >&2 | |
| for id in "${DEPLOYMENT_IDS[@]}"; do | |
| TARGET_IDS=$(aws deploy list-deployment-targets --deployment-id "$id" | jq -r '.targetIds[]?') | |
| [[ -z "$TARGET_IDS" ]] && continue | |
| aws deploy batch-get-deployment-targets --deployment-id "$id" --target-ids $TARGET_IDS \ | |
| > "$OUT_DIR/targets_${id}.json" | |
| done | |
| jq -s '[.[].deploymentTargets[]?.instanceTarget]' "$OUT_DIR"/targets_*.json > "$OUT_DIR/all_targets.json" | |
| echo "==> Looking up 'Roles' tag for each instance seen..." >&2 | |
| INSTANCE_IDS=() | |
| while IFS= read -r line; do | |
| [[ -n "$line" ]] && INSTANCE_IDS+=("$line") | |
| done < <(jq -r '[.[].targetId] | unique | .[]' "$OUT_DIR/all_targets.json") | |
| echo '{}' > "$OUT_DIR/instance_roles.json" | |
| if [[ ${#INSTANCE_IDS[@]} -gt 0 ]]; then | |
| TAG_TMP_DIR=$(mktemp -d) | |
| chunk_num=0 | |
| for ((i = 0; i < ${#INSTANCE_IDS[@]}; i += 200)); do | |
| CHUNK=("${INSTANCE_IDS[@]:i:200}") | |
| IDS_CSV=$(IFS=,; echo "${CHUNK[*]}") | |
| aws ec2 describe-tags \ | |
| --filters "Name=resource-id,Values=$IDS_CSV" "Name=key,Values=Roles" \ | |
| --query 'Tags[].{ResourceId:ResourceId,Value:Value}' \ | |
| > "$TAG_TMP_DIR/tags_${chunk_num}.json" | |
| ((chunk_num++)) | |
| done | |
| jq -s '[.[][]] | reduce .[] as $t ({}; .[$t.ResourceId] = $t.Value)' "$TAG_TMP_DIR"/tags_*.json > "$OUT_DIR/instance_roles.json" | |
| rm -rf "$TAG_TMP_DIR" | |
| fi | |
| echo "==> Deployment summary" | tee "$OUT_DIR/summary.tsv" >&2 | |
| { | |
| echo -e "DeploymentId\tStatus\tCreator\tStart\tComplete\tErrorCode\tErrorMessage" | |
| jq -r '.[] | [.deploymentId, .status, (.creator // "-"), (.startTime // "-"), (.completeTime // "-"), (.errorInformation.code // "-"), (.errorInformation.message // "-")] | @tsv' "$OUT_DIR/deployments.json" | |
| } | tee -a "$OUT_DIR/summary.tsv" | column -t -s $'\t' >&2 | |
| echo "==> Per-instance failure rate + first/last failure timing (within the ${#DEPLOYMENT_IDS[@]} deployments pulled) — this answers 'same instances vs random'" >&2 | |
| { | |
| echo -e "InstanceId\tRoles\tAttempts\tSucceeded\tFailed\tFailureRate%\tFirstFailedAt\tLastSucceededAt\tLastAttemptAt\tLastAttemptStatus\tLastFailedEvent" | |
| jq -r --slurpfile rolesFile "$OUT_DIR/instance_roles.json" ' | |
| def toepoch: if type=="number" then floor else (.[0:19] + "Z" | fromdateiso8601) end; | |
| ($rolesFile[0]) as $roles | |
| | map(.lastUpdatedAt |= (if . == null then null else toepoch end)) | |
| | group_by(.targetId) | |
| | map({ | |
| targetId: .[0].targetId, | |
| roles: ($roles[.[0].targetId] // "-"), | |
| attempts: length, | |
| succeeded: (map(select(.status=="Succeeded")) | length), | |
| failed: (map(select(.status=="Failed")) | length), | |
| lastFailedEvent: ([.[] | select(.status=="Failed") | (.lifecycleEvents[]? | select(.status=="Failed") | .lifecycleEventName)] | last // "-"), | |
| firstFailedAt: ([.[] | select(.status=="Failed" and .lastUpdatedAt != null) | .lastUpdatedAt] | min), | |
| lastSucceededAt: ([.[] | select(.status=="Succeeded" and .lastUpdatedAt != null) | .lastUpdatedAt] | max), | |
| lastAttempt: (sort_by(.lastUpdatedAt) | last) | |
| }) | |
| | map(. + {lastAttemptAt: .lastAttempt.lastUpdatedAt, lastAttemptStatus: .lastAttempt.status} | del(.lastAttempt)) | |
| | sort_by(.lastSucceededAt // -1) | |
| | .[] | [ | |
| .targetId, | |
| .roles, | |
| .attempts, | |
| .succeeded, | |
| .failed, | |
| ((.failed / .attempts * 100) | round), | |
| (if .firstFailedAt then (.firstFailedAt | todate) else "-" end), | |
| (if .lastSucceededAt then (.lastSucceededAt | todate) else "NEVER" end), | |
| (if .lastAttemptAt then (.lastAttemptAt | todate) else "-" end), | |
| .lastAttemptStatus, | |
| .lastFailedEvent | |
| ] | |
| | @tsv | |
| ' "$OUT_DIR/all_targets.json" | |
| } | tee "$OUT_DIR/instance_failure_summary.tsv" | column -t -s $'\t' >&2 | |
| echo "==> Failed lifecycle event detail (per target/deployment) — the actual script/exit reason" >&2 | |
| jq -r ' | |
| .[] | select(.status=="Failed") as $t | |
| | ($t.lifecycleEvents[]? | select(.status=="Failed")) as $ev | |
| | [$t.targetId, $t.deploymentId, $ev.lifecycleEventName, ($ev.diagnostics.scriptName // "-"), ($ev.diagnostics.message // "-"), ($ev.diagnostics.logTail // "-")] | |
| | @tsv | |
| ' "$OUT_DIR/all_targets.json" | tee "$OUT_DIR/failed_lifecycle_events.tsv" | column -t -s $'\t' >&2 | |
| echo "==> Raw data written to $OUT_DIR/" >&2 |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment