#!/usr/bin/env bash # Watch the host recover after the orphan single-node containers were removed. # 2 CPUs + ~92 containers means the 6 leftovers (one with 58 leaked chall.py # processes) were the dominant load source; SLA timeouts on a healthy service # were a symptom of that, not of the service. for i in 1 2 3 4 5 6; do LOAD=$(cut -d' ' -f1-3 /proc/loadavg) IDLE=$(vmstat 1 2 | tail -1 | awk '{print $15}') CHALL=$(ps -eo args --no-headers | grep -c '[c]hall.py') echo "t+$((i * 20))s load=$LOAD idle=${IDLE}% chall.py=$CHALL" sleep 20 done