Last active
September 28, 2026 22:07
-
-
Save lizthegrey/2209d831930588f63076bdc0ac7b78e2 to your computer and use it in GitHub Desktop.
Reproducer: writer stalls after the owning cgroup is removed (cgroup_writeback_by_id -ENOENT on dying cgwbs)
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| #!/bin/bash | |
| # SPDX-License-Identifier: GPL-2.0 | |
| # | |
| # Reproduce writer stalls after the cgroup that owned the files is removed. | |
| # | |
| # $CG/old appends to NFILES files for OLD_SECS and exits. With MODE=dead | |
| # its cgroup is then removed while the files are still dirty; MODE=alive | |
| # keeps it as a control. $CG/new then appends to the same files for | |
| # NEW_SECS and prints, once a second, its throughput, write latency and | |
| # how far behind schedule it is (lag_s). If bpftrace is available it also | |
| # counts cgroup_writeback_by_id() return values (-2 is -ENOENT). | |
| # | |
| # Run as root on cgroup v2 with the memory controller enabled. Settings | |
| # (environment, defaults in brackets): | |
| # MODE=dead|alive remove the old cgroup or keep it [dead] | |
| # NEW_FILES=same|fresh new writer uses the old files or new ones [same] | |
| # NFILES files written by both writers [1000] | |
| # RATE_MBPS append rate of each writer, MiB/s [20] | |
| # CHUNK_KB write() size [64] | |
| # OLD_SECS / NEW_SECS how long each writer runs [40/120] | |
| # MEM_MAX memory.max of both writers' cgroups [512M] | |
| # POD_MAX memory.max of their parent cgroup [max] | |
| # DIR directory on ext4/xfs/btrfs to write to; | |
| # default: a new ext4 loop device in WORK | |
| # IMG_GB size of the sparse loop image [16] | |
| # WORK root-owned scratch directory [/var/lib/wb-dying-repro] | |
| # CG parent cgroup to create [/sys/fs/cgroup/wbrepro] | |
| # TRACE=auto|0 use bpftrace if available [auto] | |
| # SYNC_OLD=0|1|2 sync the volume (2: twice) after the old writer exits [0] | |
| # DIRTY_AFTER after that, the old cgroup appends 4KiB to this | |
| # many of its files, like a final marker write [0] | |
| # SHUTDOWN_FILE=empty|offset|offset_rename | |
| # then the old cgroup writes a new shutdown file: | |
| # empty, 14 bytes (like a final Kafka offset), or | |
| # 14 bytes via a temp file and rename [unset] | |
| # LATE_RMDIR remove the old cgroup this many seconds after | |
| # the new writer starts, not before it [0] | |
| # POD_HOP run the old writer under $CG/oldpod and remove | |
| # oldpod this many seconds after old; the new | |
| # writer starts after both are gone [unset] | |
| set -euo pipefail | |
| MODE=${MODE:-dead} | |
| NEW_FILES=${NEW_FILES:-same} | |
| NFILES=${NFILES:-1000} | |
| RATE_MBPS=${RATE_MBPS:-20} | |
| CHUNK_KB=${CHUNK_KB:-64} | |
| OLD_SECS=${OLD_SECS:-40} | |
| NEW_SECS=${NEW_SECS:-120} | |
| MEM_MAX=${MEM_MAX:-512M} | |
| POD_MAX=${POD_MAX:-max} | |
| DIR=${DIR:-} | |
| IMG_GB=${IMG_GB:-16} | |
| WORK=${WORK:-/var/lib/wb-dying-repro} | |
| CG=${CG:-/sys/fs/cgroup/wbrepro} | |
| TRACE=${TRACE:-auto} | |
| SYNC_OLD=${SYNC_OLD:-0} | |
| LATE_RMDIR=${LATE_RMDIR:-0} | |
| POD_HOP=${POD_HOP:-} | |
| DIRTY_AFTER=${DIRTY_AFTER:-0} | |
| SHUTDOWN_FILE=${SHUTDOWN_FILE:-} | |
| die() { echo "error: $*" >&2; exit 1; } | |
| [ "$(id -u)" = 0 ] || die "must run as root" | |
| for v in NFILES RATE_MBPS CHUNK_KB OLD_SECS NEW_SECS IMG_GB; do | |
| [[ ${!v} =~ ^[1-9][0-9]{0,6}$ ]] || die "$v must be a positive integer" | |
| done | |
| [[ $MEM_MAX =~ ^([0-9]+[KMG]?|max)$ ]] || die "MEM_MAX must look like 512M or be max" | |
| [[ $POD_MAX =~ ^([0-9]+[KMG]?|max)$ ]] || die "POD_MAX must look like 4G or be max" | |
| case $TRACE in auto|0) ;; *) die "TRACE must be auto or 0" ;; esac | |
| case $SYNC_OLD in 0|1|2) ;; *) die "SYNC_OLD must be 0, 1 or 2" ;; esac | |
| case $SHUTDOWN_FILE in ""|empty|offset|offset_rename) ;; | |
| *) die "SHUTDOWN_FILE must be empty, offset or offset_rename" ;; esac | |
| [[ $DIRTY_AFTER =~ ^[0-9]{1,7}$ ]] || die "DIRTY_AFTER must be a number of files" | |
| [[ $LATE_RMDIR =~ ^[0-9]{1,3}$ ]] || die "LATE_RMDIR must be a number of seconds" | |
| [[ -z $POD_HOP || $POD_HOP =~ ^[0-9]{1,3}$ ]] || die "POD_HOP must be a number of seconds" | |
| [[ -n $POD_HOP && $LATE_RMDIR != 0 ]] && die "POD_HOP and LATE_RMDIR are exclusive" | |
| OLDCG=$CG/old | |
| [ -n "$POD_HOP" ] && OLDCG=$CG/oldpod/old | |
| case $CG in /sys/fs/cgroup/*) ;; *) die "CG must be under /sys/fs/cgroup" ;; esac | |
| [ "$(stat -fc %T /sys/fs/cgroup)" = cgroup2fs ] || die "/sys/fs/cgroup is not cgroup v2" | |
| grep -qw memory /sys/fs/cgroup/cgroup.subtree_control || | |
| die "memory controller not enabled in /sys/fs/cgroup/cgroup.subtree_control" | |
| case $MODE in dead|alive) ;; *) die "MODE must be dead or alive" ;; esac | |
| case $NEW_FILES in same|fresh) ;; *) die "NEW_FILES must be same or fresh" ;; esac | |
| [ -e "$CG" ] && die "$CG already exists (leftover from a previous run?)" | |
| mkdir -p -m 0700 "$WORK" | |
| [ -d "$WORK" ] && [ ! -L "$WORK" ] && [ "$(stat -c %u "$WORK")" = 0 ] && | |
| [ -z "$(find "$WORK" -maxdepth 0 -perm /022)" ] || | |
| die "$WORK must be a root-owned directory not writable by others" | |
| ulimit -n $((NFILES + 64)) | |
| LOOPDEV= | |
| MNT= | |
| BPF_PID= | |
| WRITER_PID= | |
| cleanup() { | |
| set +e | |
| [ -n "$WRITER_PID" ] && kill "$WRITER_PID" 2>/dev/null && wait "$WRITER_PID" 2>/dev/null | |
| [ -n "$BPF_PID" ] && kill -INT "$BPF_PID" 2>/dev/null && wait "$BPF_PID" 2>/dev/null | |
| for c in "$OLDCG" "$CG"/new; do | |
| [ -d "$c" ] || continue | |
| for p in $(cat "$c/cgroup.procs" 2>/dev/null); do kill -9 "$p"; done | |
| done | |
| sleep 0.5 | |
| rmdir "$OLDCG" "$CG"/oldpod "$CG"/new "$CG" 2>/dev/null | |
| if [ -n "$MNT" ]; then | |
| umount "$MNT" | |
| losetup -d "$LOOPDEV" | |
| fi | |
| } | |
| trap cleanup EXIT | |
| # ---- volume ---------------------------------------------------------------- | |
| if [ -z "$DIR" ]; then | |
| IMG=$WORK/img | |
| MNT=$WORK/mnt | |
| rm -f "$IMG" | |
| truncate -s "${IMG_GB}G" "$IMG" | |
| mkfs.ext4 -q -F "$IMG" | |
| LOOPDEV=$(losetup --direct-io=on -f --show "$IMG") | |
| mkdir -p "$MNT" | |
| mount "$LOOPDEV" "$MNT" | |
| DIR=$MNT/data | |
| fi | |
| mkdir -p "$DIR" | |
| OLD_DIR=$DIR/old-$$ | |
| NEW_DIR=$OLD_DIR | |
| [ "$NEW_FILES" = fresh ] && NEW_DIR=$DIR/new-$$ | |
| mkdir -p "$OLD_DIR" "$NEW_DIR" | |
| # ---- cgroups --------------------------------------------------------------- | |
| mkdir "$CG" | |
| echo +memory > "$CG/cgroup.subtree_control" | |
| echo +io > "$CG/cgroup.subtree_control" 2>/dev/null || | |
| echo "note: could not enable io controller in $CG; continuing with memory only" | |
| if [ -n "$POD_HOP" ]; then | |
| mkdir "$CG/oldpod" | |
| echo +memory > "$CG/oldpod/cgroup.subtree_control" | |
| fi | |
| mkdir "$OLDCG" "$CG/new" | |
| echo "$POD_MAX" > "$CG/memory.max" | |
| echo "$MEM_MAX" > "$OLDCG/memory.max" | |
| echo "$MEM_MAX" > "$CG/new/memory.max" | |
| # ---- writer ---------------------------------------------------------------- | |
| WRITER=$WORK/writer.py | |
| cat > "$WRITER" <<'EOF' | |
| import os, sys, time | |
| d, nfiles, rate_mbps, secs, chunk_kb, cg, label = sys.argv[1:8] | |
| nfiles = int(nfiles) | |
| rate = float(rate_mbps) * 1048576 | |
| secs = float(secs) | |
| chunk = int(chunk_kb) * 1024 | |
| parent = os.path.dirname(cg) | |
| def memstat(path, key): | |
| try: | |
| with open(os.path.join(path, "memory.stat")) as f: | |
| for line in f: | |
| k, v = line.split() | |
| if k == key: | |
| return int(v) >> 20 | |
| except OSError: | |
| pass | |
| return -1 | |
| def cgstat(path, key): | |
| try: | |
| with open(os.path.join(path, "cgroup.stat")) as f: | |
| for line in f: | |
| k, v = line.split() | |
| if k == key: | |
| return int(v) | |
| except OSError: | |
| pass | |
| return -1 | |
| buf = os.urandom(chunk) | |
| fds = [os.open(f"{d}/f{i:06d}", os.O_WRONLY | os.O_APPEND | os.O_CREAT, 0o644) | |
| for i in range(nfiles)] | |
| print(f"{label}: t MiB/s lag_s max_ms p99_ms n>100ms " | |
| f"dirty_self_MiB dirty_pod_MiB dying", flush=True) | |
| start = time.monotonic() | |
| written = 0 | |
| sec_bytes = 0 | |
| lats = [] | |
| sec = 0 | |
| i = 0 | |
| worst_lag = 0.0 | |
| lag_start = lag_end = None | |
| def report(): | |
| global sec_bytes, lats, worst_lag, lag_start, lag_end | |
| el = time.monotonic() - start | |
| lag = max(0.0, (rate * el - written) / rate) | |
| worst_lag = max(worst_lag, lag) | |
| if lag > 1.0: | |
| if lag_start is None: | |
| lag_start = sec | |
| lag_end = sec | |
| lats.sort() | |
| mx = lats[-1] * 1000 if lats else 0.0 | |
| p99 = lats[int(len(lats) * 0.99)] * 1000 if lats else 0.0 | |
| slow = sum(1 for l in lats if l > 0.1) | |
| print(f"{label}: {sec:4d} {sec_bytes / 1048576:6.1f} {lag:6.1f} " | |
| f"{mx:7.1f} {p99:7.1f} {slow:5d} " | |
| f"{memstat(cg, 'file_dirty'):8d} {memstat(parent, 'file_dirty'):8d} " | |
| f"{cgstat(parent, 'nr_dying_descendants'):3d}", flush=True) | |
| sec_bytes = 0 | |
| lats = [] | |
| while True: | |
| el = time.monotonic() - start | |
| if el >= secs: | |
| break | |
| while int(el) > sec: | |
| report() | |
| sec += 1 | |
| ahead = written - rate * el | |
| if ahead >= 0: | |
| time.sleep(min(0.005, ahead / rate + 0.0005)) | |
| continue | |
| t0 = time.perf_counter() | |
| os.write(fds[i % nfiles], buf) | |
| lats.append(time.perf_counter() - t0) | |
| written += chunk | |
| sec_bytes += chunk | |
| i += 1 | |
| report() | |
| if lag_start is None: | |
| print(f"{label}: summary: worst lag {worst_lag:.1f}s, never more than 1s behind") | |
| else: | |
| print(f"{label}: summary: worst lag {worst_lag:.1f}s, " | |
| f">1s behind from t={lag_start}s to t={lag_end}s") | |
| EOF | |
| # Start "$@" inside cgroup $1, in the background; sets WRITER_PID. | |
| run_in_cg() { | |
| local cg=$1 | |
| shift | |
| bash -c 'echo $$ > "$0/cgroup.procs" && exec "$@"' "$cg" "$@" & | |
| WRITER_PID=$! | |
| } | |
| # ---- tracing --------------------------------------------------------------- | |
| if [ "$TRACE" != 0 ] && command -v bpftrace >/dev/null; then | |
| BPF_LOG=$WORK/bpftrace.log | |
| bpftrace -e ' | |
| kprobe:cgroup_writeback_by_id { @in[tid] = 1; } | |
| kretprobe:cgroup_writeback_by_id /@in[tid]/ { | |
| @by_id_ret[(int32)retval] = count(); | |
| @by_id_ret_5s[(int32)retval] = count(); | |
| delete(@in[tid]); | |
| } | |
| tracepoint:writeback:flush_foreign { @flush_foreign[args.cgroup_ino] = count(); } | |
| tracepoint:writeback:inode_switch_wbs { | |
| @switched[args.old_cgroup_ino, args.new_cgroup_ino] = count(); | |
| } | |
| tracepoint:writeback:writeback_start { | |
| @wb_start[args.cgroup_ino, args.reason] = count(); | |
| } | |
| tracepoint:writeback:balance_dirty_pages { @bdp_pause_ms[args.cgroup_ino] = hist(args.pause); } | |
| interval:s:5 { time("%H:%M:%S "); print(@by_id_ret_5s); clear(@by_id_ret_5s); } | |
| END { clear(@in); clear(@by_id_ret_5s); } | |
| ' > "$BPF_LOG" 2>&1 & | |
| BPF_PID=$! | |
| sleep 3 | |
| kill -0 "$BPF_PID" 2>/dev/null || | |
| { echo "note: bpftrace failed, see $BPF_LOG"; BPF_PID=; } | |
| fi | |
| # ---- run ------------------------------------------------------------------- | |
| echo "kernel $(uname -r), MODE=$MODE NEW_FILES=$NEW_FILES NFILES=$NFILES" \ | |
| "RATE_MBPS=$RATE_MBPS MEM_MAX=$MEM_MAX POD_MAX=$POD_MAX SYNC_OLD=$SYNC_OLD LATE_RMDIR=$LATE_RMDIR POD_HOP=$POD_HOP DIRTY_AFTER=$DIRTY_AFTER SHUTDOWN_FILE=$SHUTDOWN_FILE dir=$DIR" | |
| echo "vm.dirty_expire_centisecs=$(sysctl -n vm.dirty_expire_centisecs)" \ | |
| "vm.dirty_writeback_centisecs=$(sysctl -n vm.dirty_writeback_centisecs)" | |
| echo "cgroup inos: pod=$(stat -c %i "$CG") old=$(stat -c %i "$OLDCG")" \ | |
| "new=$(stat -c %i "$CG/new")" | |
| echo "writeback reasons (enum wb_reason): 0=background 1=vmscan 2=sync 3=periodic;" \ | |
| "foreign_flush is 7 up to v6.18, 6 once WB_REASON_LAPTOP_TIMER is gone" | |
| echo | |
| echo "== old container writes for ${OLD_SECS}s" | |
| run_in_cg "$OLDCG" python3 "$WRITER" "$OLD_DIR" "$NFILES" "$RATE_MBPS" \ | |
| "$OLD_SECS" "$CHUNK_KB" "$OLDCG" old > "$WORK/old.log" | |
| wait "$WRITER_PID" | |
| WRITER_PID= | |
| tail -n 2 "$WORK/old.log" | |
| remove_old() { | |
| for _ in $(seq 50); do rmdir "$OLDCG" 2>/dev/null && break; sleep 0.1; done | |
| [ -d "$OLDCG" ] && die "could not rmdir $OLDCG" | |
| echo "== old cgroup removed $1" | |
| if [ -n "$POD_HOP" ]; then | |
| sleep "$POD_HOP" | |
| rmdir "$CG/oldpod" || die "could not rmdir $CG/oldpod" | |
| echo "== old pod cgroup removed ${POD_HOP}s later" | |
| fi | |
| } | |
| if [ "$SYNC_OLD" != 0 ]; then | |
| sync -f "$DIR" | |
| [ "$SYNC_OLD" = 2 ] && sync -f "$DIR" | |
| echo "== synced after the old writer exited" | |
| fi | |
| if [ "$DIRTY_AFTER" != 0 ]; then | |
| run_in_cg "$OLDCG" python3 -c ' | |
| import os, sys | |
| d, n = sys.argv[1], int(sys.argv[2]) | |
| for i in range(n): | |
| fd = os.open(f"{d}/f{i:06d}", os.O_WRONLY | os.O_APPEND) | |
| os.write(fd, b"C" * 4096) | |
| os.close(fd) | |
| ' "$OLD_DIR" "$DIRTY_AFTER" | |
| wait "$WRITER_PID" | |
| WRITER_PID= | |
| echo "== old cgroup appended 4KiB to $DIRTY_AFTER file(s)" | |
| fi | |
| if [ -n "$SHUTDOWN_FILE" ]; then | |
| run_in_cg "$OLDCG" python3 -c ' | |
| import os, sys | |
| d, mode = sys.argv[1], sys.argv[2] | |
| data = b"" if mode == "empty" else b"1234567890123\n" | |
| path = f"{d}/SHUTDOWN" | |
| tmp = path + ".tmp" if mode == "offset_rename" else path | |
| fd = os.open(tmp, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o644) | |
| os.write(fd, data) | |
| os.close(fd) | |
| if tmp != path: | |
| os.rename(tmp, path) | |
| ' "$OLD_DIR" "$SHUTDOWN_FILE" | |
| wait "$WRITER_PID" | |
| WRITER_PID= | |
| echo "== old cgroup wrote a shutdown file ($SHUTDOWN_FILE)" | |
| fi | |
| if [ "$MODE" = dead ] && [ "$LATE_RMDIR" = 0 ]; then | |
| remove_old "(before the new writer starts)" | |
| elif [ "$MODE" = alive ]; then | |
| echo "== old cgroup kept alive (control)" | |
| fi | |
| echo "== new container writes for ${NEW_SECS}s" | |
| run_in_cg "$CG/new" python3 "$WRITER" "$NEW_DIR" "$NFILES" "$RATE_MBPS" \ | |
| "$NEW_SECS" "$CHUNK_KB" "$CG/new" new | |
| if [ "$MODE" = dead ] && [ "$LATE_RMDIR" != 0 ]; then | |
| sleep "$LATE_RMDIR" | |
| remove_old "${LATE_RMDIR}s after the new writer started" | |
| fi | |
| wait "$WRITER_PID" | |
| WRITER_PID= | |
| if [ -n "$BPF_PID" ]; then | |
| kill -INT "$BPF_PID" | |
| wait "$BPF_PID" || true | |
| BPF_PID= | |
| echo | |
| echo "== bpftrace ($BPF_LOG)" | |
| echo " @by_id_ret[-2] = foreign flushes that found no wb (dying owner)" | |
| cat "$BPF_LOG" | |
| fi |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment