Skip to content

Instantly share code, notes, and snippets.

@lizthegrey
Last active September 28, 2026 22:07
Show Gist options
  • Select an option

  • Save lizthegrey/2209d831930588f63076bdc0ac7b78e2 to your computer and use it in GitHub Desktop.

Select an option

Save lizthegrey/2209d831930588f63076bdc0ac7b78e2 to your computer and use it in GitHub Desktop.
Reproducer: writer stalls after the owning cgroup is removed (cgroup_writeback_by_id -ENOENT on dying cgwbs)
#!/bin/bash
# SPDX-License-Identifier: GPL-2.0
#
# Reproduce writer stalls after the cgroup that owned the files is removed.
#
# $CG/old appends to NFILES files for OLD_SECS and exits. With MODE=dead
# its cgroup is then removed while the files are still dirty; MODE=alive
# keeps it as a control. $CG/new then appends to the same files for
# NEW_SECS and prints, once a second, its throughput, write latency and
# how far behind schedule it is (lag_s). If bpftrace is available it also
# counts cgroup_writeback_by_id() return values (-2 is -ENOENT).
#
# Run as root on cgroup v2 with the memory controller enabled. Settings
# (environment, defaults in brackets):
# MODE=dead|alive remove the old cgroup or keep it [dead]
# NEW_FILES=same|fresh new writer uses the old files or new ones [same]
# NFILES files written by both writers [1000]
# RATE_MBPS append rate of each writer, MiB/s [20]
# CHUNK_KB write() size [64]
# OLD_SECS / NEW_SECS how long each writer runs [40/120]
# MEM_MAX memory.max of both writers' cgroups [512M]
# POD_MAX memory.max of their parent cgroup [max]
# DIR directory on ext4/xfs/btrfs to write to;
# default: a new ext4 loop device in WORK
# IMG_GB size of the sparse loop image [16]
# WORK root-owned scratch directory [/var/lib/wb-dying-repro]
# CG parent cgroup to create [/sys/fs/cgroup/wbrepro]
# TRACE=auto|0 use bpftrace if available [auto]
# SYNC_OLD=0|1|2 sync the volume (2: twice) after the old writer exits [0]
# DIRTY_AFTER after that, the old cgroup appends 4KiB to this
# many of its files, like a final marker write [0]
# SHUTDOWN_FILE=empty|offset|offset_rename
# then the old cgroup writes a new shutdown file:
# empty, 14 bytes (like a final Kafka offset), or
# 14 bytes via a temp file and rename [unset]
# LATE_RMDIR remove the old cgroup this many seconds after
# the new writer starts, not before it [0]
# POD_HOP run the old writer under $CG/oldpod and remove
# oldpod this many seconds after old; the new
# writer starts after both are gone [unset]
set -euo pipefail
MODE=${MODE:-dead}
NEW_FILES=${NEW_FILES:-same}
NFILES=${NFILES:-1000}
RATE_MBPS=${RATE_MBPS:-20}
CHUNK_KB=${CHUNK_KB:-64}
OLD_SECS=${OLD_SECS:-40}
NEW_SECS=${NEW_SECS:-120}
MEM_MAX=${MEM_MAX:-512M}
POD_MAX=${POD_MAX:-max}
DIR=${DIR:-}
IMG_GB=${IMG_GB:-16}
WORK=${WORK:-/var/lib/wb-dying-repro}
CG=${CG:-/sys/fs/cgroup/wbrepro}
TRACE=${TRACE:-auto}
SYNC_OLD=${SYNC_OLD:-0}
LATE_RMDIR=${LATE_RMDIR:-0}
POD_HOP=${POD_HOP:-}
DIRTY_AFTER=${DIRTY_AFTER:-0}
SHUTDOWN_FILE=${SHUTDOWN_FILE:-}
die() { echo "error: $*" >&2; exit 1; }
[ "$(id -u)" = 0 ] || die "must run as root"
for v in NFILES RATE_MBPS CHUNK_KB OLD_SECS NEW_SECS IMG_GB; do
[[ ${!v} =~ ^[1-9][0-9]{0,6}$ ]] || die "$v must be a positive integer"
done
[[ $MEM_MAX =~ ^([0-9]+[KMG]?|max)$ ]] || die "MEM_MAX must look like 512M or be max"
[[ $POD_MAX =~ ^([0-9]+[KMG]?|max)$ ]] || die "POD_MAX must look like 4G or be max"
case $TRACE in auto|0) ;; *) die "TRACE must be auto or 0" ;; esac
case $SYNC_OLD in 0|1|2) ;; *) die "SYNC_OLD must be 0, 1 or 2" ;; esac
case $SHUTDOWN_FILE in ""|empty|offset|offset_rename) ;;
*) die "SHUTDOWN_FILE must be empty, offset or offset_rename" ;; esac
[[ $DIRTY_AFTER =~ ^[0-9]{1,7}$ ]] || die "DIRTY_AFTER must be a number of files"
[[ $LATE_RMDIR =~ ^[0-9]{1,3}$ ]] || die "LATE_RMDIR must be a number of seconds"
[[ -z $POD_HOP || $POD_HOP =~ ^[0-9]{1,3}$ ]] || die "POD_HOP must be a number of seconds"
[[ -n $POD_HOP && $LATE_RMDIR != 0 ]] && die "POD_HOP and LATE_RMDIR are exclusive"
OLDCG=$CG/old
[ -n "$POD_HOP" ] && OLDCG=$CG/oldpod/old
case $CG in /sys/fs/cgroup/*) ;; *) die "CG must be under /sys/fs/cgroup" ;; esac
[ "$(stat -fc %T /sys/fs/cgroup)" = cgroup2fs ] || die "/sys/fs/cgroup is not cgroup v2"
grep -qw memory /sys/fs/cgroup/cgroup.subtree_control ||
die "memory controller not enabled in /sys/fs/cgroup/cgroup.subtree_control"
case $MODE in dead|alive) ;; *) die "MODE must be dead or alive" ;; esac
case $NEW_FILES in same|fresh) ;; *) die "NEW_FILES must be same or fresh" ;; esac
[ -e "$CG" ] && die "$CG already exists (leftover from a previous run?)"
mkdir -p -m 0700 "$WORK"
[ -d "$WORK" ] && [ ! -L "$WORK" ] && [ "$(stat -c %u "$WORK")" = 0 ] &&
[ -z "$(find "$WORK" -maxdepth 0 -perm /022)" ] ||
die "$WORK must be a root-owned directory not writable by others"
ulimit -n $((NFILES + 64))
LOOPDEV=
MNT=
BPF_PID=
WRITER_PID=
cleanup() {
set +e
[ -n "$WRITER_PID" ] && kill "$WRITER_PID" 2>/dev/null && wait "$WRITER_PID" 2>/dev/null
[ -n "$BPF_PID" ] && kill -INT "$BPF_PID" 2>/dev/null && wait "$BPF_PID" 2>/dev/null
for c in "$OLDCG" "$CG"/new; do
[ -d "$c" ] || continue
for p in $(cat "$c/cgroup.procs" 2>/dev/null); do kill -9 "$p"; done
done
sleep 0.5
rmdir "$OLDCG" "$CG"/oldpod "$CG"/new "$CG" 2>/dev/null
if [ -n "$MNT" ]; then
umount "$MNT"
losetup -d "$LOOPDEV"
fi
}
trap cleanup EXIT
# ---- volume ----------------------------------------------------------------
if [ -z "$DIR" ]; then
IMG=$WORK/img
MNT=$WORK/mnt
rm -f "$IMG"
truncate -s "${IMG_GB}G" "$IMG"
mkfs.ext4 -q -F "$IMG"
LOOPDEV=$(losetup --direct-io=on -f --show "$IMG")
mkdir -p "$MNT"
mount "$LOOPDEV" "$MNT"
DIR=$MNT/data
fi
mkdir -p "$DIR"
OLD_DIR=$DIR/old-$$
NEW_DIR=$OLD_DIR
[ "$NEW_FILES" = fresh ] && NEW_DIR=$DIR/new-$$
mkdir -p "$OLD_DIR" "$NEW_DIR"
# ---- cgroups ---------------------------------------------------------------
mkdir "$CG"
echo +memory > "$CG/cgroup.subtree_control"
echo +io > "$CG/cgroup.subtree_control" 2>/dev/null ||
echo "note: could not enable io controller in $CG; continuing with memory only"
if [ -n "$POD_HOP" ]; then
mkdir "$CG/oldpod"
echo +memory > "$CG/oldpod/cgroup.subtree_control"
fi
mkdir "$OLDCG" "$CG/new"
echo "$POD_MAX" > "$CG/memory.max"
echo "$MEM_MAX" > "$OLDCG/memory.max"
echo "$MEM_MAX" > "$CG/new/memory.max"
# ---- writer ----------------------------------------------------------------
WRITER=$WORK/writer.py
cat > "$WRITER" <<'EOF'
import os, sys, time
d, nfiles, rate_mbps, secs, chunk_kb, cg, label = sys.argv[1:8]
nfiles = int(nfiles)
rate = float(rate_mbps) * 1048576
secs = float(secs)
chunk = int(chunk_kb) * 1024
parent = os.path.dirname(cg)
def memstat(path, key):
try:
with open(os.path.join(path, "memory.stat")) as f:
for line in f:
k, v = line.split()
if k == key:
return int(v) >> 20
except OSError:
pass
return -1
def cgstat(path, key):
try:
with open(os.path.join(path, "cgroup.stat")) as f:
for line in f:
k, v = line.split()
if k == key:
return int(v)
except OSError:
pass
return -1
buf = os.urandom(chunk)
fds = [os.open(f"{d}/f{i:06d}", os.O_WRONLY | os.O_APPEND | os.O_CREAT, 0o644)
for i in range(nfiles)]
print(f"{label}: t MiB/s lag_s max_ms p99_ms n>100ms "
f"dirty_self_MiB dirty_pod_MiB dying", flush=True)
start = time.monotonic()
written = 0
sec_bytes = 0
lats = []
sec = 0
i = 0
worst_lag = 0.0
lag_start = lag_end = None
def report():
global sec_bytes, lats, worst_lag, lag_start, lag_end
el = time.monotonic() - start
lag = max(0.0, (rate * el - written) / rate)
worst_lag = max(worst_lag, lag)
if lag > 1.0:
if lag_start is None:
lag_start = sec
lag_end = sec
lats.sort()
mx = lats[-1] * 1000 if lats else 0.0
p99 = lats[int(len(lats) * 0.99)] * 1000 if lats else 0.0
slow = sum(1 for l in lats if l > 0.1)
print(f"{label}: {sec:4d} {sec_bytes / 1048576:6.1f} {lag:6.1f} "
f"{mx:7.1f} {p99:7.1f} {slow:5d} "
f"{memstat(cg, 'file_dirty'):8d} {memstat(parent, 'file_dirty'):8d} "
f"{cgstat(parent, 'nr_dying_descendants'):3d}", flush=True)
sec_bytes = 0
lats = []
while True:
el = time.monotonic() - start
if el >= secs:
break
while int(el) > sec:
report()
sec += 1
ahead = written - rate * el
if ahead >= 0:
time.sleep(min(0.005, ahead / rate + 0.0005))
continue
t0 = time.perf_counter()
os.write(fds[i % nfiles], buf)
lats.append(time.perf_counter() - t0)
written += chunk
sec_bytes += chunk
i += 1
report()
if lag_start is None:
print(f"{label}: summary: worst lag {worst_lag:.1f}s, never more than 1s behind")
else:
print(f"{label}: summary: worst lag {worst_lag:.1f}s, "
f">1s behind from t={lag_start}s to t={lag_end}s")
EOF
# Start "$@" inside cgroup $1, in the background; sets WRITER_PID.
run_in_cg() {
local cg=$1
shift
bash -c 'echo $$ > "$0/cgroup.procs" && exec "$@"' "$cg" "$@" &
WRITER_PID=$!
}
# ---- tracing ---------------------------------------------------------------
if [ "$TRACE" != 0 ] && command -v bpftrace >/dev/null; then
BPF_LOG=$WORK/bpftrace.log
bpftrace -e '
kprobe:cgroup_writeback_by_id { @in[tid] = 1; }
kretprobe:cgroup_writeback_by_id /@in[tid]/ {
@by_id_ret[(int32)retval] = count();
@by_id_ret_5s[(int32)retval] = count();
delete(@in[tid]);
}
tracepoint:writeback:flush_foreign { @flush_foreign[args.cgroup_ino] = count(); }
tracepoint:writeback:inode_switch_wbs {
@switched[args.old_cgroup_ino, args.new_cgroup_ino] = count();
}
tracepoint:writeback:writeback_start {
@wb_start[args.cgroup_ino, args.reason] = count();
}
tracepoint:writeback:balance_dirty_pages { @bdp_pause_ms[args.cgroup_ino] = hist(args.pause); }
interval:s:5 { time("%H:%M:%S "); print(@by_id_ret_5s); clear(@by_id_ret_5s); }
END { clear(@in); clear(@by_id_ret_5s); }
' > "$BPF_LOG" 2>&1 &
BPF_PID=$!
sleep 3
kill -0 "$BPF_PID" 2>/dev/null ||
{ echo "note: bpftrace failed, see $BPF_LOG"; BPF_PID=; }
fi
# ---- run -------------------------------------------------------------------
echo "kernel $(uname -r), MODE=$MODE NEW_FILES=$NEW_FILES NFILES=$NFILES" \
"RATE_MBPS=$RATE_MBPS MEM_MAX=$MEM_MAX POD_MAX=$POD_MAX SYNC_OLD=$SYNC_OLD LATE_RMDIR=$LATE_RMDIR POD_HOP=$POD_HOP DIRTY_AFTER=$DIRTY_AFTER SHUTDOWN_FILE=$SHUTDOWN_FILE dir=$DIR"
echo "vm.dirty_expire_centisecs=$(sysctl -n vm.dirty_expire_centisecs)" \
"vm.dirty_writeback_centisecs=$(sysctl -n vm.dirty_writeback_centisecs)"
echo "cgroup inos: pod=$(stat -c %i "$CG") old=$(stat -c %i "$OLDCG")" \
"new=$(stat -c %i "$CG/new")"
echo "writeback reasons (enum wb_reason): 0=background 1=vmscan 2=sync 3=periodic;" \
"foreign_flush is 7 up to v6.18, 6 once WB_REASON_LAPTOP_TIMER is gone"
echo
echo "== old container writes for ${OLD_SECS}s"
run_in_cg "$OLDCG" python3 "$WRITER" "$OLD_DIR" "$NFILES" "$RATE_MBPS" \
"$OLD_SECS" "$CHUNK_KB" "$OLDCG" old > "$WORK/old.log"
wait "$WRITER_PID"
WRITER_PID=
tail -n 2 "$WORK/old.log"
remove_old() {
for _ in $(seq 50); do rmdir "$OLDCG" 2>/dev/null && break; sleep 0.1; done
[ -d "$OLDCG" ] && die "could not rmdir $OLDCG"
echo "== old cgroup removed $1"
if [ -n "$POD_HOP" ]; then
sleep "$POD_HOP"
rmdir "$CG/oldpod" || die "could not rmdir $CG/oldpod"
echo "== old pod cgroup removed ${POD_HOP}s later"
fi
}
if [ "$SYNC_OLD" != 0 ]; then
sync -f "$DIR"
[ "$SYNC_OLD" = 2 ] && sync -f "$DIR"
echo "== synced after the old writer exited"
fi
if [ "$DIRTY_AFTER" != 0 ]; then
run_in_cg "$OLDCG" python3 -c '
import os, sys
d, n = sys.argv[1], int(sys.argv[2])
for i in range(n):
fd = os.open(f"{d}/f{i:06d}", os.O_WRONLY | os.O_APPEND)
os.write(fd, b"C" * 4096)
os.close(fd)
' "$OLD_DIR" "$DIRTY_AFTER"
wait "$WRITER_PID"
WRITER_PID=
echo "== old cgroup appended 4KiB to $DIRTY_AFTER file(s)"
fi
if [ -n "$SHUTDOWN_FILE" ]; then
run_in_cg "$OLDCG" python3 -c '
import os, sys
d, mode = sys.argv[1], sys.argv[2]
data = b"" if mode == "empty" else b"1234567890123\n"
path = f"{d}/SHUTDOWN"
tmp = path + ".tmp" if mode == "offset_rename" else path
fd = os.open(tmp, os.O_WRONLY | os.O_CREAT | os.O_TRUNC, 0o644)
os.write(fd, data)
os.close(fd)
if tmp != path:
os.rename(tmp, path)
' "$OLD_DIR" "$SHUTDOWN_FILE"
wait "$WRITER_PID"
WRITER_PID=
echo "== old cgroup wrote a shutdown file ($SHUTDOWN_FILE)"
fi
if [ "$MODE" = dead ] && [ "$LATE_RMDIR" = 0 ]; then
remove_old "(before the new writer starts)"
elif [ "$MODE" = alive ]; then
echo "== old cgroup kept alive (control)"
fi
echo "== new container writes for ${NEW_SECS}s"
run_in_cg "$CG/new" python3 "$WRITER" "$NEW_DIR" "$NFILES" "$RATE_MBPS" \
"$NEW_SECS" "$CHUNK_KB" "$CG/new" new
if [ "$MODE" = dead ] && [ "$LATE_RMDIR" != 0 ]; then
sleep "$LATE_RMDIR"
remove_old "${LATE_RMDIR}s after the new writer started"
fi
wait "$WRITER_PID"
WRITER_PID=
if [ -n "$BPF_PID" ]; then
kill -INT "$BPF_PID"
wait "$BPF_PID" || true
BPF_PID=
echo
echo "== bpftrace ($BPF_LOG)"
echo " @by_id_ret[-2] = foreign flushes that found no wb (dying owner)"
cat "$BPF_LOG"
fi
Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment