From f7e61d2b19a77954514ef1dd8c438e364c8e92ba Mon Sep 17 00:00:00 2001 From: Alan Derk <62039254+aderk@users.noreply.github.com> Date: Wed, 7 Oct 2026 20:59:09 +0000 Subject: [PATCH 01/11] bcachefs: passes_no_ratelimit_set: a subvolume unlink marks a pass (no ratelimit) bch2_recovery_pass_set_no_ratelimit() wrote false, so the flag never reached the superblock. Fixed by bcachefs-tools "passes: bch2_recovery_pass_set_no_ratelimit() never sets the flag". --- tests/fs/bcachefs/single_device.ktest | 27 +++++++++++++++++++++++++++ 1 file changed, 27 insertions(+) diff --git a/tests/fs/bcachefs/single_device.ktest b/tests/fs/bcachefs/single_device.ktest index 2339f5b6..d7f32b15 100755 --- a/tests/fs/bcachefs/single_device.ktest +++ b/tests/fs/bcachefs/single_device.ktest @@ -3062,6 +3062,33 @@ test_recovery_pass_offline() fi } +# A subvolume unlink asks for check_subvols to run unratelimited next time; +# the setter wrote false, so the flag never appeared in the superblock. +test_passes_no_ratelimit_set() +{ + local dev=${ktest_scratch_dev[0]} + + set_watchdog 60 + + run_quiet "" bcachefs format -f $dev + + # the flag is only shown for a pass with a last_run: run them once + mount -t bcachefs -o fsck $dev /mnt + umount /mnt + + mount -t bcachefs $dev /mnt + bcachefs subvolume create /mnt/sub + bcachefs subvolume delete /mnt/sub + umount /mnt + + local passes=$(bcachefs show-super -f recovery_passes $dev) + echo "$passes" + if ! grep "no ratelimit" <<<"$passes" >/dev/null; then + echo "ERROR: no recovery pass is marked (no ratelimit) after a subvolume unlink" + return 1 + fi +} + # The passphrase lifecycle on an existing filesystem: adding one, changing # one, and taking one off. # From f521a4e9425a022c04eac82f0a4fc66e94efd27d Mon Sep 17 00:00:00 2001 From: Alan Derk <62039254+aderk@users.noreply.github.com> Date: Wed, 7 Oct 2026 20:59:10 +0000 Subject: [PATCH 02/11] bcachefs: passes_exclude_bitmask: recovery_passes_exclude stops a requested pass recovery_passes_exclude was masked with a pass number, so excluding scan_for_btree_nodes didn't stop it. Fixed by bcachefs-tools "passes: recovery_passes_exclude is masked with a pass number". --- tests/fs/bcachefs/single_device.ktest | 24 ++++++++++++++++++++++++ 1 file changed, 24 insertions(+) diff --git a/tests/fs/bcachefs/single_device.ktest b/tests/fs/bcachefs/single_device.ktest index d7f32b15..d7efeb5a 100755 --- a/tests/fs/bcachefs/single_device.ktest +++ b/tests/fs/bcachefs/single_device.ktest @@ -3089,6 +3089,30 @@ test_passes_no_ratelimit_set() fi } +# recovery_passes_exclude is a bitfield; it was masked with a pass number, +# which cleared the scan_for_btree_nodes bit from the exclusion. +test_passes_exclude_bitmask() +{ + local dev=${ktest_scratch_dev[0]} + + set_watchdog 60 + + run_quiet "" bcachefs format -f $dev + mount -t bcachefs $dev /mnt + umount /mnt + + dmesg_mark passes_exclude + mount -t bcachefs -o recovery_passes=scan_for_btree_nodes,recovery_passes_exclude=scan_for_btree_nodes $dev /mnt + umount /mnt + + if dmesg_since passes_exclude | grep "scan_for_btree_nodes\.\.\." >/dev/null; then + echo "ERROR: scan_for_btree_nodes ran although recovery_passes_exclude named it" + return 1 + fi + + bcachefs_test_end_checks $dev +} + # The passphrase lifecycle on an existing filesystem: adding one, changing # one, and taking one off. # From edb4da9fdc8e5ba6bcb8282c81804aa0905e4448 Mon Sep 17 00:00:00 2001 From: Alan Derk <62039254+aderk@users.noreply.github.com> Date: Wed, 7 Oct 2026 20:59:10 +0000 Subject: [PATCH 03/11] bcachefs: fallocate_pagecache_reserved: a write over cached fallocated folios isn't counted twice bch2_mark_pagecache_reserved() marked nothing, so i_blocks grew a second time. Fixed by bcachefs-tools "fallocate: bch2_mark_pagecache_reserved() marks nothing". --- tests/fs/bcachefs/single_device.ktest | 46 +++++++++++++++++++++++++++ 1 file changed, 46 insertions(+) diff --git a/tests/fs/bcachefs/single_device.ktest b/tests/fs/bcachefs/single_device.ktest index d7efeb5a..7d42cebb 100755 --- a/tests/fs/bcachefs/single_device.ktest +++ b/tests/fs/bcachefs/single_device.ktest @@ -205,6 +205,52 @@ test_extent_merge2() true } +# A buffered write into a fallocated range whose folios are cached must not +# add the sectors to i_blocks a second time: fallocate already accounted them, +# and bch2_mark_pagecache_reserved() is what flips the cached folios' sectors +# to reserved. It advanced *start before computing the range, so it marked +# nothing. +test_fallocate_pagecache_reserved() +{ + set_watchdog 60 + local dev=${ktest_scratch_dev[0]} + + run_quiet "" bcachefs format -f $dev + mount -t bcachefs $dev /mnt + + # cache zero folios over the hole, then fallocate under them + truncate -s 32M /mnt/f + cat /mnt/f > /dev/null + fallocate -l 32M /mnt/f + + local blocks=$(stat -c %b /mnt/f) + echo "st_blocks after fallocate: $blocks" + if [[ $blocks != 65536 ]]; then + echo "ERROR: expected 65536 blocks after fallocate, got $blocks" + return 1 + fi + + dd if=/dev/urandom of=/mnt/f bs=1M count=32 conv=notrunc status=none + + blocks=$(stat -c %b /mnt/f) + echo "st_blocks after a buffered write into the fallocated range: $blocks" + if [[ $blocks != 65536 ]]; then + echo "ERROR: the write was counted again: $blocks blocks" + return 1 + fi + + sync + blocks=$(stat -c %b /mnt/f) + echo "st_blocks after sync: $blocks" + if [[ $blocks != 65536 ]]; then + echo "ERROR: expected 65536 blocks after sync, got $blocks" + return 1 + fi + + umount /mnt + bcachefs_test_end_checks $dev +} + test_extent_repair_overlapping() { set_watchdog 30 From a9c1797bd5bbcef1144635be7db139e434bea397 Mon Sep 17 00:00:00 2001 From: Alan Derk <62039254+aderk@users.noreply.github.com> Date: Wed, 7 Oct 2026 20:59:10 +0000 Subject: [PATCH 04/11] bcachefs: quota_chown_counts_once: a chgrp's limit check counts the group's usage once bch2_quota_transfer() counted the destination's usage twice and refused a chgrp that fit. Fixed by bcachefs-tools "quota: a transfer's limit check counts the destination's usage twice". --- tests/fs/bcachefs/single_device.ktest | 64 +++++++++++++++++++++++++++ 1 file changed, 64 insertions(+) diff --git a/tests/fs/bcachefs/single_device.ktest b/tests/fs/bcachefs/single_device.ktest index 7d42cebb..f025d707 100755 --- a/tests/fs/bcachefs/single_device.ktest +++ b/tests/fs/bcachefs/single_device.ktest @@ -1203,6 +1203,70 @@ test_quota() bcachefs_test_end_checks ${ktest_scratch_dev[0]} } +# quotactl(Q_SETQUOTA) a block hard limit in KiB, or with no limit print +# quotactl(Q_GETQUOTA)'s " "; the VM has no quota tools. +# quotactl [] +quotactl() +{ + python3 - "$@" <<'PYEOF' +import ctypes, os, sys + +class dqblk(ctypes.Structure): + _fields_ = [(n, ctypes.c_uint64) for n in + ("bhard", "bsoft", "curspace", "ihard", "isoft", "curinodes", "btime", "itime")] \ + + [("valid", ctypes.c_uint32)] + +dev, qtype, qid = sys.argv[1], int(sys.argv[2]), int(sys.argv[3]) +Q_GETQUOTA, Q_SETQUOTA, QIF_BLIMITS = 0x800007, 0x800008, 1 +if len(sys.argv) > 4: + cmd, d = Q_SETQUOTA, dqblk(bhard=int(sys.argv[4]), valid=QIF_BLIMITS) +else: + cmd, d = Q_GETQUOTA, dqblk() +libc = ctypes.CDLL(None, use_errno=True) +if libc.quotactl((cmd << 8) | qtype, dev.encode(), qid, ctypes.byref(d)): + e = ctypes.get_errno() + raise OSError(e, os.strerror(e)) +if cmd == Q_GETQUOTA: + print(d.curspace, d.curinodes) +PYEOF +} + +# A chgrp into a group whose usage plus the file fits under its hard limit +# must succeed; one that would exceed it must fail. bch2_quota_transfer() +# counted the group's usage twice, so the first chgrp failed whenever +# 2 * usage + size was over the limit. Done by the file's owner, as a member +# of the group: root ignores hard limits. +test_quota_chown_counts_once() +{ + set_watchdog 60 + local dev=${ktest_scratch_dev[0]} + + run_quiet "" bcachefs format -f $dev + mount -t bcachefs -o grpquota $dev /mnt + + set_as_user + chmod 777 /mnt + local gid=$as_user_gid + local user="setpriv --reuid=$as_user_uid --regid=100 --groups=$gid" + quotactl $dev 1 $gid $((1024 * 1024)) + + $user dd if=/dev/zero of=/mnt/g600 bs=1M count=600 oflag=direct status=none + $user chgrp $gid /mnt/g600 + + $user dd if=/dev/zero of=/mnt/g100 bs=1M count=100 oflag=direct status=none + $user dd if=/dev/zero of=/mnt/g500 bs=1M count=500 oflag=direct status=none + + # 600M used + 100M fits under 1G; 700M used + 500M doesn't + if ! $user chgrp $gid /mnt/g100; then + echo "ERROR: chgrp of a 100M file to a group with 600M of a 1G limit was refused" + return 1 + fi + assert_fails_with "Disk quota exceeded" $user chgrp $gid /mnt/g500 + + umount /mnt + bcachefs_test_end_checks $dev +} + # test nfs exports: require-kernel-config NFSD require-kernel-config NFSD_V4 From 59f587b82da8c47c9ebb4952cf7fa8e288d07f98 Mon Sep 17 00:00:00 2001 From: Alan Derk <62039254+aderk@users.noreply.github.com> Date: Wed, 7 Oct 2026 20:59:10 +0000 Subject: [PATCH 05/11] bcachefs: quota_snapshot_inode_uncounted: snapshot inodes stay out of quota usage Deleting a file in a snapshot hit a BUG_ON in the quota code: snapshot inodes were charged at runtime but never counted. Fixed by bcachefs-tools "quota: snapshot inodes are charged at runtime but never counted". --- tests/fs/bcachefs/single_device.ktest | 61 +++++++++++++++++++++++++++ 1 file changed, 61 insertions(+) diff --git a/tests/fs/bcachefs/single_device.ktest b/tests/fs/bcachefs/single_device.ktest index f025d707..5c17be9c 100755 --- a/tests/fs/bcachefs/single_device.ktest +++ b/tests/fs/bcachefs/single_device.ktest @@ -1267,6 +1267,67 @@ test_quota_chown_counts_once() bcachefs_test_end_checks $dev } +# Quotas count master subvolume inodes only, but evict still charged snapshot +# inodes: deleting a snapshot's copy of a file whose master copy was chowned +# away took the old owner's usage negative and hit a BUG_ON. In the snapshot, +# a project id change and a rename whiteout must not move usage either, the +# inode must report its new project id, and the usage in memory must match +# what a remount reads back. +test_quota_snapshot_inode_uncounted() +{ + set_watchdog 120 + local dev=${ktest_scratch_dev[0]} + + run_quiet "" bcachefs format -f $dev + mount -t bcachefs -o usrquota,prjquota $dev /mnt + + set_as_user + chmod 777 /mnt + $as_user dd if=/dev/zero of=/mnt/f bs=1M count=64 oflag=direct status=none + bcachefs subvolume snapshot /mnt /mnt/snap + chown 1001 /mnt/f + + rm /mnt/snap/f + sync + + echo w > /mnt/snap/w + python3 -c ' +import ctypes, sys +libc = ctypes.CDLL(None, use_errno=True) +if libc.renameat2(-100, b"/mnt/snap/w", -100, b"/mnt/snap/w2", 4): + sys.exit("renameat2 RENAME_WHITEOUT: errno %d" % ctypes.get_errno())' + [[ -c /mnt/snap/w ]] + + chattr -p 42 /mnt/snap/w2 + lsattr -p /mnt/snap/w2 + local p=$(lsattr -p /mnt/snap/w2 | awk '{ print $1 }') + if [[ $p != 42 ]]; then + echo "ERROR: the snapshot inode reports project $p after it was set to 42" + return 1 + fi + + rm -f /mnt/snap/w /mnt/snap/w2 + sync + + local ids="0:0 0:65534 0:1001 2:0 2:42" before=() after=() i n=0 + for i in $ids; do before+=("$(quotactl $dev ${i%:*} ${i#*:})"); done + umount /mnt + mount -t bcachefs -o usrquota,prjquota $dev /mnt + for i in $ids; do after+=("$(quotactl $dev ${i%:*} ${i#*:})"); done + + for i in $ids; do + echo "quota $i: in memory ${before[n]}, after remount ${after[n]}" + if [[ ${before[n]} != "${after[n]}" ]]; then + echo "ERROR: quota usage in memory differs from what a remount reads" + return 1 + fi + n=$((n + 1)) + done + + umount /mnt + bcachefs_test_end_checks $dev +} + # test nfs exports: require-kernel-config NFSD require-kernel-config NFSD_V4 From 26e7b5e500ff350fa713c830ce938cec62ef81d2 Mon Sep 17 00:00:00 2001 From: Alan Derk <62039254+aderk@users.noreply.github.com> Date: Wed, 7 Oct 2026 20:59:10 +0000 Subject: [PATCH 06/11] bcachefs: sysfs_fail_next_write_ref: a rejected logged_op_fail_next write doesn't hang umount The EINVAL return leaked a c->writes ref, and umount waited for it forever. Fixed by bcachefs-tools "sysfs: a bad logged_op_fail_next write leaks c->writes, then umount hangs". --- tests/fs/bcachefs/single_device.ktest | 22 ++++++++++++++++++++++ 1 file changed, 22 insertions(+) diff --git a/tests/fs/bcachefs/single_device.ktest b/tests/fs/bcachefs/single_device.ktest index 5c17be9c..05681043 100755 --- a/tests/fs/bcachefs/single_device.ktest +++ b/tests/fs/bcachefs/single_device.ktest @@ -1639,6 +1639,28 @@ test_debugfs() bcachefs_test_end_checks ${ktest_scratch_dev[0]} } +# A write of an unknown name to logged_op_fail_next is EINVAL and must leave +# the filesystem unmountable as before. The error return skipped the +# c->writes put, so umount waited forever for the ref. +test_sysfs_fail_next_write_ref() +{ + set_watchdog 60 + local dev=${ktest_scratch_dev[0]} + + run_quiet "" bcachefs format -f $dev + mount -t bcachefs $dev /mnt + + assert_fails_with "Invalid argument" \ + bash -c 'echo nonsense > /sys/fs/bcachefs/*/internal/logged_op_fail_next' + + if ! timeout 30 umount /mnt; then + echo "ERROR: umount hung after a rejected logged_op_fail_next write" + return 1 + fi + + bcachefs_test_end_checks $dev +} + test_set_option() { set_watchdog 60 From 59f96f09a75f0dc9c9b61d6c07cc3429a3ab3e7d Mon Sep 17 00:00:00 2001 From: Alan Derk <62039254+aderk@users.noreply.github.com> Date: Wed, 7 Oct 2026 20:59:10 +0000 Subject: [PATCH 07/11] bcachefs: label_change_rescans_device: relabelling a device out of the target moves its data A label change queued only a pending scan, which never looked at the device's data. Fixed by bcachefs-tools "opts: a device label change never rescans the device's data". --- tests/fs/bcachefs/tier.ktest | 55 ++++++++++++++++++++++++++++++++++++ 1 file changed, 55 insertions(+) diff --git a/tests/fs/bcachefs/tier.ktest b/tests/fs/bcachefs/tier.ktest index bedb7f7a..c8fde4e8 100755 --- a/tests/fs/bcachefs/tier.ktest +++ b/tests/fs/bcachefs/tier.ktest @@ -1122,6 +1122,61 @@ test_reconcile_data_target_overfull() bcachefs_test_end_checks ${ktest_scratch_dev[0]} } +# user data bytes on a device: usage_dev_user +usage_dev_user() +{ + bcachefs fs usage -f devices /mnt | + awk -v hdr="(device $1):" ' + $2 == "(device" || $3 == "(device" { in_dev = index($0, hdr) > 0; next } + in_dev && $1 == "user:" { print $2; exit }' +} + +# Relabelling a device out of the background target must move its data to +# the target. Only a pending scan was queued, which never looks at it. +test_label_change_rescans_device() +{ + set_watchdog 300 + local devs=(${ktest_scratch_dev[0]} ${ktest_scratch_dev[2]} ${ktest_scratch_dev[3]}) + + run_quiet "" bcachefs format -f \ + --label=ssd.a ${ktest_scratch_dev[0]} \ + --label=hdd.b ${ktest_scratch_dev[2]} \ + --label=hdd.c ${ktest_scratch_dev[3]} \ + --foreground_target=ssd \ + --background_target=hdd + mount -t bcachefs "$(join_by : "${devs[@]}")" /mnt + + dd if=/dev/urandom of=/mnt/data bs=1M count=256 oflag=direct status=none + timeout 120 bcachefs reconcile wait /mnt + + local b_before=$(usage_dev_user 1) + echo "user data on b after reconcile: $b_before" + if ((b_before < 32 << 20)); then + echo "ERROR: the background target didn't give b its share: $b_before" + return 1 + fi + + echo ssd.b > /sys/fs/bcachefs/*/dev-1/label + + local i b_now + for i in $(seq 90); do + b_now=$(usage_dev_user 1) + ((b_now < 8 << 20)) && break + sleep 1 + done + echo "user data on b ${i}s after the relabel: $b_now" + + if ((b_now >= 8 << 20)); then + cat /sys/fs/bcachefs/*/reconcile_status + echo "ERROR: data stayed on the relabelled device" + return 1 + fi + + umount /mnt + bcachefs fsck -ny "${devs[@]}" + bcachefs_test_end_checks ${ktest_scratch_dev[0]} +} + # Disk reservations at replicas > 1 - GH #1221. Here for the devices, not the # tiering: raw sectors free say nothing about whether N copies fit on N # distinct devices, and this file already has four unequal ones. From 3d993c45071f048573af8307b613c1b50705c5da Mon Sep 17 00:00:00 2001 From: Alan Derk <62039254+aderk@users.noreply.github.com> Date: Wed, 7 Oct 2026 20:59:10 +0000 Subject: [PATCH 08/11] bcachefs: reconcile_btree_ptr_metadata_cookie: fsck accepts btree ptrs while a metadata scan is pending The btree ptr opts check read the cookie at the node key's inode field, so fsck reported every ptr. Fixed by bcachefs-tools "reconcile: a btree ptr's opt check reads the inode cookie". --- tests/fs/bcachefs/replication.ktest | 26 ++++++++++++++++++++++++++ 1 file changed, 26 insertions(+) diff --git a/tests/fs/bcachefs/replication.ktest b/tests/fs/bcachefs/replication.ktest index 696eae3b..97e5f788 100755 --- a/tests/fs/bcachefs/replication.ktest +++ b/tests/fs/bcachefs/replication.ktest @@ -1440,6 +1440,32 @@ test_rereplicate2() umount /mnt } +# With a metadata_replicas change pending (its scan not yet run) the btree +# ptrs are allowed to mismatch: the metadata cookie says so. The check read +# the cookie at the node key's inode field instead and reported them. +test_reconcile_btree_ptr_metadata_cookie() +{ + set_watchdog 120 + local devs=("${ktest_scratch_dev[0]}" "${ktest_scratch_dev[1]}") + + run_quiet "" bcachefs format -f --metadata_replicas=1 "${devs[@]}" + mount -t bcachefs -o reconcile_enabled=0 "$(join_by : "${devs[@]}")" /mnt + for i in $(seq 500); do echo $i > /mnt/f$i; done + sync + + echo 2 > /sys/fs/bcachefs/*/options/metadata_replicas + umount /mnt + + local out rc=0 + out=$(bcachefs fsck -n "${devs[@]}" 2>&1) || rc=$? + echo "fsck -n exited $rc" + if grep "incorrect/missing reconcile opts" <<<"$out" >/dev/null; then + grep -A6 "incorrect/missing reconcile opts" <<<"$out" | head -40 + echo "ERROR: btree ptrs reported while the metadata scan is pending" + return 1 + fi +} + # Bumping data_replicas on a single file (set-file-option, not the fs-wide # sysfs knob) must re-replicate that file's existing extents: it queues an # inum reconcile scan that re-derives need_rb=data_replicas per extent and From 07de162770c9c7dd480465178841acf41d61f6c0 Mon Sep 17 00:00:00 2001 From: Alan Derk <62039254+aderk@users.noreply.github.com> Date: Wed, 7 Oct 2026 20:59:10 +0000 Subject: [PATCH 09/11] bcachefs: reflink_may_update_disk_res: setting may_update_opts takes a disk reservation The reflink_p update committed without a disk reservation and warned under BCACHEFS_DEBUG. Fixed by bcachefs-tools "ioctl: set_reflink_p_may_update_opts commits without a reservation". --- tests/fs/bcachefs/tier.ktest | 39 ++++++++++++++++++++++++++++++++++++ 1 file changed, 39 insertions(+) diff --git a/tests/fs/bcachefs/tier.ktest b/tests/fs/bcachefs/tier.ktest index c8fde4e8..49f1176a 100755 --- a/tests/fs/bcachefs/tier.ktest +++ b/tests/fs/bcachefs/tier.ktest @@ -811,6 +811,45 @@ test_reflink_option_propagate_replicas() bcachefs_test_end_checks ${ktest_scratch_dev[2]} } +# The may_update_opts ioctl commits a reflink_p update on the extents btree; +# with CONFIG_BCACHEFS_DEBUG the commit path warns when that comes without a +# disk reservation. +test_reflink_may_update_disk_res() +{ + set_watchdog 60 + local dev=${ktest_scratch_dev[0]} + + run_quiet "" bcachefs format -f $dev + mount -t bcachefs $dev /mnt + + # A clone made by a user who doesn't own the source gets reflink_p keys + # without may_update_opts: the ioctl then has keys to update. + set_as_user + chmod 777 /mnt + dd if=/dev/urandom of=/mnt/orig bs=1M count=8 oflag=direct status=none + chmod 644 /mnt/orig + $as_user cp --reflink=always /mnt/orig /mnt/clone + sync + + local ino=$(stat -c %i /mnt/clone) + if ! bcachefs list -b extents /mnt | + awk -v p="reflink_p $ino:" 'index($0, p) && !/may_update_opts/ { n++ } END { exit !n }'; then + echo "ERROR: the clone has no reflink_p key without may_update_opts" + return 1 + fi + + dmesg_mark reflink_may_update + bcachefs reflink-option-propagate --set-may-update /mnt/clone + + if dmesg_since reflink_may_update | grep "without a disk reservation" >/dev/null; then + echo "ERROR: set_reflink_p_may_update_opts committed without a disk reservation" + return 1 + fi + + umount /mnt + bcachefs_test_end_checks $dev +} + # When an extent is made indirect, bch2_make_extent_indirect() must snapshot # the source inode's IO options onto the new reflink_v: a non-degraded data # extent carries no bch_extent_reconcile entry (the reconciler re-derives the From c3a1702b429428342ad4ab4fe73c775c98107e72 Mon Sep 17 00:00:00 2001 From: Alan Derk <62039254+aderk@users.noreply.github.com> Date: Wed, 7 Oct 2026 20:59:10 +0000 Subject: [PATCH 10/11] bcachefs: reconcile_pending_cookie_kept: an interrupted pass keeps the pending scan cookie The pending scan cookie was deleted in the scan phase, so stopping the pass before its pending phases lost the work. Fixed by bcachefs-tools "reconcile: the pending scan cookie is deleted before the pending phases run". --- tests/fs/bcachefs/reconcile-restart.ktest | 75 +++++++++++++++++++++++ 1 file changed, 75 insertions(+) diff --git a/tests/fs/bcachefs/reconcile-restart.ktest b/tests/fs/bcachefs/reconcile-restart.ktest index 0d87b22d..687291f9 100755 --- a/tests/fs/bcachefs/reconcile-restart.ktest +++ b/tests/fs/bcachefs/reconcile-restart.ktest @@ -32,6 +32,8 @@ . $(dirname $(readlink -e ${BASH_SOURCE[0]}))/bcachefs-test-libs.sh +config-scratch-devs 4G +config-scratch-devs 4G config-scratch-devs 4G # "The thread is running" and "we couldn't tell" are different answers, and the @@ -206,6 +208,79 @@ test_reconcile_survives_disable_during_scan() bcachefs_test_end_checks "$dev" } +# pending reconcile work in bytes, from fs usage +reconcile_pending_bytes() +{ + bcachefs fs usage -f rebalance_work /mnt | + awk '$1 == "pending:" { print $2; exit }' +} + +# A device add queues a pending scan so the extents parked for lack of +# devices are retried. If the pass is interrupted before the pending phase +# the cookie must still be there for the next pass; it was deleted in the +# scan phase, the first of the pass, so the work was lost. +test_reconcile_pending_cookie_kept() +{ + set_watchdog 600 + local devs=("${ktest_scratch_dev[0]}" "${ktest_scratch_dev[1]}") + + run_quiet "" bcachefs format -f --replicas=2 "${devs[@]}" + mount -t bcachefs "$(join_by : "${devs[@]}")" /mnt + dd if=/dev/urandom of=/mnt/data bs=1M count=1024 oflag=direct status=none + + # three replicas on two devices: every extent parks on the pending list + echo 3 > /sys/fs/bcachefs/*/options/data_replicas + timeout 300 bcachefs reconcile wait /mnt + local pending=$(reconcile_pending_bytes) + echo "pending after data_replicas=3: $pending" + if ((pending == 0)); then + echo "ERROR: no extents parked on the pending list" + return 1 + fi + + # Queue the device add's pending scan with reconcile stopped, so the pass + # starts when we start watching it. + echo 0 > /sys/fs/bcachefs/*/options/reconcile_enabled + bcachefs device add -f /mnt ${ktest_scratch_dev[2]} + echo 1 > /sys/fs/bcachefs/*/options/reconcile_enabled + + # The scan phase comes first; stop the pass once it's past it and still + # working, before the pending phases complete. + local i status + for i in $(seq 1200); do + status=$(cat /sys/fs/bcachefs/*/reconcile_status) + grep "^processing" <<<"$status" >/dev/null && break + if grep "^waiting" <<<"$status" >/dev/null && ((i > 40)); then + echo "$status" + echo "ERROR: the pass finished before the test could interrupt it" + return 1 + fi + sleep 0.05 + done + echo "$status" + echo 0 > /sys/fs/bcachefs/*/options/reconcile_enabled + sleep 1 + + local cookies=$(bcachefs list -b reconcile_scan /mnt) + echo "$cookies" + if ! grep "cookie 0:2:" <<<"$cookies" >/dev/null; then + echo "ERROR: the pending scan cookie was deleted before the pending phase ran" + return 1 + fi + + echo 1 > /sys/fs/bcachefs/*/options/reconcile_enabled + timeout 300 bcachefs reconcile wait /mnt + pending=$(reconcile_pending_bytes) + echo "pending after the retry: $pending" + if ((pending != 0)); then + echo "ERROR: pending work wasn't retried after the interrupted pass" + return 1 + fi + + umount /mnt + bcachefs_test_end_checks ${devs[0]} +} + # Suspend and hibernate failed with "Freezing of tasks failed ... bch-reconcile" # (bcachefs #700, tools #972). The freezer stops freezable workqueues before it # freezes kernel threads, and move completions run on freezable workqueues - so From 6a5c3d43339dcc1140d174d724c8637f58c81bd4 Mon Sep 17 00:00:00 2001 From: Alan Derk <62039254+aderk@users.noreply.github.com> Date: Wed, 7 Oct 2026 20:59:10 +0000 Subject: [PATCH 11/11] bcachefs: btree_scrub_stale_replica: scrub rewrites a btree node replica with the wrong seq Btree node scrub checked only checksums and called a stale replica good. Fixed by bcachefs-tools "btree node scrub: a stale or torn replica passes". --- tests/fs/bcachefs/replication.ktest | 78 +++++++++++++++++++++++++++++ 1 file changed, 78 insertions(+) diff --git a/tests/fs/bcachefs/replication.ktest b/tests/fs/bcachefs/replication.ktest index 97e5f788..cfc750af 100755 --- a/tests/fs/bcachefs/replication.ktest +++ b/tests/fs/bcachefs/replication.ktest @@ -3809,5 +3809,83 @@ test_scrub_corrupt_ec_unreferenced() bcachefs_test_end_checks ${ktest_scratch_dev[$vdev]} } +# The leaf nodes of a btree that have a replica on , one line each: +# btree_leaf_lines ... +btree_leaf_lines() +{ + local btree=$1 dev_idx=$2 + shift 2 + # a key's value continues on indented lines: join them + bcachefs list -m nodes -b $btree "$@" | + awk '{ gsub(/[ \t]+/, " ") } + /^u64s/ { if (line) print line; line = $0; next } + { line = line " " $0 } + END { if (line) print line }' | + grep "btree_ptr" | grep -E "ptr: ([^ ]+ )?$dev_idx:[0-9]+:[0-9]+" +} + +# A replica that holds another node's data (an older node left by a lost +# write) has valid checksums and a wrong seq. Scrub must rewrite the node; +# it compared nothing but the checksums and called the replica good. +test_btree_scrub_stale_replica() +{ + ktest_expect_device_errors=1 + set_watchdog 300 + local devs=("${ktest_scratch_dev[0]}" "${ktest_scratch_dev[1]}") + + run_quiet "" bcachefs format -f \ + --metadata_replicas=2 \ + --bucket_size=${BUCKET_SIZE_KB}k \ + --btree_node_size=64k \ + "${devs[@]}" + mount -t bcachefs "$(join_by : "${devs[@]}")" /mnt + + for i in $(seq 2000); do echo $i > /mnt/f$i; done + sync + umount /mnt + + local before=$(btree_leaf_lines inodes 1 "${devs[@]}") + local nr=$(echo "$before" | wc -l) + echo "inodes btree: $nr nodes with a dev 1 replica" + if ((nr < 2)); then + echo "ERROR: need two inodes leaves, got $nr" + return 1 + fi + + local n_line=$(echo "$before" | sed -n 1p) + local m_line=$(echo "$before" | sed -n 2p) + local n_sector=$(echo "$n_line" | bcachefs_ptr_sectors $BUCKET_SIZE_SECTORS 1) + local m_sector=$(echo "$m_line" | bcachefs_ptr_sectors $BUCKET_SIZE_SECTORS 1) + echo "node N: $n_line" + echo "node M: $m_line" + + # Cache every inodes leaf first, so the kernel never reads the stale + # replica itself: only scrub reads it, from disk. noatime, so nothing + # but a rewrite changes N. + mount -t bcachefs -o noatime "$(join_by : "${devs[@]}")" /mnt + ls -l /mnt > /dev/null + + # M's replica over N's + dd if=${devs[1]} of=${devs[1]} bs=512 skip=$m_sector seek=$n_sector \ + count=$((64 * 2)) iflag=direct oflag=direct conv=notrunc status=none + + local rc=0 + bcachefs scrub /mnt || rc=$? + echo "scrub exit code: $rc" + umount /mnt + + local n_pos=$(echo "$n_line" | awk '{ for (i = 1; i < NF; i++) if ($i == "btree_ptr_v2") print $(i + 1) }') + local n_after=$(btree_leaf_lines inodes 1 "${devs[@]}" | + awk -v pos="$n_pos" '{ for (i = 1; i < NF; i++) if ($i == "btree_ptr_v2" && $(i + 1) == pos) print }') + local n_sector_after=$(echo "$n_after" | bcachefs_ptr_sectors $BUCKET_SIZE_SECTORS 1) + echo "N's dev 1 replica: sector $n_sector, after scrub ${n_sector_after:-none}" + if [[ $n_sector_after == "$n_sector" ]]; then + echo "ERROR: scrub left the stale replica of node N in place" + return 1 + fi + + bcachefs fsck -ny "${devs[@]}" +} + main "$@"