diff --git a/tests/fs/bcachefs/reconcile-restart.ktest b/tests/fs/bcachefs/reconcile-restart.ktest index 0d87b22d..687291f9 100755 --- a/tests/fs/bcachefs/reconcile-restart.ktest +++ b/tests/fs/bcachefs/reconcile-restart.ktest @@ -32,6 +32,8 @@ . $(dirname $(readlink -e ${BASH_SOURCE[0]}))/bcachefs-test-libs.sh +config-scratch-devs 4G +config-scratch-devs 4G config-scratch-devs 4G # "The thread is running" and "we couldn't tell" are different answers, and the @@ -206,6 +208,79 @@ test_reconcile_survives_disable_during_scan() bcachefs_test_end_checks "$dev" } +# pending reconcile work in bytes, from fs usage +reconcile_pending_bytes() +{ + bcachefs fs usage -f rebalance_work /mnt | + awk '$1 == "pending:" { print $2; exit }' +} + +# A device add queues a pending scan so the extents parked for lack of +# devices are retried. If the pass is interrupted before the pending phase +# the cookie must still be there for the next pass; it was deleted in the +# scan phase, the first of the pass, so the work was lost. +test_reconcile_pending_cookie_kept() +{ + set_watchdog 600 + local devs=("${ktest_scratch_dev[0]}" "${ktest_scratch_dev[1]}") + + run_quiet "" bcachefs format -f --replicas=2 "${devs[@]}" + mount -t bcachefs "$(join_by : "${devs[@]}")" /mnt + dd if=/dev/urandom of=/mnt/data bs=1M count=1024 oflag=direct status=none + + # three replicas on two devices: every extent parks on the pending list + echo 3 > /sys/fs/bcachefs/*/options/data_replicas + timeout 300 bcachefs reconcile wait /mnt + local pending=$(reconcile_pending_bytes) + echo "pending after data_replicas=3: $pending" + if ((pending == 0)); then + echo "ERROR: no extents parked on the pending list" + return 1 + fi + + # Queue the device add's pending scan with reconcile stopped, so the pass + # starts when we start watching it. + echo 0 > /sys/fs/bcachefs/*/options/reconcile_enabled + bcachefs device add -f /mnt ${ktest_scratch_dev[2]} + echo 1 > /sys/fs/bcachefs/*/options/reconcile_enabled + + # The scan phase comes first; stop the pass once it's past it and still + # working, before the pending phases complete. + local i status + for i in $(seq 1200); do + status=$(cat /sys/fs/bcachefs/*/reconcile_status) + grep "^processing" <<<"$status" >/dev/null && break + if grep "^waiting" <<<"$status" >/dev/null && ((i > 40)); then + echo "$status" + echo "ERROR: the pass finished before the test could interrupt it" + return 1 + fi + sleep 0.05 + done + echo "$status" + echo 0 > /sys/fs/bcachefs/*/options/reconcile_enabled + sleep 1 + + local cookies=$(bcachefs list -b reconcile_scan /mnt) + echo "$cookies" + if ! grep "cookie 0:2:" <<<"$cookies" >/dev/null; then + echo "ERROR: the pending scan cookie was deleted before the pending phase ran" + return 1 + fi + + echo 1 > /sys/fs/bcachefs/*/options/reconcile_enabled + timeout 300 bcachefs reconcile wait /mnt + pending=$(reconcile_pending_bytes) + echo "pending after the retry: $pending" + if ((pending != 0)); then + echo "ERROR: pending work wasn't retried after the interrupted pass" + return 1 + fi + + umount /mnt + bcachefs_test_end_checks ${devs[0]} +} + # Suspend and hibernate failed with "Freezing of tasks failed ... bch-reconcile" # (bcachefs #700, tools #972). The freezer stops freezable workqueues before it # freezes kernel threads, and move completions run on freezable workqueues - so diff --git a/tests/fs/bcachefs/replication.ktest b/tests/fs/bcachefs/replication.ktest index 696eae3b..cfc750af 100755 --- a/tests/fs/bcachefs/replication.ktest +++ b/tests/fs/bcachefs/replication.ktest @@ -1440,6 +1440,32 @@ test_rereplicate2() umount /mnt } +# With a metadata_replicas change pending (its scan not yet run) the btree +# ptrs are allowed to mismatch: the metadata cookie says so. The check read +# the cookie at the node key's inode field instead and reported them. +test_reconcile_btree_ptr_metadata_cookie() +{ + set_watchdog 120 + local devs=("${ktest_scratch_dev[0]}" "${ktest_scratch_dev[1]}") + + run_quiet "" bcachefs format -f --metadata_replicas=1 "${devs[@]}" + mount -t bcachefs -o reconcile_enabled=0 "$(join_by : "${devs[@]}")" /mnt + for i in $(seq 500); do echo $i > /mnt/f$i; done + sync + + echo 2 > /sys/fs/bcachefs/*/options/metadata_replicas + umount /mnt + + local out rc=0 + out=$(bcachefs fsck -n "${devs[@]}" 2>&1) || rc=$? + echo "fsck -n exited $rc" + if grep "incorrect/missing reconcile opts" <<<"$out" >/dev/null; then + grep -A6 "incorrect/missing reconcile opts" <<<"$out" | head -40 + echo "ERROR: btree ptrs reported while the metadata scan is pending" + return 1 + fi +} + # Bumping data_replicas on a single file (set-file-option, not the fs-wide # sysfs knob) must re-replicate that file's existing extents: it queues an # inum reconcile scan that re-derives need_rb=data_replicas per extent and @@ -3783,5 +3809,83 @@ test_scrub_corrupt_ec_unreferenced() bcachefs_test_end_checks ${ktest_scratch_dev[$vdev]} } +# The leaf nodes of a btree that have a replica on , one line each: +# btree_leaf_lines ... +btree_leaf_lines() +{ + local btree=$1 dev_idx=$2 + shift 2 + # a key's value continues on indented lines: join them + bcachefs list -m nodes -b $btree "$@" | + awk '{ gsub(/[ \t]+/, " ") } + /^u64s/ { if (line) print line; line = $0; next } + { line = line " " $0 } + END { if (line) print line }' | + grep "btree_ptr" | grep -E "ptr: ([^ ]+ )?$dev_idx:[0-9]+:[0-9]+" +} + +# A replica that holds another node's data (an older node left by a lost +# write) has valid checksums and a wrong seq. Scrub must rewrite the node; +# it compared nothing but the checksums and called the replica good. +test_btree_scrub_stale_replica() +{ + ktest_expect_device_errors=1 + set_watchdog 300 + local devs=("${ktest_scratch_dev[0]}" "${ktest_scratch_dev[1]}") + + run_quiet "" bcachefs format -f \ + --metadata_replicas=2 \ + --bucket_size=${BUCKET_SIZE_KB}k \ + --btree_node_size=64k \ + "${devs[@]}" + mount -t bcachefs "$(join_by : "${devs[@]}")" /mnt + + for i in $(seq 2000); do echo $i > /mnt/f$i; done + sync + umount /mnt + + local before=$(btree_leaf_lines inodes 1 "${devs[@]}") + local nr=$(echo "$before" | wc -l) + echo "inodes btree: $nr nodes with a dev 1 replica" + if ((nr < 2)); then + echo "ERROR: need two inodes leaves, got $nr" + return 1 + fi + + local n_line=$(echo "$before" | sed -n 1p) + local m_line=$(echo "$before" | sed -n 2p) + local n_sector=$(echo "$n_line" | bcachefs_ptr_sectors $BUCKET_SIZE_SECTORS 1) + local m_sector=$(echo "$m_line" | bcachefs_ptr_sectors $BUCKET_SIZE_SECTORS 1) + echo "node N: $n_line" + echo "node M: $m_line" + + # Cache every inodes leaf first, so the kernel never reads the stale + # replica itself: only scrub reads it, from disk. noatime, so nothing + # but a rewrite changes N. + mount -t bcachefs -o noatime "$(join_by : "${devs[@]}")" /mnt + ls -l /mnt > /dev/null + + # M's replica over N's + dd if=${devs[1]} of=${devs[1]} bs=512 skip=$m_sector seek=$n_sector \ + count=$((64 * 2)) iflag=direct oflag=direct conv=notrunc status=none + + local rc=0 + bcachefs scrub /mnt || rc=$? + echo "scrub exit code: $rc" + umount /mnt + + local n_pos=$(echo "$n_line" | awk '{ for (i = 1; i < NF; i++) if ($i == "btree_ptr_v2") print $(i + 1) }') + local n_after=$(btree_leaf_lines inodes 1 "${devs[@]}" | + awk -v pos="$n_pos" '{ for (i = 1; i < NF; i++) if ($i == "btree_ptr_v2" && $(i + 1) == pos) print }') + local n_sector_after=$(echo "$n_after" | bcachefs_ptr_sectors $BUCKET_SIZE_SECTORS 1) + echo "N's dev 1 replica: sector $n_sector, after scrub ${n_sector_after:-none}" + if [[ $n_sector_after == "$n_sector" ]]; then + echo "ERROR: scrub left the stale replica of node N in place" + return 1 + fi + + bcachefs fsck -ny "${devs[@]}" +} + main "$@" diff --git a/tests/fs/bcachefs/single_device.ktest b/tests/fs/bcachefs/single_device.ktest index 2339f5b6..05681043 100755 --- a/tests/fs/bcachefs/single_device.ktest +++ b/tests/fs/bcachefs/single_device.ktest @@ -205,6 +205,52 @@ test_extent_merge2() true } +# A buffered write into a fallocated range whose folios are cached must not +# add the sectors to i_blocks a second time: fallocate already accounted them, +# and bch2_mark_pagecache_reserved() is what flips the cached folios' sectors +# to reserved. It advanced *start before computing the range, so it marked +# nothing. +test_fallocate_pagecache_reserved() +{ + set_watchdog 60 + local dev=${ktest_scratch_dev[0]} + + run_quiet "" bcachefs format -f $dev + mount -t bcachefs $dev /mnt + + # cache zero folios over the hole, then fallocate under them + truncate -s 32M /mnt/f + cat /mnt/f > /dev/null + fallocate -l 32M /mnt/f + + local blocks=$(stat -c %b /mnt/f) + echo "st_blocks after fallocate: $blocks" + if [[ $blocks != 65536 ]]; then + echo "ERROR: expected 65536 blocks after fallocate, got $blocks" + return 1 + fi + + dd if=/dev/urandom of=/mnt/f bs=1M count=32 conv=notrunc status=none + + blocks=$(stat -c %b /mnt/f) + echo "st_blocks after a buffered write into the fallocated range: $blocks" + if [[ $blocks != 65536 ]]; then + echo "ERROR: the write was counted again: $blocks blocks" + return 1 + fi + + sync + blocks=$(stat -c %b /mnt/f) + echo "st_blocks after sync: $blocks" + if [[ $blocks != 65536 ]]; then + echo "ERROR: expected 65536 blocks after sync, got $blocks" + return 1 + fi + + umount /mnt + bcachefs_test_end_checks $dev +} + test_extent_repair_overlapping() { set_watchdog 30 @@ -1157,6 +1203,131 @@ test_quota() bcachefs_test_end_checks ${ktest_scratch_dev[0]} } +# quotactl(Q_SETQUOTA) a block hard limit in KiB, or with no limit print +# quotactl(Q_GETQUOTA)'s " "; the VM has no quota tools. +# quotactl [] +quotactl() +{ + python3 - "$@" <<'PYEOF' +import ctypes, os, sys + +class dqblk(ctypes.Structure): + _fields_ = [(n, ctypes.c_uint64) for n in + ("bhard", "bsoft", "curspace", "ihard", "isoft", "curinodes", "btime", "itime")] \ + + [("valid", ctypes.c_uint32)] + +dev, qtype, qid = sys.argv[1], int(sys.argv[2]), int(sys.argv[3]) +Q_GETQUOTA, Q_SETQUOTA, QIF_BLIMITS = 0x800007, 0x800008, 1 +if len(sys.argv) > 4: + cmd, d = Q_SETQUOTA, dqblk(bhard=int(sys.argv[4]), valid=QIF_BLIMITS) +else: + cmd, d = Q_GETQUOTA, dqblk() +libc = ctypes.CDLL(None, use_errno=True) +if libc.quotactl((cmd << 8) | qtype, dev.encode(), qid, ctypes.byref(d)): + e = ctypes.get_errno() + raise OSError(e, os.strerror(e)) +if cmd == Q_GETQUOTA: + print(d.curspace, d.curinodes) +PYEOF +} + +# A chgrp into a group whose usage plus the file fits under its hard limit +# must succeed; one that would exceed it must fail. bch2_quota_transfer() +# counted the group's usage twice, so the first chgrp failed whenever +# 2 * usage + size was over the limit. Done by the file's owner, as a member +# of the group: root ignores hard limits. +test_quota_chown_counts_once() +{ + set_watchdog 60 + local dev=${ktest_scratch_dev[0]} + + run_quiet "" bcachefs format -f $dev + mount -t bcachefs -o grpquota $dev /mnt + + set_as_user + chmod 777 /mnt + local gid=$as_user_gid + local user="setpriv --reuid=$as_user_uid --regid=100 --groups=$gid" + quotactl $dev 1 $gid $((1024 * 1024)) + + $user dd if=/dev/zero of=/mnt/g600 bs=1M count=600 oflag=direct status=none + $user chgrp $gid /mnt/g600 + + $user dd if=/dev/zero of=/mnt/g100 bs=1M count=100 oflag=direct status=none + $user dd if=/dev/zero of=/mnt/g500 bs=1M count=500 oflag=direct status=none + + # 600M used + 100M fits under 1G; 700M used + 500M doesn't + if ! $user chgrp $gid /mnt/g100; then + echo "ERROR: chgrp of a 100M file to a group with 600M of a 1G limit was refused" + return 1 + fi + assert_fails_with "Disk quota exceeded" $user chgrp $gid /mnt/g500 + + umount /mnt + bcachefs_test_end_checks $dev +} + +# Quotas count master subvolume inodes only, but evict still charged snapshot +# inodes: deleting a snapshot's copy of a file whose master copy was chowned +# away took the old owner's usage negative and hit a BUG_ON. In the snapshot, +# a project id change and a rename whiteout must not move usage either, the +# inode must report its new project id, and the usage in memory must match +# what a remount reads back. +test_quota_snapshot_inode_uncounted() +{ + set_watchdog 120 + local dev=${ktest_scratch_dev[0]} + + run_quiet "" bcachefs format -f $dev + mount -t bcachefs -o usrquota,prjquota $dev /mnt + + set_as_user + chmod 777 /mnt + $as_user dd if=/dev/zero of=/mnt/f bs=1M count=64 oflag=direct status=none + bcachefs subvolume snapshot /mnt /mnt/snap + chown 1001 /mnt/f + + rm /mnt/snap/f + sync + + echo w > /mnt/snap/w + python3 -c ' +import ctypes, sys +libc = ctypes.CDLL(None, use_errno=True) +if libc.renameat2(-100, b"/mnt/snap/w", -100, b"/mnt/snap/w2", 4): + sys.exit("renameat2 RENAME_WHITEOUT: errno %d" % ctypes.get_errno())' + [[ -c /mnt/snap/w ]] + + chattr -p 42 /mnt/snap/w2 + lsattr -p /mnt/snap/w2 + local p=$(lsattr -p /mnt/snap/w2 | awk '{ print $1 }') + if [[ $p != 42 ]]; then + echo "ERROR: the snapshot inode reports project $p after it was set to 42" + return 1 + fi + + rm -f /mnt/snap/w /mnt/snap/w2 + sync + + local ids="0:0 0:65534 0:1001 2:0 2:42" before=() after=() i n=0 + for i in $ids; do before+=("$(quotactl $dev ${i%:*} ${i#*:})"); done + umount /mnt + mount -t bcachefs -o usrquota,prjquota $dev /mnt + for i in $ids; do after+=("$(quotactl $dev ${i%:*} ${i#*:})"); done + + for i in $ids; do + echo "quota $i: in memory ${before[n]}, after remount ${after[n]}" + if [[ ${before[n]} != "${after[n]}" ]]; then + echo "ERROR: quota usage in memory differs from what a remount reads" + return 1 + fi + n=$((n + 1)) + done + + umount /mnt + bcachefs_test_end_checks $dev +} + # test nfs exports: require-kernel-config NFSD require-kernel-config NFSD_V4 @@ -1468,6 +1639,28 @@ test_debugfs() bcachefs_test_end_checks ${ktest_scratch_dev[0]} } +# A write of an unknown name to logged_op_fail_next is EINVAL and must leave +# the filesystem unmountable as before. The error return skipped the +# c->writes put, so umount waited forever for the ref. +test_sysfs_fail_next_write_ref() +{ + set_watchdog 60 + local dev=${ktest_scratch_dev[0]} + + run_quiet "" bcachefs format -f $dev + mount -t bcachefs $dev /mnt + + assert_fails_with "Invalid argument" \ + bash -c 'echo nonsense > /sys/fs/bcachefs/*/internal/logged_op_fail_next' + + if ! timeout 30 umount /mnt; then + echo "ERROR: umount hung after a rejected logged_op_fail_next write" + return 1 + fi + + bcachefs_test_end_checks $dev +} + test_set_option() { set_watchdog 60 @@ -3062,6 +3255,57 @@ test_recovery_pass_offline() fi } +# A subvolume unlink asks for check_subvols to run unratelimited next time; +# the setter wrote false, so the flag never appeared in the superblock. +test_passes_no_ratelimit_set() +{ + local dev=${ktest_scratch_dev[0]} + + set_watchdog 60 + + run_quiet "" bcachefs format -f $dev + + # the flag is only shown for a pass with a last_run: run them once + mount -t bcachefs -o fsck $dev /mnt + umount /mnt + + mount -t bcachefs $dev /mnt + bcachefs subvolume create /mnt/sub + bcachefs subvolume delete /mnt/sub + umount /mnt + + local passes=$(bcachefs show-super -f recovery_passes $dev) + echo "$passes" + if ! grep "no ratelimit" <<<"$passes" >/dev/null; then + echo "ERROR: no recovery pass is marked (no ratelimit) after a subvolume unlink" + return 1 + fi +} + +# recovery_passes_exclude is a bitfield; it was masked with a pass number, +# which cleared the scan_for_btree_nodes bit from the exclusion. +test_passes_exclude_bitmask() +{ + local dev=${ktest_scratch_dev[0]} + + set_watchdog 60 + + run_quiet "" bcachefs format -f $dev + mount -t bcachefs $dev /mnt + umount /mnt + + dmesg_mark passes_exclude + mount -t bcachefs -o recovery_passes=scan_for_btree_nodes,recovery_passes_exclude=scan_for_btree_nodes $dev /mnt + umount /mnt + + if dmesg_since passes_exclude | grep "scan_for_btree_nodes\.\.\." >/dev/null; then + echo "ERROR: scan_for_btree_nodes ran although recovery_passes_exclude named it" + return 1 + fi + + bcachefs_test_end_checks $dev +} + # The passphrase lifecycle on an existing filesystem: adding one, changing # one, and taking one off. # diff --git a/tests/fs/bcachefs/tier.ktest b/tests/fs/bcachefs/tier.ktest index bedb7f7a..49f1176a 100755 --- a/tests/fs/bcachefs/tier.ktest +++ b/tests/fs/bcachefs/tier.ktest @@ -811,6 +811,45 @@ test_reflink_option_propagate_replicas() bcachefs_test_end_checks ${ktest_scratch_dev[2]} } +# The may_update_opts ioctl commits a reflink_p update on the extents btree; +# with CONFIG_BCACHEFS_DEBUG the commit path warns when that comes without a +# disk reservation. +test_reflink_may_update_disk_res() +{ + set_watchdog 60 + local dev=${ktest_scratch_dev[0]} + + run_quiet "" bcachefs format -f $dev + mount -t bcachefs $dev /mnt + + # A clone made by a user who doesn't own the source gets reflink_p keys + # without may_update_opts: the ioctl then has keys to update. + set_as_user + chmod 777 /mnt + dd if=/dev/urandom of=/mnt/orig bs=1M count=8 oflag=direct status=none + chmod 644 /mnt/orig + $as_user cp --reflink=always /mnt/orig /mnt/clone + sync + + local ino=$(stat -c %i /mnt/clone) + if ! bcachefs list -b extents /mnt | + awk -v p="reflink_p $ino:" 'index($0, p) && !/may_update_opts/ { n++ } END { exit !n }'; then + echo "ERROR: the clone has no reflink_p key without may_update_opts" + return 1 + fi + + dmesg_mark reflink_may_update + bcachefs reflink-option-propagate --set-may-update /mnt/clone + + if dmesg_since reflink_may_update | grep "without a disk reservation" >/dev/null; then + echo "ERROR: set_reflink_p_may_update_opts committed without a disk reservation" + return 1 + fi + + umount /mnt + bcachefs_test_end_checks $dev +} + # When an extent is made indirect, bch2_make_extent_indirect() must snapshot # the source inode's IO options onto the new reflink_v: a non-degraded data # extent carries no bch_extent_reconcile entry (the reconciler re-derives the @@ -1122,6 +1161,61 @@ test_reconcile_data_target_overfull() bcachefs_test_end_checks ${ktest_scratch_dev[0]} } +# user data bytes on a device: usage_dev_user +usage_dev_user() +{ + bcachefs fs usage -f devices /mnt | + awk -v hdr="(device $1):" ' + $2 == "(device" || $3 == "(device" { in_dev = index($0, hdr) > 0; next } + in_dev && $1 == "user:" { print $2; exit }' +} + +# Relabelling a device out of the background target must move its data to +# the target. Only a pending scan was queued, which never looks at it. +test_label_change_rescans_device() +{ + set_watchdog 300 + local devs=(${ktest_scratch_dev[0]} ${ktest_scratch_dev[2]} ${ktest_scratch_dev[3]}) + + run_quiet "" bcachefs format -f \ + --label=ssd.a ${ktest_scratch_dev[0]} \ + --label=hdd.b ${ktest_scratch_dev[2]} \ + --label=hdd.c ${ktest_scratch_dev[3]} \ + --foreground_target=ssd \ + --background_target=hdd + mount -t bcachefs "$(join_by : "${devs[@]}")" /mnt + + dd if=/dev/urandom of=/mnt/data bs=1M count=256 oflag=direct status=none + timeout 120 bcachefs reconcile wait /mnt + + local b_before=$(usage_dev_user 1) + echo "user data on b after reconcile: $b_before" + if ((b_before < 32 << 20)); then + echo "ERROR: the background target didn't give b its share: $b_before" + return 1 + fi + + echo ssd.b > /sys/fs/bcachefs/*/dev-1/label + + local i b_now + for i in $(seq 90); do + b_now=$(usage_dev_user 1) + ((b_now < 8 << 20)) && break + sleep 1 + done + echo "user data on b ${i}s after the relabel: $b_now" + + if ((b_now >= 8 << 20)); then + cat /sys/fs/bcachefs/*/reconcile_status + echo "ERROR: data stayed on the relabelled device" + return 1 + fi + + umount /mnt + bcachefs fsck -ny "${devs[@]}" + bcachefs_test_end_checks ${ktest_scratch_dev[0]} +} + # Disk reservations at replicas > 1 - GH #1221. Here for the devices, not the # tiering: raw sectors free say nothing about whether N copies fit on N # distinct devices, and this file already has four unequal ones.