A mount option is an awkward fit for write-hole protection: it occupies
a mount-option bit, and protection silently lapses whenever the option
is forgotten -- an fstab edit, a rescue mount, a recovery boot -- which
is exactly when a degraded raid56 is most likely to be written.
Control the policy with a "stripe_alloc" filesystem property instead,
following the property system's compression precedent: a btrfs.
namespace xattr, user-visible and admin-controlled, on the top-level
subvolume's root directory. Set it once (setfattr -n btrfs.stripe_alloc
-v 1, or btrfs property once btrfs-progs learns the name) and it is
persistent: the kernel applies it when the root directory inode loads
its properties during mount, before any user IO. Deleting the xattr
disables the policy; open runs drain at the next commit's retirement,
which runs unconditionally. Stray copies of the xattr -- a received or
cloned subvolume -- are ignored: only the top-level root carries the
policy, so receiving a stream from a stripe_alloc filesystem cannot
flip the policy on the destination.
The support checks (free space tree, not zoned, no remap-tree, not
mixed block groups) are enforced both when the property is set and when
it is applied at mount. The mixed block group test matters most here.
btrfs_check_mountopts() refuses mixed for the mount option, and without
the same test this would be a second way in that skips it -- enabling a
policy the series does not support there, after the block group read
that would have warned about uncovered metadata has already run, so the
user is told the filesystem is protected and it is not.
This stays within the series' no-on-disk-format-change constraint, and
that is the compatibility story: an older kernel mounts the filesystem
read-write and simply uses the legacy allocator, which is fully
compatible because the on-disk layout is unchanged. New kernels honor
the property during the feature's long-tail testing period; if no use
case surfaces where stripe-exclusive allocation is worse than the write
hole it protects against, it can eventually become the only allocation
mode and the property a no-op.
Known old-kernel interactions with the btrfs. namespace: existing
kernels list and read the xattr (the btrfs. get path is a plain xattr
read) but refuse to set or remove unknown property names, so the flag
can only be managed from a kernel that knows it. btrfs-progs
interaction (check, property list) with an unrecognized property is a
userspace compatibility item to verify and, if needed, patch.
The mount option is kept for now as a non-persistent override. Later
protection stages with different risk profiles (the log-tree
full-stripe relocation) should be gated by their own property rather
than widening this one, so their testing exposure can be controlled
independently.
Assisted-by: Claude:claude-fable-5
(cherry picked from commit
ecd4d3de8a0493f7c406d919d533f88e8b45390c)
return true;
}
+/*
+ * Can stripe-exclusive allocation be enabled on this filesystem? Shared by
+ * the mount option validation and the "stripe_alloc" filesystem property.
+ * Called on a mounted (or mounting, options finalized) filesystem.
+ */
+int btrfs_stripe_alloc_check_support(struct btrfs_fs_info *fs_info)
+{
+ if (!btrfs_test_opt(fs_info, FREE_SPACE_TREE)) {
+ btrfs_err(fs_info,
+ "stripe_alloc requires the free space tree (space_cache=v2)");
+ return -EINVAL;
+ }
+ if (btrfs_fs_incompat(fs_info, REMAP_TREE)) {
+ btrfs_err(fs_info,
+ "stripe_alloc is not supported together with the remap-tree feature");
+ return -EINVAL;
+ }
+ if (btrfs_is_zoned(fs_info)) {
+ btrfs_err(fs_info,
+ "stripe_alloc is not supported on zoned filesystems");
+ return -EINVAL;
+ }
+ /*
+ * Mixed block groups are refused for the mount option in
+ * btrfs_check_mountopts(); refuse them here too, or the property
+ * becomes a way to enable the policy that skips that check. A
+ * mixed group is excluded from the claim rule anyway
+ * (btrfs_is_stripe_alloc_bg() requires DATA without METADATA), so
+ * enabling here would protect nothing while reporting success.
+ */
+ if (btrfs_fs_incompat(fs_info, MIXED_GROUPS)) {
+ btrfs_err(fs_info,
+ "stripe_alloc is not supported on mixed block groups");
+ return -EINVAL;
+ }
+ return 0;
+}
+
+/*
+ * Arm or disarm stripe_unusable accounting across all raid56 data block
+ * groups whose caches are loaded. Same walk as the commit-time rescan.
+ */
+static void stripe_alloc_sweep_groups(struct btrfs_fs_info *fs_info, bool arm)
+{
+ struct btrfs_space_info *sinfo;
+
+ list_for_each_entry(sinfo, &fs_info->space_info, list) {
+ int raid;
+
+ if (!(sinfo->flags & BTRFS_BLOCK_GROUP_DATA))
+ continue;
+ down_read(&sinfo->groups_sem);
+ for (raid = 0; raid < BTRFS_NR_RAID_TYPES; raid++) {
+ struct btrfs_block_group *bg;
+
+ if (!(btrfs_raid_array[raid].bg_flag &
+ BTRFS_BLOCK_GROUP_RAID56_MASK))
+ continue;
+ list_for_each_entry(bg, &sinfo->block_groups[raid],
+ list) {
+ if (!arm)
+ btrfs_block_group_disarm_stripe_unusable(bg);
+ else if (bg->cached == BTRFS_CACHE_FINISHED)
+ btrfs_block_group_init_stripe_unusable(bg);
+ }
+ }
+ up_read(&sinfo->groups_sem);
+ }
+}
+
+/*
+ * Turn on stripe-exclusive allocation, from the "stripe_alloc" filesystem
+ * property on the top-level subvolume's root directory. Runs during mount
+ * when the root directory inode loads its properties -- before any user IO
+ * -- or at runtime from btrfs_set_prop().
+ */
+int btrfs_enable_stripe_alloc(struct btrfs_fs_info *fs_info)
+{
+ int ret;
+
+ if (btrfs_test_opt(fs_info, STRIPE_ALLOC))
+ return 0;
+ ret = btrfs_stripe_alloc_check_support(fs_info);
+ if (ret)
+ return ret;
+ btrfs_set_opt(fs_info->mount_opt, STRIPE_ALLOC);
+ /*
+ * Arm stripe_unusable accounting for raid56 data groups whose caches
+ * loaded before the policy was enabled: block groups are read before
+ * the root directory's properties apply during mount, and a runtime
+ * property set can arrive at any point. Groups still caching arm
+ * when their caching thread finishes, now that the policy is on.
+ */
+ stripe_alloc_sweep_groups(fs_info, true);
+ btrfs_info(fs_info,
+ "using stripe-exclusive allocation for raid56 data");
+ return 0;
+}
+
+/*
+ * Turn off stripe-exclusive allocation (property deleted). Open runs are
+ * not touched here: the commit-time retirement runs unconditionally and
+ * drains them at the next transaction commit.
+ */
+void btrfs_disable_stripe_alloc(struct btrfs_fs_info *fs_info)
+{
+ if (!btrfs_test_opt(fs_info, STRIPE_ALLOC))
+ return;
+ /* Disarm while the policy bit still passes the per-group gate. */
+ stripe_alloc_sweep_groups(fs_info, false);
+ btrfs_clear_opt(fs_info->mount_opt, STRIPE_ALLOC);
+ btrfs_info(fs_info,
+ "stripe-exclusive allocation for raid56 data disabled");
+}
+
/*
* Does @logical lie within a stripe run (open or draining)? Used by the
* raid56 layer to decide whether a partial write to this stripe may be
bool btrfs_stripe_in_open_run(struct btrfs_fs_info *fs_info, u64 logical);
bool btrfs_stripe_run_range_usable(struct btrfs_block_group *bg,
u64 run_start, u64 *run_len);
+int btrfs_stripe_alloc_check_support(struct btrfs_fs_info *fs_info);
+int btrfs_enable_stripe_alloc(struct btrfs_fs_info *fs_info);
+void btrfs_disable_stripe_alloc(struct btrfs_fs_info *fs_info);
void btrfs_retire_open_stripes(struct btrfs_fs_info *fs_info,
struct btrfs_transaction *trans);
void btrfs_clear_data_reloc_bg(struct btrfs_block_group *bg);
goto fail_qgroup;
}
+ /*
+ * Load the top-level root directory's properties (the stripe_alloc
+ * policy) before anything that can allocate data -- the resumed
+ * balance or dev replace kicked off below -- so filesystem-wide
+ * policy is in force for every allocation this mount ever makes.
+ * Log replay above needed no policy: it allocates no new data, only
+ * re-references extents written before the crash. The iget applies
+ * the properties as a side effect; the later iget from fill_super
+ * finds them already applied.
+ */
+ {
+ struct btrfs_inode *root_dir;
+
+ root_dir = btrfs_iget(BTRFS_FIRST_FREE_OBJECTID,
+ fs_info->fs_root);
+ if (IS_ERR(root_dir)) {
+ ret = PTR_ERR(root_dir);
+ btrfs_err(fs_info,
+ "failed to read root directory: %d", ret);
+ goto fail_qgroup;
+ }
+ iput(&root_dir->vfs_inode);
+ }
+
if (sb_rdonly(sb))
return 0;
#include "space-info.h"
#include "fs.h"
#include "accessors.h"
+#include "block-group.h"
#include "super.h"
#include "dir-item.h"
return false;
}
+/*
+ * The "stripe_alloc" property is filesystem-wide policy: stripe-exclusive
+ * (raid56 write hole safe) data allocation. It lives on the top-level
+ * subvolume's root directory only, where it is persistent, admin-controlled
+ * and visible with plain xattr tools; the kernel applies it when that inode
+ * loads during mount, before any user IO. Unlike a mount option it cannot
+ * be forgotten from fstab or a rescue mount, and older kernels still mount
+ * the filesystem (with their original raid56 write behavior).
+ */
+static bool prop_stripe_alloc_scope_ok(const struct btrfs_inode *inode)
+{
+ return btrfs_root_id(inode->root) == BTRFS_FS_TREE_OBJECTID &&
+ btrfs_ino(inode) == BTRFS_FIRST_FREE_OBJECTID;
+}
+
+static int prop_stripe_alloc_validate(const struct btrfs_inode *inode,
+ const char *value, size_t len)
+{
+ if (!prop_stripe_alloc_scope_ok(inode))
+ return -EINVAL;
+ if (!value || len == 0)
+ return 0;
+ if (len == 1 && value[0] == '1')
+ return btrfs_stripe_alloc_check_support(inode->root->fs_info);
+ return -EINVAL;
+}
+
+static int prop_stripe_alloc_apply(struct btrfs_inode *inode,
+ const char *value, size_t len)
+{
+ struct btrfs_fs_info *fs_info = inode->root->fs_info;
+
+ /*
+ * Ignore stray copies of the xattr, e.g. on a received or cloned
+ * subvolume: only the top-level root directory carries the policy.
+ */
+ if (!prop_stripe_alloc_scope_ok(inode))
+ return 0;
+ if (!value || len == 0) {
+ btrfs_disable_stripe_alloc(fs_info);
+ return 0;
+ }
+ return btrfs_enable_stripe_alloc(fs_info);
+}
+
+static const char *prop_stripe_alloc_extract(const struct btrfs_inode *inode)
+{
+ /* Not inheritable; nothing regenerates the value from inode state. */
+ return NULL;
+}
+
+static bool prop_stripe_alloc_ignore(const struct btrfs_inode *inode)
+{
+ return false;
+}
+
static const char *prop_compression_extract(const struct btrfs_inode *inode)
{
switch (inode->prop_compress) {
.ignore = prop_compression_ignore,
.inheritable = 1
},
+ {
+ .xattr_name = XATTR_BTRFS_PREFIX "stripe_alloc",
+ .validate = prop_stripe_alloc_validate,
+ .apply = prop_stripe_alloc_apply,
+ .extract = prop_stripe_alloc_extract,
+ .ignore = prop_stripe_alloc_ignore,
+ .inheritable = 0
+ },
};
int btrfs_inode_inherit_props(struct btrfs_trans_handle *trans,