* make the decision when submitting the bio.
*
* The pattern between do_readpage(), submit_one_bio() and
- * submit_extent_folio() is quite subtle, so tracking this is tricky.
+ * submit_one_block() is quite subtle, so tracking this is tricky.
*
* As we process extent E, we might submit a bio with existing built up
* extents before adding E to a new bio, or we might just add E to the
* bio. As a result, E's generation could apply to the current bio or
* to the next one, so we need to be careful to update the bio_ctrl's
* generation with E's only when we are sure E is added to bio_ctrl->bbio
- * in submit_extent_folio().
+ * in submit_one_block().
*
* See the comment in btrfs_lookup_bio_sums() for more detail on the
* need for this optimization.
}
/*
- * @disk_bytenr: logical bytenr where the write will be
- * @page: page to add to the bio
- * @size: portion of page that we want to write to
- * @pg_offset: offset of the new bio or to check whether we are adding
- * a contiguous page to the previous one
+ * @disk_bytenr: logical bytenr where the read/write will be
+ * @folio: the folio the block belongs to
+ * @pg_offset: the offset inside the folio
* @read_em_generation: generation of the extent_map we are submitting
* (only used for read)
*
- * The will either add the page into the existing @bio_ctrl->bbio, or allocate a
+ * This will either add the block into the existing @bio_ctrl->bbio, or allocate a
* new one in @bio_ctrl->bbio.
* The mirror number for this IO should already be initialized in
* @bio_ctrl->mirror_num.
*
- * Return the number of bytes that are queued into a bio.
- * If the returned bytes is smaller than @size, it means we hit a critical error
- * for data write, where there is no ordered extent for the range.
+ * Return 0 if the block is queued or submitted.
+ * Return <0 for error.
*/
-static unsigned int submit_extent_folio(struct btrfs_bio_ctrl *bio_ctrl,
- u64 disk_bytenr, struct folio *folio,
- size_t size, unsigned long pg_offset,
- u64 read_em_generation)
+static int submit_one_block(struct btrfs_bio_ctrl *bio_ctrl,
+ u64 disk_bytenr, struct folio *folio,
+ unsigned long pg_offset, u64 read_em_generation)
{
struct btrfs_inode *inode = folio_to_inode(folio);
+ const struct btrfs_fs_info *fs_info = inode->root->fs_info;
+ const u32 blocksize = fs_info->sectorsize;
loff_t file_offset = folio_pos(folio) + pg_offset;
- unsigned int queued = 0;
- ASSERT(pg_offset + size <= folio_size(folio));
+ ASSERT(pg_offset + blocksize <= folio_size(folio));
ASSERT(bio_ctrl->end_io_func);
if (bio_ctrl->bbio &&
!btrfs_bio_is_contig(bio_ctrl, disk_bytenr, file_offset))
submit_one_bio(bio_ctrl);
- do {
- u32 len = size;
-
- /* Allocate new bio if needed */
- if (!bio_ctrl->bbio) {
- int ret;
-
- ret = alloc_new_bio(inode, bio_ctrl, disk_bytenr, file_offset);
- if (ret < 0)
- break;
- }
+again:
+ /* Allocate new bio if needed */
+ if (!bio_ctrl->bbio) {
+ int ret;
- /* Cap to the current ordered extent boundary if there is one. */
- if (len > bio_ctrl->len_to_oe_boundary) {
- ASSERT(bio_ctrl->compress_type == BTRFS_COMPRESS_NONE);
- ASSERT(is_data_inode(inode));
- len = bio_ctrl->len_to_oe_boundary;
- }
+ ret = alloc_new_bio(inode, bio_ctrl, disk_bytenr, file_offset);
+ if (ret < 0)
+ return ret;
+ }
- if (!bio_add_folio(&bio_ctrl->bbio->bio, folio, len, pg_offset)) {
- /* bio full: move on to a new one */
- submit_one_bio(bio_ctrl);
- continue;
- }
- /*
- * Now that the folio is definitely added to the bio, include its
- * generation in the max generation calculation.
- */
- bio_ctrl->generation = max(bio_ctrl->generation, read_em_generation);
- bio_ctrl->next_file_offset += len;
+ if (!bio_add_folio(&bio_ctrl->bbio->bio, folio, blocksize, pg_offset)) {
+ /* bio full: move on to a new one */
+ submit_one_bio(bio_ctrl);
+ goto again;
+ }
- if (bio_ctrl->wbc)
- wbc_account_cgroup_owner(bio_ctrl->wbc, folio, len);
+ /*
+ * Now that the folio is definitely added to the bio, include its
+ * generation in the max generation calculation.
+ */
+ bio_ctrl->generation = max(bio_ctrl->generation, read_em_generation);
+ bio_ctrl->next_file_offset += blocksize;
- size -= len;
- pg_offset += len;
- disk_bytenr += len;
- file_offset += len;
- queued += len;
+ if (bio_ctrl->wbc)
+ wbc_account_cgroup_owner(bio_ctrl->wbc, folio, blocksize);
- /*
- * len_to_oe_boundary defaults to U32_MAX, which isn't folio or
- * sector aligned. alloc_new_bio() then sets it to the end of
- * our ordered extent for writes into zoned devices.
- *
- * When len_to_oe_boundary is tracking an ordered extent, we
- * trust the ordered extent code to align things properly, and
- * the check above to cap our write to the ordered extent
- * boundary is correct.
- *
- * When len_to_oe_boundary is U32_MAX, the cap above would
- * result in a 4095 byte IO for the last folio right before
- * we hit the bio limit of UINT_MAX. bio_add_folio() has all
- * the checks required to make sure we don't overflow the bio,
- * and we should just ignore len_to_oe_boundary completely
- * unless we're using it to track an ordered extent.
- *
- * It's pretty hard to make a bio sized U32_MAX, but it can
- * happen when the page cache is able to feed us contiguous
- * folios for large extents.
- */
- if (bio_ctrl->len_to_oe_boundary != U32_MAX)
- bio_ctrl->len_to_oe_boundary -= len;
-
- /* Ordered extent boundary: move on to a new bio. */
- if (bio_ctrl->len_to_oe_boundary == 0)
- submit_one_bio(bio_ctrl);
- /*
- * If we have accumulated decent amount of IO, send it to the
- * block layer so that IO can run while we are accumulating
- * more folios to write.
- */
- else if (bio_ctrl->wbc &&
- bio_ctrl->bbio->bio.bi_iter.bi_size >=
- inode->root->fs_info->writeback_bio_size)
- submit_one_bio(bio_ctrl);
+ /*
+ * len_to_oe_boundary defaults to U32_MAX, which isn't folio or sector
+ * aligned. alloc_new_bio() then sets it to the end of our ordered
+ * extent for writes into zoned devices.
+ *
+ * When len_to_oe_boundary is tracking an ordered extent, the
+ * len_to_oe_boundary should follow that OE and never go beyond the max
+ * extent size (128MiB).
+ *
+ * When len_to_oe_boundary is U32_MAX, decreasing the length by
+ * blocksize will never make it reach 0, thus skipping the later
+ * submit_one_bio() call. So if len_to_oe_boundary() is not tracking
+ * an OE, do not decrease it.
+ *
+ * It's pretty hard to make a bio sized U32_MAX, but it can happen when
+ * the page cache is able to feed us contiguous folios for large
+ * extents.
+ */
+ if (bio_ctrl->len_to_oe_boundary != U32_MAX)
+ bio_ctrl->len_to_oe_boundary -= blocksize;
- } while (size);
- return queued;
+ /* Ordered extent boundary: move on to a new bio. */
+ if (bio_ctrl->len_to_oe_boundary == 0)
+ submit_one_bio(bio_ctrl);
+ /*
+ * If we have accumulated decent amount of IO, send it to the block
+ * layer so that IO can run while we are accumulating more folios to
+ * write.
+ */
+ else if (bio_ctrl->wbc &&
+ bio_ctrl->bbio->bio.bi_iter.bi_size >= fs_info->writeback_bio_size)
+ submit_one_bio(bio_ctrl);
+ return 0;
}
static int attach_extent_buffer_folio(struct extent_buffer *eb,
u64 disk_bytenr;
u64 block_start;
u64 em_gen;
- unsigned int queued;
ASSERT(IS_ALIGNED(cur, fs_info->sectorsize));
if (cur >= last_byte) {
if (force_bio_submit)
submit_one_bio(bio_ctrl);
- queued = submit_extent_folio(bio_ctrl, disk_bytenr, folio, blocksize,
- pg_offset, em_gen);
+ ret = submit_one_block(bio_ctrl, disk_bytenr, folio, pg_offset, em_gen);
/* Read submission should not fail. */
- ASSERT(queued == blocksize);
+ ASSERT(ret == 0);
}
return 0;
}
*
* Caller should make sure filepos < i_size and handle filepos >= i_size case.
*/
-static int submit_one_sector(struct btrfs_inode *inode,
- struct folio *folio,
- u64 filepos, struct btrfs_bio_ctrl *bio_ctrl,
- loff_t i_size)
+static int submit_write_sector(struct btrfs_inode *inode,
+ struct folio *folio,
+ u64 filepos, struct btrfs_bio_ctrl *bio_ctrl,
+ loff_t i_size)
{
struct btrfs_fs_info *fs_info = inode->root->fs_info;
struct btrfs_ordered_extent *oe;
u64 disk_bytenr;
u64 extent_offset;
const u32 sectorsize = fs_info->sectorsize;
- unsigned int queued;
+ int ret;
ASSERT(IS_ALIGNED(filepos, sectorsize));
*/
ASSERT(folio_test_writeback(folio));
- queued = submit_extent_folio(bio_ctrl, disk_bytenr, folio,
- sectorsize, filepos - folio_pos(folio), 0);
- if (unlikely(queued < sectorsize)) {
+ ret = submit_one_block(bio_ctrl, disk_bytenr, folio,
+ offset_in_folio(folio, filepos), 0);
+ if (unlikely(ret < 0)) {
btrfs_folio_clear_writeback(fs_info, folio, filepos, sectorsize);
btrfs_mark_ordered_io_finished(inode, filepos, fs_info->sectorsize,
false);
btrfs_err_rl(fs_info,
- "failed to queue sector for root %lld ino %llu filepos %llu",
+ "failed to queue sector for root %lld ino %llu filepos %llu: %pe",
btrfs_root_id(inode->root),
- btrfs_ino(inode), filepos);
- return -EUCLEAN;
+ btrfs_ino(inode), filepos, ERR_PTR(ret));
+ return ret;
}
return 0;
}
btrfs_folio_clear_dirty(fs_info, folio, cur, fs_info->sectorsize);
continue;
}
- ret = submit_one_sector(inode, folio, cur, bio_ctrl, i_size);
+ ret = submit_write_sector(inode, folio, cur, bio_ctrl, i_size);
if (unlikely(ret < 0)) {
if (!found_error)
found_error = ret;