diff mbox series

[v2,04/11] virtio-blk: check the write granularity of writes to sequential zones

Message ID 20260902194423.759355-5-cassel@kernel.org
State New
Headers show
Series block: fix the zone write granularity and the zone append limit | expand

Commit Message

Niklas Cassel Sept. 2, 2026, 7:44 p.m. UTC
All VIRTIO_BLK_T_OUT requests issued to sequential zones and all
VIRTIO_BLK_T_ZONE_APPEND requests must have an offset and a data size
that are multiples of the write granularity reported by the device
(virtio 1.4, 5.2.6.1), and a violation is reported as
VIRTIO_BLK_S_ZONE_UNALIGNED_WP (virtio 1.4, 5.2.6).

Neither request type was fully checked. Zone appends validated only the
offset, while writes were not checked at all.

Check the size of the appended data, and both the offset and the size of
a write, against blkconf_zone_write_granularity(), so that every request the
device accepts is one that the guest driver was told is valid. Writes to
conventional zones keep no alignment constraint beyond the logical block
size. The write path performs the check after virtio_blk_sect_range_ok()
so that the zone index derived from the guest supplied sector is known to
be in range.

Signed-off-by: Niklas Cassel <cassel@kernel.org>
---
 hw/block/virtio-blk.c | 37 ++++++++++++++++++++++++++++++++++++-
 1 file changed, 36 insertions(+), 1 deletion(-)

Comments

Damien Le Moal Sept. 3, 2026, 12:51 a.m. UTC | #1
On 9/3/26 04:44, Niklas Cassel wrote:
> All VIRTIO_BLK_T_OUT requests issued to sequential zones and all
> VIRTIO_BLK_T_ZONE_APPEND requests must have an offset and a data size
> that are multiples of the write granularity reported by the device
> (virtio 1.4, 5.2.6.1), and a violation is reported as
> VIRTIO_BLK_S_ZONE_UNALIGNED_WP (virtio 1.4, 5.2.6).
> 
> Neither request type was fully checked. Zone appends validated only the
> offset, while writes were not checked at all.
> 
> Check the size of the appended data, and both the offset and the size of
> a write, against blkconf_zone_write_granularity(), so that every request the
> device accepts is one that the guest driver was told is valid. Writes to
> conventional zones keep no alignment constraint beyond the logical block
> size. The write path performs the check after virtio_blk_sect_range_ok()
> so that the zone index derived from the guest supplied sector is known to
> be in range.
> 
> Signed-off-by: Niklas Cassel <cassel@kernel.org>
> ---
>  hw/block/virtio-blk.c | 37 ++++++++++++++++++++++++++++++++++++-
>  1 file changed, 36 insertions(+), 1 deletion(-)
> 
> diff --git a/hw/block/virtio-blk.c b/hw/block/virtio-blk.c
> index b22aed04a1..247af53ae2 100644
> --- a/hw/block/virtio-blk.c
> +++ b/hw/block/virtio-blk.c
> @@ -398,6 +398,32 @@ static bool virtio_blk_sect_range_ok(VirtIOBlock *dev,
>      return true;
>  }
>  
> +/*
> + * Both the offset and the size of a write to a sequential zone must be a
> + * multiple of the write granularity that the device reports. Conventional
> + * zones are not constrained. The zone index is derived from a guest supplied
> + * sector, so this must only be called once virtio_blk_sect_range_ok() has
> + * bounded it.
> + */
> +static bool virtio_blk_zone_write_granularity_ok(VirtIOBlock *dev,
> +                                                 uint64_t sector, size_t size)
> +{
> +    BlockDriverState *bs = blk_bs(dev->blk);
> +    uint64_t offset = sector << BDRV_SECTOR_BITS;
> +    uint32_t wg_mask;
> +
> +    if (bs->bl.zoned == BLK_Z_NONE) {
> +        return true;
> +    }
> +
> +    wg_mask = blkconf_zone_write_granularity(&dev->conf.conf) - 1;
> +    if (!(offset & wg_mask) && !(size & wg_mask)) {
> +        return true;
> +    }
> +
> +    return BDRV_ZT_IS_CONV(bs->wps->wp[offset / bs->bl.zone_size]);
> +}

Even though in practice I do not think it would ever happen, technically,
conventional zones are not bound by the physical sector size/write granularity
and can accept LBA aligned writes even when LBA size < physical block size
(write granularity).

SO I think this should be:

+static bool virtio_blk_zone_write_granularity_ok(VirtIOBlock *dev,
+                                                 uint64_t sector, size_t size)
+{
+    BlockDriverState *bs = blk_bs(dev->blk);
+    uint64_t offset = sector << BDRV_SECTOR_BITS;
+    uint32_t wg_mask;
+
+    if (bs->bl.zoned == BLK_Z_NONE) {
+        return true;
+    }
+
+    if (BDRV_ZT_IS_CONV(bs->wps->wp[offset / bs->bl.zone_size]) {
+        return true;
+    }
+
+    wg_mask = blkconf_zone_write_granularity(&dev->conf.conf) - 1;
+    return !(offset & wg_mask) && !(size & wg_mask);
+}

Side note not for this patch: it really would be nice to have a helper that
gives a zone number from an offset, using a bit shift (bs->bl.zone_size_shift)
instead of a costly division.

> +
>  static uint8_t virtio_blk_handle_discard_write_zeroes(VirtIOBlockReq *req,
>      struct virtio_blk_discard_write_zeroes *dwz_hdr, bool is_write_zeroes)
>  {
> @@ -522,7 +548,7 @@ static bool check_zoned_request(VirtIOBlock *s, int64_t offset, int64_t len,
>      if (append) {
>          uint32_t wg_mask = blkconf_zone_write_granularity(&s->conf.conf) - 1;
>  
> -        if (offset & wg_mask) {
> +        if (offset & wg_mask || len & wg_mask) {
>              *status = VIRTIO_BLK_S_ZONE_UNALIGNED_WP;
>              return false;
>          }
> @@ -911,6 +937,15 @@ static int virtio_blk_handle_request(VirtIOBlockReq *req, MultiReqBuffer *mrb)
>              return 0;
>          }
>  
> +        if (is_write &&
> +            !virtio_blk_zone_write_granularity_ok(s, req->sector_num,
> +                                                  req->qiov.size)) {
> +            virtio_blk_req_complete(req, VIRTIO_BLK_S_ZONE_UNALIGNED_WP);
> +            block_acct_invalid(blk_get_stats(s->blk), BLOCK_ACCT_WRITE);
> +            g_free(req);
> +            return 0;
> +        }
> +
>          block_acct_start(blk_get_stats(s->blk), &req->acct, req->qiov.size,
>                           is_write ? BLOCK_ACCT_WRITE : BLOCK_ACCT_READ);
>
Niklas Cassel Sept. 4, 2026, 4:05 p.m. UTC | #2
On Thu, Sep 03, 2026 at 09:51:01AM +0900, Damien Le Moal wrote:
> On 9/3/26 04:44, Niklas Cassel wrote:
> > All VIRTIO_BLK_T_OUT requests issued to sequential zones and all
> > VIRTIO_BLK_T_ZONE_APPEND requests must have an offset and a data size
> > that are multiples of the write granularity reported by the device
> > (virtio 1.4, 5.2.6.1), and a violation is reported as
> > VIRTIO_BLK_S_ZONE_UNALIGNED_WP (virtio 1.4, 5.2.6).
> > 
> > Neither request type was fully checked. Zone appends validated only the
> > offset, while writes were not checked at all.
> > 
> > Check the size of the appended data, and both the offset and the size of
> > a write, against blkconf_zone_write_granularity(), so that every request the
> > device accepts is one that the guest driver was told is valid. Writes to
> > conventional zones keep no alignment constraint beyond the logical block
> > size. The write path performs the check after virtio_blk_sect_range_ok()
> > so that the zone index derived from the guest supplied sector is known to
> > be in range.
> > 
> > Signed-off-by: Niklas Cassel <cassel@kernel.org>
> > ---
> >  hw/block/virtio-blk.c | 37 ++++++++++++++++++++++++++++++++++++-
> >  1 file changed, 36 insertions(+), 1 deletion(-)
> > 
> > diff --git a/hw/block/virtio-blk.c b/hw/block/virtio-blk.c
> > index b22aed04a1..247af53ae2 100644
> > --- a/hw/block/virtio-blk.c
> > +++ b/hw/block/virtio-blk.c
> > @@ -398,6 +398,32 @@ static bool virtio_blk_sect_range_ok(VirtIOBlock *dev,
> >      return true;
> >  }
> >  
> > +/*
> > + * Both the offset and the size of a write to a sequential zone must be a
> > + * multiple of the write granularity that the device reports. Conventional
> > + * zones are not constrained. The zone index is derived from a guest supplied
> > + * sector, so this must only be called once virtio_blk_sect_range_ok() has
> > + * bounded it.
> > + */
> > +static bool virtio_blk_zone_write_granularity_ok(VirtIOBlock *dev,
> > +                                                 uint64_t sector, size_t size)
> > +{
> > +    BlockDriverState *bs = blk_bs(dev->blk);
> > +    uint64_t offset = sector << BDRV_SECTOR_BITS;
> > +    uint32_t wg_mask;
> > +
> > +    if (bs->bl.zoned == BLK_Z_NONE) {
> > +        return true;
> > +    }
> > +
> > +    wg_mask = blkconf_zone_write_granularity(&dev->conf.conf) - 1;
> > +    if (!(offset & wg_mask) && !(size & wg_mask)) {
> > +        return true;
> > +    }
> > +
> > +    return BDRV_ZT_IS_CONV(bs->wps->wp[offset / bs->bl.zone_size]);
> > +}
> 
> Even though in practice I do not think it would ever happen, technically,
> conventional zones are not bound by the physical sector size/write granularity
> and can accept LBA aligned writes even when LBA size < physical block size
> (write granularity).

Well, that is how the code works already :)

And the function comment does also already say that conventional
zones are not constrained.

Yes, I guess swapping the order does make the code slightly easier to read,
so let me do that.


> 
> SO I think this should be:
> 
> +static bool virtio_blk_zone_write_granularity_ok(VirtIOBlock *dev,
> +                                                 uint64_t sector, size_t size)
> +{
> +    BlockDriverState *bs = blk_bs(dev->blk);
> +    uint64_t offset = sector << BDRV_SECTOR_BITS;
> +    uint32_t wg_mask;
> +
> +    if (bs->bl.zoned == BLK_Z_NONE) {
> +        return true;
> +    }
> +
> +    if (BDRV_ZT_IS_CONV(bs->wps->wp[offset / bs->bl.zone_size]) {
> +        return true;
> +    }
> +
> +    wg_mask = blkconf_zone_write_granularity(&dev->conf.conf) - 1;
> +    return !(offset & wg_mask) && !(size & wg_mask);
> +}
> 
> Side note not for this patch: it really would be nice to have a helper that
> gives a zone number from an offset, using a bit shift (bs->bl.zone_size_shift)
> instead of a costly division.

Will add a helper as a new patch.


Kind regards,
Niklas
diff mbox series

Patch

diff --git a/hw/block/virtio-blk.c b/hw/block/virtio-blk.c
index b22aed04a1..247af53ae2 100644
--- a/hw/block/virtio-blk.c
+++ b/hw/block/virtio-blk.c
@@ -398,6 +398,32 @@  static bool virtio_blk_sect_range_ok(VirtIOBlock *dev,
     return true;
 }
 
+/*
+ * Both the offset and the size of a write to a sequential zone must be a
+ * multiple of the write granularity that the device reports. Conventional
+ * zones are not constrained. The zone index is derived from a guest supplied
+ * sector, so this must only be called once virtio_blk_sect_range_ok() has
+ * bounded it.
+ */
+static bool virtio_blk_zone_write_granularity_ok(VirtIOBlock *dev,
+                                                 uint64_t sector, size_t size)
+{
+    BlockDriverState *bs = blk_bs(dev->blk);
+    uint64_t offset = sector << BDRV_SECTOR_BITS;
+    uint32_t wg_mask;
+
+    if (bs->bl.zoned == BLK_Z_NONE) {
+        return true;
+    }
+
+    wg_mask = blkconf_zone_write_granularity(&dev->conf.conf) - 1;
+    if (!(offset & wg_mask) && !(size & wg_mask)) {
+        return true;
+    }
+
+    return BDRV_ZT_IS_CONV(bs->wps->wp[offset / bs->bl.zone_size]);
+}
+
 static uint8_t virtio_blk_handle_discard_write_zeroes(VirtIOBlockReq *req,
     struct virtio_blk_discard_write_zeroes *dwz_hdr, bool is_write_zeroes)
 {
@@ -522,7 +548,7 @@  static bool check_zoned_request(VirtIOBlock *s, int64_t offset, int64_t len,
     if (append) {
         uint32_t wg_mask = blkconf_zone_write_granularity(&s->conf.conf) - 1;
 
-        if (offset & wg_mask) {
+        if (offset & wg_mask || len & wg_mask) {
             *status = VIRTIO_BLK_S_ZONE_UNALIGNED_WP;
             return false;
         }
@@ -911,6 +937,15 @@  static int virtio_blk_handle_request(VirtIOBlockReq *req, MultiReqBuffer *mrb)
             return 0;
         }
 
+        if (is_write &&
+            !virtio_blk_zone_write_granularity_ok(s, req->sector_num,
+                                                  req->qiov.size)) {
+            virtio_blk_req_complete(req, VIRTIO_BLK_S_ZONE_UNALIGNED_WP);
+            block_acct_invalid(blk_get_stats(s->blk), BLOCK_ACCT_WRITE);
+            g_free(req);
+            return 0;
+        }
+
         block_acct_start(blk_get_stats(s->blk), &req->acct, req->qiov.size,
                          is_write ? BLOCK_ACCT_WRITE : BLOCK_ACCT_READ);