Virtuozzo QEMU development (svt-core@virtuozzo.com)
 help / color / mirror / Atom feed
From: Andrey Zhadchenko <andrey.zhadchenko@virtuozzo.com>
To: svt-core@virtuozzo.com
Cc: den@openvz.org, andrey.drobyshev@virtuozzo.com
Subject: [QEMU HCI-8.0 PATCH 4/5] vhost-blk: watch the device for resize events
Date: Thu,  3 Sep 2026 15:32:03 +0300	[thread overview]
Message-ID: <20260903123204.24035-5-andrey.zhadchenko@virtuozzo.com> (raw)
In-Reply-To: <20260903123204.24035-1-andrey.zhadchenko@virtuozzo.com>

Resize was tied to block node, which we removed some time ago.
Luckily we can make resize automated: watch netlink for relevant
events and call virtio_notify_config() if we detect capacity
change.
Failed netlink setup during creation leads to failure, but this
is a price we are ready to pay for consistency.

https://virtuozzo.atlassian.net/browse/VSTOR-143437
Signed-off-by: Andrey Zhadchenko <andrey.zhadchenko@virtuozzo.com>
---
 hw/block/vhost-blk.c          | 193 ++++++++++++++++++++++++++++++++++
 include/hw/virtio/vhost-blk.h |   5 +
 2 files changed, 198 insertions(+)

diff --git a/hw/block/vhost-blk.c b/hw/block/vhost-blk.c
index 47b6e560d4..f8eca4d58a 100644
--- a/hw/block/vhost-blk.c
+++ b/hw/block/vhost-blk.c
@@ -10,7 +10,9 @@
 
 #include "qemu/osdep.h"
 #include "qapi/error.h"
+#include "qemu/cutils.h"
 #include "qemu/error-report.h"
+#include "qemu/main-loop.h"
 #include "qom/object.h"
 #include "hw/qdev-core.h"
 #include "hw/boards.h"
@@ -25,8 +27,13 @@
 #include "linux-headers/linux/vhost.h"
 #include <sys/ioctl.h>
 #include <linux/fs.h>
+#include <linux/netlink.h>
 #include "system/runstate.h"
 
+static int vhost_blk_uevent_fd = -1;
+static QLIST_HEAD(, VHostBlk) vhost_blk_uevent_watchers =
+    QLIST_HEAD_INITIALIZER(vhost_blk_uevent_watchers);
+
 static int vhost_blk_start(VirtIODevice *vdev)
 {
     VHostBlk *s = VHOST_BLK(vdev);
@@ -230,6 +237,182 @@ static int vhost_blk_update_size(VHostBlk *s, Error **errp)
     return changed;
 }
 
+static void vhost_blk_resize_bh(void *opaque)
+{
+    VHostBlk *s = opaque;
+    Error *local_err = NULL;
+    int ret;
+
+    ret = vhost_blk_update_size(s, &local_err);
+    if (ret < 0) {
+        error_report_err(local_err);
+        return;
+    }
+
+    if (ret) {
+        virtio_notify_config(VIRTIO_DEVICE(s));
+    }
+}
+
+static void vhost_blk_uevent_read(void *opaque)
+{
+    char buffer[64 * 1024 + 1];
+
+    for (;;) {
+        struct sockaddr_nl source;
+        socklen_t source_len = sizeof(source);
+        uint64_t event_major = UINT64_MAX;
+        uint64_t event_minor = UINT64_MAX;
+        bool action_change = false;
+        bool subsystem_block = false;
+        bool resize = false;
+        char *field;
+        char *end;
+        ssize_t len;
+
+        memset(&source, 0, sizeof(source));
+        len = recvfrom(vhost_blk_uevent_fd, buffer, sizeof(buffer) - 1,
+                       MSG_DONTWAIT, (struct sockaddr *)&source, &source_len);
+        if (len < 0) {
+            if (errno == EINTR) {
+                continue;
+            }
+            if (errno == ENOBUFS) {
+                VHostBlk *s;
+
+                /* uevents dropped. Re-check just to be sure */
+                QLIST_FOREACH(s, &vhost_blk_uevent_watchers, uevent_node) {
+                    qemu_bh_schedule(s->resize_bh);
+                }
+                continue;
+            }
+            if (errno != EAGAIN && errno != EWOULDBLOCK) {
+                error_report("vhost-blk: unable to receive uevent: %s",
+                             strerror(errno));
+            }
+            return;
+        }
+
+        if (source.nl_family != AF_NETLINK || source.nl_pid != 0) {
+            continue;
+        }
+
+        buffer[len] = '\0';
+        field = buffer;
+        end = buffer + len;
+        while (field < end) {
+            size_t field_len = strnlen(field, end - field);
+
+            if (!strcmp(field, "ACTION=change")) {
+                action_change = true;
+            } else if (!strcmp(field, "SUBSYSTEM=block")) {
+                subsystem_block = true;
+            } else if (!strcmp(field, "RESIZE=1")) {
+                resize = true;
+            } else if (g_str_has_prefix(field, "MAJOR=")) {
+                uint64_t value;
+
+                if (!qemu_strtou64(field + strlen("MAJOR="), NULL, 10,
+                                   &value)) {
+                    event_major = value;
+                }
+            } else if (g_str_has_prefix(field, "MINOR=")) {
+                uint64_t value;
+
+                if (!qemu_strtou64(field + strlen("MINOR="), NULL, 10,
+                                   &value)) {
+                    event_minor = value;
+                }
+            }
+
+            if (field_len == end - field) {
+                break;
+            }
+            field += field_len + 1;
+        }
+
+        if (action_change && subsystem_block && resize &&
+            event_major <= UINT_MAX && event_minor <= UINT_MAX) {
+            VHostBlk *s;
+
+            QLIST_FOREACH(s, &vhost_blk_uevent_watchers, uevent_node) {
+                dev_t rdev = s->backend_rdev;
+
+                if (major(rdev) == event_major &&
+                    minor(rdev) == event_minor) {
+                    qemu_bh_schedule(s->resize_bh);
+                }
+            }
+        }
+    }
+}
+
+static bool vhost_blk_uevent_init(Error **errp)
+{
+    struct sockaddr_nl address = {
+        .nl_family = AF_NETLINK,
+        .nl_groups = 1,
+    };
+
+    if (vhost_blk_uevent_fd >= 0) {
+        return true;
+    }
+
+    vhost_blk_uevent_fd = socket(AF_NETLINK,
+                                 SOCK_DGRAM | SOCK_NONBLOCK | SOCK_CLOEXEC,
+                                 NETLINK_KOBJECT_UEVENT);
+    if (vhost_blk_uevent_fd < 0) {
+        error_setg_errno(errp, errno,
+                         "vhost-blk: unable to create uevent socket");
+        return false;
+    }
+
+    if (bind(vhost_blk_uevent_fd, (struct sockaddr *)&address,
+             sizeof(address)) < 0) {
+        error_setg_errno(errp, errno,
+                         "vhost-blk: unable to bind uevent socket");
+        qemu_close(vhost_blk_uevent_fd);
+        vhost_blk_uevent_fd = -1;
+        return false;
+    }
+
+    qemu_set_fd_handler(vhost_blk_uevent_fd, vhost_blk_uevent_read,
+                        NULL, NULL);
+    return true;
+}
+
+static void vhost_blk_uevent_cleanup_if_unused(void)
+{
+    if (vhost_blk_uevent_fd < 0 ||
+        !QLIST_EMPTY(&vhost_blk_uevent_watchers)) {
+        return;
+    }
+
+    qemu_set_fd_handler(vhost_blk_uevent_fd, NULL, NULL, NULL);
+    qemu_close(vhost_blk_uevent_fd);
+    vhost_blk_uevent_fd = -1;
+}
+
+static void vhost_blk_uevent_register(VHostBlk *s)
+{
+    s->resize_bh = qemu_bh_new(vhost_blk_resize_bh, s);
+    QLIST_INSERT_HEAD(&vhost_blk_uevent_watchers, s, uevent_node);
+    s->uevent_registered = true;
+}
+
+static void vhost_blk_uevent_unregister(VHostBlk *s)
+{
+    if (!s->uevent_registered) {
+        return;
+    }
+
+    QLIST_REMOVE(s, uevent_node);
+    s->uevent_registered = false;
+    qemu_bh_delete(s->resize_bh);
+    s->resize_bh = NULL;
+    vhost_blk_uevent_cleanup_if_unused();
+}
+
 static bool vhost_blk_open_backend(VHostBlk *s, Error **errp)
 {
     BlockConf *conf = &s->conf.conf;
@@ -254,6 +437,7 @@ static bool vhost_blk_open_backend(VHostBlk *s, Error **errp)
                    s->conf.devpath);
         goto fail;
     }
+    s->backend_rdev = st.st_rdev;
 
     if (ioctl(s->backend_fd, BLKROGET, &readonly) < 0) {
         error_setg_errno(errp, errno,
@@ -332,6 +516,12 @@ static void vhost_blk_device_realize(DeviceState *dev, Error **errp)
         return;
     }
 
+    if (!vhost_blk_uevent_init(errp)) {
+        qemu_close(s->backend_fd);
+        s->backend_fd = -1;
+        return;
+    }
+
     s->dev.nvqs = conf->num_queues;
     s->dev.max_queues = conf->num_queues;
     s->dev.vqs = g_new0(struct vhost_virtqueue, s->dev.nvqs);
@@ -366,6 +556,7 @@ static void vhost_blk_device_realize(DeviceState *dev, Error **errp)
         goto cleanup;
     }
 
+    vhost_blk_uevent_register(s);
     return;
 
 cleanup:
@@ -385,6 +576,7 @@ cleanup:
         virtio_del_queue(vdev, i);
     }
     virtio_cleanup(vdev);
+    vhost_blk_uevent_cleanup_if_unused();
     return;
 }
 
@@ -393,6 +585,7 @@ static void vhost_blk_device_unrealize(DeviceState *dev)
     VirtIODevice *vdev = VIRTIO_DEVICE(dev);
     VHostBlk *s = VHOST_BLK(dev);
 
+    vhost_blk_uevent_unregister(s);
     qemu_del_vm_change_state_handler(s->mighand);
     vhost_blk_set_status(vdev, 0);
     vhost_dev_cleanup(&s->dev);
diff --git a/include/hw/virtio/vhost-blk.h b/include/hw/virtio/vhost-blk.h
index c6646f5845..815939419f 100644
--- a/include/hw/virtio/vhost-blk.h
+++ b/include/hw/virtio/vhost-blk.h
@@ -14,6 +14,7 @@
 #include "standard-headers/linux/virtio_blk.h"
 #include "hw/block/block.h"
 #include "hw/virtio/vhost.h"
+#include "qemu/queue.h"
 
 #define TYPE_VHOST_BLK "vhost-blk"
 #define VHOST_BLK(obj) \
@@ -43,6 +44,10 @@ typedef struct VHostBlk {
     struct vhost_dev dev;
     bool vhost_started;
     uint64_t length;
+    uint64_t backend_rdev;
+    QEMUBH *resize_bh;
+    QLIST_ENTRY(VHostBlk) uevent_node;
+    bool uevent_registered;
 } VHostBlk;
 
 #endif
-- 
2.43.5


  parent reply	other threads:[~2026-09-03 12:35 UTC|newest]

Thread overview: 15+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-03 12:31 [QEMU HCI-8.0 PATCH 0/5] vhost-blk change backend setup Andrey Zhadchenko
2026-09-03 12:32 ` [QEMU HCI-8.0 PATCH 1/5] blk: factor out validation Andrey Zhadchenko
2026-09-03 14:56   ` Andrey Drobyshev
2026-09-03 12:32 ` [QEMU HCI-8.0 PATCH 2/5] vhost-blk: change backend setup Andrey Zhadchenko
2026-09-03 14:56   ` Andrey Drobyshev
2026-09-03 15:27     ` Andrey Zhadchenko
2026-09-03 15:34       ` Andrey Drobyshev
2026-09-03 12:32 ` [QEMU HCI-8.0 PATCH 3/5] vhost-blk: add read-only flag Andrey Zhadchenko
2026-09-03 14:56   ` Andrey Drobyshev
2026-09-03 12:32 ` Andrey Zhadchenko [this message]
2026-09-03 14:56   ` [QEMU HCI-8.0 PATCH 4/5] vhost-blk: watch the device for resize events Andrey Drobyshev
2026-09-03 15:30     ` Andrey Zhadchenko
2026-09-03 15:45       ` Andrey Drobyshev
2026-09-03 15:50         ` Andrey Zhadchenko
2026-09-03 12:32 ` [QEMU HCI-8.0 PATCH 5/5] vhost-blk: filter uevents in the kernel Andrey Zhadchenko

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260903123204.24035-5-andrey.zhadchenko@virtuozzo.com \
    --to=andrey.zhadchenko@virtuozzo.com \
    --cc=andrey.drobyshev@virtuozzo.com \
    --cc=den@openvz.org \
    --cc=svt-core@virtuozzo.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox