aboutsummaryrefslogtreecommitdiff
diff options
context:
space:
mode:
authorLinus Torvalds <torvalds@linux-foundation.org>2026-08-20 17:07:50 -0700
committerLinus Torvalds <torvalds@linux-foundation.org>2026-08-20 17:07:50 -0700
commit7f063b2f17eaba2a35e251aa53627f2a70d536e2 (patch)
tree82676aefc15baa911c835e69c840184ee45607c7
parenta625b2a387628df94e385faf5c81bf252f304ed9 (diff)
parentb282418bc366194677eafd1dad180d92254586ac (diff)
Merge tag 'for_linus' of git://git.kernel.org/pub/scm/linux/kernel/git/mst/vhost
Pull vhost,vdpa,virtio updates from Michael Tsirkin: - transport v3 support in virtio-mmio - suspend support in vduse - fixes, cleanups all over the place * tag 'for_linus' of git://git.kernel.org/pub/scm/linux/kernel/git/mst/vhost: (54 commits) vduse: Add suspend vduse: do not take rwsem at reset work flush vduse: add F_QUEUE_READY feature vduse: add VDUSE_SET_FEATURES ioctl vduse: add VDUSE_GET_FEATURES ioctl vduse: store control device pointer tools/virtio: Fix control typo in trace agent comment tools/virtio: Fix userspace typo in vringh test comment vhost: reject zero-size IOTLB INVALIDATE vdpa: Remove redundant dev_err() virtio_ring: fix infinite loop in virtnet_poll_cleantx when device is broken vdpa/mlx5: roll back MR update after VQ setup failure MAINTAINERS: remove Gabriel from LiteX and fw-cfg drivers virtio_mem: fix typo in comment vdpa/solidrun: fix typos in snet_ctrl comments virtio: fix article before virtio in dma-buf comment vhost: fix inaccurate kdoc in iotlb helpers virtio: rtc: time out alarm requests vdpa/mlx5: fix wrong MLX5_ADDR_OF struct type in alloc_inout() vdpa: octeon_ep: add missing MODULE_DEVICE_TABLE() ...
-rw-r--r--MAINTAINERS2
-rw-r--r--drivers/crypto/virtio/virtio_crypto_akcipher_algs.c6
-rw-r--r--drivers/crypto/virtio/virtio_crypto_skcipher_algs.c3
-rw-r--r--drivers/nvdimm/nd_virtio.c265
-rw-r--r--drivers/nvdimm/pmem.c51
-rw-r--r--drivers/nvdimm/region_devs.c5
-rw-r--r--drivers/nvdimm/virtio_pmem.c65
-rw-r--r--drivers/nvdimm/virtio_pmem.h22
-rw-r--r--drivers/vdpa/alibaba/eni_vdpa.c1
-rw-r--r--drivers/vdpa/mlx5/core/mr.c2
-rw-r--r--drivers/vdpa/mlx5/net/mlx5_vnet.c27
-rw-r--r--drivers/vdpa/octeon_ep/octep_vdpa_main.c5
-rw-r--r--drivers/vdpa/solidrun/snet_ctrl.c4
-rw-r--r--drivers/vdpa/vdpa_sim/vdpa_sim.c29
-rw-r--r--drivers/vdpa/vdpa_user/vduse_dev.c265
-rw-r--r--drivers/vdpa/virtio_pci/vp_vdpa.c12
-rw-r--r--drivers/vhost/iotlb.c6
-rw-r--r--drivers/vhost/net.c3
-rw-r--r--drivers/vhost/vhost.c4
-rw-r--r--drivers/virtio/virtio.c41
-rw-r--r--drivers/virtio/virtio_balloon.c51
-rw-r--r--drivers/virtio/virtio_dma_buf.c2
-rw-r--r--drivers/virtio/virtio_mem.c10
-rw-r--r--drivers/virtio/virtio_mmio.c13
-rw-r--r--drivers/virtio/virtio_pci_common.c2
-rw-r--r--drivers/virtio/virtio_ring.c8
-rw-r--r--drivers/virtio/virtio_rtc_driver.c14
-rw-r--r--include/linux/libnvdimm.h9
-rw-r--r--include/linux/virtio.h1
-rw-r--r--include/linux/virtio_dma_buf.h2
-rw-r--r--include/uapi/linux/vduse.h28
-rw-r--r--include/uapi/linux/virtio_ids.h1
-rw-r--r--lib/iov_iter.c1
-rw-r--r--net/vmw_vsock/virtio_transport_common.c13
-rw-r--r--tools/virtio/vhost_net_test.c5
-rw-r--r--tools/virtio/virtio-trace/trace-agent-ctl.c2
-rw-r--r--tools/virtio/vringh_test.c2
37 files changed, 760 insertions, 222 deletions
diff --git a/MAINTAINERS b/MAINTAINERS
index a830d3b252e2..b7614cc2c752 100644
--- a/MAINTAINERS
+++ b/MAINTAINERS
@@ -15032,7 +15032,6 @@ F: lib/tests/list-test.c
LITEX PLATFORM
M: Karol Gugala <kgugala@antmicro.com>
M: Mateusz Holenko <mholenko@antmicro.com>
-M: Gabriel Somlo <gsomlo@gmail.com>
M: Joel Stanley <joel@jms.id.au>
S: Maintained
F: Documentation/devicetree/bindings/*/litex,*.yaml
@@ -21987,7 +21986,6 @@ S: Maintained
F: drivers/net/ipa/
QEMU MACHINE EMULATOR AND VIRTUALIZER SUPPORT
-M: Gabriel Somlo <somlo@cmu.edu>
M: "Michael S. Tsirkin" <mst@redhat.com>
L: qemu-devel@nongnu.org
S: Maintained
diff --git a/drivers/crypto/virtio/virtio_crypto_akcipher_algs.c b/drivers/crypto/virtio/virtio_crypto_akcipher_algs.c
index d8d452cac391..9078f22978b7 100644
--- a/drivers/crypto/virtio/virtio_crypto_akcipher_algs.c
+++ b/drivers/crypto/virtio/virtio_crypto_akcipher_algs.c
@@ -88,7 +88,8 @@ static void virtio_crypto_dataq_akcipher_callback(struct virtio_crypto_request *
}
/* actual length may be less than dst buffer */
- akcipher_req->dst_len = len - sizeof(vc_req->status);
+ akcipher_req->dst_len = min_t(unsigned int, len - sizeof(vc_req->status),
+ akcipher_req->dst_len);
sg_copy_from_buffer(akcipher_req->dst, sg_nents(akcipher_req->dst),
vc_akcipher_req->dst_buf, akcipher_req->dst_len);
virtio_crypto_akcipher_finalize_req(vc_akcipher_req, akcipher_req, error);
@@ -194,7 +195,8 @@ static int virtio_crypto_alg_akcipher_close_session(struct virtio_crypto_akciphe
if (ctrl_status->status != VIRTIO_CRYPTO_OK) {
pr_err("virtio_crypto: Close session failed status: %u, session_id: 0x%llx\n",
- ctrl_status->status, destroy_session->session_id);
+ ctrl_status->status,
+ le64_to_cpu(destroy_session->session_id));
err = -EINVAL;
goto out;
}
diff --git a/drivers/crypto/virtio/virtio_crypto_skcipher_algs.c b/drivers/crypto/virtio/virtio_crypto_skcipher_algs.c
index e82fc16cab25..3ca441ae2759 100644
--- a/drivers/crypto/virtio/virtio_crypto_skcipher_algs.c
+++ b/drivers/crypto/virtio/virtio_crypto_skcipher_algs.c
@@ -232,7 +232,8 @@ static int virtio_crypto_alg_skcipher_close_session(
if (ctrl_status->status != VIRTIO_CRYPTO_OK) {
pr_err("virtio_crypto: Close session failed status: %u, session_id: 0x%llx\n",
- ctrl_status->status, destroy_session->session_id);
+ ctrl_status->status,
+ le64_to_cpu(destroy_session->session_id));
err = -EINVAL;
goto out;
diff --git a/drivers/nvdimm/nd_virtio.c b/drivers/nvdimm/nd_virtio.c
index 4176046627be..3b8be79a20a0 100644
--- a/drivers/nvdimm/nd_virtio.c
+++ b/drivers/nvdimm/nd_virtio.c
@@ -9,26 +9,130 @@
#include "virtio_pmem.h"
#include "nd.h"
+struct virtio_pmem_flush_work {
+ struct work_struct work;
+ struct nd_region *nd_region;
+ struct bio *bio;
+};
+
+static void virtio_pmem_req_release(struct kref *kref)
+{
+ struct virtio_pmem_request *req;
+
+ req = container_of(kref, struct virtio_pmem_request, kref);
+ kfree(req);
+}
+
+static void virtio_pmem_signal_done(struct virtio_pmem_request *req)
+{
+ /* Pairs with smp_load_acquire() in virtio_pmem_req_done(). */
+ smp_store_release(&req->done, true);
+ wake_up(&req->host_acked);
+}
+
+static bool virtio_pmem_req_done(struct virtio_pmem_request *req)
+{
+ /* Pairs with smp_store_release() in virtio_pmem_signal_done(). */
+ return smp_load_acquire(&req->done);
+}
+
+static void virtio_pmem_complete_err(struct virtio_pmem_request *req)
+{
+ req->resp.ret = cpu_to_le32(1);
+ virtio_pmem_signal_done(req);
+}
+
+static void virtio_pmem_wake_one_waiter(struct virtio_pmem *vpmem)
+{
+ struct virtio_pmem_request *req_buf;
+
+ if (list_empty(&vpmem->req_list))
+ return;
+
+ req_buf = list_first_entry(&vpmem->req_list,
+ struct virtio_pmem_request, list);
+ list_del_init(&req_buf->list);
+ WRITE_ONCE(req_buf->wq_buf_avail, true);
+ wake_up(&req_buf->wq_buf);
+}
+
+static void virtio_pmem_wake_all_waiters(struct virtio_pmem *vpmem)
+{
+ struct virtio_pmem_request *req, *tmp;
+
+ list_for_each_entry_safe(req, tmp, &vpmem->req_list, list) {
+ list_del_init(&req->list);
+ WRITE_ONCE(req->wq_buf_avail, true);
+ wake_up(&req->wq_buf);
+ }
+}
+
+static void virtio_pmem_clear_inflight(struct virtio_pmem *vpmem,
+ struct virtio_pmem_request *req)
+{
+ if (vpmem->req_inflight == req)
+ vpmem->req_inflight = NULL;
+}
+
+static void virtio_pmem_wake_inflight(struct virtio_pmem *vpmem)
+{
+ struct virtio_pmem_request *req = vpmem->req_inflight;
+
+ if (req)
+ wake_up(&req->host_acked);
+}
+
+void virtio_pmem_mark_broken(struct virtio_pmem *vpmem)
+{
+ if (!READ_ONCE(vpmem->broken)) {
+ WRITE_ONCE(vpmem->broken, true);
+ dev_err_once(&vpmem->vdev->dev, "virtqueue is broken\n");
+ }
+
+ virtio_pmem_wake_inflight(vpmem);
+ virtio_pmem_wake_all_waiters(vpmem);
+}
+EXPORT_SYMBOL_GPL(virtio_pmem_mark_broken);
+
+void virtio_pmem_drain(struct virtio_pmem *vpmem)
+{
+ struct virtio_pmem_request *req;
+ unsigned int len;
+
+ if (!vpmem->req_vq)
+ return;
+
+ while ((req = virtqueue_get_buf(vpmem->req_vq, &len)) != NULL) {
+ virtio_pmem_clear_inflight(vpmem, req);
+ virtio_pmem_complete_err(req);
+ kref_put(&req->kref, virtio_pmem_req_release);
+ }
+
+ while ((req = virtqueue_detach_unused_buf(vpmem->req_vq)) != NULL) {
+ virtio_pmem_clear_inflight(vpmem, req);
+ virtio_pmem_complete_err(req);
+ kref_put(&req->kref, virtio_pmem_req_release);
+ }
+}
+EXPORT_SYMBOL_GPL(virtio_pmem_drain);
+
/* The interrupt handler */
void virtio_pmem_host_ack(struct virtqueue *vq)
{
struct virtio_pmem *vpmem = vq->vdev->priv;
- struct virtio_pmem_request *req_data, *req_buf;
+ struct virtio_pmem_request *req_data;
unsigned long flags;
unsigned int len;
spin_lock_irqsave(&vpmem->pmem_lock, flags);
while ((req_data = virtqueue_get_buf(vq, &len)) != NULL) {
- req_data->done = true;
- wake_up(&req_data->host_acked);
-
- if (!list_empty(&vpmem->req_list)) {
- req_buf = list_first_entry(&vpmem->req_list,
- struct virtio_pmem_request, list);
- req_buf->wq_buf_avail = true;
- wake_up(&req_buf->wq_buf);
- list_del(&req_buf->list);
- }
+ virtio_pmem_clear_inflight(vpmem, req_data);
+ virtio_pmem_wake_one_waiter(vpmem);
+ if (READ_ONCE(vpmem->broken))
+ virtio_pmem_complete_err(req_data);
+ else
+ virtio_pmem_signal_done(req_data);
+ kref_put(&req_data->kref, virtio_pmem_req_release);
}
spin_unlock_irqrestore(&vpmem->pmem_lock, flags);
}
@@ -55,11 +159,15 @@ static int virtio_pmem_flush(struct nd_region *nd_region)
return -EIO;
}
- req_data = kmalloc_obj(*req_data);
+ if (READ_ONCE(vpmem->broken))
+ return -EIO;
+
+ req_data = kmalloc_obj(*req_data, GFP_NOIO);
if (!req_data)
return -ENOMEM;
- req_data->done = false;
+ kref_init(&req_data->kref);
+ WRITE_ONCE(req_data->done, false);
init_waitqueue_head(&req_data->host_acked);
init_waitqueue_head(&req_data->wq_buf);
INIT_LIST_HEAD(&req_data->list);
@@ -70,67 +178,132 @@ static int virtio_pmem_flush(struct nd_region *nd_region)
sgs[1] = &ret;
spin_lock_irqsave(&vpmem->pmem_lock, flags);
- /*
- * If virtqueue_add_sgs returns -ENOSPC then req_vq virtual
- * queue does not have free descriptor. We add the request
- * to req_list and wait for host_ack to wake us up when free
- * slots are available.
- */
- while ((err = virtqueue_add_sgs(vpmem->req_vq, sgs, 1, 1, req_data,
- GFP_ATOMIC)) == -ENOSPC) {
-
- dev_info(&vdev->dev, "failed to send command to virtio pmem device, no free slots in the virtqueue\n");
- req_data->wq_buf_avail = false;
+ /*
+ * If virtqueue_add_sgs returns -ENOSPC then req_vq virtual
+ * queue does not have free descriptor. We add the request
+ * to req_list and wait for host_ack to wake us up when free
+ * slots are available.
+ */
+ for (;;) {
+ if (READ_ONCE(vpmem->broken)) {
+ err = -EIO;
+ break;
+ }
+
+ err = virtqueue_add_sgs(vpmem->req_vq, sgs, 1, 1, req_data,
+ GFP_ATOMIC);
+ if (!err) {
+ /*
+ * Take the virtqueue reference while @pmem_lock is
+ * held so completion cannot run concurrently.
+ */
+ kref_get(&req_data->kref);
+ vpmem->req_inflight = req_data;
+ break;
+ }
+
+ if (err != -ENOSPC)
+ break;
+
+ dev_info_ratelimited(&vdev->dev,
+ "failed to send command to virtio pmem device, no free slots in the virtqueue\n");
+ WRITE_ONCE(req_data->wq_buf_avail, false);
list_add_tail(&req_data->list, &vpmem->req_list);
spin_unlock_irqrestore(&vpmem->pmem_lock, flags);
/* A host response results in "host_ack" getting called */
- wait_event(req_data->wq_buf, req_data->wq_buf_avail);
+ wait_event(req_data->wq_buf,
+ READ_ONCE(req_data->wq_buf_avail) ||
+ READ_ONCE(vpmem->broken));
spin_lock_irqsave(&vpmem->pmem_lock, flags);
+
+ if (READ_ONCE(vpmem->broken))
+ break;
+ }
+
+ if (READ_ONCE(vpmem->broken))
+ err = -EIO;
+ if (err == -EIO || virtqueue_is_broken(vpmem->req_vq))
+ virtio_pmem_mark_broken(vpmem);
+
+ err1 = true;
+ if (!err && !READ_ONCE(vpmem->broken)) {
+ err1 = virtqueue_kick(vpmem->req_vq);
+ if (!err1)
+ virtio_pmem_mark_broken(vpmem);
}
- err1 = virtqueue_kick(vpmem->req_vq);
spin_unlock_irqrestore(&vpmem->pmem_lock, flags);
/*
* virtqueue_add_sgs failed with error different than -ENOSPC, we can't
* do anything about that.
*/
- if (err || !err1) {
+ if (READ_ONCE(vpmem->broken) || err || !err1) {
dev_info(&vdev->dev, "failed to send command to virtio pmem device\n");
err = -EIO;
} else {
/* A host response results in "host_ack" getting called */
- wait_event(req_data->host_acked, req_data->done);
- err = le32_to_cpu(req_data->resp.ret);
+ wait_event(req_data->host_acked,
+ virtio_pmem_req_done(req_data) ||
+ READ_ONCE(vpmem->broken));
+ if (virtio_pmem_req_done(req_data))
+ err = le32_to_cpu(req_data->resp.ret);
+ else
+ err = -EIO;
}
- kfree(req_data);
+ kref_put(&req_data->kref, virtio_pmem_req_release);
return err;
};
+static void virtio_pmem_flush_work(struct work_struct *work)
+{
+ struct virtio_pmem_flush_work *flush;
+ int err;
+
+ flush = container_of(work, struct virtio_pmem_flush_work, work);
+ err = virtio_pmem_flush(flush->nd_region);
+ if (err > 0)
+ err = -EIO;
+ if (err)
+ flush->bio->bi_status = errno_to_blk_status(err);
+ bio_endio(flush->bio);
+ kfree(flush);
+}
+
/* The asynchronous flush callback function */
int async_pmem_flush(struct nd_region *nd_region, struct bio *bio)
{
- /*
- * Create child bio for asynchronous flush and chain with
- * parent bio. Otherwise directly call nd_region flush.
- */
- if (bio && bio->bi_iter.bi_sector != -1) {
- struct bio *child = bio_alloc(bio->bi_bdev, 0,
- REQ_OP_WRITE | REQ_PREFLUSH,
- GFP_ATOMIC);
+ struct virtio_device *vdev = nd_region->provider_data;
+ struct virtio_pmem *vpmem = vdev->priv;
+ struct virtio_pmem_flush_work *flush;
+ unsigned long flags;
+ int err;
- if (!child)
+ if (bio && bio->bi_iter.bi_sector != -1) {
+ flush = kmalloc_obj(*flush, GFP_NOIO);
+ if (!flush)
return -ENOMEM;
- bio_clone_blkg_association(child, bio);
- child->bi_iter.bi_sector = -1;
- bio_chain(child, bio);
- submit_bio(child);
- return 0;
+
+ INIT_WORK(&flush->work, virtio_pmem_flush_work);
+ flush->nd_region = nd_region;
+ flush->bio = bio;
+
+ spin_lock_irqsave(&vpmem->pmem_lock, flags);
+ if (READ_ONCE(vpmem->broken)) {
+ spin_unlock_irqrestore(&vpmem->pmem_lock, flags);
+ kfree(flush);
+ return -EIO;
+ }
+ queue_work(vpmem->flush_wq, &flush->work);
+ spin_unlock_irqrestore(&vpmem->pmem_lock, flags);
+ return NVDIMM_FLUSH_ASYNC;
}
- if (virtio_pmem_flush(nd_region))
+
+ err = virtio_pmem_flush(nd_region);
+ if (err > 0)
return -EIO;
- return 0;
+ return err;
};
EXPORT_SYMBOL_GPL(async_pmem_flush);
MODULE_DESCRIPTION("Virtio Persistent Memory Driver");
diff --git a/drivers/nvdimm/pmem.c b/drivers/nvdimm/pmem.c
index 92c67fbbc1c8..30a51c365ce8 100644
--- a/drivers/nvdimm/pmem.c
+++ b/drivers/nvdimm/pmem.c
@@ -208,29 +208,44 @@ static void pmem_submit_bio(struct bio *bio)
struct pmem_device *pmem = bio->bi_bdev->bd_disk->private_data;
struct nd_region *nd_region = to_region(pmem);
- if (bio->bi_opf & REQ_PREFLUSH)
- ret = nvdimm_flush(nd_region, bio);
+ if (bio->bi_opf & REQ_PREFLUSH) {
+ ret = nvdimm_flush(nd_region, NULL);
+ if (ret) {
+ bio->bi_status = errno_to_blk_status(ret);
+ bio_endio(bio);
+ return;
+ }
+ }
- do_acct = blk_queue_io_stat(bio->bi_bdev->bd_disk->queue);
- if (do_acct)
- start = bio_start_io_acct(bio);
- bio_for_each_segment(bvec, bio, iter) {
- if (op_is_write(bio_op(bio)))
- rc = pmem_do_write(pmem, bvec.bv_page, bvec.bv_offset,
- iter.bi_sector, bvec.bv_len);
- else
- rc = pmem_do_read(pmem, bvec.bv_page, bvec.bv_offset,
- iter.bi_sector, bvec.bv_len);
- if (rc) {
- bio->bi_status = rc;
- break;
+ if (bio_has_data(bio)) {
+ do_acct = blk_queue_io_stat(bio->bi_bdev->bd_disk->queue);
+ if (do_acct)
+ start = bio_start_io_acct(bio);
+ bio_for_each_segment(bvec, bio, iter) {
+ if (op_is_write(bio_op(bio)))
+ rc = pmem_do_write(pmem, bvec.bv_page,
+ bvec.bv_offset,
+ iter.bi_sector,
+ bvec.bv_len);
+ else
+ rc = pmem_do_read(pmem, bvec.bv_page,
+ bvec.bv_offset,
+ iter.bi_sector,
+ bvec.bv_len);
+ if (rc) {
+ bio->bi_status = rc;
+ break;
+ }
}
+ if (do_acct)
+ bio_end_io_acct(bio, start);
}
- if (do_acct)
- bio_end_io_acct(bio, start);
- if (bio->bi_opf & REQ_FUA)
+ if ((bio->bi_opf & REQ_FUA) && !bio->bi_status) {
ret = nvdimm_flush(nd_region, bio);
+ if (ret == NVDIMM_FLUSH_ASYNC)
+ return;
+ }
if (ret)
bio->bi_status = errno_to_blk_status(ret);
diff --git a/drivers/nvdimm/region_devs.c b/drivers/nvdimm/region_devs.c
index 5e079d61cbaa..24f42b4650ba 100644
--- a/drivers/nvdimm/region_devs.c
+++ b/drivers/nvdimm/region_devs.c
@@ -1093,7 +1093,10 @@ int nvdimm_flush(struct nd_region *nd_region, struct bio *bio)
if (!nd_region->flush)
rc = generic_nvdimm_flush(nd_region);
else {
- if (nd_region->flush(nd_region, bio))
+ rc = nd_region->flush(nd_region, bio);
+ if (rc > 0)
+ return rc;
+ if (rc && rc != -ENOMEM)
rc = -EIO;
}
diff --git a/drivers/nvdimm/virtio_pmem.c b/drivers/nvdimm/virtio_pmem.c
index 77b196661905..7ee3fb1779f7 100644
--- a/drivers/nvdimm/virtio_pmem.c
+++ b/drivers/nvdimm/virtio_pmem.c
@@ -17,18 +17,34 @@ static struct virtio_device_id id_table[] = {
/* Initialize virt queue */
static int init_vq(struct virtio_pmem *vpmem)
{
+ int err;
+
/* single vq */
vpmem->req_vq = virtio_find_single_vq(vpmem->vdev,
virtio_pmem_host_ack, "flush_queue");
- if (IS_ERR(vpmem->req_vq))
- return PTR_ERR(vpmem->req_vq);
+ if (IS_ERR(vpmem->req_vq)) {
+ err = PTR_ERR(vpmem->req_vq);
+ vpmem->req_vq = NULL;
+ return err;
+ }
spin_lock_init(&vpmem->pmem_lock);
INIT_LIST_HEAD(&vpmem->req_list);
+ vpmem->req_inflight = NULL;
+ WRITE_ONCE(vpmem->broken, false);
return 0;
};
+static void virtio_pmem_del_vqs(struct virtio_pmem *vpmem)
+{
+ if (!vpmem->req_vq)
+ return;
+
+ vpmem->vdev->config->del_vqs(vpmem->vdev);
+ vpmem->req_vq = NULL;
+}
+
static int virtio_pmem_validate(struct virtio_device *vdev)
{
struct virtio_shm_region shm_reg;
@@ -67,10 +83,17 @@ static int virtio_pmem_probe(struct virtio_device *vdev)
mutex_init(&vpmem->flush_lock);
vpmem->vdev = vdev;
vdev->priv = vpmem;
+ vpmem->flush_wq = alloc_ordered_workqueue("virtio-pmem-flush",
+ WQ_MEM_RECLAIM);
+ if (!vpmem->flush_wq) {
+ err = -ENOMEM;
+ goto out_err;
+ }
+
err = init_vq(vpmem);
if (err) {
dev_err(&vdev->dev, "failed to initialize virtio pmem vq's\n");
- goto out_err;
+ goto out_wq;
}
if (virtio_has_feature(vdev, VIRTIO_PMEM_F_SHMEM_REGION)) {
@@ -130,7 +153,9 @@ out_nd:
virtio_reset_device(vdev);
nvdimm_bus_unregister(vpmem->nvdimm_bus);
out_vq:
- vdev->config->del_vqs(vdev);
+ virtio_pmem_del_vqs(vpmem);
+out_wq:
+ destroy_workqueue(vpmem->flush_wq);
out_err:
return err;
}
@@ -138,17 +163,43 @@ out_err:
static void virtio_pmem_remove(struct virtio_device *vdev)
{
struct nvdimm_bus *nvdimm_bus = dev_get_drvdata(&vdev->dev);
+ struct virtio_pmem *vpmem = vdev->priv;
+ unsigned long flags;
- nvdimm_bus_unregister(nvdimm_bus);
- vdev->config->del_vqs(vdev);
+ spin_lock_irqsave(&vpmem->pmem_lock, flags);
+ virtio_pmem_mark_broken(vpmem);
+ spin_unlock_irqrestore(&vpmem->pmem_lock, flags);
+
+ drain_workqueue(vpmem->flush_wq);
virtio_reset_device(vdev);
+
+ spin_lock_irqsave(&vpmem->pmem_lock, flags);
+ virtio_pmem_drain(vpmem);
+ spin_unlock_irqrestore(&vpmem->pmem_lock, flags);
+
+ nvdimm_bus_unregister(nvdimm_bus);
+ virtio_pmem_del_vqs(vpmem);
+ destroy_workqueue(vpmem->flush_wq);
}
static int virtio_pmem_freeze(struct virtio_device *vdev)
{
- vdev->config->del_vqs(vdev);
+ struct virtio_pmem *vpmem = vdev->priv;
+ unsigned long flags;
+
+ spin_lock_irqsave(&vpmem->pmem_lock, flags);
+ virtio_pmem_mark_broken(vpmem);
+ spin_unlock_irqrestore(&vpmem->pmem_lock, flags);
+
+ drain_workqueue(vpmem->flush_wq);
virtio_reset_device(vdev);
+ spin_lock_irqsave(&vpmem->pmem_lock, flags);
+ virtio_pmem_drain(vpmem);
+ spin_unlock_irqrestore(&vpmem->pmem_lock, flags);
+
+ virtio_pmem_del_vqs(vpmem);
+
return 0;
}
diff --git a/drivers/nvdimm/virtio_pmem.h b/drivers/nvdimm/virtio_pmem.h
index f72cf17f9518..0b90777d7658 100644
--- a/drivers/nvdimm/virtio_pmem.h
+++ b/drivers/nvdimm/virtio_pmem.h
@@ -10,15 +10,17 @@
#ifndef _LINUX_VIRTIO_PMEM_H
#define _LINUX_VIRTIO_PMEM_H
+#include <linux/dma-mapping.h>
#include <linux/module.h>
#include <uapi/linux/virtio_pmem.h>
+#include <linux/kref.h>
#include <linux/libnvdimm.h>
#include <linux/mutex.h>
#include <linux/spinlock.h>
+#include <linux/workqueue.h>
struct virtio_pmem_request {
- struct virtio_pmem_req req;
- struct virtio_pmem_resp resp;
+ struct kref kref;
/* Wait queue to process deferred work after ack from host */
wait_queue_head_t host_acked;
@@ -28,6 +30,11 @@ struct virtio_pmem_request {
wait_queue_head_t wq_buf;
bool wq_buf_avail;
struct list_head list;
+
+ struct virtio_pmem_req req;
+ __dma_from_device_group_begin(resp);
+ struct virtio_pmem_resp resp;
+ __dma_from_device_group_end(resp);
};
struct virtio_pmem {
@@ -39,6 +46,9 @@ struct virtio_pmem {
/* Serialize flush requests to the device. */
struct mutex flush_lock;
+ /* Complete asynchronous FUA flushes outside the submit path. */
+ struct workqueue_struct *flush_wq;
+
/* nvdimm bus registers virtio pmem device */
struct nvdimm_bus *nvdimm_bus;
struct nvdimm_bus_descriptor nd_desc;
@@ -46,6 +56,12 @@ struct virtio_pmem {
/* List to store deferred work if virtqueue is full */
struct list_head req_list;
+ /* Request currently owned by the virtqueue. */
+ struct virtio_pmem_request *req_inflight;
+
+ /* Fail fast and wake waiters if the request virtqueue is broken. */
+ bool broken;
+
/* Synchronize virtqueue data */
spinlock_t pmem_lock;
@@ -55,5 +71,7 @@ struct virtio_pmem {
};
void virtio_pmem_host_ack(struct virtqueue *vq);
+void virtio_pmem_mark_broken(struct virtio_pmem *vpmem);
+void virtio_pmem_drain(struct virtio_pmem *vpmem);
int async_pmem_flush(struct nd_region *nd_region, struct bio *bio);
#endif
diff --git a/drivers/vdpa/alibaba/eni_vdpa.c b/drivers/vdpa/alibaba/eni_vdpa.c
index e476504db0c8..fd6fdba46094 100644
--- a/drivers/vdpa/alibaba/eni_vdpa.c
+++ b/drivers/vdpa/alibaba/eni_vdpa.c
@@ -545,6 +545,7 @@ static struct pci_device_id eni_pci_ids[] = {
VIRTIO_ID_NET) },
{ 0 },
};
+MODULE_DEVICE_TABLE(pci, eni_pci_ids);
static struct pci_driver eni_vdpa_driver = {
.name = "alibaba-eni-vdpa",
diff --git a/drivers/vdpa/mlx5/core/mr.c b/drivers/vdpa/mlx5/core/mr.c
index 77a479aeaa85..b0c5ff23d022 100644
--- a/drivers/vdpa/mlx5/core/mr.c
+++ b/drivers/vdpa/mlx5/core/mr.c
@@ -481,7 +481,7 @@ static int add_direct_chain(struct mlx5_vdpa_dev *mvdev,
return 0;
err_alloc:
- list_for_each_entry_safe(dmr, n, &mr->head, list) {
+ list_for_each_entry_safe(dmr, n, &tmp, list) {
list_del_init(&dmr->list);
unmap_direct_mr(mvdev, dmr);
kfree(dmr);
diff --git a/drivers/vdpa/mlx5/net/mlx5_vnet.c b/drivers/vdpa/mlx5/net/mlx5_vnet.c
index ad0d5fbbbca8..8563fec2855d 100644
--- a/drivers/vdpa/mlx5/net/mlx5_vnet.c
+++ b/drivers/vdpa/mlx5/net/mlx5_vnet.c
@@ -1080,7 +1080,7 @@ static void alloc_inout(struct mlx5_vdpa_net *ndev, int cmd, void **in, int *inl
MLX5_SET(init2rtr_qp_in, *in, opcode, cmd);
MLX5_SET(init2rtr_qp_in, *in, uid, ndev->mvdev.res.uid);
MLX5_SET(init2rtr_qp_in, *in, qpn, qpn);
- qpc = MLX5_ADDR_OF(rst2init_qp_in, *in, qpc);
+ qpc = MLX5_ADDR_OF(init2rtr_qp_in, *in, qpc);
MLX5_SET(qpc, qpc, mtu, MLX5_QPC_MTU_256_BYTES);
MLX5_SET(qpc, qpc, log_msg_max, 30);
MLX5_SET(qpc, qpc, remote_qpn, rqpn);
@@ -1098,7 +1098,7 @@ static void alloc_inout(struct mlx5_vdpa_net *ndev, int cmd, void **in, int *inl
MLX5_SET(rtr2rts_qp_in, *in, opcode, cmd);
MLX5_SET(rtr2rts_qp_in, *in, uid, ndev->mvdev.res.uid);
MLX5_SET(rtr2rts_qp_in, *in, qpn, qpn);
- qpc = MLX5_ADDR_OF(rst2init_qp_in, *in, qpc);
+ qpc = MLX5_ADDR_OF(rtr2rts_qp_in, *in, qpc);
pp = MLX5_ADDR_OF(qpc, qpc, primary_address_path);
MLX5_SET(ads, pp, ack_timeout, 14);
MLX5_SET(qpc, qpc, retry_count, 7);
@@ -3055,18 +3055,24 @@ static int mlx5_vdpa_change_map(struct mlx5_vdpa_dev *mvdev,
unsigned int asid)
{
struct mlx5_vdpa_net *ndev = to_mlx5_vdpa_ndev(mvdev);
+ struct mlx5_vdpa_mr *old_mr;
bool teardown = !is_resumable(ndev);
int err;
suspend_vqs(ndev, 0, ndev->cur_num_vqs);
if (teardown) {
err = save_channels_info(ndev);
- if (err)
+ if (err) {
+ mlx5_vdpa_put_mr(mvdev, new_mr);
return err;
+ }
teardown_vq_resources(ndev);
}
+ /* Keep the old MR alive in case rebuilding the VQs fails. */
+ old_mr = mvdev->mres.mr[asid];
+ mlx5_vdpa_get_mr(mvdev, old_mr);
mlx5_vdpa_update_mr(mvdev, new_mr, asid);
for (int i = 0; i < mvdev->max_vqs; i++)
@@ -3074,17 +3080,22 @@ static int mlx5_vdpa_change_map(struct mlx5_vdpa_dev *mvdev,
MLX5_VIRTQ_MODIFY_MASK_DESC_GROUP_MKEY;
if (!(mvdev->status & VIRTIO_CONFIG_S_DRIVER_OK) || mvdev->suspended)
- return 0;
+ goto out;
if (teardown) {
restore_channels_info(ndev);
err = setup_vq_resources(ndev, true);
- if (err)
+ if (err) {
+ /* The saved reference becomes the restored map reference. */
+ mlx5_vdpa_update_mr(mvdev, old_mr, asid);
return err;
+ }
}