summaryrefslogtreecommitdiffstats
path: root/lib/timerqueue.c
diff options
context:
space:
mode:
authorNiklas Cassel <cassel@kernel.org>2026-09-18 16:06:42 +0200
committerJens Axboe <axboe@kernel.dk>2026-09-29 12:54:47 -0600
commit684b413b5483f57c890c171b9400076a0143b918 (patch)
treef167daddad3dce9af44773541373eafbccc9ac61 /lib/timerqueue.c
downloadlinux-stable-684b413b5483f57c890c171b9400076a0143b918.tar.gz
linux-stable-684b413b5483f57c890c171b9400076a0143b918.zip
virtio_blk: set the zone write granularitygrafted
virtblk_read_zoned_limits() reads the write granularity that the device reports in virtio_blk_zoned_characteristics and assigns it to the physical block size and to io_min, but never to the limit that is named after it. queue_limits.zone_write_granularity is left at zero, so blk_validate_zoned_limits() raises it to the logical block size: if (lim->zone_write_granularity < lim->logical_block_size) lim->zone_write_granularity = lim->logical_block_size; A device that reports a granularity coarser than its logical block size, which is what the field exists to express, therefore has it silently reduced. A 512e host managed disk passed through to a guest reports a logical block size of 512 and a write granularity of 4096, and the guest ends up with a zone write granularity of 512. bio_split_alignment() returns lim->zone_write_granularity if it is non-zero and bio_split_io_at() may split a bio with as per bio_split_alignment(). This can real to the write getting rejected by the host drive, as the write is not aligned to the physical block size. zonefs also takes its block size from bdev_zone_write_granularity(), so it would incorrectly use 512 on a disk that requires 4096. sd_zbc_read_zones() sets the limit from the physical block size for the same reason. NVMe ZNS and null_blk leave it unset, but the fallback gives the right answer for them, as their write granularity is the logical block size. virtio carries a separate value that may exceed it. Set the zone write granularity from the value that the device reports. Fixes: 95bfec41bd3d ("virtio-blk: add support for zoned block devices") Signed-off-by: Niklas Cassel <cassel@kernel.org> Reviewed-by: Stefan Hajnoczi <stefanha@redhat.com> Link: https://patch.msgid.link/20260918140641.2031075-2-cassel@kernel.org Signed-off-by: Jens Axboe <axboe@kernel.dk>
Diffstat (limited to 'lib/timerqueue.c')
-rw-r--r--lib/timerqueue.c98
1 files changed, 98 insertions, 0 deletions
diff --git a/lib/timerqueue.c b/lib/timerqueue.c
new file mode 100644
index 000000000..e2a1e08cb
--- /dev/null
+++ b/lib/timerqueue.c
@@ -0,0 +1,98 @@
+// SPDX-License-Identifier: GPL-2.0-or-later
+/*
+ * Generic Timer-queue
+ *
+ * Manages a simple queue of timers, ordered by expiration time.
+ * Uses rbtrees for quick list adds and expiration.
+ *
+ * NOTE: All of the following functions need to be serialized
+ * to avoid races. No locking is done by this library code.
+ */
+
+#include <linux/bug.h>
+#include <linux/timerqueue.h>
+#include <linux/rbtree.h>
+#include <linux/export.h>
+
+#define __node_2_tq(_n) \
+ rb_entry((_n), struct timerqueue_node, node)
+
+static inline bool __timerqueue_less(struct rb_node *a, const struct rb_node *b)
+{
+ return __node_2_tq(a)->expires < __node_2_tq(b)->expires;
+}
+
+/**
+ * timerqueue_add - Adds timer to timerqueue.
+ *
+ * @head: head of timerqueue
+ * @node: timer node to be added
+ *
+ * Adds the timer node to the timerqueue, sorted by the node's expires
+ * value. Returns true if the newly added timer is the first expiring timer in
+ * the queue.
+ */
+bool timerqueue_add(struct timerqueue_head *head, struct timerqueue_node *node)
+{
+ /* Make sure we don't add nodes that are already added */
+ WARN_ON_ONCE(!RB_EMPTY_NODE(&node->node));
+
+ return rb_add_cached(&node->node, &head->rb_root, __timerqueue_less);
+}
+EXPORT_SYMBOL_GPL(timerqueue_add);
+
+/**
+ * timerqueue_del - Removes a timer from the timerqueue.
+ *
+ * @head: head of timerqueue
+ * @node: timer node to be removed
+ *
+ * Removes the timer node from the timerqueue. Returns true if the queue is
+ * not empty after the remove.
+ */
+bool timerqueue_del(struct timerqueue_head *head, struct timerqueue_node *node)
+{
+ WARN_ON_ONCE(RB_EMPTY_NODE(&node->node));
+
+ rb_erase_cached(&node->node, &head->rb_root);
+ RB_CLEAR_NODE(&node->node);
+
+ return !RB_EMPTY_ROOT(&head->rb_root.rb_root);
+}
+EXPORT_SYMBOL_GPL(timerqueue_del);
+
+/**
+ * timerqueue_iterate_next - Returns the timer after the provided timer
+ *
+ * @node: Pointer to a timer.
+ *
+ * Provides the timer that is after the given node. This is used, when
+ * necessary, to iterate through the list of timers in a timer list
+ * without modifying the list.
+ */
+struct timerqueue_node *timerqueue_iterate_next(struct timerqueue_node *node)
+{
+ struct rb_node *next;
+
+ if (!node)
+ return NULL;
+ next = rb_next(&node->node);
+ if (!next)
+ return NULL;
+ return container_of(next, struct timerqueue_node, node);
+}
+EXPORT_SYMBOL_GPL(timerqueue_iterate_next);
+
+#define __node_2_tq_linked(_n) \
+ container_of(rb_entry((_n), struct rb_node_linked, node), struct timerqueue_linked_node, node)
+
+static __always_inline bool __tq_linked_less(struct rb_node *a, const struct rb_node *b)
+{
+ return __node_2_tq_linked(a)->expires < __node_2_tq_linked(b)->expires;
+}
+
+bool timerqueue_linked_add(struct timerqueue_linked_head *head, struct timerqueue_linked_node *node)
+{
+ return rb_add_linked(&node->node, &head->rb_root, __tq_linked_less);
+}
+EXPORT_SYMBOL_GPL(timerqueue_linked_add);