summaryrefslogtreecommitdiffstats
path: root/block
diff options
context:
space:
mode:
authorLinus Torvalds <torvalds@linux-foundation.org>2026-10-02 12:25:40 -0700
committerLinus Torvalds <torvalds@linux-foundation.org>2026-10-02 12:25:40 -0700
commit3b7cab693ba2bab63774bf5b988e8a61b2ef0f32 (patch)
tree90e364f2004af7eb6c8ff0c205ffd15fd0c2144b /block
parent3f1fe48a36b0b6722dc3fd421d93512bac138e9a (diff)
parent684b413b5483f57c890c171b9400076a0143b918 (diff)
downloadlinux-stable-3b7cab693ba2bab63774bf5b988e8a61b2ef0f32.tar.gz
linux-stable-3b7cab693ba2bab63774bf5b988e8a61b2ef0f32.zip
Merge tag 'block-7.3-20261002' of git://git.kernel.org/pub/scm/linux/kernel/git/axboe/linux
Pull block fixes from Jens Axboe: - NVMe fixes via Keith: - Fix an out-of-bounds write in nvmet_auth_challenge(), where sizeof() on a void pointer undercounted the challenge header and let a short AUTH_RECEIVE buffer pass the check - nvme-multipath fixes for an ANA log bounds check underflow, the command effects log lifetime for multipath heads, and only setting BLK_FEAT_ZONED after the zone info is known. - nvmet fixes for ns->enabled teardown ordering, rejecting I/O after the percpu ns reference is killed, device path preservation on allocation failure, and too-short SGL segments in pci-epf - nvme-tcp: revert the per-socket dynamic lockdep keys, and delay the socket reclassification - A DMA pool alignment quirk for the Micron 4100AT - Controller state/reset race fixes, and -Wformat-security workarounds - blk-mq: set RQF_USE_SCHED when the operation is known, and allow cached requests to be used for flush operations - Reject polled dio with user integrity metadata - Save the IRQ state in blkg_tryget_closest() - Set the zone write granularity in virtio_blk - ublk selftest fixes * tag 'block-7.3-20261002' of git://git.kernel.org/pub/scm/linux/kernel/git/axboe/linux: (23 commits) virtio_blk: set the zone write granularity nvme-multipath: set BLK_FEAT_ZONED only after the zone info is known nvme: fix command effects log lifetime for multipath heads nvmet: don't allow I/O admission after percpu ns reference is killed nvmet: defer setting ns->enabled to false in nvmet_ns_disable() nvmet: copy the hostid into the ctrl before creating PR pc_refs nvmet-auth: fix out-of-bounds write in nvmet_auth_challenge() nvmet: pci-epf: reject too-short SGL segments nvme-multipath: fix underflow in ANA log bounds checks nvme: work around all -Wformat-security warnings nvme: work around -Wformat-security warning nvme: do not reset controllers in NVME_CTRL_NEW state nvme-tcp: delay nvme_tcp_reclassify_socket() Revert "nvme-tcp: lockdep: use dynamic lockdep keys per socket instance" drbd: remove unused drbd_nl_mcgrps[] array blk-mq: allow cached requests to be used for flush operations blk-mq: set RQF_USE_SCHED when the operation is known block: reject polled dio with user integrity metadata selftests: ublk: fix unused_result error blk-cgroup: save IRQ state in blkg_tryget_closest() ...
Diffstat (limited to 'block')
-rw-r--r--block/blk-cgroup.c5
-rw-r--r--block/blk-mq.c63
-rw-r--r--block/fops.c5
-rw-r--r--block/mq-deadline.c2
4 files changed, 46 insertions, 29 deletions
diff --git a/block/blk-cgroup.c b/block/blk-cgroup.c
index 2b5c29434..b56db1cc6 100644
--- a/block/blk-cgroup.c
+++ b/block/blk-cgroup.c
@@ -2091,6 +2091,7 @@ static inline struct blkcg_gq *blkg_tryget_closest(struct bio *bio,
struct request_queue *q = bio->bi_bdev->bd_queue;
struct blkcg *blkcg = css_to_blkcg(css);
struct blkcg_gq *blkg;
+ unsigned long flags;
rcu_read_lock();
blkg = blkg_lookup(blkcg, q);
@@ -2105,11 +2106,11 @@ static inline struct blkcg_gq *blkg_tryget_closest(struct bio *bio,
* Fast path failed, we're probably issuing IO in this cgroup the first
* time, hold lock to create new blkg.
*/
- spin_lock_irq(&q->queue_lock);
+ spin_lock_irqsave(&q->queue_lock, flags);
blkg = blkg_lookup_create(blkcg, bio->bi_bdev->bd_disk);
if (blkg)
blkg = blkg_lookup_tryget(blkg);
- spin_unlock_irq(&q->queue_lock);
+ spin_unlock_irqrestore(&q->queue_lock, flags);
return blkg;
}
diff --git a/block/blk-mq.c b/block/blk-mq.c
index a26a11c73..025a799f3 100644
--- a/block/blk-mq.c
+++ b/block/blk-mq.c
@@ -447,16 +447,6 @@ static struct request *blk_mq_rq_ctx_init(struct blk_mq_alloc_data *data,
WRITE_ONCE(rq->deadline, 0);
req_ref_set(rq, 1);
- if (rq->rq_flags & RQF_USE_SCHED) {
- struct elevator_queue *e = data->q->elevator;
-
- INIT_HLIST_NODE(&rq->hash);
- RB_CLEAR_NODE(&rq->rb_node);
-
- if (e->type->ops.prepare_request)
- e->type->ops.prepare_request(rq);
- }
-
return rq;
}
@@ -498,6 +488,12 @@ __blk_mq_alloc_requests_batch(struct blk_mq_alloc_data *data)
return rq_list_pop(data->cached_rqs);
}
+static bool blk_op_bypass_sched(blk_opf_t opf)
+{
+ return (opf & REQ_OP_MASK) == REQ_OP_FLUSH ||
+ blk_op_is_passthrough(opf);
+}
+
static void blk_mq_limit_depth(struct blk_mq_alloc_data *data)
{
struct elevator_mq_ops *ops;
@@ -518,12 +514,10 @@ static void blk_mq_limit_depth(struct blk_mq_alloc_data *data)
* Flush/passthrough requests are special and go directly to the
* dispatch list, they are not subject to the async_depth limit.
*/
- if ((data->cmd_flags & REQ_OP_MASK) == REQ_OP_FLUSH ||
- blk_op_is_passthrough(data->cmd_flags))
+ if (blk_op_bypass_sched(data->cmd_flags))
return;
WARN_ON_ONCE(data->flags & BLK_MQ_REQ_RESERVED);
- data->rq_flags |= RQF_USE_SCHED;
/*
* By default, sync requests have no limit, and async requests are
@@ -534,6 +528,29 @@ static void blk_mq_limit_depth(struct blk_mq_alloc_data *data)
ops->limit_depth(data->cmd_flags, data);
}
+/*
+ * Finish initializing a request once it has been claimed for an operation.
+ * Cached requests are allocated before that operation is known.
+ */
+static void blk_mq_rq_late_init(struct request *rq, u64 alloc_time_ns)
+{
+ struct elevator_queue *e;
+
+ blk_mq_rq_time_init(rq, alloc_time_ns);
+
+ if (!(rq->rq_flags & RQF_SCHED_TAGS) || (rq->rq_flags & RQF_RESV) ||
+ blk_op_bypass_sched(rq->cmd_flags))
+ return;
+
+ rq->rq_flags |= RQF_USE_SCHED;
+ INIT_HLIST_NODE(&rq->hash);
+ RB_CLEAR_NODE(&rq->rb_node);
+
+ e = rq->q->elevator;
+ if (e->type->ops.prepare_request)
+ e->type->ops.prepare_request(rq);
+}
+
static struct request *__blk_mq_alloc_requests(struct blk_mq_alloc_data *data)
{
struct request_queue *q = data->q;
@@ -562,7 +579,7 @@ retry:
if (data->nr_tags > 1) {
rq = __blk_mq_alloc_requests_batch(data);
if (rq) {
- blk_mq_rq_time_init(rq, alloc_time_ns);
+ blk_mq_rq_late_init(rq, alloc_time_ns);
return rq;
}
data->nr_tags = 1;
@@ -590,7 +607,7 @@ retry:
if (!(data->rq_flags & RQF_SCHED_TAGS))
blk_mq_inc_active_requests(data->hctx);
rq = blk_mq_rq_ctx_init(data, blk_mq_tags_from_data(data), tag);
- blk_mq_rq_time_init(rq, alloc_time_ns);
+ blk_mq_rq_late_init(rq, alloc_time_ns);
return rq;
}
@@ -637,8 +654,6 @@ static struct request *blk_mq_alloc_cached_request(struct request_queue *q,
if (plug->nr_ios == 1)
return NULL;
rq = blk_mq_rq_cache_fill(q, plug, opf, flags);
- if (!rq)
- return NULL;
} else {
rq = rq_list_peek(&plug->cached_rqs);
if (!rq || rq->q != q)
@@ -646,15 +661,13 @@ static struct request *blk_mq_alloc_cached_request(struct request_queue *q,
if (blk_mq_get_hctx_type(opf) != rq->mq_hctx->type)
return NULL;
- if (op_is_flush(rq->cmd_flags) != op_is_flush(opf))
- return NULL;
rq_list_pop(&plug->cached_rqs);
- blk_mq_rq_time_init(rq, blk_time_get_ns());
+ rq->cmd_flags = opf;
+ INIT_LIST_HEAD(&rq->queuelist);
+ blk_mq_rq_late_init(rq, blk_time_get_ns());
}
- rq->cmd_flags = opf;
- INIT_LIST_HEAD(&rq->queuelist);
return rq;
}
@@ -766,7 +779,7 @@ struct request *blk_mq_alloc_request_hctx(struct request_queue *q,
if (!(data.rq_flags & RQF_SCHED_TAGS))
blk_mq_inc_active_requests(data.hctx);
rq = blk_mq_rq_ctx_init(&data, blk_mq_tags_from_data(&data), tag);
- blk_mq_rq_time_init(rq, alloc_time_ns);
+ blk_mq_rq_late_init(rq, alloc_time_ns);
rq->__data_len = 0;
rq->phys_gap_bit = 0;
rq->__sector = (sector_t) -1;
@@ -3060,8 +3073,6 @@ static struct request *blk_mq_get_cached_request(struct blk_plug *plug,
if (type != rq->mq_hctx->type &&
(type != HCTX_TYPE_READ || rq->mq_hctx->type != HCTX_TYPE_DEFAULT))
return NULL;
- if (op_is_flush(rq->cmd_flags) != op_is_flush(opf))
- return NULL;
rq_list_pop(&plug->cached_rqs);
return rq;
}
@@ -3163,9 +3174,9 @@ void blk_mq_submit_bio(struct bio *bio)
new_request:
if (rq) {
rq_qos_throttle(rq->q, bio);
- blk_mq_rq_time_init(rq, blk_time_get_ns());
rq->cmd_flags = bio->bi_opf;
INIT_LIST_HEAD(&rq->queuelist);
+ blk_mq_rq_late_init(rq, blk_time_get_ns());
} else {
rq = blk_mq_get_new_requests(q, plug, bio);
if (unlikely(!rq)) {
diff --git a/block/fops.c b/block/fops.c
index 2ce7c6c47..c915ec012 100644
--- a/block/fops.c
+++ b/block/fops.c
@@ -402,6 +402,11 @@ static ssize_t blkdev_direct_IO(struct kiocb *iocb, struct iov_iter *iter)
if (blkdev_dio_invalid(bdev, iocb, iter))
return -EINVAL;
+ /* HIPRI needs private as bio; HAS_METADATA keeps it as uio_meta */
+ if ((iocb->ki_flags & IOCB_HIPRI) &&
+ (iocb->ki_flags & IOCB_HAS_METADATA))
+ return -EOPNOTSUPP;
+
if (iov_iter_rw(iter) == WRITE) {
u16 max_write_streams = bdev_max_write_streams(bdev);
diff --git a/block/mq-deadline.c b/block/mq-deadline.c
index 5f643c0ce..e5db1ee09 100644
--- a/block/mq-deadline.c
+++ b/block/mq-deadline.c
@@ -685,7 +685,7 @@ static void dd_insert_requests(struct blk_mq_hw_ctx *hctx,
blk_mq_free_requests(&free);
}
-/* Callback from inside blk_mq_rq_ctx_init(). */
+/* Callback from inside blk_mq_rq_late_init(). */
static void dd_prepare_request(struct request *rq)
{
rq->elv.priv[0] = NULL;