diff options
| author | Niklas Cassel <cassel@kernel.org> | 2026-09-18 16:06:42 +0200 |
|---|---|---|
| committer | Jens Axboe <axboe@kernel.dk> | 2026-09-29 12:54:47 -0600 |
| commit | 684b413b5483f57c890c171b9400076a0143b918 (patch) | |
| tree | f167daddad3dce9af44773541373eafbccc9ac61 /samples/pidfd | |
| download | linux-stable-684b413b5483f57c890c171b9400076a0143b918.tar.gz linux-stable-684b413b5483f57c890c171b9400076a0143b918.zip | |
virtio_blk: set the zone write granularitygrafted
virtblk_read_zoned_limits() reads the write granularity that the device
reports in virtio_blk_zoned_characteristics and assigns it to the
physical block size and to io_min, but never to the limit that is named
after it. queue_limits.zone_write_granularity is left at zero, so
blk_validate_zoned_limits() raises it to the logical block size:
if (lim->zone_write_granularity < lim->logical_block_size)
lim->zone_write_granularity = lim->logical_block_size;
A device that reports a granularity coarser than its logical block size,
which is what the field exists to express, therefore has it silently
reduced. A 512e host managed disk passed through to a guest reports a
logical block size of 512 and a write granularity of 4096, and the guest
ends up with a zone write granularity of 512.
bio_split_alignment() returns lim->zone_write_granularity if it is non-zero
and bio_split_io_at() may split a bio with as per bio_split_alignment().
This can real to the write getting rejected by the host drive, as the write
is not aligned to the physical block size.
zonefs also takes its block size from bdev_zone_write_granularity(), so it
would incorrectly use 512 on a disk that requires 4096.
sd_zbc_read_zones() sets the limit from the physical block size for the
same reason. NVMe ZNS and null_blk leave it unset, but the fallback
gives the right answer for them, as their write granularity is the
logical block size. virtio carries a separate value that may exceed it.
Set the zone write granularity from the value that the device reports.
Fixes: 95bfec41bd3d ("virtio-blk: add support for zoned block devices")
Signed-off-by: Niklas Cassel <cassel@kernel.org>
Reviewed-by: Stefan Hajnoczi <stefanha@redhat.com>
Link: https://patch.msgid.link/20260918140641.2031075-2-cassel@kernel.org
Signed-off-by: Jens Axboe <axboe@kernel.dk>
Diffstat (limited to 'samples/pidfd')
| -rw-r--r-- | samples/pidfd/.gitignore | 2 | ||||
| -rw-r--r-- | samples/pidfd/Makefile | 4 | ||||
| -rw-r--r-- | samples/pidfd/pidfd-metadata.c | 120 |
3 files changed, 126 insertions, 0 deletions
diff --git a/samples/pidfd/.gitignore b/samples/pidfd/.gitignore new file mode 100644 index 000000000..d4cfa3176 --- /dev/null +++ b/samples/pidfd/.gitignore @@ -0,0 +1,2 @@ +# SPDX-License-Identifier: GPL-2.0-only +/pidfd-metadata diff --git a/samples/pidfd/Makefile b/samples/pidfd/Makefile new file mode 100644 index 000000000..9754e2d81 --- /dev/null +++ b/samples/pidfd/Makefile @@ -0,0 +1,4 @@ +# SPDX-License-Identifier: GPL-2.0 +usertprogs-always-y += pidfd-metadata + +userccflags += -I usr/include diff --git a/samples/pidfd/pidfd-metadata.c b/samples/pidfd/pidfd-metadata.c new file mode 100644 index 000000000..c459155da --- /dev/null +++ b/samples/pidfd/pidfd-metadata.c @@ -0,0 +1,120 @@ +// SPDX-License-Identifier: GPL-2.0 + +#define _GNU_SOURCE +#include <err.h> +#include <errno.h> +#include <fcntl.h> +#include <inttypes.h> +#include <limits.h> +#include <sched.h> +#include <signal.h> +#include <stdio.h> +#include <stdlib.h> +#include <string.h> +#include <sys/stat.h> +#include <sys/syscall.h> +#include <sys/types.h> +#include <sys/wait.h> +#include <unistd.h> + +#ifndef CLONE_PIDFD +#define CLONE_PIDFD 0x00001000 +#endif + +#ifndef __NR_pidfd_send_signal +#define __NR_pidfd_send_signal -1 +#endif + +static int do_child(void *args) +{ + printf("%d\n", getpid()); + _exit(EXIT_SUCCESS); +} + +static pid_t pidfd_clone(int flags, int *pidfd) +{ + size_t stack_size = 1024; + char *stack[1024] = { 0 }; + +#ifdef __ia64__ + return __clone2(do_child, stack, stack_size, flags | SIGCHLD, NULL, pidfd); +#else + return clone(do_child, stack + stack_size, flags | SIGCHLD, NULL, pidfd); +#endif +} + +static inline int sys_pidfd_send_signal(int pidfd, int sig, siginfo_t *info, + unsigned int flags) +{ + return syscall(__NR_pidfd_send_signal, pidfd, sig, info, flags); +} + +static int pidfd_metadata_fd(pid_t pid, int pidfd) +{ + int procfd, ret; + char path[100]; + + snprintf(path, sizeof(path), "/proc/%d", pid); + procfd = open(path, O_DIRECTORY | O_RDONLY | O_CLOEXEC); + if (procfd < 0) { + warn("Failed to open %s\n", path); + return -1; + } + + /* + * Verify that the pid has not been recycled and our /proc/<pid> handle + * is still valid. + */ + ret = sys_pidfd_send_signal(pidfd, 0, NULL, 0); + if (ret < 0) { + switch (errno) { + case EPERM: + /* Process exists, just not allowed to signal it. */ + break; + default: + warn("Failed to signal process\n"); + close(procfd); + procfd = -1; + } + } + + return procfd; +} + +int main(int argc, char *argv[]) +{ + int pidfd = -1, ret = EXIT_FAILURE; + char buf[4096] = { 0 }; + pid_t pid; + int procfd, statusfd; + ssize_t bytes; + + pid = pidfd_clone(CLONE_PIDFD, &pidfd); + if (pid < 0) + err(ret, "CLONE_PIDFD"); + if (pidfd == -1) { + warnx("CLONE_PIDFD is not supported by the kernel"); + goto out; + } + + procfd = pidfd_metadata_fd(pid, pidfd); + close(pidfd); + if (procfd < 0) + goto out; + + statusfd = openat(procfd, "status", O_RDONLY | O_CLOEXEC); + close(procfd); + if (statusfd < 0) + goto out; + + bytes = read(statusfd, buf, sizeof(buf)); + if (bytes > 0) + bytes = write(STDOUT_FILENO, buf, bytes); + close(statusfd); + ret = EXIT_SUCCESS; + +out: + (void)wait(NULL); + + exit(ret); +} |
