diff options
| author | Niklas Cassel <cassel@kernel.org> | 2026-09-18 16:06:42 +0200 |
|---|---|---|
| committer | Jens Axboe <axboe@kernel.dk> | 2026-09-29 12:54:47 -0600 |
| commit | 684b413b5483f57c890c171b9400076a0143b918 (patch) | |
| tree | f167daddad3dce9af44773541373eafbccc9ac61 /include/clocksource/hyperv_timer.h | |
| download | linux-stable-684b413b5483f57c890c171b9400076a0143b918.tar.gz linux-stable-684b413b5483f57c890c171b9400076a0143b918.zip | |
virtio_blk: set the zone write granularitygrafted
virtblk_read_zoned_limits() reads the write granularity that the device
reports in virtio_blk_zoned_characteristics and assigns it to the
physical block size and to io_min, but never to the limit that is named
after it. queue_limits.zone_write_granularity is left at zero, so
blk_validate_zoned_limits() raises it to the logical block size:
if (lim->zone_write_granularity < lim->logical_block_size)
lim->zone_write_granularity = lim->logical_block_size;
A device that reports a granularity coarser than its logical block size,
which is what the field exists to express, therefore has it silently
reduced. A 512e host managed disk passed through to a guest reports a
logical block size of 512 and a write granularity of 4096, and the guest
ends up with a zone write granularity of 512.
bio_split_alignment() returns lim->zone_write_granularity if it is non-zero
and bio_split_io_at() may split a bio with as per bio_split_alignment().
This can real to the write getting rejected by the host drive, as the write
is not aligned to the physical block size.
zonefs also takes its block size from bdev_zone_write_granularity(), so it
would incorrectly use 512 on a disk that requires 4096.
sd_zbc_read_zones() sets the limit from the physical block size for the
same reason. NVMe ZNS and null_blk leave it unset, but the fallback
gives the right answer for them, as their write granularity is the
logical block size. virtio carries a separate value that may exceed it.
Set the zone write granularity from the value that the device reports.
Fixes: 95bfec41bd3d ("virtio-blk: add support for zoned block devices")
Signed-off-by: Niklas Cassel <cassel@kernel.org>
Reviewed-by: Stefan Hajnoczi <stefanha@redhat.com>
Link: https://patch.msgid.link/20260918140641.2031075-2-cassel@kernel.org
Signed-off-by: Jens Axboe <axboe@kernel.dk>
Diffstat (limited to 'include/clocksource/hyperv_timer.h')
| -rw-r--r-- | include/clocksource/hyperv_timer.h | 111 |
1 files changed, 111 insertions, 0 deletions
diff --git a/include/clocksource/hyperv_timer.h b/include/clocksource/hyperv_timer.h new file mode 100644 index 000000000..8d3befb7e --- /dev/null +++ b/include/clocksource/hyperv_timer.h @@ -0,0 +1,111 @@ +/* SPDX-License-Identifier: GPL-2.0 */ + +/* + * Definitions for the clocksource provided by the Hyper-V + * hypervisor to guest VMs, as described in the Hyper-V Top + * Level Functional Spec (TLFS). + * + * Copyright (C) 2019, Microsoft, Inc. + * + * Author: Michael Kelley <mikelley@microsoft.com> + */ + +#ifndef __CLKSOURCE_HYPERV_TIMER_H +#define __CLKSOURCE_HYPERV_TIMER_H + +#include <linux/clocksource.h> +#include <linux/math64.h> +#include <hyperv/hvhdk.h> + +#define HV_MAX_MAX_DELTA_TICKS 0xffffffff +#define HV_MIN_DELTA_TICKS 1 + +#ifdef CONFIG_HYPERV_TIMER + +#include <asm/hyperv_timer.h> + +/* Routines called by the VMbus driver */ +extern int hv_stimer_alloc(bool have_percpu_irqs); +extern int hv_stimer_cleanup(unsigned int cpu); +extern void hv_stimer_global_cleanup(void); + +extern void hv_init_clocksource(void); +extern void hv_remap_tsc_clocksource(void); + +extern unsigned long hv_get_tsc_pfn(void); +extern struct ms_hyperv_tsc_page *hv_get_tsc_page(void); + +extern void hv_adj_sched_clock_offset(u64 offset); + +static __always_inline bool +hv_read_tsc_page_tsc(const struct ms_hyperv_tsc_page *tsc_pg, + u64 *cur_tsc, u64 *time) +{ + u64 scale, offset; + u32 sequence; + + /* + * The protocol for reading Hyper-V TSC page is specified in Hypervisor + * Top-Level Functional Specification ver. 3.0 and above. To get the + * reference time we must do the following: + * - READ ReferenceTscSequence + * A special '0' value indicates the time source is unreliable and we + * need to use something else. The currently published specification + * versions (up to 4.0b) contain a mistake and wrongly claim '-1' + * instead of '0' as the special value, see commit c35b82ef0294. + * - ReferenceTime = + * ((RDTSC() * ReferenceTscScale) >> 64) + ReferenceTscOffset + * - READ ReferenceTscSequence again. In case its value has changed + * since our first reading we need to discard ReferenceTime and repeat + * the whole sequence as the hypervisor was updating the page in + * between. + */ + do { + sequence = READ_ONCE(tsc_pg->tsc_sequence); + if (!sequence) + return false; + /* + * Make sure we read sequence before we read other values from + * TSC page. + */ + smp_rmb(); + + scale = READ_ONCE(tsc_pg->tsc_scale); + offset = READ_ONCE(tsc_pg->tsc_offset); + *cur_tsc = hv_get_raw_timer(); + + /* + * Make sure we read sequence after we read all other values + * from TSC page. + */ + smp_rmb(); + + } while (READ_ONCE(tsc_pg->tsc_sequence) != sequence); + + *time = mul_u64_u64_shr(*cur_tsc, scale, 64) + offset; + return true; +} + +#else /* CONFIG_HYPERV_TIMER */ +static inline unsigned long hv_get_tsc_pfn(void) +{ + return 0; +} + +static inline struct ms_hyperv_tsc_page *hv_get_tsc_page(void) +{ + return NULL; +} + +static __always_inline bool +hv_read_tsc_page_tsc(const struct ms_hyperv_tsc_page *tsc_pg, u64 *cur_tsc, u64 *time) +{ + return false; +} + +static inline int hv_stimer_cleanup(unsigned int cpu) { return 0; } +static inline void hv_stimer_global_cleanup(void) {} + +#endif /* CONFIG_HYPERV_TIMER */ + +#endif |
