From 8f1da3536b8f8420988fe6e6ae15d58afa2ba2d8 Mon Sep 17 00:00:00 2001 From: Kameron Carr Date: Tue, 11 Aug 2026 09:04:45 -0700 Subject: Drivers: hv: vmbus: add vmbus_establish_gpadl_caller_decrypted() Add a new vmbus_establish_gpadl_caller_decrypted() for callers that want to decrypt their own buffers. Add a new hv_gpadl_type, HV_GPADL_BUFFER_DECRYPTED, to communicate the decryption status of the buffer. No functional change for existing callers. Signed-off-by: Kameron Carr Reviewed-by: Michael Kelley Signed-off-by: Wei Liu --- include/linux/hyperv.h | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) (limited to 'include') diff --git a/include/linux/hyperv.h b/include/linux/hyperv.h index a2b484679eb4..08b927fab610 100644 --- a/include/linux/hyperv.h +++ b/include/linux/hyperv.h @@ -70,7 +70,8 @@ */ enum hv_gpadl_type { HV_GPADL_BUFFER, - HV_GPADL_RING + HV_GPADL_RING, + HV_GPADL_BUFFER_DECRYPTED }; /* Single-page buffer */ @@ -1205,6 +1206,11 @@ extern int vmbus_establish_gpadl(struct vmbus_channel *channel, u32 size, struct vmbus_gpadl *gpadl); +extern int vmbus_establish_gpadl_caller_decrypted(struct vmbus_channel *channel, + void *kbuffer, + u32 size, + struct vmbus_gpadl *gpadl); + extern int vmbus_teardown_gpadl(struct vmbus_channel *channel, struct vmbus_gpadl *gpadl); -- cgit From 73fe42af955a24b6cc792b9a4867e809196f05ed Mon Sep 17 00:00:00 2001 From: Kameron Carr Date: Tue, 11 Aug 2026 09:04:46 -0700 Subject: Drivers: hv: vmbus: Add vmbus_alloc_buffer()/vmbus_free_buffer() for CoCo VMs On CoCo VMs without confidential VMBus, the netvsc send and receive buffers must be made host-visible by decrypting them. These buffers are vmalloc'ed, but set_memory_decrypted()/encrypted() do not work on vmalloc'ed memory. This use case is (so far) unique to netvsc, so solve it locally rather than changing the set_memory() or allocation APIs. Add vmbus_alloc_buffer()/vmbus_free_buffer() to the VMBus core. When the guest's isolation model requires it, allocate the buffer as a list of physically-contiguous chunks via alloc_pages_node(), starting at MAX_PAGE_ORDER and falling back to smaller orders so the allocation still succeeds under memory fragmentation. Each chunk is decrypted in place via set_memory_decrypted() on its direct-map address, and the chunks are then stitched into a single virtually-contiguous range with vmap(). Buffers that do not need decryption keep using vzalloc(). To free the buffer, vmbus_free_buffer() calls vunmap() on the range then re-encrypts and frees each chunk individually; any chunk that fails re-encryption is leaked to prevent accidentally freeing decrypted memory. This approach minimizes scattering of decrypted 4 KiB pages through the kernel direct map and the resulting shattering of large page mappings. Signed-off-by: Kameron Carr Reviewed-by: Michael Kelley Signed-off-by: Wei Liu --- drivers/hv/channel.c | 155 +++++++++++++++++++++++++++++++++++++++++++++++++ include/linux/hyperv.h | 7 +++ 2 files changed, 162 insertions(+) (limited to 'include') diff --git a/drivers/hv/channel.c b/drivers/hv/channel.c index 4782f5070bba..f4370617deac 100644 --- a/drivers/hv/channel.c +++ b/drivers/hv/channel.c @@ -13,11 +13,13 @@ #include #include #include +#include #include #include #include #include #include +#include #include #include #include @@ -608,6 +610,159 @@ int vmbus_establish_gpadl_caller_decrypted(struct vmbus_channel *channel, } EXPORT_SYMBOL_GPL(vmbus_establish_gpadl_caller_decrypted); +/** + * vmbus_free_buffer - release a buffer allocated by vmbus_alloc_buffer(). + * + * @addr: buffer address, or NULL if none was allocated (e.g. cleanup from a + * failed allocation) + * @chunks: chunks array from vmbus_alloc_buffer(), or NULL + * @chunk_cnt: number of entries in @chunks + * + * When @chunks is NULL the buffer is a plain vzalloc() allocation. + * + * Otherwise tear down the vmap, and for each chunk re-encrypt and free + * the underlying pages. Any chunk that cannot be re-encrypted is leaked. + */ +void vmbus_free_buffer(void *addr, struct page **chunks, u32 chunk_cnt) +{ + u32 i; + + if (!chunks) { + vfree(addr); + return; + } + + vunmap(addr); + + for (i = 0; i < chunk_cnt; i++) { + unsigned long vaddr = + (unsigned long)page_address(chunks[i]); + unsigned int order = folio_order(page_folio(chunks[i])); + + if (set_memory_encrypted(vaddr, 1U << order)) + continue; + __free_pages(chunks[i], order); + } + + kvfree(chunks); +} +EXPORT_SYMBOL_GPL(vmbus_free_buffer); + +/** + * vmbus_alloc_buffer - allocate a host-visible, virtually-contiguous buffer. + * + * @channel: the channel the buffer will be attached to + * @size: requested buffer size in bytes (will be rounded up to PAGE_SIZE) + * @chunks_out: on success, set to the array of underlying chunks, or NULL when + * the buffer was allocated with vzalloc() + * @chunk_cnt_out: on success, set to the number of chunks + * + * Buffers not requiring decryption are allocated with vzalloc(). + * + * Buffers requiring decryption are allocated as a series of + * physically-contiguous chunks, starting at MAX_PAGE_ORDER and falling back to + * smaller orders on allocation failure. Each chunk is transitioned to + * host-visible via set_memory_decrypted() on its direct-map address, then all + * chunks are combined into a virtually-contiguous range via vmap(). + * + * Return: the buffer's virtual address, or NULL on failure. + */ +void *vmbus_alloc_buffer(struct vmbus_channel *channel, + u32 size, + struct page ***chunks_out, + u32 *chunk_cnt_out) +{ + unsigned long nr_pages = PFN_UP(size); + unsigned long remaining = nr_pages; + unsigned long page_idx = 0; + struct page **chunks = NULL; + struct page **pages = NULL; + int order = MAX_PAGE_ORDER; + u32 chunk_cnt = 0; + void *addr; + u32 i; + int ret; + + *chunks_out = NULL; + *chunk_cnt_out = 0; + + if (!nr_pages) + return NULL; + + /* If the buffer does not need to be decrypted, just use vzalloc() */ + if (!hv_is_isolation_supported() || channel->co_external_memory) + return vzalloc(nr_pages << PAGE_SHIFT); + + /* Worst case: every chunk is a single page. */ + chunks = kvmalloc_array(nr_pages, sizeof(*chunks), + GFP_KERNEL | __GFP_ZERO); + if (!chunks) + goto err; + + pages = kvmalloc_array(nr_pages, sizeof(*pages), GFP_KERNEL); + if (!pages) + goto err; + + while (remaining) { + struct page *page; + gfp_t gfp; + + order = min(order, ilog2(remaining)); + + /* + * Use __GFP_NORETRY | __GFP_NOWARN to avoid OOM-killing, + * but try harder at order 0 since that is the final + * fallback. + * __GFP_COMP stores order information in the page folio. + */ + gfp = GFP_KERNEL | __GFP_ZERO; + if (order) + gfp |= __GFP_COMP | __GFP_NORETRY | __GFP_NOWARN; + + page = alloc_pages_node(cpu_to_node(channel->target_cpu), + gfp, order); + if (!page) { + if (!order--) + goto err; + continue; + } + + ret = set_memory_decrypted((unsigned long)page_address(page), + 1U << order); + if (ret) { + /* + * set_memory_decrypted() failed; the page state is + * unknown so it must be leaked rather than freed. + */ + goto err; + } + + chunks[chunk_cnt++] = page; + + for (i = 0; i < (1U << order); i++) + pages[page_idx++] = page + i; + + remaining -= 1U << order; + } + + addr = vmap(pages, nr_pages, VM_MAP, pgprot_decrypted(PAGE_KERNEL)); + if (!addr) + goto err; + + memset(addr, 0, nr_pages << PAGE_SHIFT); + + kvfree(pages); + *chunks_out = chunks; + *chunk_cnt_out = chunk_cnt; + return addr; + +err: + kvfree(pages); + vmbus_free_buffer(NULL, chunks, chunk_cnt); + return NULL; +} +EXPORT_SYMBOL_GPL(vmbus_alloc_buffer); + /** * request_arr_init - Allocates memory for the requestor array. Each slot * keeps track of the next available slot in the array. Initially, each diff --git a/include/linux/hyperv.h b/include/linux/hyperv.h index 08b927fab610..a97be78653e8 100644 --- a/include/linux/hyperv.h +++ b/include/linux/hyperv.h @@ -1214,6 +1214,13 @@ extern int vmbus_establish_gpadl_caller_decrypted(struct vmbus_channel *channel, extern int vmbus_teardown_gpadl(struct vmbus_channel *channel, struct vmbus_gpadl *gpadl); +extern void *vmbus_alloc_buffer(struct vmbus_channel *channel, + u32 size, + struct page ***chunks_out, + u32 *chunk_cnt_out); + +extern void vmbus_free_buffer(void *addr, struct page **chunks, u32 chunk_cnt); + void vmbus_reset_channel_cb(struct vmbus_channel *channel); extern int vmbus_recvpacket(struct vmbus_channel *channel, -- cgit From a9a05e801efffe3a2f949fef2a83f855e733d6e1 Mon Sep 17 00:00:00 2001 From: Michael Kelley Date: Wed, 5 Aug 2026 13:37:46 -0700 Subject: Drivers: hv: Remove support for WS2012/2012R2 & Win8/8.1 version of Hyper-V Linux code for running as a Hyper-V guest includes special cases for running on Hyper-V in WS2012/2012R2 and Windows 8/8.1. These versions were initially released 14 years ago, and official support ended in 2023 (unless a customer has contracted for extended security updates). Given the release of subsequent versions with improved functionality, there's no need to continue to support the latest Linux kernels on these versions of Hyper-V. If someone is running Linux on one of these older Hyper-V versions and doesn't want to upgrade, they can continue to do so as presumably they don't want upgrade the Linux version either. Simplify Linux code by removing special cases for running on these old versions of Hyper-V. Remove the negotiation of the VMBus protocol versions for WS2012/Win8, and remove special case code based on those VMBus protocol versions. Also update the balloon and snapshot drivers to no longer negotiate driver-specific protocol versions for these older Hyper-V versions, and remove any related special cases. Signed-off-by: Michael Kelley Signed-off-by: Wei Liu --- drivers/hv/channel_mgmt.c | 8 +++----- drivers/hv/connection.c | 11 +++++++---- drivers/hv/hv_balloon.c | 36 +++++++++++++++--------------------- drivers/hv/hv_snapshot.c | 5 ----- include/linux/hyperv.h | 5 ++--- 5 files changed, 27 insertions(+), 38 deletions(-) (limited to 'include') diff --git a/drivers/hv/channel_mgmt.c b/drivers/hv/channel_mgmt.c index 89d214dda360..a044fd3b3c4e 100644 --- a/drivers/hv/channel_mgmt.c +++ b/drivers/hv/channel_mgmt.c @@ -929,12 +929,10 @@ static void vmbus_unload_response(struct vmbus_channel_message_header *hdr) void vmbus_initiate_unload(bool crash) { struct vmbus_channel_message_header hdr; + enum vmbus_connect_state old_state; - if (xchg(&vmbus_connection.conn_state, DISCONNECTED) == DISCONNECTED) - return; - - /* Pre-Win2012R2 hosts don't support reconnect */ - if (vmbus_proto_version < VERSION_WIN8_1) + old_state = xchg(&vmbus_connection.conn_state, DISCONNECTED); + if (old_state == DISCONNECTED || old_state == CONNECTING) return; reinit_completion(&vmbus_connection.unload_event); diff --git a/drivers/hv/connection.c b/drivers/hv/connection.c index 0fd50d4cb573..1ab3581b096a 100644 --- a/drivers/hv/connection.c +++ b/drivers/hv/connection.c @@ -47,7 +47,9 @@ EXPORT_SYMBOL_GPL(vmbus_proto_version); /* * Table of VMBus versions listed from newest to oldest. - * VERSION_WIN7 and VERSION_WS2008 are no longer supported in + * VERSION_WIN7,VERSION_WS2008, VERSION_WIN8 (which is + * Windows Server 2012) and VERSION_WIN8_1 (which is + * Windows Server 2012 R2) are no longer supported in * Linux guests and are not listed. */ static __u32 vmbus_versions[] = { @@ -57,9 +59,7 @@ static __u32 vmbus_versions[] = { VERSION_WIN10_V5_1, VERSION_WIN10_V5, VERSION_WIN10_V4_1, - VERSION_WIN10, - VERSION_WIN8_1, - VERSION_WIN8 + VERSION_WIN10 }; /* @@ -304,6 +304,9 @@ int vmbus_connect(void) for (i = 0; ; i++) { if (i == ARRAY_SIZE(vmbus_versions)) { ret = -EDOM; + pr_err("Hyper-V host does not support VMBus version %d.%d or higher;\n\ + the host may be an older version no longer supported by Linux\n", + vmbus_versions[i-1] >> 16, vmbus_versions[i-1] & 0xFFFF); goto cleanup; } diff --git a/drivers/hv/hv_balloon.c b/drivers/hv/hv_balloon.c index 42ce27be344d..9cba97e81111 100644 --- a/drivers/hv/hv_balloon.c +++ b/drivers/hv/hv_balloon.c @@ -58,6 +58,10 @@ #define DYNMEM_MAJOR_VERSION(Version) ((__u32)(Version) >> 16) #define DYNMEM_MINOR_VERSION(Version) ((__u32)(Version) & 0xff) +/* + * VERSION_1 and VERSION_2 are retained for the historical record, + * but are no longer supported in Linux guests. + */ enum { DYNMEM_PROTOCOL_VERSION_1 = DYNMEM_MAKE_VERSION(0, 3), DYNMEM_PROTOCOL_VERSION_2 = DYNMEM_MAKE_VERSION(1, 0), @@ -65,9 +69,7 @@ enum { DYNMEM_PROTOCOL_VERSION_WIN7 = DYNMEM_PROTOCOL_VERSION_1, DYNMEM_PROTOCOL_VERSION_WIN8 = DYNMEM_PROTOCOL_VERSION_2, - DYNMEM_PROTOCOL_VERSION_WIN10 = DYNMEM_PROTOCOL_VERSION_3, - - DYNMEM_PROTOCOL_VERSION_CURRENT = DYNMEM_PROTOCOL_VERSION_WIN10 + DYNMEM_PROTOCOL_VERSION_WIN10 = DYNMEM_PROTOCOL_VERSION_3 }; /* @@ -1434,19 +1436,9 @@ static void version_resp(struct hv_dynmem_device *dm, version_req.version.version = dm->next_version; dm->version = version_req.version.version; - /* - * Set the next version to try in case current version fails. - * Win7 protocol ought to be the last one to try. - */ - switch (version_req.version.version) { - case DYNMEM_PROTOCOL_VERSION_WIN8: - dm->next_version = DYNMEM_PROTOCOL_VERSION_WIN7; - version_req.is_last_attempt = 0; - break; - default: - dm->next_version = 0; - version_req.is_last_attempt = 1; - } + /* Set the next version to try in case current version fails. */ + dm->next_version = 0; + version_req.is_last_attempt = 1; ret = vmbus_sendpacket(dm->dev->channel, &version_req, sizeof(struct dm_version_request), @@ -1735,16 +1727,18 @@ static int balloon_connect_vsp(struct hv_device *dev) /* * Initiate the hand shake with the host and negotiate - * a version that the host can support. We start with the - * highest version number and go down if the host cannot - * support it. + * a version that the host can support. The mechanism is in place + * to start with the highest version number and go down if the host + * cannot support it. But currently we only try the WIN10 version + * since support for older Hyper-V versions has been removed from + * Linux. */ memset(&version_req, 0, sizeof(struct dm_version_request)); version_req.hdr.type = DM_VERSION_REQUEST; version_req.hdr.size = sizeof(struct dm_version_request); version_req.hdr.trans_id = atomic_inc_return(&trans_id); version_req.version.version = DYNMEM_PROTOCOL_VERSION_WIN10; - version_req.is_last_attempt = 0; + version_req.is_last_attempt = 1; dm_device.version = version_req.version.version; ret = vmbus_sendpacket(dev->channel, &version_req, @@ -1964,7 +1958,7 @@ static int balloon_probe(struct hv_device *dev, #endif dm_device.dev = dev; dm_device.state = DM_INITIALIZING; - dm_device.next_version = DYNMEM_PROTOCOL_VERSION_WIN8; + dm_device.next_version = 0; init_completion(&dm_device.host_event); init_completion(&dm_device.config_event); INIT_LIST_HEAD(&dm_device.ha_region_list); diff --git a/drivers/hv/hv_snapshot.c b/drivers/hv/hv_snapshot.c index 506871aeacf0..847dc69e8250 100644 --- a/drivers/hv/hv_snapshot.c +++ b/drivers/hv/hv_snapshot.c @@ -372,11 +372,6 @@ static void vss_on_reset(void) int hv_vss_init(struct hv_util_service *srv) { - if (vmbus_proto_version < VERSION_WIN8_1) { - pr_warn("Integration service 'Backup (volume snapshot)'" - " not supported on this host version.\n"); - return -ENOTSUPP; - } recv_buffer = srv->recv_buffer; vss_transaction.recv_channel = srv->channel; vss_transaction.recv_channel->max_pkt_size = VSS_MAX_PKT_SIZE; diff --git a/include/linux/hyperv.h b/include/linux/hyperv.h index a97be78653e8..9e109d91aa14 100644 --- a/include/linux/hyperv.h +++ b/include/linux/hyperv.h @@ -261,9 +261,8 @@ static inline u32 hv_get_avail_to_write_percent( * 5 . 2 (Windows Server 2019, RS5) * 5 . 3 (Windows Server 2022) * - * The WS2008 and WIN7 versions are listed here for - * completeness but are no longer supported in the - * Linux kernel. + * The WS2008, WIN7, WIN8, and WIN8_1 versions are listed here for + * completeness but are no longer supported in the Linux kernel. */ #define VMBUS_MAKE_VERSION(MAJ, MIN) ((((u32)MAJ) << 16) | (MIN)) -- cgit From be0cfab740e58b70047ef6e7e3d578f00ed5d258 Mon Sep 17 00:00:00 2001 From: Michael Kelley Date: Wed, 5 Aug 2026 13:37:51 -0700 Subject: clocksource: hyper-v: Remove support for stimer interrupts in message mode In Hyper-V versions prior to WS2016/Win10, Hyper-V synthetic timers interrupt the guest by delivering a message that is initially handled by the Linux VMBus driver. Starting with WS2016/Win10, Hyper-V can deliver stimer interrupts directly to an assigned interrupt vector without involving the VMBus driver. This is called "Direct Mode". With the overall removal of Linux support for running on Hyper-V hosts earlier than WS2016 and Windows 10, it's no longer necessary to support the legacy message-based delivery. Remove that delivery mechanism and always use Direct Mode. If for some reason, the Hyper-V host does not enumerate Direct Mode, output an error message but continue to run using the LAPIC timer instead of an stimer. With these changes, the VMBus driver no longer calls the stimer interrupt service routine. This removal has a broader benefit in unblocking the disentangling of VMBus code and stimer code, as they should be independent of each other. The final disentangling will come as a follow-on patch set. Signed-off-by: Michael Kelley Signed-off-by: Wei Liu --- arch/x86/hyperv/hv_init.c | 17 +++-- arch/x86/kernel/cpu/mshyperv.c | 4 +- drivers/clocksource/hyperv_timer.c | 150 +++++-------------------------------- drivers/hv/hv.c | 4 - drivers/hv/vmbus_drv.c | 10 +-- include/clocksource/hyperv_timer.h | 6 -- 6 files changed, 32 insertions(+), 159 deletions(-) (limited to 'include') diff --git a/arch/x86/hyperv/hv_init.c b/arch/x86/hyperv/hv_init.c index 55a8b6de2865..0b4a1c0b0b16 100644 --- a/arch/x86/hyperv/hv_init.c +++ b/arch/x86/hyperv/hv_init.c @@ -171,8 +171,7 @@ static int hv_cpu_init(unsigned int cpu) } /* Allow Hyper-V stimer vector to be injected from Hypervisor. */ - if (ms_hyperv.misc_features & HV_STIMER_DIRECT_MODE_AVAILABLE) - apic_update_vector(cpu, HYPERV_STIMER0_VECTOR, true); + apic_update_vector(cpu, HYPERV_STIMER0_VECTOR, true); return hyperv_init_ghcb(); } @@ -281,8 +280,7 @@ static int hv_cpu_die(unsigned int cpu) *ghcb_va = NULL; } - if (ms_hyperv.misc_features & HV_STIMER_DIRECT_MODE_AVAILABLE) - apic_update_vector(cpu, HYPERV_STIMER0_VECTOR, false); + apic_update_vector(cpu, HYPERV_STIMER0_VECTOR, false); hv_common_cpu_die(cpu); @@ -425,15 +423,18 @@ static void (* __initdata old_setup_percpu_clockev)(void); static void __init hv_stimer_setup_percpu_clockev(void) { + int ret; + /* - * Ignore any errors in setting up stimer clockevents + * Continue afters errors in setting up stimer clockevents * as we can run with the LAPIC timer as a fallback. */ - (void)hv_stimer_alloc(false); + ret = hv_stimer_alloc(false); + if (ret) + pr_warn("stimer setup failed with error %d\n", ret); /* - * Still register the LAPIC timer, because the direct-mode STIMER is - * not supported by old versions of Hyper-V. This also allows users + * Still register the LAPIC timer to allows users * to switch to LAPIC timer via /sys, if they want to. */ if (old_setup_percpu_clockev) diff --git a/arch/x86/kernel/cpu/mshyperv.c b/arch/x86/kernel/cpu/mshyperv.c index 229c7377a980..b4af7c0a70ac 100644 --- a/arch/x86/kernel/cpu/mshyperv.c +++ b/arch/x86/kernel/cpu/mshyperv.c @@ -731,9 +731,7 @@ static void __init ms_hyperv_init_platform(void) } /* Install system interrupt handler for stimer0 */ - if (ms_hyperv.misc_features & HV_STIMER_DIRECT_MODE_AVAILABLE) { - sysvec_install(HYPERV_STIMER0_VECTOR, sysvec_hyperv_stimer0); - } + sysvec_install(HYPERV_STIMER0_VECTOR, sysvec_hyperv_stimer0); # ifdef CONFIG_SMP smp_ops.smp_prepare_boot_cpu = hv_smp_prepare_boot_cpu; diff --git a/drivers/clocksource/hyperv_timer.c b/drivers/clocksource/hyperv_timer.c index df567795d175..dddfff458ebf 100644 --- a/drivers/clocksource/hyperv_timer.c +++ b/drivers/clocksource/hyperv_timer.c @@ -31,44 +31,20 @@ static struct clock_event_device __percpu *hv_clock_event; /* Note: offset can hold negative values after hibernation. */ static u64 hv_sched_clock_offset __read_mostly; -/* - * If false, we're using the old mechanism for stimer0 interrupts - * where it sends a VMbus message when it expires. The old - * mechanism is used when running on older versions of Hyper-V - * that don't support Direct Mode. While Hyper-V provides - * four stimer's per CPU, Linux uses only stimer0. - * - * Because Direct Mode does not require processing a VMbus - * message, stimer interrupts can be enabled earlier in the - * process of booting a CPU, and consistent with when timer - * interrupts are enabled for other clocksource drivers. - * However, for legacy versions of Hyper-V when Direct Mode - * is not enabled, setting up stimer interrupts must be - * delayed until VMbus is initialized and can process the - * interrupt message. - */ -static bool direct_mode_enabled; - static int stimer0_irq = -1; -static int stimer0_message_sint; static __maybe_unused DEFINE_PER_CPU(long, stimer0_evt); -/* - * Common code for stimer0 interrupts coming via Direct Mode or - * as a VMbus message. - */ -void hv_stimer0_isr(void) +static void hv_stimer0_isr(void) { struct clock_event_device *ce; ce = this_cpu_ptr(hv_clock_event); ce->event_handler(ce); } -EXPORT_SYMBOL_GPL(hv_stimer0_isr); /* * stimer0 interrupt handler for architectures that support - * per-cpu interrupts, which also implies Direct Mode. + * per-cpu interrupts */ static irqreturn_t __maybe_unused hv_stimer0_percpu_isr(int irq, void *dev_id) { @@ -91,7 +67,7 @@ static int hv_ce_shutdown(struct clock_event_device *evt) { hv_set_msr(HV_MSR_STIMER0_COUNT, 0); hv_set_msr(HV_MSR_STIMER0_CONFIG, 0); - if (direct_mode_enabled && stimer0_irq >= 0) + if (stimer0_irq >= 0) disable_percpu_irq(stimer0_irq); return 0; @@ -104,23 +80,16 @@ static int hv_ce_set_oneshot(struct clock_event_device *evt) timer_cfg.as_uint64 = 0; timer_cfg.enable = 1; timer_cfg.auto_enable = 1; - if (direct_mode_enabled) { - /* - * When it expires, the timer will directly interrupt - * on the specified hardware vector/IRQ. - */ - timer_cfg.direct_mode = 1; - timer_cfg.apic_vector = HYPERV_STIMER0_VECTOR; - if (stimer0_irq >= 0) - enable_percpu_irq(stimer0_irq, IRQ_TYPE_NONE); - } else { - /* - * When it expires, the timer will generate a VMbus message, - * to be handled by the normal VMbus interrupt handler. - */ - timer_cfg.direct_mode = 0; - timer_cfg.sintx = stimer0_message_sint; - } + + /* + * When it expires, the timer will directly interrupt + * on the specified hardware vector/IRQ. + */ + timer_cfg.direct_mode = 1; + timer_cfg.apic_vector = HYPERV_STIMER0_VECTOR; + if (stimer0_irq >= 0) + enable_percpu_irq(stimer0_irq, IRQ_TYPE_NONE); + hv_set_msr(HV_MSR_STIMER0_CONFIG, timer_cfg.as_uint64); return 0; } @@ -175,25 +144,8 @@ int hv_stimer_cleanup(unsigned int cpu) if (!hv_clock_event) return 0; - /* - * In the legacy case where Direct Mode is not enabled - * (which can only be on x86/64), stimer cleanup happens - * relatively early in the CPU offlining process. We - * must unbind the stimer-based clockevent device so - * that the LAPIC timer can take over until clockevents - * are no longer needed in the offlining process. Note - * that clockevents_unbind_device() eventually calls - * hv_ce_shutdown(). - * - * The unbind should not be done when Direct Mode is - * enabled because we may be on an architecture where - * there are no other clockevent devices to fallback to. - */ ce = per_cpu_ptr(hv_clock_event, cpu); - if (direct_mode_enabled) - hv_ce_shutdown(ce); - else - clockevents_unbind_device(ce, cpu); + hv_ce_shutdown(ce); return 0; } @@ -268,23 +220,14 @@ int hv_stimer_alloc(bool have_percpu_irqs) * Hyper-V on x86. In that case, return as error as Linux will use a * clockevent based on emulated LAPIC timer hardware. */ - if (!(ms_hyperv.features & HV_MSR_SYNTIMER_AVAILABLE)) + if (!(ms_hyperv.features & HV_MSR_SYNTIMER_AVAILABLE) || + !(ms_hyperv.misc_features & HV_STIMER_DIRECT_MODE_AVAILABLE)) return -EINVAL; hv_clock_event = alloc_percpu(struct clock_event_device); if (!hv_clock_event) return -ENOMEM; - direct_mode_enabled = ms_hyperv.misc_features & - HV_STIMER_DIRECT_MODE_AVAILABLE; - - /* - * If Direct Mode isn't enabled, the remainder of the initialization - * is done later by hv_stimer_legacy_init() - */ - if (!direct_mode_enabled) - return 0; - if (have_percpu_irqs) { ret = hv_setup_stimer0_irq(); if (ret) @@ -293,11 +236,6 @@ int hv_stimer_alloc(bool have_percpu_irqs) hv_setup_stimer0_handler(hv_stimer0_isr); } - /* - * Since we are in Direct Mode, stimer initialization - * can be done now with a CPUHP value in the same range - * as other clockevent devices. - */ ret = cpuhp_setup_state(CPUHP_AP_HYPERV_TIMER_STARTING, "clockevents/hyperv/stimer:starting", hv_stimer_init, hv_stimer_cleanup); @@ -314,67 +252,19 @@ free_clock_event: } EXPORT_SYMBOL_GPL(hv_stimer_alloc); -/* - * hv_stimer_legacy_init -- Called from the VMbus driver to handle - * the case when Direct Mode is not enabled, and the stimer - * must be initialized late in the CPU onlining process. - * - */ -void hv_stimer_legacy_init(unsigned int cpu, int sint) -{ - if (direct_mode_enabled) - return; - - /* - * This function gets called by each vCPU, so setting the - * global stimer_message_sint value each time is conceptually - * not ideal, but the value passed in is always the same and - * it avoids introducing yet another interface into this - * clocksource driver just to set the sint in the legacy case. - */ - stimer0_message_sint = sint; - (void)hv_stimer_init(cpu); -} -EXPORT_SYMBOL_GPL(hv_stimer_legacy_init); - -/* - * hv_stimer_legacy_cleanup -- Called from the VMbus driver to - * handle the case when Direct Mode is not enabled, and the - * stimer must be cleaned up early in the CPU offlining - * process. - */ -void hv_stimer_legacy_cleanup(unsigned int cpu) -{ - if (direct_mode_enabled) - return; - (void)hv_stimer_cleanup(cpu); -} -EXPORT_SYMBOL_GPL(hv_stimer_legacy_cleanup); - /* * Do a global cleanup of clockevents for the cases of kexec and * vmbus exit */ void hv_stimer_global_cleanup(void) { - int cpu; - - /* - * hv_stime_legacy_cleanup() will stop the stimer if Direct - * Mode is not enabled, and fallback to the LAPIC timer. - */ - for_each_present_cpu(cpu) { - hv_stimer_legacy_cleanup(cpu); - } - if (!hv_clock_event) return; - if (direct_mode_enabled) { - cpuhp_remove_state(CPUHP_AP_HYPERV_TIMER_STARTING); - hv_remove_stimer0_irq(); - stimer0_irq = -1; - } + cpuhp_remove_state(CPUHP_AP_HYPERV_TIMER_STARTING); + hv_remove_stimer0_irq(); + stimer0_irq = -1; + free_percpu(hv_clock_event); hv_clock_event = NULL; diff --git a/drivers/hv/hv.c b/drivers/hv/hv.c index ef4b1b03395d..fe50090dcc01 100644 --- a/drivers/hv/hv.c +++ b/drivers/hv/hv.c @@ -399,8 +399,6 @@ int hv_synic_init(unsigned int cpu) else hv_hyp_synic_enable_interrupts(); - hv_stimer_legacy_init(cpu, VMBUS_MESSAGE_SINT); - return 0; } @@ -630,8 +628,6 @@ int hv_synic_cleanup(unsigned int cpu) return -EBUSY; always_cleanup: - hv_stimer_legacy_cleanup(cpu); - /* * First, disable the event and message pages * used for communicating with the host, and then diff --git a/drivers/hv/vmbus_drv.c b/drivers/hv/vmbus_drv.c index 18ee549d9880..5ebdbe24b5a1 100644 --- a/drivers/hv/vmbus_drv.c +++ b/drivers/hv/vmbus_drv.c @@ -1320,14 +1320,8 @@ static void vmbus_message_sched(struct hv_per_cpu_context *hv_cpu, void *message msg = (struct hv_message *)message_page_addr + VMBUS_MESSAGE_SINT; /* Check if there are actual msgs to be processed */ - if (msg->header.message_type != HVMSG_NONE) { - if (msg->header.message_type == HVMSG_TIMER_EXPIRED) { - hv_stimer0_isr(); - vmbus_signal_eom(msg, HVMSG_TIMER_EXPIRED); - } else { - tasklet_schedule(&hv_cpu->msg_dpc); - } - } + if (msg->header.message_type != HVMSG_NONE) + tasklet_schedule(&hv_cpu->msg_dpc); } static void __vmbus_isr(void) diff --git a/include/clocksource/hyperv_timer.h b/include/clocksource/hyperv_timer.h index d48dd4176fd3..8d3befb7e667 100644 --- a/include/clocksource/hyperv_timer.h +++ b/include/clocksource/hyperv_timer.h @@ -27,10 +27,7 @@ /* Routines called by the VMbus driver */ extern int hv_stimer_alloc(bool have_percpu_irqs); extern int hv_stimer_cleanup(unsigned int cpu); -extern void hv_stimer_legacy_init(unsigned int cpu, int sint); -extern void hv_stimer_legacy_cleanup(unsigned int cpu); extern void hv_stimer_global_cleanup(void); -extern void hv_stimer0_isr(void); extern void hv_init_clocksource(void); extern void hv_remap_tsc_clocksource(void); @@ -107,10 +104,7 @@ hv_read_tsc_page_tsc(const struct ms_hyperv_tsc_page *tsc_pg, u64 *cur_tsc, u64 } static inline int hv_stimer_cleanup(unsigned int cpu) { return 0; } -static inline void hv_stimer_legacy_init(unsigned int cpu, int sint) {} -static inline void hv_stimer_legacy_cleanup(unsigned int cpu) {} static inline void hv_stimer_global_cleanup(void) {} -static inline void hv_stimer0_isr(void) {} #endif /* CONFIG_HYPERV_TIMER */ -- cgit