summaryrefslogtreecommitdiff
path: root/drivers
diff options
context:
space:
mode:
authorLinus Torvalds <torvalds@linux-foundation.org>2026-08-26 16:47:40 -0700
committerLinus Torvalds <torvalds@linux-foundation.org>2026-08-26 16:47:40 -0700
commit5e6ff28676dd92a608eb00eeb8d1319ad34024dc (patch)
tree64a1056df82ae73af0d05ae70dba0f2c38f3c8c5 /drivers
parent73e3f0710014fe6d4ed98cfc02292f6121db7558 (diff)
parentbe0cfab740e58b70047ef6e7e3d578f00ed5d258 (diff)
Merge tag 'hyperv-next-signed-20260826' of git://git.kernel.org/pub/scm/linux/kernel/git/hyperv/linux
Pull hyperv updates from Wei Liu: - Decrypt netvsc buffer on contiguous direct-map addresses (Kameron Carr) - Drop WS2012/2012R2 & Win8/8.1 Hyper-V support (Michael Kelley) - Use more meaningful errnos for hypercall status code (Hardik Garg) - Fix lost interrupts on CPU hot-unplug for Hyper-V PCI/MSI (Naman Jain) - Reserve more MSHV vectors for Linux root partition (Wei Liu) * tag 'hyperv-next-signed-20260826' of git://git.kernel.org/pub/scm/linux/kernel/git/hyperv/linux: clocksource: hyper-v: Remove support for stimer interrupts in message mode scsi: storvsc: Remove support for storvsc protocol of old Hyper-V hosts hv_netvsc: Remove GPADL teardown special case for old Hyper-V hosts hv_sock: Remove check for old Hyper-V hosts Drivers: hv: Remove support for WS2012/2012R2 & Win8/8.1 version of Hyper-V hv_netvsc: Allocate send/receive buffers using vmbus_alloc_buffer() Drivers: hv: vmbus: Add vmbus_alloc_buffer()/vmbus_free_buffer() for CoCo VMs Drivers: hv: vmbus: add vmbus_establish_gpadl_caller_decrypted() Drivers: hv: vmbus: Skip VMBus module cleanup for non-nested root partition x86/hyperv: reserve more vectors PCI: hv: Set irq_retrigger callback for the Hyper-V PCI MSI irqchip Drivers: hv: Use meaningful errnos for hypercall status codes
Diffstat (limited to 'drivers')
-rw-r--r--drivers/clocksource/hyperv_timer.c150
-rw-r--r--drivers/hv/channel.c182
-rw-r--r--drivers/hv/channel_mgmt.c8
-rw-r--r--drivers/hv/connection.c11
-rw-r--r--drivers/hv/hv.c4
-rw-r--r--drivers/hv/hv_balloon.c36
-rw-r--r--drivers/hv/hv_common.c12
-rw-r--r--drivers/hv/hv_snapshot.c5
-rw-r--r--drivers/hv/vmbus_drv.c25
-rw-r--r--drivers/net/hyperv/hyperv_net.h8
-rw-r--r--drivers/net/hyperv/netvsc.c123
-rw-r--r--drivers/net/hyperv/netvsc_drv.c6
-rw-r--r--drivers/pci/controller/pci-hyperv.c1
-rw-r--r--drivers/scsi/storvsc_drv.c53
14 files changed, 344 insertions, 280 deletions
diff --git a/drivers/clocksource/hyperv_timer.c b/drivers/clocksource/hyperv_timer.c
index df567795d175..dddfff458ebf 100644
--- a/drivers/clocksource/hyperv_timer.c
+++ b/drivers/clocksource/hyperv_timer.c
@@ -31,44 +31,20 @@ static struct clock_event_device __percpu *hv_clock_event;
/* Note: offset can hold negative values after hibernation. */
static u64 hv_sched_clock_offset __read_mostly;
-/*
- * If false, we're using the old mechanism for stimer0 interrupts
- * where it sends a VMbus message when it expires. The old
- * mechanism is used when running on older versions of Hyper-V
- * that don't support Direct Mode. While Hyper-V provides
- * four stimer's per CPU, Linux uses only stimer0.
- *
- * Because Direct Mode does not require processing a VMbus
- * message, stimer interrupts can be enabled earlier in the
- * process of booting a CPU, and consistent with when timer
- * interrupts are enabled for other clocksource drivers.
- * However, for legacy versions of Hyper-V when Direct Mode
- * is not enabled, setting up stimer interrupts must be
- * delayed until VMbus is initialized and can process the
- * interrupt message.
- */
-static bool direct_mode_enabled;
-
static int stimer0_irq = -1;
-static int stimer0_message_sint;
static __maybe_unused DEFINE_PER_CPU(long, stimer0_evt);
-/*
- * Common code for stimer0 interrupts coming via Direct Mode or
- * as a VMbus message.
- */
-void hv_stimer0_isr(void)
+static void hv_stimer0_isr(void)
{
struct clock_event_device *ce;
ce = this_cpu_ptr(hv_clock_event);
ce->event_handler(ce);
}
-EXPORT_SYMBOL_GPL(hv_stimer0_isr);
/*
* stimer0 interrupt handler for architectures that support
- * per-cpu interrupts, which also implies Direct Mode.
+ * per-cpu interrupts
*/
static irqreturn_t __maybe_unused hv_stimer0_percpu_isr(int irq, void *dev_id)
{
@@ -91,7 +67,7 @@ static int hv_ce_shutdown(struct clock_event_device *evt)
{
hv_set_msr(HV_MSR_STIMER0_COUNT, 0);
hv_set_msr(HV_MSR_STIMER0_CONFIG, 0);
- if (direct_mode_enabled && stimer0_irq >= 0)
+ if (stimer0_irq >= 0)
disable_percpu_irq(stimer0_irq);
return 0;
@@ -104,23 +80,16 @@ static int hv_ce_set_oneshot(struct clock_event_device *evt)
timer_cfg.as_uint64 = 0;
timer_cfg.enable = 1;
timer_cfg.auto_enable = 1;
- if (direct_mode_enabled) {
- /*
- * When it expires, the timer will directly interrupt
- * on the specified hardware vector/IRQ.
- */
- timer_cfg.direct_mode = 1;
- timer_cfg.apic_vector = HYPERV_STIMER0_VECTOR;
- if (stimer0_irq >= 0)
- enable_percpu_irq(stimer0_irq, IRQ_TYPE_NONE);
- } else {
- /*
- * When it expires, the timer will generate a VMbus message,
- * to be handled by the normal VMbus interrupt handler.
- */
- timer_cfg.direct_mode = 0;
- timer_cfg.sintx = stimer0_message_sint;
- }
+
+ /*
+ * When it expires, the timer will directly interrupt
+ * on the specified hardware vector/IRQ.
+ */
+ timer_cfg.direct_mode = 1;
+ timer_cfg.apic_vector = HYPERV_STIMER0_VECTOR;
+ if (stimer0_irq >= 0)
+ enable_percpu_irq(stimer0_irq, IRQ_TYPE_NONE);
+
hv_set_msr(HV_MSR_STIMER0_CONFIG, timer_cfg.as_uint64);
return 0;
}
@@ -175,25 +144,8 @@ int hv_stimer_cleanup(unsigned int cpu)
if (!hv_clock_event)
return 0;
- /*
- * In the legacy case where Direct Mode is not enabled
- * (which can only be on x86/64), stimer cleanup happens
- * relatively early in the CPU offlining process. We
- * must unbind the stimer-based clockevent device so
- * that the LAPIC timer can take over until clockevents
- * are no longer needed in the offlining process. Note
- * that clockevents_unbind_device() eventually calls
- * hv_ce_shutdown().
- *
- * The unbind should not be done when Direct Mode is
- * enabled because we may be on an architecture where
- * there are no other clockevent devices to fallback to.
- */
ce = per_cpu_ptr(hv_clock_event, cpu);
- if (direct_mode_enabled)
- hv_ce_shutdown(ce);
- else
- clockevents_unbind_device(ce, cpu);
+ hv_ce_shutdown(ce);
return 0;
}
@@ -268,23 +220,14 @@ int hv_stimer_alloc(bool have_percpu_irqs)
* Hyper-V on x86. In that case, return as error as Linux will use a
* clockevent based on emulated LAPIC timer hardware.
*/
- if (!(ms_hyperv.features & HV_MSR_SYNTIMER_AVAILABLE))
+ if (!(ms_hyperv.features & HV_MSR_SYNTIMER_AVAILABLE) ||
+ !(ms_hyperv.misc_features & HV_STIMER_DIRECT_MODE_AVAILABLE))
return -EINVAL;
hv_clock_event = alloc_percpu(struct clock_event_device);
if (!hv_clock_event)
return -ENOMEM;
- direct_mode_enabled = ms_hyperv.misc_features &
- HV_STIMER_DIRECT_MODE_AVAILABLE;
-
- /*
- * If Direct Mode isn't enabled, the remainder of the initialization
- * is done later by hv_stimer_legacy_init()
- */
- if (!direct_mode_enabled)
- return 0;
-
if (have_percpu_irqs) {
ret = hv_setup_stimer0_irq();
if (ret)
@@ -293,11 +236,6 @@ int hv_stimer_alloc(bool have_percpu_irqs)
hv_setup_stimer0_handler(hv_stimer0_isr);
}
- /*
- * Since we are in Direct Mode, stimer initialization
- * can be done now with a CPUHP value in the same range
- * as other clockevent devices.
- */
ret = cpuhp_setup_state(CPUHP_AP_HYPERV_TIMER_STARTING,
"clockevents/hyperv/stimer:starting",
hv_stimer_init, hv_stimer_cleanup);
@@ -315,66 +253,18 @@ free_clock_event:
EXPORT_SYMBOL_GPL(hv_stimer_alloc);
/*
- * hv_stimer_legacy_init -- Called from the VMbus driver to handle
- * the case when Direct Mode is not enabled, and the stimer
- * must be initialized late in the CPU onlining process.
- *
- */
-void hv_stimer_legacy_init(unsigned int cpu, int sint)
-{
- if (direct_mode_enabled)
- return;
-
- /*
- * This function gets called by each vCPU, so setting the
- * global stimer_message_sint value each time is conceptually
- * not ideal, but the value passed in is always the same and
- * it avoids introducing yet another interface into this
- * clocksource driver just to set the sint in the legacy case.
- */
- stimer0_message_sint = sint;
- (void)hv_stimer_init(cpu);
-}
-EXPORT_SYMBOL_GPL(hv_stimer_legacy_init);
-
-/*
- * hv_stimer_legacy_cleanup -- Called from the VMbus driver to
- * handle the case when Direct Mode is not enabled, and the
- * stimer must be cleaned up early in the CPU offlining
- * process.
- */
-void hv_stimer_legacy_cleanup(unsigned int cpu)
-{
- if (direct_mode_enabled)
- return;
- (void)hv_stimer_cleanup(cpu);
-}
-EXPORT_SYMBOL_GPL(hv_stimer_legacy_cleanup);
-
-/*
* Do a global cleanup of clockevents for the cases of kexec and
* vmbus exit
*/
void hv_stimer_global_cleanup(void)
{
- int cpu;
-
- /*
- * hv_stime_legacy_cleanup() will stop the stimer if Direct
- * Mode is not enabled, and fallback to the LAPIC timer.
- */
- for_each_present_cpu(cpu) {
- hv_stimer_legacy_cleanup(cpu);
- }
-
if (!hv_clock_event)
return;
- if (direct_mode_enabled) {
- cpuhp_remove_state(CPUHP_AP_HYPERV_TIMER_STARTING);
- hv_remove_stimer0_irq();
- stimer0_irq = -1;
- }
+ cpuhp_remove_state(CPUHP_AP_HYPERV_TIMER_STARTING);
+ hv_remove_stimer0_irq();
+ stimer0_irq = -1;
+
free_percpu(hv_clock_event);
hv_clock_event = NULL;
diff --git a/drivers/hv/channel.c b/drivers/hv/channel.c
index 6821f225248b..f4370617deac 100644
--- a/drivers/hv/channel.c
+++ b/drivers/hv/channel.c
@@ -13,11 +13,13 @@
#include <linux/wait.h>
#include <linux/mm.h>
#include <linux/slab.h>
+#include <linux/log2.h>
#include <linux/module.h>
#include <linux/hyperv.h>
#include <linux/uio.h>
#include <linux/interrupt.h>
#include <linux/set_memory.h>
+#include <linux/vmalloc.h>
#include <linux/export.h>
#include <asm/page.h>
#include <asm/mshyperv.h>
@@ -40,6 +42,7 @@ static inline u32 hv_gpadl_size(enum hv_gpadl_type type, u32 size)
{
switch (type) {
case HV_GPADL_BUFFER:
+ case HV_GPADL_BUFFER_DECRYPTED:
return size;
case HV_GPADL_RING:
/* The size of a ringbuffer must be page-aligned */
@@ -100,6 +103,7 @@ static inline u64 hv_gpadl_hvpfn(enum hv_gpadl_type type, void *kbuffer,
switch (type) {
case HV_GPADL_BUFFER:
+ case HV_GPADL_BUFFER_DECRYPTED:
break;
case HV_GPADL_RING:
if (i == 0)
@@ -460,7 +464,8 @@ static int __vmbus_establish_gpadl(struct vmbus_channel *channel,
}
gpadl->decrypted = !((channel->co_external_memory && type == HV_GPADL_BUFFER) ||
- (channel->co_ring_buffer && type == HV_GPADL_RING));
+ (channel->co_ring_buffer && type == HV_GPADL_RING) ||
+ (type == HV_GPADL_BUFFER_DECRYPTED));
if (gpadl->decrypted) {
/*
* The "decrypted" flag being true assumes that set_memory_decrypted() succeeds.
@@ -575,7 +580,7 @@ cleanup:
* @channel: a channel
* @kbuffer: from kmalloc or vmalloc
* @size: page-size multiple
- * @gpadl_handle: some funky thing
+ * @gpadl: output gpadl
*/
int vmbus_establish_gpadl(struct vmbus_channel *channel, void *kbuffer,
u32 size, struct vmbus_gpadl *gpadl)
@@ -585,6 +590,179 @@ int vmbus_establish_gpadl(struct vmbus_channel *channel, void *kbuffer,
}
EXPORT_SYMBOL_GPL(vmbus_establish_gpadl);
+/*
+ * vmbus_establish_gpadl_caller_decrypted - Establish a GPADL for a buffer
+ * that has already been decrypted by the caller.
+ *
+ * @channel: a channel
+ * @kbuffer: from kmalloc or vmalloc; must already be decrypted by the caller
+ * @size: page-size multiple
+ * @gpadl: output gpadl
+ *
+ * The caller is responsible for re-encrypting the buffer before freeing it.
+ */
+int vmbus_establish_gpadl_caller_decrypted(struct vmbus_channel *channel,
+ void *kbuffer, u32 size,
+ struct vmbus_gpadl *gpadl)
+{
+ return __vmbus_establish_gpadl(channel, HV_GPADL_BUFFER_DECRYPTED,
+ kbuffer, size, 0U, gpadl);
+}
+EXPORT_SYMBOL_GPL(vmbus_establish_gpadl_caller_decrypted);
+
+/**
+ * vmbus_free_buffer - release a buffer allocated by vmbus_alloc_buffer().
+ *
+ * @addr: buffer address, or NULL if none was allocated (e.g. cleanup from a
+ * failed allocation)
+ * @chunks: chunks array from vmbus_alloc_buffer(), or NULL
+ * @chunk_cnt: number of entries in @chunks
+ *
+ * When @chunks is NULL the buffer is a plain vzalloc() allocation.
+ *
+ * Otherwise tear down the vmap, and for each chunk re-encrypt and free
+ * the underlying pages. Any chunk that cannot be re-encrypted is leaked.
+ */
+void vmbus_free_buffer(void *addr, struct page **chunks, u32 chunk_cnt)
+{
+ u32 i;
+
+ if (!chunks) {
+ vfree(addr);
+ return;
+ }
+
+ vunmap(addr);
+
+ for (i = 0; i < chunk_cnt; i++) {
+ unsigned long vaddr =
+ (unsigned long)page_address(chunks[i]);
+ unsigned int order = folio_order(page_folio(chunks[i]));
+
+ if (set_memory_encrypted(vaddr, 1U << order))
+ continue;
+ __free_pages(chunks[i], order);
+ }
+
+ kvfree(chunks);
+}
+EXPORT_SYMBOL_GPL(vmbus_free_buffer);
+
+/**
+ * vmbus_alloc_buffer - allocate a host-visible, virtually-contiguous buffer.
+ *
+ * @channel: the channel the buffer will be attached to
+ * @size: requested buffer size in bytes (will be rounded up to PAGE_SIZE)
+ * @chunks_out: on success, set to the array of underlying chunks, or NULL when
+ * the buffer was allocated with vzalloc()
+ * @chunk_cnt_out: on success, set to the number of chunks
+ *
+ * Buffers not requiring decryption are allocated with vzalloc().
+ *
+ * Buffers requiring decryption are allocated as a series of
+ * physically-contiguous chunks, starting at MAX_PAGE_ORDER and falling back to
+ * smaller orders on allocation failure. Each chunk is transitioned to
+ * host-visible via set_memory_decrypted() on its direct-map address, then all
+ * chunks are combined into a virtually-contiguous range via vmap().
+ *
+ * Return: the buffer's virtual address, or NULL on failure.
+ */
+void *vmbus_alloc_buffer(struct vmbus_channel *channel,
+ u32 size,
+ struct page ***chunks_out,
+ u32 *chunk_cnt_out)
+{
+ unsigned long nr_pages = PFN_UP(size);
+ unsigned long remaining = nr_pages;
+ unsigned long page_idx = 0;
+ struct page **chunks = NULL;
+ struct page **pages = NULL;
+ int order = MAX_PAGE_ORDER;
+ u32 chunk_cnt = 0;
+ void *addr;
+ u32 i;
+ int ret;
+
+ *chunks_out = NULL;
+ *chunk_cnt_out = 0;
+
+ if (!nr_pages)
+ return NULL;
+
+ /* If the buffer does not need to be decrypted, just use vzalloc() */
+ if (!hv_is_isolation_supported() || channel->co_external_memory)
+ return vzalloc(nr_pages << PAGE_SHIFT);
+
+ /* Worst case: every chunk is a single page. */
+ chunks = kvmalloc_array(nr_pages, sizeof(*chunks),
+ GFP_KERNEL | __GFP_ZERO);
+ if (!chunks)
+ goto err;
+
+ pages = kvmalloc_array(nr_pages, sizeof(*pages), GFP_KERNEL);
+ if (!pages)
+ goto err;
+
+ while (remaining) {
+ struct page *page;
+ gfp_t gfp;
+
+ order = min(order, ilog2(remaining));
+
+ /*
+ * Use __GFP_NORETRY | __GFP_NOWARN to avoid OOM-killing,
+ * but try harder at order 0 since that is the final
+ * fallback.
+ * __GFP_COMP stores order information in the page folio.
+ */
+ gfp = GFP_KERNEL | __GFP_ZERO;
+ if (order)
+ gfp |= __GFP_COMP | __GFP_NORETRY | __GFP_NOWARN;
+
+ page = alloc_pages_node(cpu_to_node(channel->target_cpu),
+ gfp, order);
+ if (!page) {
+ if (!order--)
+ goto err;
+ continue;
+ }
+
+ ret = set_memory_decrypted((unsigned long)page_address(page),
+ 1U << order);
+ if (ret) {
+ /*
+ * set_memory_decrypted() failed; the page state is
+ * unknown so it must be leaked rather than freed.
+ */
+ goto err;
+ }
+
+ chunks[chunk_cnt++] = page;
+
+ for (i = 0; i < (1U << order); i++)
+ pages[page_idx++] = page + i;
+
+ remaining -= 1U << order;
+ }
+
+ addr = vmap(pages, nr_pages, VM_MAP, pgprot_decrypted(PAGE_KERNEL));
+ if (!addr)
+ goto err;
+
+ memset(addr, 0, nr_pages << PAGE_SHIFT);
+
+ kvfree(pages);
+ *chunks_out = chunks;
+ *chunk_cnt_out = chunk_cnt;
+ return addr;
+
+err:
+ kvfree(pages);
+ vmbus_free_buffer(NULL, chunks, chunk_cnt);
+ return NULL;
+}
+EXPORT_SYMBOL_GPL(vmbus_alloc_buffer);
+
/**
* request_arr_init - Allocates memory for the requestor array. Each slot
* keeps track of the next available slot in the array. Initially, each
diff --git a/drivers/hv/channel_mgmt.c b/drivers/hv/channel_mgmt.c
index 89d214dda360..a044fd3b3c4e 100644
--- a/drivers/hv/channel_mgmt.c
+++ b/drivers/hv/channel_mgmt.c
@@ -929,12 +929,10 @@ static void vmbus_unload_response(struct vmbus_channel_message_header *hdr)
void vmbus_initiate_unload(bool crash)
{
struct vmbus_channel_message_header hdr;
+ enum vmbus_connect_state old_state;
- if (xchg(&vmbus_connection.conn_state, DISCONNECTED) == DISCONNECTED)
- return;
-
- /* Pre-Win2012R2 hosts don't support reconnect */
- if (vmbus_proto_version < VERSION_WIN8_1)
+ old_state = xchg(&vmbus_connection.conn_state, DISCONNECTED);
+ if (old_state == DISCONNECTED || old_state == CONNECTING)
return;
reinit_completion(&vmbus_connection.unload_event);
diff --git a/drivers/hv/connection.c b/drivers/hv/connection.c
index 0fd50d4cb573..1ab3581b096a 100644
--- a/drivers/hv/connection.c
+++ b/drivers/hv/connection.c
@@ -47,7 +47,9 @@ EXPORT_SYMBOL_GPL(vmbus_proto_version);
/*
* Table of VMBus versions listed from newest to oldest.
- * VERSION_WIN7 and VERSION_WS2008 are no longer supported in
+ * VERSION_WIN7,VERSION_WS2008, VERSION_WIN8 (which is
+ * Windows Server 2012) and VERSION_WIN8_1 (which is
+ * Windows Server 2012 R2) are no longer supported in
* Linux guests and are not listed.
*/
static __u32 vmbus_versions[] = {
@@ -57,9 +59,7 @@ static __u32 vmbus_versions[] = {
VERSION_WIN10_V5_1,
VERSION_WIN10_V5,
VERSION_WIN10_V4_1,
- VERSION_WIN10,
- VERSION_WIN8_1,
- VERSION_WIN8
+ VERSION_WIN10
};
/*
@@ -304,6 +304,9 @@ int vmbus_connect(void)
for (i = 0; ; i++) {
if (i == ARRAY_SIZE(vmbus_versions)) {
ret = -EDOM;
+ pr_err("Hyper-V host does not support VMBus version %d.%d or higher;\n\
+ the host may be an older version no longer supported by Linux\n",
+ vmbus_versions[i-1] >> 16, vmbus_versions[i-1] & 0xFFFF);
goto cleanup;
}
diff --git a/drivers/hv/hv.c b/drivers/hv/hv.c
index ef4b1b03395d..fe50090dcc01 100644
--- a/drivers/hv/hv.c
+++ b/drivers/hv/hv.c
@@ -399,8 +399,6 @@ int hv_synic_init(unsigned int cpu)
else
hv_hyp_synic_enable_interrupts();
- hv_stimer_legacy_init(cpu, VMBUS_MESSAGE_SINT);
-
return 0;
}
@@ -630,8 +628,6 @@ int hv_synic_cleanup(unsigned int cpu)
return -EBUSY;
always_cleanup:
- hv_stimer_legacy_cleanup(cpu);
-
/*
* First, disable the event and message pages
* used for communicating with the host, and then
diff --git a/drivers/hv/hv_balloon.c b/drivers/hv/hv_balloon.c
index 42ce27be344d..9cba97e81111 100644
--- a/drivers/hv/hv_balloon.c
+++ b/drivers/hv/hv_balloon.c
@@ -58,6 +58,10 @@
#define DYNMEM_MAJOR_VERSION(Version) ((__u32)(Version) >> 16)
#define DYNMEM_MINOR_VERSION(Version) ((__u32)(Version) & 0xff)
+/*
+ * VERSION_1 and VERSION_2 are retained for the historical record,
+ * but are no longer supported in Linux guests.
+ */
enum {
DYNMEM_PROTOCOL_VERSION_1 = DYNMEM_MAKE_VERSION(0, 3),
DYNMEM_PROTOCOL_VERSION_2 = DYNMEM_MAKE_VERSION(1, 0),
@@ -65,9 +69,7 @@ enum {
DYNMEM_PROTOCOL_VERSION_WIN7 = DYNMEM_PROTOCOL_VERSION_1,
DYNMEM_PROTOCOL_VERSION_WIN8 = DYNMEM_PROTOCOL_VERSION_2,
- DYNMEM_PROTOCOL_VERSION_WIN10 = DYNMEM_PROTOCOL_VERSION_3,
-
- DYNMEM_PROTOCOL_VERSION_CURRENT = DYNMEM_PROTOCOL_VERSION_WIN10
+ DYNMEM_PROTOCOL_VERSION_WIN10 = DYNMEM_PROTOCOL_VERSION_3
};
/*
@@ -1434,19 +1436,9 @@ static void version_resp(struct hv_dynmem_device *dm,
version_req.version.version = dm->next_version;
dm->version = version_req.version.version;
- /*
- * Set the next version to try in case current version fails.
- * Win7 protocol ought to be the last one to try.
- */
- switch (version_req.version.version) {
- case DYNMEM_PROTOCOL_VERSION_WIN8:
- dm->next_version = DYNMEM_PROTOCOL_VERSION_WIN7;
- version_req.is_last_attempt = 0;
- break;
- default:
- dm->next_version = 0;
- version_req.is_last_attempt = 1;
- }
+ /* Set the next version to try in case current version fails. */
+ dm->next_version = 0;
+ version_req.is_last_attempt = 1;
ret = vmbus_sendpacket(dm->dev->channel, &version_req,
sizeof(struct dm_version_request),
@@ -1735,16 +1727,18 @@ static int balloon_connect_vsp(struct hv_device *dev)
/*
* Initiate the hand shake with the host and negotiate
- * a version that the host can support. We start with the
- * highest version number and go down if the host cannot
- * support it.
+ * a version that the host can support. The mechanism is in place
+ * to start with the highest version number and go down if the host
+ * cannot support it. But currently we only try the WIN10 version
+ * since support for older Hyper-V versions has been removed from
+ * Linux.
*/
memset(&version_req, 0, sizeof(struct dm_version_request));
version_req.hdr.type = DM_VERSION_REQUEST;
version_req.hdr.size = sizeof(struct dm_version_request);
version_req.hdr.trans_id = atomic_inc_return(&trans_id);
version_req.version.version = DYNMEM_PROTOCOL_VERSION_WIN10;
- version_req.is_last_attempt = 0;
+ version_req.is_last_attempt = 1;
dm_device.version = version_req.version.version;
ret = vmbus_sendpacket(dev->channel, &version_req,
@@ -1964,7 +1958,7 @@ static int balloon_probe(struct hv_device *dev,
#endif
dm_device.dev = dev;
dm_device.state = DM_INITIALIZING;
- dm_device.next_version = DYNMEM_PROTOCOL_VERSION_WIN8;
+ dm_device.next_version = 0;
init_completion(&dm_device.host_event);
init_completion(&dm_device.config_event);
INIT_LIST_HEAD(&dm_device.ha_region_list);
diff --git a/drivers/hv/hv_common.c b/drivers/hv/hv_common.c
index 6b67ac616789..31256cb22b39 100644
--- a/drivers/hv/hv_common.c
+++ b/drivers/hv/hv_common.c
@@ -787,11 +787,11 @@ static const struct hv_status_info hv_status_infos[] = {
_STATUS_INFO(HV_STATUS_INVALID_HYPERCALL_INPUT, -EINVAL),
_STATUS_INFO(HV_STATUS_INVALID_ALIGNMENT, -EIO),
_STATUS_INFO(HV_STATUS_INVALID_PARAMETER, -EINVAL),
- _STATUS_INFO(HV_STATUS_ACCESS_DENIED, -EIO),
+ _STATUS_INFO(HV_STATUS_ACCESS_DENIED, -EACCES),
_STATUS_INFO(HV_STATUS_INVALID_PARTITION_STATE, -EIO),
- _STATUS_INFO(HV_STATUS_OPERATION_DENIED, -EIO),
- _STATUS_INFO(HV_STATUS_UNKNOWN_PROPERTY, -EIO),
- _STATUS_INFO(HV_STATUS_PROPERTY_VALUE_OUT_OF_RANGE, -EIO),
+ _STATUS_INFO(HV_STATUS_OPERATION_DENIED, -EACCES),
+ _STATUS_INFO(HV_STATUS_UNKNOWN_PROPERTY, -EINVAL),
+ _STATUS_INFO(HV_STATUS_PROPERTY_VALUE_OUT_OF_RANGE, -EINVAL),
_STATUS_INFO(HV_STATUS_INSUFFICIENT_MEMORY, -ENOMEM),
_STATUS_INFO(HV_STATUS_INSUFFICIENT_CONTIGUOUS_MEMORY, -ENOMEM),
_STATUS_INFO(HV_STATUS_INSUFFICIENT_ROOT_MEMORY, -ENOMEM),
@@ -805,11 +805,9 @@ static const struct hv_status_info hv_status_infos[] = {
_STATUS_INFO(HV_STATUS_NOT_ACKNOWLEDGED, -EIO),
_STATUS_INFO(HV_STATUS_INVALID_VP_STATE, -EIO),
_STATUS_INFO(HV_STATUS_NO_RESOURCES, -EIO),
- _STATUS_INFO(HV_STATUS_PROCESSOR_FEATURE_NOT_SUPPORTED, -EIO),
+ _STATUS_INFO(HV_STATUS_PROCESSOR_FEATURE_NOT_SUPPORTED, -EOPNOTSUPP),
_STATUS_INFO(HV_STATUS_INVALID_LP_INDEX, -EINVAL),
_STATUS_INFO(HV_STATUS_INVALID_REGISTER_VALUE, -EINVAL),
- _STATUS_INFO(HV_STATUS_INVALID_LP_INDEX, -EIO),
- _STATUS_INFO(HV_STATUS_INVALID_REGISTER_VALUE, -EIO),
_STATUS_INFO(HV_STATUS_OPERATION_FAILED, -EIO),
_STATUS_INFO(HV_STATUS_TIME_OUT, -EIO),
_STATUS_INFO(HV_STATUS_CALL_PENDING, -EIO),
diff --git a/drivers/hv/hv_snapshot.c b/drivers/hv/hv_snapshot.c
index 506871aeacf0..847dc69e8250 100644
--- a/drivers/hv/hv_snapshot.c
+++ b/drivers/hv/hv_snapshot.c
@@ -372,11 +372,6 @@ static void vss_on_reset(void)
int
hv_vss_init(struct hv_util_service *srv)
{
- if (vmbus_proto_version < VERSION_WIN8_1) {
- pr_warn("Integration service 'Backup (volume snapshot)'"
- " not supported on this host version.\n");
- return -ENOTSUPP;
- }
recv_buffer = srv->recv_buffer;
vss_transaction.recv_channel = srv->channel;
vss_transaction.recv_channel->max_pkt_size = VSS_MAX_PKT_SIZE;
diff --git a/drivers/hv/vmbus_drv.c b/drivers/hv/vmbus_drv.c
index 6824bd7cb3c4..5ebdbe24b5a1 100644
--- a/drivers/hv/vmbus_drv.c
+++ b/drivers/hv/vmbus_drv.c
@@ -1320,14 +1320,8 @@ static void vmbus_message_sched(struct hv_per_cpu_context *hv_cpu, void *message
msg = (struct hv_message *)message_page_addr + VMBUS_MESSAGE_SINT;
/* Check if there are actual msgs to be processed */
- if (msg->header.message_type != HVMSG_NONE) {
- if (msg->header.message_type == HVMSG_TIMER_EXPIRED) {
- hv_stimer0_isr();
- vmbus_signal_eom(msg, HVMSG_TIMER_EXPIRED);
- } else {
- tasklet_schedule(&hv_cpu->msg_dpc);
- }
- }
+ if (msg->header.message_type != HVMSG_NONE)
+ tasklet_schedule(&hv_cpu->msg_dpc);
}
static void __vmbus_isr(void)
@@ -2982,6 +2976,13 @@ static int __init hv_acpi_init(void)
return -ENODEV;
if (hv_root_partition() && !hv_nested)
+ /*
+ * A non-nested root partition does not need VMBus client
+ * functionality. However, the mshv_root module may have
+ * a dependency on the VMBus module as described in
+ * commit 840b740a35bf. Return success so the module
+ * loads even though no VMBus initialization is done.
+ */
return 0;
/*
@@ -3030,6 +3031,14 @@ static void __exit vmbus_exit(void)
{
int cpu;
+ if (hv_root_partition() && !hv_nested)
+ /*
+ * If a non-nested root partition loaded the VMBus module,
+ * hv_acpi_init() did not do any VMBus initialization.
+ * There's nothing to clean up, so just return.
+ */
+ return;
+
unregister_syscore(&hv_synic_syscore);
hv_remove_kexec_handler();
diff --git a/drivers/net/hyperv/hyperv_net.h b/drivers/net/hyperv/hyperv_net.h
index 7397c693f984..4841367fdab2 100644
--- a/drivers/net/hyperv/hyperv_net.h
+++ b/drivers/net/hyperv/hyperv_net.h
@@ -220,6 +220,8 @@ struct net_device_context;
extern u32 netvsc_ring_bytes;
+int netvsc_workqueue_init(void);
+void netvsc_workqueue_destroy(void);
struct netvsc_device *netvsc_device_add(struct hv_device *device,
const struct netvsc_device_info *info);
int netvsc_alloc_recv_comp_ring(struct netvsc_device *net_device, u32 q_idx);
@@ -1158,6 +1160,8 @@ struct netvsc_device {
/* Receive buffer allocated by us but manages by NetVSP */
void *recv_buf;
u32 recv_buf_size; /* allocated bytes */
+ struct page **recv_buf_chunks;
+ u32 recv_buf_chunk_cnt;
struct vmbus_gpadl recv_buf_gpadl_handle;
u32 recv_section_cnt;
u32 recv_section_size;
@@ -1166,6 +1170,8 @@ struct netvsc_device {
/* Send buffer allocated by us */
void *send_buf;
u32 send_buf_size;
+ struct page **send_buf_chunks;
+ u32 send_buf_chunk_cnt;
struct vmbus_gpadl send_buf_gpadl_handle;
u32 send_section_cnt;
u32 send_section_size;
@@ -1193,7 +1199,7 @@ struct netvsc_device {
struct netvsc_channel chan_table[VRSS_CHANNEL_MAX];
- struct rcu_head rcu;
+ struct rcu_work rwork;
};
/* NdisInitialize message */
diff --git a/drivers/net/hyperv/netvsc.c b/drivers/net/hyperv/netvsc.c
index 4d319c50955e..5cd084e5696c 100644
--- a/drivers/net/hyperv/netvsc.c
+++ b/drivers/net/hyperv/netvsc.c
@@ -29,6 +29,8 @@
#include "hyperv_net.h"
#include "netvsc_trace.h"
+static struct workqueue_struct *netvsc_wq;
+
/*
* Switch the data path from the synthetic interface to the VF
* interface.
@@ -126,6 +128,47 @@ static void netvsc_subchan_work(struct work_struct *w)
rtnl_unlock();
}
+static void __free_netvsc_device(struct netvsc_device *nvdev)
+{
+ int i;
+
+ kfree(nvdev->extension);
+
+ vmbus_free_buffer(nvdev->recv_buf, nvdev->recv_buf_chunks,
+ nvdev->recv_buf_chunk_cnt);
+ vmbus_free_buffer(nvdev->send_buf, nvdev->send_buf_chunks,
+ nvdev->send_buf_chunk_cnt);
+ bitmap_free(nvdev->send_section_map);
+
+ for (i = 0; i < VRSS_CHANNEL_MAX; i++) {
+ xdp_rxq_info_unreg(&nvdev->chan_table[i].xdp_rxq);
+ kfree(nvdev->chan_table[i].recv_buf);
+ vfree(nvdev->chan_table[i].mrc.slots);
+ }
+
+ kfree(nvdev);
+}
+
+static void free_netvsc_device(struct work_struct *w)
+{
+ struct rcu_work *rwork = to_rcu_work(w);
+
+ __free_netvsc_device(container_of(rwork, struct netvsc_device, rwork));
+}
+
+int netvsc_workqueue_init(void)
+{
+ netvsc_wq = alloc_workqueue("hv_netvsc", WQ_UNBOUND, 0);
+
+ return netvsc_wq ? 0 : -ENOMEM;
+}
+
+void netvsc_workqueue_destroy(void)
+{
+ rcu_barrier();
+ destroy_workqueue(netvsc_wq);
+}
+
static struct netvsc_device *alloc_net_device(void)
{
struct netvsc_device *net_device;
@@ -144,36 +187,18 @@ static struct netvsc_device *alloc_net_device(void)
init_completion(&net_device->channel_init_wait);
init_waitqueue_head(&net_device->subchan_open);
INIT_WORK(&net_device->subchan_work, netvsc_subchan_work);
+ INIT_RCU_WORK(&net_device->rwork, free_netvsc_device);
return net_device;
}
-static void free_netvsc_device(struct rcu_head *head)
-{
- struct netvsc_device *nvdev
- = container_of(head, struct netvsc_device, rcu);
- int i;
-
- kfree(nvdev->extension);
-
- if (!nvdev->recv_buf_gpadl_handle.decrypted)
- vfree(nvdev->recv_buf);
- if (!nvdev->send_buf_gpadl_handle.decrypted)
- vfree(nvdev->send_buf);
- bitmap_free(nvdev->send_section_map);
-
- for (i = 0; i < VRSS_CHANNEL_MAX; i++) {
- xdp_rxq_info_unreg(&nvdev->chan_table[i].xdp_rxq);
- kfree(nvdev->chan_table[i].recv_buf);
- vfree(nvdev->chan_table[i].mrc.slots);
- }
-
- kfree(nvdev);
-}
-
static void free_netvsc_device_rcu(struct netvsc_device *nvdev)
{
- call_rcu(&nvdev->rcu, free_netvsc_device);
+ /*
+ * Defer the actual free to process context: vunmap() and
+ * set_memory_encrypted() cannot run from RCU softirq context.
+ */
+ queue_rcu_work(netvsc_wq, &nvdev->rwork);
}
static void netvsc_revoke_recv_buf(struct hv_device *device,
@@ -352,7 +377,10 @@ static int netvsc_init_buf(struct hv_device *device,
buf_size = min_t(unsigned int, buf_size,
NETVSC_RECEIVE_BUFFER_SIZE_LEGACY);
- net_device->recv_buf = vzalloc(buf_size);
+ net_device->recv_buf =
+ vmbus_alloc_buffer(device->channel, buf_size,
+ &net_device->recv_buf_chunks,
+ &net_device->recv_buf_chunk_cnt);
if (!net_device->recv_buf) {
netdev_err(ndev,
"unable to allocate receive buffer of size %u\n",
@@ -368,9 +396,10 @@ static int netvsc_init_buf(struct hv_device *device,
* channel. Note: This call uses the vmbus connection rather
* than the channel to establish the gpadl handle.
*/
- ret = vmbus_establish_gpadl(device->channel, net_device->recv_buf,
- buf_size,
- &net_device->recv_buf_gpadl_handle);
+ ret = vmbus_establish_gpadl_caller_decrypted(device->channel,
+ net_device->recv_buf,
+ buf_size,
+ &net_device->recv_buf_gpadl_handle);
if (ret != 0) {
netdev_err(ndev,
"unable to establish receive buffer's gpadl\n");
@@ -458,7 +487,10 @@ static int netvsc_init_buf(struct hv_device *device,
buf_size = device_info->send_sections * device_info->send_section_size;
buf_size = round_up(buf_size, PAGE_SIZE);
- net_device->send_buf = vzalloc(buf_size);
+ net_device->send_buf =
+ vmbus_alloc_buffer(device->channel, buf_size,
+ &net_device->send_buf_chunks,
+ &net_device->send_buf_chunk_cnt);
if (!net_device->send_buf) {
netdev_err(ndev, "unable to allocate send buffer of size %u\n",
buf_size);
@@ -471,9 +503,10 @@ static int netvsc_init_buf(struct hv_device *device,
* channel. Note: This call uses the vmbus connection rather
* than the channel to establish the gpadl handle.
*/
- ret = vmbus_establish_gpadl(device->channel, net_device->send_buf,
- buf_size,
- &net_device->send_buf_gpadl_handle);
+ ret = vmbus_establish_gpadl_caller_decrypted(device->channel,
+ net_device->send_buf,
+ buf_size,
+ &net_device->send_buf_gpadl_handle);
if (ret != 0) {
netdev_err(ndev,
"unable to establish send buffer's gpadl\n");
@@ -695,17 +728,8 @@ void netvsc_device_remove(struct hv_device *device)
= rtnl_dereference(net_device_ctx->nvdev);
int i;
- /*
- * Revoke receive buffer. If host is pre-Win2016 then tear down
- * receive buffer GPADL. Do the same for send buffer.
- */
netvsc_revoke_recv_buf(device, net_device, ndev);
- if (vmbus_proto_version < VERSION_WIN10)
- netvsc_teardown_recv_gpadl(device, net_device, ndev);
-
netvsc_revoke_send_buf(device, net_device, ndev);
- if (vmbus_proto_version < VERSION_WIN10)
- netvsc_teardown_send_gpadl(device, net_device, ndev);
RCU_INIT_POINTER(net_device_ctx->nvdev, NULL);
@@ -733,14 +757,9 @@ void netvsc_device_remove(struct hv_device *device)
/* Now, we can close the channel safely */
vmbus_close(device->channel);
- /*
- * If host is Win2016 or higher then we do the GPADL tear down
- * here after VMBus is closed.
- */
- if (vmbus_proto_version >= VERSION_WIN10) {
- netvsc_teardown_recv_gpadl(device, net_device, ndev);
- netvsc_teardown_send_gpadl(device, net_device, ndev);
- }
+ /* Must do the GPADL teardown after channel is closed */
+ netvsc_teardown_recv_gpadl(device, net_device, ndev);
+ netvsc_teardown_send_gpadl(device, net_device, ndev);
/* Release all resources */
free_netvsc_device_rcu(net_device);
@@ -1874,7 +1893,11 @@ cleanup:
netif_napi_del(&net_device->chan_table[0].napi);
cleanup2:
- free_netvsc_device(&net_device->rcu);
+ /*
+ * net_device was never published, so we don't need to wait for an
+ * RCU grace period -- call the free routine synchronously.
+ */
+ __free_netvsc_device(net_device);
return ERR_PTR(ret);
}
diff --git a/drivers/net/hyperv/netvsc_drv.c b/drivers/net/hyperv/netvsc_drv.c
index ee5ab5ceb2be..1d43c73fd73f 100644
--- a/drivers/net/hyperv/netvsc_drv.c
+++ b/drivers/net/hyperv/netvsc_drv.c
@@ -2867,12 +2867,17 @@ static void __exit netvsc_drv_exit(void)
{
unregister_netdevice_notifier(&netvsc_netdev_notifier);
vmbus_driver_unregister(&netvsc_drv);
+ netvsc_workqueue_destroy();
}
static int __init netvsc_drv_init(void)
{
int ret;
+ ret = netvsc_workqueue_init();
+ if (ret)
+ return ret;
+
if (ring_size < RING_SIZE_MIN) {
ring_size = RING_SIZE_MIN;
pr_info("Increased ring_size to %u (min allowed)\n",
@@ -2890,6 +2895,7 @@ static int __init netvsc_drv_init(void)
err_vmbus_reg:
unregister_netdevice_notifier(&netvsc_netdev_notifier);
+ netvsc_workqueue_destroy();
return ret;
}
diff --git a/drivers/pci/controller/pci-hyperv.c b/drivers/pci/controller/pci-hyperv.c
index cfc8fa403dad..89816a2bd7cd 100644
--- a/drivers/pci/controller/pci-hyperv.c
+++ b/drivers/pci/controller/pci-hyperv.c
@@ -2119,6 +2119,7 @@ static bool hv_pcie_init_dev_msi_info(struct device *dev, struct irq_domain *dom
info->ops->msi_prepare = hv_msi_prepare;
chip->irq_set_affinity = irq_chip_set_affinity_parent;
+ chip->irq_retrigger = irq_chip_retrigger_hierarchy;
if (IS_ENABLED(CONFIG_X86))
chip->flags |= IRQCHIP_MOVE_DEFERRED;
diff --git a/drivers/scsi/storvsc_drv.c b/drivers/scsi/storvsc_drv.c
index 58b6ae96d03f..ef3cd3bc7390 100644
--- a/drivers/scsi/storvsc_drv.c
+++ b/drivers/scsi/storvsc_drv.c
@@ -47,9 +47,15 @@
* V1 RC < 2008/1/31: 1.0
* V1 RC > 2008/1/31: 2.0
* Win7: 4.2
- * Win8: 5.1
- * Win8.1: 6.0
- * Win10: 6.2
+ * Win8/WS2012: 5.1
+ * Win8.1/WS2012R2: 6.0 (also for HvLite paravisor in Azure)
+ * Win10/WS2016: 6.2
+ *
+ * Protocol versions earlier than Win8.1 are no longer supported since
+ * Win8.1/WS2012R2 and earlier hosts are no longer supported by Linux.
+ * But protocol version 6.0 is retained since it is used by the HvLite
+ * paravisor in Azure. The #define's for the earlier versions remain
+ * for the historical record.
*/
#define VMSTOR_PROTO_VERSION(MAJOR_, MINOR_) ((((MAJOR_) & 0xff) << 8) | \
@@ -218,7 +224,6 @@ struct vmscsi_request {
static const int protocol_version[] = {
VMSTOR_PROTO_VERSION_WIN10,
VMSTOR_PROTO_VERSION_WIN8_1,
- VMSTOR_PROTO_VERSION_WIN8,
};
@@ -1631,13 +1636,12 @@ static int storvsc_sdev_configure(struct scsi_device *sdevice,
sdevice->no_write_same = 1;
/*
- * If the host is WIN8 or WIN8 R2, claim conformance to SPC-3
+ * If the host is WIN8 R2, claim conformance to SPC-3
* if the device is a MSFT virtual device. If the host is
* WIN10 or newer, allow write_same.
*/
if (!strncmp(sdevice->vendor, "Msft", 4)) {
switch (vmstor_proto_version) {
- case VMSTOR_PROTO_VERSION_WIN8:
case VMSTOR_PROTO_VERSION_WIN8_1:
sdevice->scsi_level = SCSI_SPC_3;
break;
@@ -1733,28 +1737,6 @@ static enum scsi_timeout_action storvsc_eh_timed_out(struct scsi_cmnd *scmnd)
return SCSI_EH_RESET_TIMER;
}
-static bool storvsc_scsi_cmd_ok(struct scsi_cmnd *scmnd)
-{
- bool allowed = true;
- u8 scsi_op = scmnd->cmnd[0];
-
- switch (scsi_op) {
- /* the host does not handle WRITE_SAME, log accident usage */
- case WRITE_SAME:
- /*
- * smartd sends this command and the host does not handle
- * this. So, don't send it.
- */
- case SET_WINDOW:
- set_host_byte(scmnd, DID_ERROR);
- allowed = false;
- break;
- default:
- break;
- }
- return allowed;
-}
-
static enum scsi_qc_status storvsc_queuecommand(struct Scsi_Host *host,
struct scsi_cmnd *scmnd)
{
@@ -1768,21 +1750,6 @@ static enum scsi_qc_status storvsc_queuecommand(struct Scsi_Host *host,
u32 payload_sz;
u32 length;
- if (vmstor_proto_version <= VMSTOR_PROTO_VERSION_WIN8) {
- /*
- * On legacy hosts filter unimplemented commands.
- * Future hosts are expected to correctly handle
- * unsupported commands. Furthermore, it is
- * possible that some of the currently
- * unsupported commands maybe supported in
- * future versions of the host.
- */
- if (!storvsc_scsi_cmd_ok(scmnd)) {
- scsi_done(scmnd);
- return 0;
- }
- }
-
/* Setup the cmd request */
cmd_request->cmd = scmnd;