* [PATCH v3] drivers/hv: remove deposited pages from direct map
@ 2026-09-17 20:10 Magnus Kulke
2026-09-18 0:23 ` Mukesh R
0 siblings, 1 reply; 2+ messages in thread
From: Magnus Kulke @ 2026-09-17 20:10 UTC (permalink / raw)
To: linux-hyperv
Cc: Paolo Bonzini, Souradeep Chakrabarti, Wei Liu, Haiyang Zhang,
Dexuan Cui, Magnus Kulke, Long Li, linux-arch, K. Y. Srinivasan,
Anirudh Rayabharam, Arnd Bergmann, linux-kernel, Wei Liu
hv_call_deposit_pages() donates pages (deposit) to hypervisor for L2
guest on L1VH systems via HVCALL_DEPOSIT_MEMORY. The hypervisor takes
ownership of those pages and per contract revokes root partition
access to them, raising a #GP on access from the L1VH root partition.
However, the pages remain mapped in the kernel direct map, so kernel
code may still access them even though the hypervisor has revoked
access.
Helpers such as "load_unaligned_zeropad()" deliberately read past the
end of a buffer and across page boundaries. A read into an unmapped
page is tolerated and triggers a #PF, for which the kernel executed
a fixup in the exception table.
If such a call steps into a page that has been deposited, the access
raises a #GP by the hypervisor from which the above handler cannot
recover and the kernel panics:
Oops: general protection fault, maybe for address 0xff1100941a3dfffc
RIP: 0010:csum_partial+0xe5/0x110
This condition will appear on L1VH system that have created L2
partitions (and hence deposited pages) and exercise networking code
paths such as csum_partial() can trigger this condition when a buffer
ends close to a page boundary (e.g fffc in the above example).
The fix is to remove the deposited pages from the direct map before
they are passed to the hypervisor, and restore them when the hypervisor
returns them again.
We want to avoid flushing the TLB in the loop, so we use the _noflush()
variant of set_direct_map_valid() when marking a deposited page invalid
and flush the affected page ranges in one go ourselves. In the opposite
direction this is not required:
> If a paging-structure entry is modified to change the P flag from
> 0 to 1, no invalidation is necessary. This is because no TLB entry
> or paging-structure cache entry is created with information from a
> paging-structure entry in which the P flag is 0.
(Intel SDM Vol. 3, 4.10.4.3)
Signed-off-by: Magnus Kulke <magnuskulke@linux.microsoft.com>
---
Changes since v2:
- Checkpatch format fix
Changes since RFC:
- Handle direct-map restoration failures without returning unmapped
pages to the allocator.
- Move freeing of withdrawn pages into the restoration helper.
---
drivers/hv/hv_proc.c | 77 +++++++++++++++++++++++++++++++++-
drivers/hv/mshv_root_hv_call.c | 4 +-
include/asm-generic/mshyperv.h | 4 ++
3 files changed, 81 insertions(+), 4 deletions(-)
diff --git a/drivers/hv/hv_proc.c b/drivers/hv/hv_proc.c
index 57b2c64197cb..2a392b45205d 100644
--- a/drivers/hv/hv_proc.c
+++ b/drivers/hv/hv_proc.c
@@ -7,7 +7,9 @@
#include <linux/cpuhotplug.h>
#include <linux/minmax.h>
#include <linux/export.h>
+#include <linux/set_memory.h>
#include <asm/mshyperv.h>
+#include <asm/tlbflush.h>
/*
* See struct hv_deposit_memory. The first u64 is partition ID, the rest
@@ -15,6 +17,35 @@
*/
#define HV_DEPOSIT_MAX (HV_HYP_PAGE_SIZE / sizeof(u64) - 1)
+/*
+ * Add or remove a set of physically contiguous page runs from the kernel
+ * direct map. Once a page has been deposited the hypervisor owns it and
+ * revokes root partition access to it.
+ */
+static int hv_deposit_update_direct_map(struct page **pages, int *counts,
+ int num_allocations, bool valid)
+{
+ int i, err, ret = 0;
+
+ for (i = 0; i < num_allocations; ++i) {
+ err = set_direct_map_valid_noflush(pages[i], counts[i], valid);
+ if (err && !ret)
+ ret = err;
+ }
+
+ if (valid)
+ return ret;
+
+ for (i = 0; i < num_allocations; ++i) {
+ unsigned long addr = (unsigned long)page_address(pages[i]);
+ unsigned long size = (unsigned long)counts[i] << PAGE_SHIFT;
+
+ flush_tlb_kernel_range(addr, addr + size);
+ }
+
+ return ret;
+}
+
/* Deposits exact number of pages. Must be called with interrupts enabled. */
int hv_call_deposit_pages(int node, u64 partition_id, u32 num_pages)
{
@@ -72,6 +103,10 @@ int hv_call_deposit_pages(int node, u64 partition_id, u32 num_pages)
}
num_allocations = i;
+ ret = hv_deposit_update_direct_map(pages, counts, num_allocations, false);
+ if (ret)
+ goto err_restore_direct_map;
+
local_irq_save(flags);
input_page = *this_cpu_ptr(hyperv_pcpu_input_arg);
@@ -90,12 +125,23 @@ int hv_call_deposit_pages(int node, u64 partition_id, u32 num_pages)
if (!hv_result_success(status)) {
hv_status_err(status, "\n");
ret = hv_result_to_errno(status);
- goto err_free_allocations;
+ goto err_restore_direct_map;
}
ret = 0;
goto free_buf;
+err_restore_direct_map:
+ /*
+ * We don't want to return pages to the allocator if weren't able to
+ * mark them valid in the direct map.
+ */
+ if (hv_deposit_update_direct_map(pages, counts, num_allocations, true)) {
+ WARN(1, "leaking %d page block(s) that could not be set to valid\n",
+ num_allocations);
+ goto free_buf;
+ }
+
err_free_allocations:
for (i = 0; i < num_allocations; ++i) {
base_pfn = page_to_pfn(pages[i]);
@@ -110,6 +156,35 @@ int hv_call_deposit_pages(int node, u64 partition_id, u32 num_pages)
}
EXPORT_SYMBOL_GPL(hv_call_deposit_pages);
+/*
+ * Put withdrawn pages back in the direct map. Counterpart to the direct map
+ * removal done by hv_call_deposit_pages().
+ */
+void hv_restore_withdrawn_pages(const u64 *pfns, int count)
+{
+ int i, ret = 0;
+ struct page *page;
+
+ for (i = 0; i < count; ++i) {
+ page = pfn_to_page(pfns[i]);
+ ret = set_direct_map_valid_noflush(page, 1, true);
+ /*
+ * HV_DEPOSIT_MAX is capped at 511, so a deposit range cannot cover
+ * a 2MiB page, so deposited pages are of 4k granularity and cannot
+ * be collapses into a 2MiB page, which would require an allocation
+ * and can potentially fail.
+ *
+ * Should it fail anyway we leak the page, if we would hand it
+ * back to the allocator we would introduce faults into random other
+ * parts.
+ */
+ if (WARN_ON_ONCE(ret))
+ continue;
+ __free_page(page);
+ }
+}
+EXPORT_SYMBOL_GPL(hv_restore_withdrawn_pages);
+
int hv_deposit_memory_node(int node, u64 partition_id,
u64 hv_status)
{
diff --git a/drivers/hv/mshv_root_hv_call.c b/drivers/hv/mshv_root_hv_call.c
index cb55d4d4be2e..150a0c63ebc8 100644
--- a/drivers/hv/mshv_root_hv_call.c
+++ b/drivers/hv/mshv_root_hv_call.c
@@ -46,7 +46,6 @@ int hv_call_withdraw_memory(u64 count, int node, u64 partition_id)
struct page *page;
u16 completed;
u64 status, withdrawn = 0;
- int i;
unsigned long flags;
page = alloc_page(GFP_KERNEL);
@@ -69,8 +68,7 @@ int hv_call_withdraw_memory(u64 count, int node, u64 partition_id)
completed = hv_repcomp(status);
- for (i = 0; i < completed; i++)
- __free_page(pfn_to_page(output_page->gpa_page_list[i]));
+ hv_restore_withdrawn_pages(output_page->gpa_page_list, completed);
if (!hv_result_success(status)) {
if (hv_result(status) == HV_STATUS_NO_RESOURCES)
diff --git a/include/asm-generic/mshyperv.h b/include/asm-generic/mshyperv.h
index bf601d67cecb..397c8ec0ce9a 100644
--- a/include/asm-generic/mshyperv.h
+++ b/include/asm-generic/mshyperv.h
@@ -346,6 +346,7 @@ static inline bool hv_parent_partition(void)
bool hv_result_needs_memory(u64 status);
int hv_deposit_memory_node(int node, u64 partition_id, u64 status);
int hv_call_deposit_pages(int node, u64 partition_id, u32 num_pages);
+void hv_restore_withdrawn_pages(const u64 *pfns, int count);
int hv_call_add_logical_proc(int node, u32 lp_index, u32 acpi_id);
int hv_call_notify_all_processors_started(void);
bool hv_lp_exists(u32 lp_index);
@@ -364,6 +365,9 @@ static inline int hv_call_deposit_pages(int node, u64 partition_id, u32 num_page
{
return -EOPNOTSUPP;
}
+
+static inline void hv_restore_withdrawn_pages(const u64 *pfns, int count) { }
+
static inline int hv_call_add_logical_proc(int node, u32 lp_index, u32 acpi_id)
{
return -EOPNOTSUPP;
--
2.34.1
^ permalink raw reply [flat|nested] 2+ messages in thread
* Re: [PATCH v3] drivers/hv: remove deposited pages from direct map
2026-09-17 20:10 [PATCH v3] drivers/hv: remove deposited pages from direct map Magnus Kulke
@ 2026-09-18 0:23 ` Mukesh R
0 siblings, 0 replies; 2+ messages in thread
From: Mukesh R @ 2026-09-18 0:23 UTC (permalink / raw)
To: Magnus Kulke, linux-hyperv
Cc: Paolo Bonzini, Souradeep Chakrabarti, Wei Liu, Haiyang Zhang,
Dexuan Cui, Magnus Kulke, Long Li, linux-arch, K. Y. Srinivasan,
Anirudh Rayabharam, Arnd Bergmann, linux-kernel, Wei Liu
On 9/17/26 13:10, Magnus Kulke wrote:
> hv_call_deposit_pages() donates pages (deposit) to hypervisor for L2
> guest on L1VH systems via HVCALL_DEPOSIT_MEMORY. The hypervisor takes
> ownership of those pages and per contract revokes root partition
> access to them, raising a #GP on access from the L1VH root partition.
>
> However, the pages remain mapped in the kernel direct map, so kernel
> code may still access them even though the hypervisor has revoked
> access.
>
> Helpers such as "load_unaligned_zeropad()" deliberately read past the
> end of a buffer and across page boundaries. A read into an unmapped
> page is tolerated and triggers a #PF, for which the kernel executed
> a fixup in the exception table.
>
> If such a call steps into a page that has been deposited, the access
> raises a #GP by the hypervisor from which the above handler cannot
> recover and the kernel panics:
>
> Oops: general protection fault, maybe for address 0xff1100941a3dfffc
> RIP: 0010:csum_partial+0xe5/0x110
>
> This condition will appear on L1VH system that have created L2
> partitions (and hence deposited pages) and exercise networking code
> paths such as csum_partial() can trigger this condition when a buffer
> ends close to a page boundary (e.g fffc in the above example).
>
> The fix is to remove the deposited pages from the direct map before
> they are passed to the hypervisor, and restore them when the hypervisor
> returns them again.
>
> We want to avoid flushing the TLB in the loop, so we use the _noflush()
> variant of set_direct_map_valid() when marking a deposited page invalid
> and flush the affected page ranges in one go ourselves. In the opposite
> direction this is not required:
>
> > If a paging-structure entry is modified to change the P flag from
> > 0 to 1, no invalidation is necessary. This is because no TLB entry
> > or paging-structure cache entry is created with information from a
> > paging-structure entry in which the P flag is 0.
>
> (Intel SDM Vol. 3, 4.10.4.3)
>
> Signed-off-by: Magnus Kulke <magnuskulke@linux.microsoft.com>
FYI:
https://lore.kernel.org/linux-hyperv/20260912000318.2959621-1-mrathor@linux.microsoft.com/
Thanks,
-Mukesh
> ---
> Changes since v2:
> - Checkpatch format fix
>
> Changes since RFC:
> - Handle direct-map restoration failures without returning unmapped
> pages to the allocator.
> - Move freeing of withdrawn pages into the restoration helper.
> ---
> drivers/hv/hv_proc.c | 77 +++++++++++++++++++++++++++++++++-
> drivers/hv/mshv_root_hv_call.c | 4 +-
> include/asm-generic/mshyperv.h | 4 ++
> 3 files changed, 81 insertions(+), 4 deletions(-)
>
> diff --git a/drivers/hv/hv_proc.c b/drivers/hv/hv_proc.c
> index 57b2c64197cb..2a392b45205d 100644
> --- a/drivers/hv/hv_proc.c
> +++ b/drivers/hv/hv_proc.c
> @@ -7,7 +7,9 @@
> #include <linux/cpuhotplug.h>
> #include <linux/minmax.h>
> #include <linux/export.h>
> +#include <linux/set_memory.h>
> #include <asm/mshyperv.h>
> +#include <asm/tlbflush.h>
>
> /*
> * See struct hv_deposit_memory. The first u64 is partition ID, the rest
> @@ -15,6 +17,35 @@
> */
> #define HV_DEPOSIT_MAX (HV_HYP_PAGE_SIZE / sizeof(u64) - 1)
>
> +/*
> + * Add or remove a set of physically contiguous page runs from the kernel
> + * direct map. Once a page has been deposited the hypervisor owns it and
> + * revokes root partition access to it.
> + */
> +static int hv_deposit_update_direct_map(struct page **pages, int *counts,
> + int num_allocations, bool valid)
> +{
> + int i, err, ret = 0;
> +
> + for (i = 0; i < num_allocations; ++i) {
> + err = set_direct_map_valid_noflush(pages[i], counts[i], valid);
> + if (err && !ret)
> + ret = err;
> + }
> +
> + if (valid)
> + return ret;
> +
> + for (i = 0; i < num_allocations; ++i) {
> + unsigned long addr = (unsigned long)page_address(pages[i]);
> + unsigned long size = (unsigned long)counts[i] << PAGE_SHIFT;
> +
> + flush_tlb_kernel_range(addr, addr + size);
> + }
> +
> + return ret;
> +}
> +
> /* Deposits exact number of pages. Must be called with interrupts enabled. */
> int hv_call_deposit_pages(int node, u64 partition_id, u32 num_pages)
> {
> @@ -72,6 +103,10 @@ int hv_call_deposit_pages(int node, u64 partition_id, u32 num_pages)
> }
> num_allocations = i;
>
> + ret = hv_deposit_update_direct_map(pages, counts, num_allocations, false);
> + if (ret)
> + goto err_restore_direct_map;
> +
> local_irq_save(flags);
>
> input_page = *this_cpu_ptr(hyperv_pcpu_input_arg);
> @@ -90,12 +125,23 @@ int hv_call_deposit_pages(int node, u64 partition_id, u32 num_pages)
> if (!hv_result_success(status)) {
> hv_status_err(status, "\n");
> ret = hv_result_to_errno(status);
> - goto err_free_allocations;
> + goto err_restore_direct_map;
> }
>
> ret = 0;
> goto free_buf;
>
> +err_restore_direct_map:
> + /*
> + * We don't want to return pages to the allocator if weren't able to
> + * mark them valid in the direct map.
> + */
> + if (hv_deposit_update_direct_map(pages, counts, num_allocations, true)) {
> + WARN(1, "leaking %d page block(s) that could not be set to valid\n",
> + num_allocations);
> + goto free_buf;
> + }
> +
> err_free_allocations:
> for (i = 0; i < num_allocations; ++i) {
> base_pfn = page_to_pfn(pages[i]);
> @@ -110,6 +156,35 @@ int hv_call_deposit_pages(int node, u64 partition_id, u32 num_pages)
> }
> EXPORT_SYMBOL_GPL(hv_call_deposit_pages);
>
> +/*
> + * Put withdrawn pages back in the direct map. Counterpart to the direct map
> + * removal done by hv_call_deposit_pages().
> + */
> +void hv_restore_withdrawn_pages(const u64 *pfns, int count)
> +{
> + int i, ret = 0;
> + struct page *page;
> +
> + for (i = 0; i < count; ++i) {
> + page = pfn_to_page(pfns[i]);
> + ret = set_direct_map_valid_noflush(page, 1, true);
> + /*
> + * HV_DEPOSIT_MAX is capped at 511, so a deposit range cannot cover
> + * a 2MiB page, so deposited pages are of 4k granularity and cannot
> + * be collapses into a 2MiB page, which would require an allocation
> + * and can potentially fail.
> + *
> + * Should it fail anyway we leak the page, if we would hand it
> + * back to the allocator we would introduce faults into random other
> + * parts.
> + */
> + if (WARN_ON_ONCE(ret))
> + continue;
> + __free_page(page);
> + }
> +}
> +EXPORT_SYMBOL_GPL(hv_restore_withdrawn_pages);
> +
> int hv_deposit_memory_node(int node, u64 partition_id,
> u64 hv_status)
> {
> diff --git a/drivers/hv/mshv_root_hv_call.c b/drivers/hv/mshv_root_hv_call.c
> index cb55d4d4be2e..150a0c63ebc8 100644
> --- a/drivers/hv/mshv_root_hv_call.c
> +++ b/drivers/hv/mshv_root_hv_call.c
> @@ -46,7 +46,6 @@ int hv_call_withdraw_memory(u64 count, int node, u64 partition_id)
> struct page *page;
> u16 completed;
> u64 status, withdrawn = 0;
> - int i;
> unsigned long flags;
>
> page = alloc_page(GFP_KERNEL);
> @@ -69,8 +68,7 @@ int hv_call_withdraw_memory(u64 count, int node, u64 partition_id)
>
> completed = hv_repcomp(status);
>
> - for (i = 0; i < completed; i++)
> - __free_page(pfn_to_page(output_page->gpa_page_list[i]));
> + hv_restore_withdrawn_pages(output_page->gpa_page_list, completed);
>
> if (!hv_result_success(status)) {
> if (hv_result(status) == HV_STATUS_NO_RESOURCES)
> diff --git a/include/asm-generic/mshyperv.h b/include/asm-generic/mshyperv.h
> index bf601d67cecb..397c8ec0ce9a 100644
> --- a/include/asm-generic/mshyperv.h
> +++ b/include/asm-generic/mshyperv.h
> @@ -346,6 +346,7 @@ static inline bool hv_parent_partition(void)
> bool hv_result_needs_memory(u64 status);
> int hv_deposit_memory_node(int node, u64 partition_id, u64 status);
> int hv_call_deposit_pages(int node, u64 partition_id, u32 num_pages);
> +void hv_restore_withdrawn_pages(const u64 *pfns, int count);
> int hv_call_add_logical_proc(int node, u32 lp_index, u32 acpi_id);
> int hv_call_notify_all_processors_started(void);
> bool hv_lp_exists(u32 lp_index);
> @@ -364,6 +365,9 @@ static inline int hv_call_deposit_pages(int node, u64 partition_id, u32 num_page
> {
> return -EOPNOTSUPP;
> }
> +
> +static inline void hv_restore_withdrawn_pages(const u64 *pfns, int count) { }
> +
> static inline int hv_call_add_logical_proc(int node, u32 lp_index, u32 acpi_id)
> {
> return -EOPNOTSUPP;
^ permalink raw reply [flat|nested] 2+ messages in thread
end of thread, other threads:[~2026-09-18 0:23 UTC | newest]
Thread overview: 2+ messages (download: mbox.gz / follow: Atom feed)
-- links below jump to the message on this page --
2026-09-17 20:10 [PATCH v3] drivers/hv: remove deposited pages from direct map Magnus Kulke
2026-09-18 0:23 ` Mukesh R
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®