]> git.ipfire.org Git - thirdparty/linux.git/commitdiff
x86/e820: Fix handling of subpage regions when calculating nosave ranges in e820__reg...
authorMyrrh Periwinkle <myrrhperiwinkle@qtmlabs.xyz>
Sun, 6 Apr 2025 04:45:22 +0000 (11:45 +0700)
committerIngo Molnar <mingo@kernel.org>
Mon, 7 Apr 2025 17:20:08 +0000 (19:20 +0200)
While debugging kexec/hibernation hangs and crashes, it turned out that
the current implementation of e820__register_nosave_regions() suffers from
multiple serious issues:

 - The end of last region is tracked by PFN, causing it to find holes
   that aren't there if two consecutive subpage regions are present

 - The nosave PFN ranges derived from holes are rounded out (instead of
   rounded in) which makes it inconsistent with how explicitly reserved
   regions are handled

Fix this by:

 - Treating reserved regions as if they were holes, to ensure consistent
   handling (rounding out nosave PFN ranges is more correct as the
   kernel does not use partial pages)

 - Tracking the end of the last RAM region by address instead of pages
   to detect holes more precisely

These bugs appear to have been introduced about ~18 years ago with the very
first version of e820_mark_nosave_regions(), and its flawed assumptions were
carried forward uninterrupted through various waves of rewrites and renames.

[ mingo: Added Git archeology details, for kicks and giggles. ]

Fixes: e8eff5ac294e ("[PATCH] Make swsusp avoid memory holes and reserved memory regions on x86_64")
Reported-by: Roberto Ricci <io@r-ricci.it>
Tested-by: Roberto Ricci <io@r-ricci.it>
Signed-off-by: Myrrh Periwinkle <myrrhperiwinkle@qtmlabs.xyz>
Signed-off-by: Ingo Molnar <mingo@kernel.org>
Cc: Rafael J. Wysocki <rafael.j.wysocki@intel.com>
Cc: Ard Biesheuvel <ardb@kernel.org>
Cc: H. Peter Anvin <hpa@zytor.com>
Cc: Kees Cook <keescook@chromium.org>
Cc: Linus Torvalds <torvalds@linux-foundation.org>
Cc: David Woodhouse <dwmw@amazon.co.uk>
Cc: Len Brown <len.brown@intel.com>
Cc: stable@vger.kernel.org
Link: https://lore.kernel.org/r/20250406-fix-e820-nosave-v3-1-f3787bc1ee1d@qtmlabs.xyz
Closes: https://lore.kernel.org/all/Z4WFjBVHpndct7br@desktop0a/
arch/x86/kernel/e820.c

index 57120f0749cc3c23844eeb36820705687e08bbf7..9d8dd8deb2a702bd34b961ca4f5eba8a8d9643d0 100644 (file)
@@ -753,22 +753,21 @@ void __init e820__memory_setup_extended(u64 phys_addr, u32 data_len)
 void __init e820__register_nosave_regions(unsigned long limit_pfn)
 {
        int i;
-       unsigned long pfn = 0;
+       u64 last_addr = 0;
 
        for (i = 0; i < e820_table->nr_entries; i++) {
                struct e820_entry *entry = &e820_table->entries[i];
 
-               if (pfn < PFN_UP(entry->addr))
-                       register_nosave_region(pfn, PFN_UP(entry->addr));
-
-               pfn = PFN_DOWN(entry->addr + entry->size);
-
                if (entry->type != E820_TYPE_RAM)
-                       register_nosave_region(PFN_UP(entry->addr), pfn);
+                       continue;
 
-               if (pfn >= limit_pfn)
-                       break;
+               if (last_addr < entry->addr)
+                       register_nosave_region(PFN_DOWN(last_addr), PFN_UP(entry->addr));
+
+               last_addr = entry->addr + entry->size;
        }
+
+       register_nosave_region(PFN_DOWN(last_addr), limit_pfn);
 }
 
 #ifdef CONFIG_ACPI