| From 527ee15abae93858c8655aac033a8cc520f506bf Mon Sep 17 00:00:00 2001 |
| From: Naoya Horiguchi <n-horiguchi@ah.jp.nec.com> |
| Date: Fri, 31 Mar 2017 15:11:55 -0700 |
| Subject: [PATCH] mm, hugetlb: use pte_present() instead of pmd_present() in |
| follow_huge_pmd() |
| |
| commit c9d398fa237882ea07167e23bcfc5e6847066518 upstream. |
| |
| I found the race condition which triggers the following bug when |
| move_pages() and soft offline are called on a single hugetlb page |
| concurrently. |
| |
| Soft offlining page 0x119400 at 0x700000000000 |
| BUG: unable to handle kernel paging request at ffffea0011943820 |
| IP: follow_huge_pmd+0x143/0x190 |
| PGD 7ffd2067 |
| PUD 7ffd1067 |
| PMD 0 |
| [61163.582052] Oops: 0000 [#1] SMP |
| Modules linked in: binfmt_misc ppdev virtio_balloon parport_pc pcspkr i2c_piix4 parport i2c_core acpi_cpufreq ip_tables xfs libcrc32c ata_generic pata_acpi virtio_blk 8139too crc32c_intel ata_piix serio_raw libata virtio_pci 8139cp virtio_ring virtio mii floppy dm_mirror dm_region_hash dm_log dm_mod [last unloaded: cap_check] |
| CPU: 0 PID: 22573 Comm: iterate_numa_mo Tainted: P OE 4.11.0-rc2-mm1+ #2 |
| Hardware name: Red Hat KVM, BIOS 0.5.1 01/01/2011 |
| RIP: 0010:follow_huge_pmd+0x143/0x190 |
| RSP: 0018:ffffc90004bdbcd0 EFLAGS: 00010202 |
| RAX: 0000000465003e80 RBX: ffffea0004e34d30 RCX: 00003ffffffff000 |
| RDX: 0000000011943800 RSI: 0000000000080001 RDI: 0000000465003e80 |
| RBP: ffffc90004bdbd18 R08: 0000000000000000 R09: ffff880138d34000 |
| R10: ffffea0004650000 R11: 0000000000c363b0 R12: ffffea0011943800 |
| R13: ffff8801b8d34000 R14: ffffea0000000000 R15: 000077ff80000000 |
| FS: 00007fc977710740(0000) GS:ffff88007dc00000(0000) knlGS:0000000000000000 |
| CS: 0010 DS: 0000 ES: 0000 CR0: 0000000080050033 |
| CR2: ffffea0011943820 CR3: 000000007a746000 CR4: 00000000001406f0 |
| Call Trace: |
| follow_page_mask+0x270/0x550 |
| SYSC_move_pages+0x4ea/0x8f0 |
| SyS_move_pages+0xe/0x10 |
| do_syscall_64+0x67/0x180 |
| entry_SYSCALL64_slow_path+0x25/0x25 |
| RIP: 0033:0x7fc976e03949 |
| RSP: 002b:00007ffe72221d88 EFLAGS: 00000246 ORIG_RAX: 0000000000000117 |
| RAX: ffffffffffffffda RBX: 0000000000000000 RCX: 00007fc976e03949 |
| RDX: 0000000000c22390 RSI: 0000000000001400 RDI: 0000000000005827 |
| RBP: 00007ffe72221e00 R08: 0000000000c2c3a0 R09: 0000000000000004 |
| R10: 0000000000c363b0 R11: 0000000000000246 R12: 0000000000400650 |
| R13: 00007ffe72221ee0 R14: 0000000000000000 R15: 0000000000000000 |
| Code: 81 e4 ff ff 1f 00 48 21 c2 49 c1 ec 0c 48 c1 ea 0c 4c 01 e2 49 bc 00 00 00 00 00 ea ff ff 48 c1 e2 06 49 01 d4 f6 45 bc 04 74 90 <49> 8b 7c 24 20 40 f6 c7 01 75 2b 4c 89 e7 8b 47 1c 85 c0 7e 2a |
| RIP: follow_huge_pmd+0x143/0x190 RSP: ffffc90004bdbcd0 |
| CR2: ffffea0011943820 |
| ---[ end trace e4f81353a2d23232 ]--- |
| Kernel panic - not syncing: Fatal exception |
| Kernel Offset: disabled |
| |
| This bug is triggered when pmd_present() returns true for non-present |
| hugetlb, so fixing the present check in follow_huge_pmd() prevents it. |
| Using pmd_present() to determine present/non-present for hugetlb is not |
| correct, because pmd_present() checks multiple bits (not only |
| _PAGE_PRESENT) for historical reason and it can misjudge hugetlb state. |
| |
| Fixes: e66f17ff7177 ("mm/hugetlb: take page table lock in follow_huge_pmd()") |
| Link: http://lkml.kernel.org/r/1490149898-20231-1-git-send-email-n-horiguchi@ah.jp.nec.com |
| Signed-off-by: Naoya Horiguchi <n-horiguchi@ah.jp.nec.com> |
| Acked-by: Hillf Danton <hillf.zj@alibaba-inc.com> |
| Cc: Hugh Dickins <hughd@google.com> |
| Cc: Michal Hocko <mhocko@kernel.org> |
| Cc: "Kirill A. Shutemov" <kirill.shutemov@linux.intel.com> |
| Cc: Mike Kravetz <mike.kravetz@oracle.com> |
| Cc: Christian Borntraeger <borntraeger@de.ibm.com> |
| Cc: Gerald Schaefer <gerald.schaefer@de.ibm.com> |
| Cc: <stable@vger.kernel.org> [4.0+] |
| Signed-off-by: Andrew Morton <akpm@linux-foundation.org> |
| Signed-off-by: Linus Torvalds <torvalds@linux-foundation.org> |
| Signed-off-by: Paul Gortmaker <paul.gortmaker@windriver.com> |
| |
| diff --git a/mm/hugetlb.c b/mm/hugetlb.c |
| index 44e6d9c19af6..0a030023693f 100644 |
| --- a/mm/hugetlb.c |
| +++ b/mm/hugetlb.c |
| @@ -4471,6 +4471,7 @@ follow_huge_pmd(struct mm_struct *mm, unsigned long address, |
| { |
| struct page *page = NULL; |
| spinlock_t *ptl; |
| + pte_t pte; |
| retry: |
| ptl = pmd_lockptr(mm, pmd); |
| spin_lock(ptl); |
| @@ -4480,12 +4481,13 @@ retry: |
| */ |
| if (!pmd_huge(*pmd)) |
| goto out; |
| - if (pmd_present(*pmd)) { |
| + pte = huge_ptep_get((pte_t *)pmd); |
| + if (pte_present(pte)) { |
| page = pmd_page(*pmd) + ((address & ~PMD_MASK) >> PAGE_SHIFT); |
| if (flags & FOLL_GET) |
| get_page(page); |
| } else { |
| - if (is_hugetlb_entry_migration(huge_ptep_get((pte_t *)pmd))) { |
| + if (is_hugetlb_entry_migration(pte)) { |
| spin_unlock(ptl); |
| __migration_entry_wait(mm, (pte_t *)pmd, ptl); |
| goto retry; |
| -- |
| 2.12.0 |
| |