mirror of
https://git.kernel.org/pub/scm/linux/kernel/git/torvalds/linux.git
synced 2026-09-18 22:19:30 +02:00
mm/huge_memory: bypass THP tuneables for huge pfnmap mappings
The sysfs THP tuneables at /sys/kernel/mm/transparent_huge_pages/ rather confusingly only control the behaviour of THP in some instances. They are not applicable to MADV_COLLAPSE operations, nor to DAX mappings. Long-term, THP is predicated upon compaction being able to obtain large folios to populate THP ranges. However, vm_normal_folio() returns NULL for PFN map mappings, thus their reference count is maintained by the driver, not core mm. As a consequence, the folios are not subject to reclaim nor compaction, so are not truly part of the THP mechanism at all. However, since commit5dd40721f1("mm: allow THP orders for PFNMAPs") introduced the ability to establish huge PFN maps, they have been subject to THP tuneables. This is incorrect - if a huge PFN map is available (defined by vma->vm_ops->huge_fault being non-NULL for a VMA_PFNMAP_BIT VMA), then it should be mapped huge upon fault-in. Correct this by explicitly checking for this while ensuring that smaps continues to accurately report THPeligible statistics. While here, abstract the entire file-backed THP check in vma_can_map_huge_file(), with sensible separation of logic into helper functions. Note that drm_gem_shmem_mmap() and panthor_gem_mmap() establish huge PFN maps of shmem folios, however they are marked unevictable in drm_gem_get_pages(), and in any case would fail the reference check in __remove_mapping() even if they weren't. Failing to map huge PFN maps has resulted in significant real-world performance degradation, see links for details. [ziy@nvidia.com: rename some functions] Link: https://lore.kernel.org/DL1HIHWYJ7TB.1CY76SJS0V03L@nvidia.com Link: https://lore.kernel.org/20260827-hugepfn-allowable-orders-v1-1-94819c8807c8@kernel.org Fixes:5dd40721f1("mm: allow THP orders for PFNMAPs") Signed-off-by: Lorenzo Stoakes (ARM) <ljs@kernel.org> Signed-off-by: Zi Yan <ziy@nvidia.com> Reported-by: Cedric Le Goater <clg@redhat.com> Closes: https://lore.kernel.org/linux-mm/20260805055544.1568534-1-clg@redhat.com/ Reported-by: Saravanan D <saravanand@crusoe.ai> Closes: https://lore.kernel.org/linux-mm/20260821070520.25759-1-saravanand@crusoe.ai/ Reviewed-by: Zi Yan <ziy@nvidia.com> Tested-by: Saravanan D <saravanand@crusoe.ai> Tested-by: Lance Yang <lance.yang@linux.dev> Reviewed-by: SJ Park <sj@kernel.org> Reviewed-by: Baolin Wang <baolin.wang@linux.alibaba.com> Cc: Barry Song <baohua@kernel.org> Cc: David Hildenbrand <david@kernel.org> Cc: Dev Jain <dev.jain@arm.com> Cc: Jason Gunthorpe <jgg@ziepe.ca> Cc: Liam R. Howlett <liam@infradead.org> Cc: Peter Xu <peterx@redhat.com> Cc: Ryan Roberts <ryan.roberts@arm.com> Cc: <stable@vger.kernel.org> Signed-off-by: Andrew Morton <akpm@linux-foundation.org>
This commit is contained in:
committed by
Andrew Morton
parent
0d9ff90a54
commit
e384abeb55
+64
-22
@@ -92,7 +92,7 @@ unsigned long huge_anon_orders_madvise __read_mostly;
|
||||
unsigned long huge_anon_orders_inherit __read_mostly;
|
||||
static bool anon_orders_configured __initdata;
|
||||
|
||||
static inline bool file_thp_enabled(struct vm_area_struct *vma)
|
||||
static inline bool file_thp_enabled(const struct vm_area_struct *vma)
|
||||
{
|
||||
struct inode *inode;
|
||||
|
||||
@@ -118,6 +118,67 @@ static bool vma_is_special_huge(const struct vm_area_struct *vma)
|
||||
return vma_test_any(vma, VMA_PFNMAP_BIT, VMA_MIXEDMAP_BIT);
|
||||
}
|
||||
|
||||
static bool vma_file_bypass_thp_tuneables(const struct vm_area_struct *vma,
|
||||
enum tva_type type)
|
||||
{
|
||||
const bool has_huge_fault = vma->vm_ops->huge_fault;
|
||||
|
||||
/* MADV_COLLAPSE ignores tuneables. */
|
||||
if (type == TVA_FORCED_COLLAPSE)
|
||||
return true;
|
||||
/* Huge PFN mappings are uncompactable so the policy doesn't apply. */
|
||||
if (vma_test(vma, VMA_PFNMAP_BIT) && has_huge_fault)
|
||||
return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
static bool vma_file_allow_thp_tuneables(vm_flags_t vm_flags)
|
||||
{
|
||||
/* THP=always? */
|
||||
if (hugepage_global_always())
|
||||
return true;
|
||||
/* THP=madvise and marked MADV_HUGEPAGE? */
|
||||
if (hugepage_global_enabled() && (vm_flags & VM_HUGEPAGE))
|
||||
return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
static bool vma_file_check_thp_tuneables(const struct vm_area_struct *vma,
|
||||
vm_flags_t vm_flags, enum tva_type type)
|
||||
{
|
||||
return vma_file_bypass_thp_tuneables(vma, type) ||
|
||||
vma_file_allow_thp_tuneables(vm_flags);
|
||||
}
|
||||
|
||||
static bool vma_can_map_huge_file(const struct vm_area_struct *vma,
|
||||
vm_flags_t vm_flags, enum tva_type type)
|
||||
{
|
||||
const bool has_huge_fault = vma->vm_ops->huge_fault;
|
||||
|
||||
/*
|
||||
* Enforce THP collapse requirements as necessary. Anonymous vmas
|
||||
* were already handled in thp_vma_allowable_orders().
|
||||
*/
|
||||
if (!vma_file_check_thp_tuneables(vma, vm_flags, type))
|
||||
return false;
|
||||
|
||||
switch (type) {
|
||||
case TVA_PAGEFAULT:
|
||||
/*
|
||||
* Trust that ->huge_fault() handlers know what they are doing
|
||||
* in fault path.
|
||||
*/
|
||||
return has_huge_fault;
|
||||
case TVA_SMAPS:
|
||||
if (has_huge_fault)
|
||||
return true;
|
||||
fallthrough;
|
||||
default:
|
||||
/* Only regular file is valid in collapse path. */
|
||||
return file_thp_enabled(vma);
|
||||
}
|
||||
}
|
||||
|
||||
unsigned long __thp_vma_allowable_orders(struct vm_area_struct *vma,
|
||||
vm_flags_t vm_flags,
|
||||
enum tva_type type,
|
||||
@@ -190,27 +251,8 @@ unsigned long __thp_vma_allowable_orders(struct vm_area_struct *vma,
|
||||
vma, vma_start_pgoff(vma), 0,
|
||||
forced_collapse);
|
||||
|
||||
if (!vma_is_anonymous(vma)) {
|
||||
/*
|
||||
* Enforce THP collapse requirements as necessary. Anonymous vmas
|
||||
* were already handled in thp_vma_allowable_orders().
|
||||
*/
|
||||
if (!forced_collapse &&
|
||||
(!hugepage_global_enabled() || (!(vm_flags & VM_HUGEPAGE) &&
|
||||
!hugepage_global_always())))
|
||||
return 0;
|
||||
|
||||
/*
|
||||
* Trust that ->huge_fault() handlers know what they are doing
|
||||
* in fault path.
|
||||
*/
|
||||
if (((in_pf || smaps)) && vma->vm_ops->huge_fault)
|
||||
return orders;
|
||||
/* Only regular file is valid in collapse path */
|
||||
if (((!in_pf || smaps)) && file_thp_enabled(vma))
|
||||
return orders;
|
||||
return 0;
|
||||
}
|
||||
if (!vma_is_anonymous(vma))
|
||||
return vma_can_map_huge_file(vma, vm_flags, type) ? orders : 0;
|
||||
|
||||
if (vma_is_temporary_stack(vma))
|
||||
return 0;
|
||||
|
||||
Reference in New Issue
Block a user