1b2441318SGreg Kroah-Hartman // SPDX-License-Identifier: GPL-2.0 227137e52SSam Ravnborg /* 327137e52SSam Ravnborg * SPARC64 Huge TLB page support. 427137e52SSam Ravnborg * 527137e52SSam Ravnborg * Copyright (C) 2002, 2003, 2006 David S. Miller (davem@davemloft.net) 627137e52SSam Ravnborg */ 727137e52SSam Ravnborg 827137e52SSam Ravnborg #include <linux/fs.h> 927137e52SSam Ravnborg #include <linux/mm.h> 1001042607SIngo Molnar #include <linux/sched/mm.h> 1127137e52SSam Ravnborg #include <linux/hugetlb.h> 1227137e52SSam Ravnborg #include <linux/pagemap.h> 1327137e52SSam Ravnborg #include <linux/sysctl.h> 1427137e52SSam Ravnborg 1527137e52SSam Ravnborg #include <asm/mman.h> 1627137e52SSam Ravnborg #include <asm/pgalloc.h> 1727137e52SSam Ravnborg #include <asm/tlb.h> 1827137e52SSam Ravnborg #include <asm/tlbflush.h> 1927137e52SSam Ravnborg #include <asm/cacheflush.h> 2027137e52SSam Ravnborg #include <asm/mmu_context.h> 2127137e52SSam Ravnborg 2227137e52SSam Ravnborg /* Slightly simplified from the non-hugepage variant because by 2327137e52SSam Ravnborg * definition we don't have to worry about any page coloring stuff 2427137e52SSam Ravnborg */ 2527137e52SSam Ravnborg 2627137e52SSam Ravnborg static unsigned long hugetlb_get_unmapped_area_bottomup(struct file *filp, 2727137e52SSam Ravnborg unsigned long addr, 2827137e52SSam Ravnborg unsigned long len, 2927137e52SSam Ravnborg unsigned long pgoff, 3027137e52SSam Ravnborg unsigned long flags) 3127137e52SSam Ravnborg { 32c7d9f77dSNitin Gupta struct hstate *h = hstate_file(filp); 3327137e52SSam Ravnborg unsigned long task_size = TASK_SIZE; 342aea28b9SMichel Lespinasse struct vm_unmapped_area_info info; 3527137e52SSam Ravnborg 3627137e52SSam Ravnborg if (test_thread_flag(TIF_32BIT)) 3727137e52SSam Ravnborg task_size = STACK_TOP32; 3827137e52SSam Ravnborg 392aea28b9SMichel Lespinasse info.flags = 0; 402aea28b9SMichel Lespinasse info.length = len; 412aea28b9SMichel Lespinasse info.low_limit = TASK_UNMAPPED_BASE; 422aea28b9SMichel Lespinasse info.high_limit = min(task_size, VA_EXCLUDE_START); 43c7d9f77dSNitin Gupta info.align_mask = PAGE_MASK & ~huge_page_mask(h); 442aea28b9SMichel Lespinasse info.align_offset = 0; 452aea28b9SMichel Lespinasse addr = vm_unmapped_area(&info); 462aea28b9SMichel Lespinasse 472aea28b9SMichel Lespinasse if ((addr & ~PAGE_MASK) && task_size > VA_EXCLUDE_END) { 482aea28b9SMichel Lespinasse VM_BUG_ON(addr != -ENOMEM); 492aea28b9SMichel Lespinasse info.low_limit = VA_EXCLUDE_END; 502aea28b9SMichel Lespinasse info.high_limit = task_size; 512aea28b9SMichel Lespinasse addr = vm_unmapped_area(&info); 5227137e52SSam Ravnborg } 5327137e52SSam Ravnborg 5427137e52SSam Ravnborg return addr; 5527137e52SSam Ravnborg } 5627137e52SSam Ravnborg 5727137e52SSam Ravnborg static unsigned long 5827137e52SSam Ravnborg hugetlb_get_unmapped_area_topdown(struct file *filp, const unsigned long addr0, 5927137e52SSam Ravnborg const unsigned long len, 6027137e52SSam Ravnborg const unsigned long pgoff, 6127137e52SSam Ravnborg const unsigned long flags) 6227137e52SSam Ravnborg { 63c7d9f77dSNitin Gupta struct hstate *h = hstate_file(filp); 6427137e52SSam Ravnborg struct mm_struct *mm = current->mm; 6527137e52SSam Ravnborg unsigned long addr = addr0; 662aea28b9SMichel Lespinasse struct vm_unmapped_area_info info; 6727137e52SSam Ravnborg 6827137e52SSam Ravnborg /* This should only ever run for 32-bit processes. */ 6927137e52SSam Ravnborg BUG_ON(!test_thread_flag(TIF_32BIT)); 7027137e52SSam Ravnborg 712aea28b9SMichel Lespinasse info.flags = VM_UNMAPPED_AREA_TOPDOWN; 722aea28b9SMichel Lespinasse info.length = len; 732aea28b9SMichel Lespinasse info.low_limit = PAGE_SIZE; 742aea28b9SMichel Lespinasse info.high_limit = mm->mmap_base; 75c7d9f77dSNitin Gupta info.align_mask = PAGE_MASK & ~huge_page_mask(h); 762aea28b9SMichel Lespinasse info.align_offset = 0; 772aea28b9SMichel Lespinasse addr = vm_unmapped_area(&info); 7827137e52SSam Ravnborg 7927137e52SSam Ravnborg /* 8027137e52SSam Ravnborg * A failed mmap() very likely causes application failure, 8127137e52SSam Ravnborg * so fall back to the bottom-up function here. This scenario 8227137e52SSam Ravnborg * can happen with large stack limits and large mmap() 8327137e52SSam Ravnborg * allocations. 8427137e52SSam Ravnborg */ 852aea28b9SMichel Lespinasse if (addr & ~PAGE_MASK) { 862aea28b9SMichel Lespinasse VM_BUG_ON(addr != -ENOMEM); 872aea28b9SMichel Lespinasse info.flags = 0; 882aea28b9SMichel Lespinasse info.low_limit = TASK_UNMAPPED_BASE; 892aea28b9SMichel Lespinasse info.high_limit = STACK_TOP32; 902aea28b9SMichel Lespinasse addr = vm_unmapped_area(&info); 912aea28b9SMichel Lespinasse } 9227137e52SSam Ravnborg 9327137e52SSam Ravnborg return addr; 9427137e52SSam Ravnborg } 9527137e52SSam Ravnborg 9627137e52SSam Ravnborg unsigned long 9727137e52SSam Ravnborg hugetlb_get_unmapped_area(struct file *file, unsigned long addr, 9827137e52SSam Ravnborg unsigned long len, unsigned long pgoff, unsigned long flags) 9927137e52SSam Ravnborg { 100c7d9f77dSNitin Gupta struct hstate *h = hstate_file(file); 10127137e52SSam Ravnborg struct mm_struct *mm = current->mm; 10227137e52SSam Ravnborg struct vm_area_struct *vma; 10327137e52SSam Ravnborg unsigned long task_size = TASK_SIZE; 10427137e52SSam Ravnborg 10527137e52SSam Ravnborg if (test_thread_flag(TIF_32BIT)) 10627137e52SSam Ravnborg task_size = STACK_TOP32; 10727137e52SSam Ravnborg 108c7d9f77dSNitin Gupta if (len & ~huge_page_mask(h)) 10927137e52SSam Ravnborg return -EINVAL; 11027137e52SSam Ravnborg if (len > task_size) 11127137e52SSam Ravnborg return -ENOMEM; 11227137e52SSam Ravnborg 11327137e52SSam Ravnborg if (flags & MAP_FIXED) { 11427137e52SSam Ravnborg if (prepare_hugepage_range(file, addr, len)) 11527137e52SSam Ravnborg return -EINVAL; 11627137e52SSam Ravnborg return addr; 11727137e52SSam Ravnborg } 11827137e52SSam Ravnborg 11927137e52SSam Ravnborg if (addr) { 120c7d9f77dSNitin Gupta addr = ALIGN(addr, huge_page_size(h)); 12127137e52SSam Ravnborg vma = find_vma(mm, addr); 12227137e52SSam Ravnborg if (task_size - len >= addr && 1231be7107fSHugh Dickins (!vma || addr + len <= vm_start_gap(vma))) 12427137e52SSam Ravnborg return addr; 12527137e52SSam Ravnborg } 12627137e52SSam Ravnborg if (mm->get_unmapped_area == arch_get_unmapped_area) 12727137e52SSam Ravnborg return hugetlb_get_unmapped_area_bottomup(file, addr, len, 12827137e52SSam Ravnborg pgoff, flags); 12927137e52SSam Ravnborg else 13027137e52SSam Ravnborg return hugetlb_get_unmapped_area_topdown(file, addr, len, 13127137e52SSam Ravnborg pgoff, flags); 13227137e52SSam Ravnborg } 13327137e52SSam Ravnborg 134c7d9f77dSNitin Gupta static pte_t sun4u_hugepage_shift_to_tte(pte_t entry, unsigned int shift) 135c7d9f77dSNitin Gupta { 136c7d9f77dSNitin Gupta return entry; 137c7d9f77dSNitin Gupta } 138c7d9f77dSNitin Gupta 139c7d9f77dSNitin Gupta static pte_t sun4v_hugepage_shift_to_tte(pte_t entry, unsigned int shift) 140c7d9f77dSNitin Gupta { 141c7d9f77dSNitin Gupta unsigned long hugepage_size = _PAGE_SZ4MB_4V; 142c7d9f77dSNitin Gupta 143c7d9f77dSNitin Gupta pte_val(entry) = pte_val(entry) & ~_PAGE_SZALL_4V; 144c7d9f77dSNitin Gupta 145c7d9f77dSNitin Gupta switch (shift) { 146df7b2155SNitin Gupta case HPAGE_16GB_SHIFT: 147df7b2155SNitin Gupta hugepage_size = _PAGE_SZ16GB_4V; 148df7b2155SNitin Gupta pte_val(entry) |= _PAGE_PUD_HUGE; 149df7b2155SNitin Gupta break; 15085b1da7cSNitin Gupta case HPAGE_2GB_SHIFT: 15185b1da7cSNitin Gupta hugepage_size = _PAGE_SZ2GB_4V; 15285b1da7cSNitin Gupta pte_val(entry) |= _PAGE_PMD_HUGE; 15385b1da7cSNitin Gupta break; 154c7d9f77dSNitin Gupta case HPAGE_256MB_SHIFT: 155c7d9f77dSNitin Gupta hugepage_size = _PAGE_SZ256MB_4V; 156c7d9f77dSNitin Gupta pte_val(entry) |= _PAGE_PMD_HUGE; 157c7d9f77dSNitin Gupta break; 158c7d9f77dSNitin Gupta case HPAGE_SHIFT: 159c7d9f77dSNitin Gupta pte_val(entry) |= _PAGE_PMD_HUGE; 160c7d9f77dSNitin Gupta break; 161dcd1912dSNitin Gupta case HPAGE_64K_SHIFT: 162dcd1912dSNitin Gupta hugepage_size = _PAGE_SZ64K_4V; 163dcd1912dSNitin Gupta break; 164c7d9f77dSNitin Gupta default: 165c7d9f77dSNitin Gupta WARN_ONCE(1, "unsupported hugepage shift=%u\n", shift); 166c7d9f77dSNitin Gupta } 167c7d9f77dSNitin Gupta 168c7d9f77dSNitin Gupta pte_val(entry) = pte_val(entry) | hugepage_size; 169c7d9f77dSNitin Gupta return entry; 170c7d9f77dSNitin Gupta } 171c7d9f77dSNitin Gupta 172c7d9f77dSNitin Gupta static pte_t hugepage_shift_to_tte(pte_t entry, unsigned int shift) 173c7d9f77dSNitin Gupta { 174c7d9f77dSNitin Gupta if (tlb_type == hypervisor) 175c7d9f77dSNitin Gupta return sun4v_hugepage_shift_to_tte(entry, shift); 176c7d9f77dSNitin Gupta else 177c7d9f77dSNitin Gupta return sun4u_hugepage_shift_to_tte(entry, shift); 178c7d9f77dSNitin Gupta } 179c7d9f77dSNitin Gupta 180c7d9f77dSNitin Gupta pte_t arch_make_huge_pte(pte_t entry, struct vm_area_struct *vma, 181c7d9f77dSNitin Gupta struct page *page, int writeable) 182c7d9f77dSNitin Gupta { 183c7d9f77dSNitin Gupta unsigned int shift = huge_page_shift(hstate_vma(vma)); 18474a04967SKhalid Aziz pte_t pte; 185c7d9f77dSNitin Gupta 18674a04967SKhalid Aziz pte = hugepage_shift_to_tte(entry, shift); 18774a04967SKhalid Aziz 18874a04967SKhalid Aziz #ifdef CONFIG_SPARC64 18974a04967SKhalid Aziz /* If this vma has ADI enabled on it, turn on TTE.mcd 19074a04967SKhalid Aziz */ 19174a04967SKhalid Aziz if (vma->vm_flags & VM_SPARC_ADI) 19274a04967SKhalid Aziz return pte_mkmcd(pte); 19374a04967SKhalid Aziz else 19474a04967SKhalid Aziz return pte_mknotmcd(pte); 19574a04967SKhalid Aziz #else 19674a04967SKhalid Aziz return pte; 19774a04967SKhalid Aziz #endif 198c7d9f77dSNitin Gupta } 199c7d9f77dSNitin Gupta 200c7d9f77dSNitin Gupta static unsigned int sun4v_huge_tte_to_shift(pte_t entry) 201c7d9f77dSNitin Gupta { 202c7d9f77dSNitin Gupta unsigned long tte_szbits = pte_val(entry) & _PAGE_SZALL_4V; 203c7d9f77dSNitin Gupta unsigned int shift; 204c7d9f77dSNitin Gupta 205c7d9f77dSNitin Gupta switch (tte_szbits) { 206df7b2155SNitin Gupta case _PAGE_SZ16GB_4V: 207df7b2155SNitin Gupta shift = HPAGE_16GB_SHIFT; 208df7b2155SNitin Gupta break; 20985b1da7cSNitin Gupta case _PAGE_SZ2GB_4V: 21085b1da7cSNitin Gupta shift = HPAGE_2GB_SHIFT; 21185b1da7cSNitin Gupta break; 212c7d9f77dSNitin Gupta case _PAGE_SZ256MB_4V: 213c7d9f77dSNitin Gupta shift = HPAGE_256MB_SHIFT; 214c7d9f77dSNitin Gupta break; 215c7d9f77dSNitin Gupta case _PAGE_SZ4MB_4V: 216c7d9f77dSNitin Gupta shift = REAL_HPAGE_SHIFT; 217c7d9f77dSNitin Gupta break; 218dcd1912dSNitin Gupta case _PAGE_SZ64K_4V: 219dcd1912dSNitin Gupta shift = HPAGE_64K_SHIFT; 220dcd1912dSNitin Gupta break; 221c7d9f77dSNitin Gupta default: 222c7d9f77dSNitin Gupta shift = PAGE_SHIFT; 223c7d9f77dSNitin Gupta break; 224c7d9f77dSNitin Gupta } 225c7d9f77dSNitin Gupta return shift; 226c7d9f77dSNitin Gupta } 227c7d9f77dSNitin Gupta 228c7d9f77dSNitin Gupta static unsigned int sun4u_huge_tte_to_shift(pte_t entry) 229c7d9f77dSNitin Gupta { 230c7d9f77dSNitin Gupta unsigned long tte_szbits = pte_val(entry) & _PAGE_SZALL_4U; 231c7d9f77dSNitin Gupta unsigned int shift; 232c7d9f77dSNitin Gupta 233c7d9f77dSNitin Gupta switch (tte_szbits) { 234c7d9f77dSNitin Gupta case _PAGE_SZ256MB_4U: 235c7d9f77dSNitin Gupta shift = HPAGE_256MB_SHIFT; 236c7d9f77dSNitin Gupta break; 237c7d9f77dSNitin Gupta case _PAGE_SZ4MB_4U: 238c7d9f77dSNitin Gupta shift = REAL_HPAGE_SHIFT; 239c7d9f77dSNitin Gupta break; 240dcd1912dSNitin Gupta case _PAGE_SZ64K_4U: 241dcd1912dSNitin Gupta shift = HPAGE_64K_SHIFT; 242dcd1912dSNitin Gupta break; 243c7d9f77dSNitin Gupta default: 244c7d9f77dSNitin Gupta shift = PAGE_SHIFT; 245c7d9f77dSNitin Gupta break; 246c7d9f77dSNitin Gupta } 247c7d9f77dSNitin Gupta return shift; 248c7d9f77dSNitin Gupta } 249c7d9f77dSNitin Gupta 250*e6e4f42eSPeter Zijlstra static unsigned long tte_to_shift(pte_t entry) 251*e6e4f42eSPeter Zijlstra { 252*e6e4f42eSPeter Zijlstra if (tlb_type == hypervisor) 253*e6e4f42eSPeter Zijlstra return sun4v_huge_tte_to_shift(entry); 254*e6e4f42eSPeter Zijlstra 255*e6e4f42eSPeter Zijlstra return sun4u_huge_tte_to_shift(entry); 256*e6e4f42eSPeter Zijlstra } 257*e6e4f42eSPeter Zijlstra 258c7d9f77dSNitin Gupta static unsigned int huge_tte_to_shift(pte_t entry) 259c7d9f77dSNitin Gupta { 260*e6e4f42eSPeter Zijlstra unsigned long shift = tte_to_shift(entry); 261c7d9f77dSNitin Gupta 262c7d9f77dSNitin Gupta if (shift == PAGE_SHIFT) 263c7d9f77dSNitin Gupta WARN_ONCE(1, "tto_to_shift: invalid hugepage tte=0x%lx\n", 264c7d9f77dSNitin Gupta pte_val(entry)); 265c7d9f77dSNitin Gupta 266c7d9f77dSNitin Gupta return shift; 267c7d9f77dSNitin Gupta } 268c7d9f77dSNitin Gupta 269c7d9f77dSNitin Gupta static unsigned long huge_tte_to_size(pte_t pte) 270c7d9f77dSNitin Gupta { 271c7d9f77dSNitin Gupta unsigned long size = 1UL << huge_tte_to_shift(pte); 272c7d9f77dSNitin Gupta 273c7d9f77dSNitin Gupta if (size == REAL_HPAGE_SIZE) 274c7d9f77dSNitin Gupta size = HPAGE_SIZE; 275c7d9f77dSNitin Gupta return size; 276c7d9f77dSNitin Gupta } 277c7d9f77dSNitin Gupta 278*e6e4f42eSPeter Zijlstra unsigned long pud_leaf_size(pud_t pud) { return 1UL << tte_to_shift(*(pte_t *)&pud); } 279*e6e4f42eSPeter Zijlstra unsigned long pmd_leaf_size(pmd_t pmd) { return 1UL << tte_to_shift(*(pte_t *)&pmd); } 280*e6e4f42eSPeter Zijlstra unsigned long pte_leaf_size(pte_t pte) { return 1UL << tte_to_shift(pte); } 281*e6e4f42eSPeter Zijlstra 28227137e52SSam Ravnborg pte_t *huge_pte_alloc(struct mm_struct *mm, 28327137e52SSam Ravnborg unsigned long addr, unsigned long sz) 28427137e52SSam Ravnborg { 28527137e52SSam Ravnborg pgd_t *pgd; 2865637bc50SMike Rapoport p4d_t *p4d; 28727137e52SSam Ravnborg pud_t *pud; 288dcd1912dSNitin Gupta pmd_t *pmd; 28927137e52SSam Ravnborg 29027137e52SSam Ravnborg pgd = pgd_offset(mm, addr); 2915637bc50SMike Rapoport p4d = p4d_offset(pgd, addr); 2925637bc50SMike Rapoport pud = pud_alloc(mm, p4d, addr); 293df7b2155SNitin Gupta if (!pud) 294df7b2155SNitin Gupta return NULL; 295df7b2155SNitin Gupta if (sz >= PUD_SIZE) 2964dbe87d5SNitin Gupta return (pte_t *)pud; 297dcd1912dSNitin Gupta pmd = pmd_alloc(mm, pud, addr); 298dcd1912dSNitin Gupta if (!pmd) 299dcd1912dSNitin Gupta return NULL; 30059f1183dSNitin Gupta if (sz >= PMD_SIZE) 3014dbe87d5SNitin Gupta return (pte_t *)pmd; 3024dbe87d5SNitin Gupta return pte_alloc_map(mm, pmd, addr); 30327137e52SSam Ravnborg } 30427137e52SSam Ravnborg 3057868a208SPunit Agrawal pte_t *huge_pte_offset(struct mm_struct *mm, 3067868a208SPunit Agrawal unsigned long addr, unsigned long sz) 30727137e52SSam Ravnborg { 30827137e52SSam Ravnborg pgd_t *pgd; 3095637bc50SMike Rapoport p4d_t *p4d; 31027137e52SSam Ravnborg pud_t *pud; 311dcd1912dSNitin Gupta pmd_t *pmd; 31227137e52SSam Ravnborg 31327137e52SSam Ravnborg pgd = pgd_offset(mm, addr); 3144dbe87d5SNitin Gupta if (pgd_none(*pgd)) 3154dbe87d5SNitin Gupta return NULL; 3165637bc50SMike Rapoport p4d = p4d_offset(pgd, addr); 3175637bc50SMike Rapoport if (p4d_none(*p4d)) 3185637bc50SMike Rapoport return NULL; 3195637bc50SMike Rapoport pud = pud_offset(p4d, addr); 3204dbe87d5SNitin Gupta if (pud_none(*pud)) 3214dbe87d5SNitin Gupta return NULL; 322df7b2155SNitin Gupta if (is_hugetlb_pud(*pud)) 3234dbe87d5SNitin Gupta return (pte_t *)pud; 324dcd1912dSNitin Gupta pmd = pmd_offset(pud, addr); 3254dbe87d5SNitin Gupta if (pmd_none(*pmd)) 3264dbe87d5SNitin Gupta return NULL; 327dcd1912dSNitin Gupta if (is_hugetlb_pmd(*pmd)) 3284dbe87d5SNitin Gupta return (pte_t *)pmd; 3294dbe87d5SNitin Gupta return pte_offset_map(pmd, addr); 33027137e52SSam Ravnborg } 33127137e52SSam Ravnborg 33227137e52SSam Ravnborg void set_huge_pte_at(struct mm_struct *mm, unsigned long addr, 33327137e52SSam Ravnborg pte_t *ptep, pte_t entry) 33427137e52SSam Ravnborg { 335df7b2155SNitin Gupta unsigned int nptes, orig_shift, shift; 336df7b2155SNitin Gupta unsigned long i, size; 3377bc3777cSNitin Gupta pte_t orig; 33827137e52SSam Ravnborg 339c7d9f77dSNitin Gupta size = huge_tte_to_size(entry); 340df7b2155SNitin Gupta 341df7b2155SNitin Gupta shift = PAGE_SHIFT; 342df7b2155SNitin Gupta if (size >= PUD_SIZE) 343df7b2155SNitin Gupta shift = PUD_SHIFT; 344df7b2155SNitin Gupta else if (size >= PMD_SIZE) 345df7b2155SNitin Gupta shift = PMD_SHIFT; 346df7b2155SNitin Gupta else 347df7b2155SNitin Gupta shift = PAGE_SHIFT; 348df7b2155SNitin Gupta 349dcd1912dSNitin Gupta nptes = size >> shift; 350c7d9f77dSNitin Gupta 35127137e52SSam Ravnborg if (!pte_present(*ptep) && pte_present(entry)) 352c7d9f77dSNitin Gupta mm->context.hugetlb_pte_count += nptes; 35327137e52SSam Ravnborg 354c7d9f77dSNitin Gupta addr &= ~(size - 1); 3557bc3777cSNitin Gupta orig = *ptep; 356ac65e282SNitin Gupta orig_shift = pte_none(orig) ? PAGE_SHIFT : huge_tte_to_shift(orig); 35724e49ee3SNitin Gupta 358c7d9f77dSNitin Gupta for (i = 0; i < nptes; i++) 359dcd1912dSNitin Gupta ptep[i] = __pte(pte_val(entry) + (i << shift)); 360c7d9f77dSNitin Gupta 361dcd1912dSNitin Gupta maybe_tlb_batch_add(mm, addr, ptep, orig, 0, orig_shift); 362c7d9f77dSNitin Gupta /* An HPAGE_SIZE'ed page is composed of two REAL_HPAGE_SIZE'ed pages */ 363c7d9f77dSNitin Gupta if (size == HPAGE_SIZE) 364c7d9f77dSNitin Gupta maybe_tlb_batch_add(mm, addr + REAL_HPAGE_SIZE, ptep, orig, 0, 365dcd1912dSNitin Gupta orig_shift); 36627137e52SSam Ravnborg } 36727137e52SSam Ravnborg 36827137e52SSam Ravnborg pte_t huge_ptep_get_and_clear(struct mm_struct *mm, unsigned long addr, 36927137e52SSam Ravnborg pte_t *ptep) 37027137e52SSam Ravnborg { 371df7b2155SNitin Gupta unsigned int i, nptes, orig_shift, shift; 372c7d9f77dSNitin Gupta unsigned long size; 37327137e52SSam Ravnborg pte_t entry; 37427137e52SSam Ravnborg 37527137e52SSam Ravnborg entry = *ptep; 376c7d9f77dSNitin Gupta size = huge_tte_to_size(entry); 377f10bb007SNitin Gupta 378df7b2155SNitin Gupta shift = PAGE_SHIFT; 379df7b2155SNitin Gupta if (size >= PUD_SIZE) 380df7b2155SNitin Gupta shift = PUD_SHIFT; 381df7b2155SNitin Gupta else if (size >= PMD_SIZE) 382df7b2155SNitin Gupta shift = PMD_SHIFT; 383df7b2155SNitin Gupta else 384df7b2155SNitin Gupta shift = PAGE_SHIFT; 385df7b2155SNitin Gupta 386df7b2155SNitin Gupta nptes = size >> shift; 387df7b2155SNitin Gupta orig_shift = pte_none(entry) ? PAGE_SHIFT : huge_tte_to_shift(entry); 388c7d9f77dSNitin Gupta 38927137e52SSam Ravnborg if (pte_present(entry)) 390c7d9f77dSNitin Gupta mm->context.hugetlb_pte_count -= nptes; 39127137e52SSam Ravnborg 392c7d9f77dSNitin Gupta addr &= ~(size - 1); 393c7d9f77dSNitin Gupta for (i = 0; i < nptes; i++) 394c7d9f77dSNitin Gupta ptep[i] = __pte(0UL); 39527137e52SSam Ravnborg 396df7b2155SNitin Gupta maybe_tlb_batch_add(mm, addr, ptep, entry, 0, orig_shift); 397c7d9f77dSNitin Gupta /* An HPAGE_SIZE'ed page is composed of two REAL_HPAGE_SIZE'ed pages */ 398c7d9f77dSNitin Gupta if (size == HPAGE_SIZE) 399c7d9f77dSNitin Gupta maybe_tlb_batch_add(mm, addr + REAL_HPAGE_SIZE, ptep, entry, 0, 400df7b2155SNitin Gupta orig_shift); 40124e49ee3SNitin Gupta 40227137e52SSam Ravnborg return entry; 40327137e52SSam Ravnborg } 40427137e52SSam Ravnborg 40527137e52SSam Ravnborg int pmd_huge(pmd_t pmd) 40627137e52SSam Ravnborg { 4077bc3777cSNitin Gupta return !pmd_none(pmd) && 4087bc3777cSNitin Gupta (pmd_val(pmd) & (_PAGE_VALID|_PAGE_PMD_HUGE)) != _PAGE_VALID; 40927137e52SSam Ravnborg } 41027137e52SSam Ravnborg 41127137e52SSam Ravnborg int pud_huge(pud_t pud) 41227137e52SSam Ravnborg { 413df7b2155SNitin Gupta return !pud_none(pud) && 414df7b2155SNitin Gupta (pud_val(pud) & (_PAGE_VALID|_PAGE_PUD_HUGE)) != _PAGE_VALID; 41527137e52SSam Ravnborg } 4167bc3777cSNitin Gupta 4177bc3777cSNitin Gupta static void hugetlb_free_pte_range(struct mmu_gather *tlb, pmd_t *pmd, 4187bc3777cSNitin Gupta unsigned long addr) 4197bc3777cSNitin Gupta { 4207bc3777cSNitin Gupta pgtable_t token = pmd_pgtable(*pmd); 4217bc3777cSNitin Gupta 4227bc3777cSNitin Gupta pmd_clear(pmd); 4237bc3777cSNitin Gupta pte_free_tlb(tlb, token, addr); 424c4812909SKirill A. Shutemov mm_dec_nr_ptes(tlb->mm); 4257bc3777cSNitin Gupta } 4267bc3777cSNitin Gupta 4277bc3777cSNitin Gupta static void hugetlb_free_pmd_range(struct mmu_gather *tlb, pud_t *pud, 4287bc3777cSNitin Gupta unsigned long addr, unsigned long end, 4297bc3777cSNitin Gupta unsigned long floor, unsigned long ceiling) 4307bc3777cSNitin Gupta { 4317bc3777cSNitin Gupta pmd_t *pmd; 4327bc3777cSNitin Gupta unsigned long next; 4337bc3777cSNitin Gupta unsigned long start; 4347bc3777cSNitin Gupta 4357bc3777cSNitin Gupta start = addr; 4367bc3777cSNitin Gupta pmd = pmd_offset(pud, addr); 4377bc3777cSNitin Gupta do { 4387bc3777cSNitin Gupta next = pmd_addr_end(addr, end); 4397bc3777cSNitin Gupta if (pmd_none(*pmd)) 4407bc3777cSNitin Gupta continue; 4417bc3777cSNitin Gupta if (is_hugetlb_pmd(*pmd)) 4427bc3777cSNitin Gupta pmd_clear(pmd); 4437bc3777cSNitin Gupta else 4447bc3777cSNitin Gupta hugetlb_free_pte_range(tlb, pmd, addr); 4457bc3777cSNitin Gupta } while (pmd++, addr = next, addr != end); 4467bc3777cSNitin Gupta 4477bc3777cSNitin Gupta start &= PUD_MASK; 4487bc3777cSNitin Gupta if (start < floor) 4497bc3777cSNitin Gupta return; 4507bc3777cSNitin Gupta if (ceiling) { 4517bc3777cSNitin Gupta ceiling &= PUD_MASK; 4527bc3777cSNitin Gupta if (!ceiling) 4537bc3777cSNitin Gupta return; 4547bc3777cSNitin Gupta } 4557bc3777cSNitin Gupta if (end - 1 > ceiling - 1) 4567bc3777cSNitin Gupta return; 4577bc3777cSNitin Gupta 4587bc3777cSNitin Gupta pmd = pmd_offset(pud, start); 4597bc3777cSNitin Gupta pud_clear(pud); 4607bc3777cSNitin Gupta pmd_free_tlb(tlb, pmd, start); 4617bc3777cSNitin Gupta mm_dec_nr_pmds(tlb->mm); 4627bc3777cSNitin Gupta } 4637bc3777cSNitin Gupta 4645637bc50SMike Rapoport static void hugetlb_free_pud_range(struct mmu_gather *tlb, p4d_t *p4d, 4657bc3777cSNitin Gupta unsigned long addr, unsigned long end, 4667bc3777cSNitin Gupta unsigned long floor, unsigned long ceiling) 4677bc3777cSNitin Gupta { 4687bc3777cSNitin Gupta pud_t *pud; 4697bc3777cSNitin Gupta unsigned long next; 4707bc3777cSNitin Gupta unsigned long start; 4717bc3777cSNitin Gupta 4727bc3777cSNitin Gupta start = addr; 4735637bc50SMike Rapoport pud = pud_offset(p4d, addr); 4747bc3777cSNitin Gupta do { 4757bc3777cSNitin Gupta next = pud_addr_end(addr, end); 4767bc3777cSNitin Gupta if (pud_none_or_clear_bad(pud)) 4777bc3777cSNitin Gupta continue; 478df7b2155SNitin Gupta if (is_hugetlb_pud(*pud)) 479df7b2155SNitin Gupta pud_clear(pud); 480df7b2155SNitin Gupta else 4817bc3777cSNitin Gupta hugetlb_free_pmd_range(tlb, pud, addr, next, floor, 4827bc3777cSNitin Gupta ceiling); 4837bc3777cSNitin Gupta } while (pud++, addr = next, addr != end); 4847bc3777cSNitin Gupta 4857bc3777cSNitin Gupta start &= PGDIR_MASK; 4867bc3777cSNitin Gupta if (start < floor) 4877bc3777cSNitin Gupta return; 4887bc3777cSNitin Gupta if (ceiling) { 4897bc3777cSNitin Gupta ceiling &= PGDIR_MASK; 4907bc3777cSNitin Gupta if (!ceiling) 4917bc3777cSNitin Gupta return; 4927bc3777cSNitin Gupta } 4937bc3777cSNitin Gupta if (end - 1 > ceiling - 1) 4947bc3777cSNitin Gupta return; 4957bc3777cSNitin Gupta 4965637bc50SMike Rapoport pud = pud_offset(p4d, start); 4975637bc50SMike Rapoport p4d_clear(p4d); 4987bc3777cSNitin Gupta pud_free_tlb(tlb, pud, start); 499b4e98d9aSKirill A. Shutemov mm_dec_nr_puds(tlb->mm); 5007bc3777cSNitin Gupta } 5017bc3777cSNitin Gupta 5027bc3777cSNitin Gupta void hugetlb_free_pgd_range(struct mmu_gather *tlb, 5037bc3777cSNitin Gupta unsigned long addr, unsigned long end, 5047bc3777cSNitin Gupta unsigned long floor, unsigned long ceiling) 5057bc3777cSNitin Gupta { 5067bc3777cSNitin Gupta pgd_t *pgd; 5075637bc50SMike Rapoport p4d_t *p4d; 5087bc3777cSNitin Gupta unsigned long next; 5097bc3777cSNitin Gupta 510544f8f93SNitin Gupta addr &= PMD_MASK; 511544f8f93SNitin Gupta if (addr < floor) { 512544f8f93SNitin Gupta addr += PMD_SIZE; 513544f8f93SNitin Gupta if (!addr) 514544f8f93SNitin Gupta return; 515544f8f93SNitin Gupta } 516544f8f93SNitin Gupta if (ceiling) { 517544f8f93SNitin Gupta ceiling &= PMD_MASK; 518544f8f93SNitin Gupta if (!ceiling) 519544f8f93SNitin Gupta return; 520544f8f93SNitin Gupta } 521544f8f93SNitin Gupta if (end - 1 > ceiling - 1) 522544f8f93SNitin Gupta end -= PMD_SIZE; 523544f8f93SNitin Gupta if (addr > end - 1) 524544f8f93SNitin Gupta return; 525544f8f93SNitin Gupta 5267bc3777cSNitin Gupta pgd = pgd_offset(tlb->mm, addr); 5275637bc50SMike Rapoport p4d = p4d_offset(pgd, addr); 5287bc3777cSNitin Gupta do { 5295637bc50SMike Rapoport next = p4d_addr_end(addr, end); 5305637bc50SMike Rapoport if (p4d_none_or_clear_bad(p4d)) 5317bc3777cSNitin Gupta continue; 5325637bc50SMike Rapoport hugetlb_free_pud_range(tlb, p4d, addr, next, floor, ceiling); 5335637bc50SMike Rapoport } while (p4d++, addr = next, addr != end); 5347bc3777cSNitin Gupta } 535