From: Matt Turner <mattst88@gmail.com>
To: Richard Henderson <richard.henderson@linaro.org>,
Magnus Lindholm <linmag7@gmail.com>
Cc: linux-alpha@vger.kernel.org, linux-kernel@vger.kernel.org,
Matt Turner <mattst88@gmail.com>
Subject: [PATCH 6/6] alpha: implement hugetlb support
Date: Tue, 06 Oct 2026 10:04:18 -0400 [thread overview]
Message-ID: <20261006-alpha-hugepages-v1-6-a673a18aaa70@gmail.com> (raw)
In-Reply-To: <20261006-alpha-hugepages-v1-0-a673a18aaa70@gmail.com>
Alpha has no leaf entry above the last page table level, so the only
huge pages it can offer are granularity hint blocks: 64KB, 512KB and 4MB
with 8KB pages. Register all three as hstates, with 4MB as the default.
There is no PMD sized huge page, hence no transparent huge pages, no PMD
page table sharing and no gigantic pages.
As with arm64's contiguous PTEs, each PTE of a block keeps the frame
number of its own page, so a block is written with set_ptes(). Each
hstate's buddy order is its hint order, so a hugetlb folio is always
naturally aligned physically, as the hint requires, including after a
demote.
All PTEs of a block must agree in bits <15:0>, and both __ACCESS_BITS
and __DIRTY_BITS reach into that range, so every update rewrites the
whole block and huge_ptep_get() merges the young and dirty state back
together. Changes to a valid block go through break before make, as
section 11.6.1 of the architecture manual requires.
huge_ptep_set_access_flags() skips the break when no entry would change:
hugetlb_fault() calls it on every fault on a present entry, and breaking
an unchanged block makes threads on other CPUs refault and break it
again.
The hint is advisory, so this is safe on implementations that ignore it,
and gup_fast needs no changes since it walks the individual PTEs.
Enabling CONFIG_HUGETLB_PAGE derives pageblock_order from
HUGETLB_PAGE_ORDER, which lowers it from 10 to 9 and so halves the
anti-fragmentation and compaction granularity from 8MB to 4MB.
Hugepage migration is left disabled, since it has not been tested.
Co-developed-by: Magnus Lindholm <linmag7@gmail.com>
Signed-off-by: Magnus Lindholm <linmag7@gmail.com>
Signed-off-by: Matt Turner <mattst88@gmail.com>
---
arch/alpha/Kconfig | 1 +
arch/alpha/include/asm/hugetlb.h | 43 ++++++
arch/alpha/include/asm/page.h | 13 ++
arch/alpha/mm/Makefile | 2 +
arch/alpha/mm/hugetlbpage.c | 306 +++++++++++++++++++++++++++++++++++++++
5 files changed, 365 insertions(+)
diff --git a/arch/alpha/Kconfig b/arch/alpha/Kconfig
index b5b02bb42aca..b2c359b947f6 100644
--- a/arch/alpha/Kconfig
+++ b/arch/alpha/Kconfig
@@ -19,6 +19,7 @@ config ALPHA
select ARCH_NO_PREEMPT
select ARCH_NO_SG_CHAIN
select ARCH_SUPPORTS_ATOMIC_RMW
+ select ARCH_SUPPORTS_HUGETLBFS
select ARCH_SUPPORTS_INT128 if CC_HAS_INT128
select ARCH_SUPPORTS_PAGE_TABLE_CHECK
select ARCH_USE_CMPXCHG_LOCKREF
diff --git a/arch/alpha/include/asm/hugetlb.h b/arch/alpha/include/asm/hugetlb.h
new file mode 100644
index 000000000000..ff69e24c775d
--- /dev/null
+++ b/arch/alpha/include/asm/hugetlb.h
@@ -0,0 +1,43 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+#ifndef _ALPHA_HUGETLB_H
+#define _ALPHA_HUGETLB_H
+
+#include <asm/page.h>
+
+pte_t arch_make_huge_pte(pte_t entry, unsigned int shift, vm_flags_t flags);
+#define arch_make_huge_pte arch_make_huge_pte
+
+#define __HAVE_ARCH_HUGE_PTEP_GET
+extern pte_t huge_ptep_get(struct mm_struct *mm, unsigned long addr,
+ pte_t *ptep);
+#define __HAVE_ARCH_HUGE_SET_HUGE_PTE_AT
+extern void set_huge_pte_at(struct mm_struct *mm, unsigned long addr,
+ pte_t *ptep, pte_t pte, unsigned long sz);
+#define __HAVE_ARCH_HUGE_PTEP_GET_AND_CLEAR
+extern pte_t huge_ptep_get_and_clear(struct mm_struct *mm, unsigned long addr,
+ pte_t *ptep, unsigned long sz);
+#define __HAVE_ARCH_HUGE_PTEP_CLEAR_FLUSH
+extern pte_t huge_ptep_clear_flush(struct vm_area_struct *vma,
+ unsigned long addr, pte_t *ptep);
+#define __HAVE_ARCH_HUGE_PTEP_SET_WRPROTECT
+extern void huge_ptep_set_wrprotect(struct mm_struct *mm, unsigned long addr,
+ pte_t *ptep);
+#define __HAVE_ARCH_HUGE_PTEP_SET_ACCESS_FLAGS
+extern int huge_ptep_set_access_flags(struct vm_area_struct *vma,
+ unsigned long addr, pte_t *ptep,
+ pte_t pte, int dirty);
+#define __HAVE_ARCH_HUGE_PTE_CLEAR
+extern void huge_pte_clear(struct mm_struct *mm, unsigned long addr,
+ pte_t *ptep, unsigned long sz);
+
+/*
+ * The generic version does not flush, which would leave the old translation
+ * live across a restrictive change to PTE bits <15:0>.
+ */
+#define huge_ptep_modify_prot_start huge_ptep_modify_prot_start
+extern pte_t huge_ptep_modify_prot_start(struct vm_area_struct *vma,
+ unsigned long addr, pte_t *ptep);
+
+#include <asm-generic/hugetlb.h>
+
+#endif /* _ALPHA_HUGETLB_H */
diff --git a/arch/alpha/include/asm/page.h b/arch/alpha/include/asm/page.h
index 59d01f9b77f6..67053c52953f 100644
--- a/arch/alpha/include/asm/page.h
+++ b/arch/alpha/include/asm/page.h
@@ -6,6 +6,19 @@
#include <asm/pal.h>
#include <vdso/page.h>
+#ifdef CONFIG_HUGETLB_PAGE
+/*
+ * The default huge page is the largest granularity hint block, 4MB. All
+ * three hint sizes are registered as hstates, hence HUGE_MAX_HSTATE, which
+ * linux/hugetlb.h needs before it includes asm/hugetlb.h.
+ */
+#define HPAGE_SHIFT 22
+#define HPAGE_SIZE (_AC(1, UL) << HPAGE_SHIFT)
+#define HPAGE_MASK (~(HPAGE_SIZE - 1))
+#define HUGETLB_PAGE_ORDER (HPAGE_SHIFT - PAGE_SHIFT)
+#define HUGE_MAX_HSTATE 3
+#endif
+
#ifndef __ASSEMBLER__
#define STRICT_MM_TYPECHECKS
diff --git a/arch/alpha/mm/Makefile b/arch/alpha/mm/Makefile
index 2d05664058f6..c022e55f231a 100644
--- a/arch/alpha/mm/Makefile
+++ b/arch/alpha/mm/Makefile
@@ -4,3 +4,5 @@
#
obj-y := init.o fault.o tlbflush.o
+
+obj-$(CONFIG_HUGETLB_PAGE) += hugetlbpage.o
diff --git a/arch/alpha/mm/hugetlbpage.c b/arch/alpha/mm/hugetlbpage.c
new file mode 100644
index 000000000000..94b02de4b370
--- /dev/null
+++ b/arch/alpha/mm/hugetlbpage.c
@@ -0,0 +1,306 @@
+// SPDX-License-Identifier: GPL-2.0
+/*
+ * Alpha huge page support, built on the page table entry granularity hint
+ * described in asm/pgtable.h. All PTEs of a block have to agree in bits
+ * <15:0>, so every update below rewrites the whole block.
+ */
+
+#include <linux/hugetlb.h>
+#include <linux/mm.h>
+#include <linux/pgtable.h>
+
+#include <asm/tlbflush.h>
+
+/*
+ * Every hint size is smaller than PMD_SIZE, so a block always fits inside a
+ * single last level page table and its entry count is just its page count.
+ */
+static inline unsigned long num_contig_ptes(unsigned long sz)
+{
+ return sz >> PAGE_SHIFT;
+}
+
+/* Hint order that maps @sz, or 0 if @sz is not a hint size. */
+static unsigned int gh_order_from_size(unsigned long sz)
+{
+ unsigned int order;
+
+ for_each_gh_order(order)
+ if (gh_cont_size(order) == sz)
+ return order;
+
+ return 0;
+}
+
+pte_t *huge_pte_alloc(struct mm_struct *mm, struct vm_area_struct *vma,
+ unsigned long addr, unsigned long sz)
+{
+ unsigned int order = gh_order_from_size(sz);
+ pgd_t *pgd;
+ p4d_t *p4d;
+ pud_t *pud;
+ pmd_t *pmd;
+
+ if (!order)
+ return NULL;
+
+ pgd = pgd_offset(mm, addr);
+ p4d = p4d_alloc(mm, pgd, addr);
+ if (!p4d)
+ return NULL;
+
+ pud = pud_alloc(mm, p4d, addr);
+ if (!pud)
+ return NULL;
+
+ pmd = pmd_alloc(mm, pud, addr);
+ if (!pmd)
+ return NULL;
+
+ return pte_alloc_huge(mm, pmd, addr & gh_cont_mask(order));
+}
+
+pte_t *huge_pte_offset(struct mm_struct *mm, unsigned long addr,
+ unsigned long sz)
+{
+ unsigned int order = gh_order_from_size(sz);
+ pgd_t *pgd;
+ p4d_t *p4d;
+ pud_t *pud;
+ pmd_t *pmd;
+
+ if (!order)
+ return NULL;
+
+ pgd = pgd_offset(mm, addr);
+ if (!pgd_present(pgdp_get(pgd)))
+ return NULL;
+
+ p4d = p4d_offset(pgd, addr);
+ if (!p4d_present(p4dp_get(p4d)))
+ return NULL;
+
+ pud = pud_offset(p4d, addr);
+ if (!pud_present(pudp_get(pud)))
+ return NULL;
+
+ pmd = pmd_offset(pud, addr);
+ if (!pmd_present(pmdp_get(pmd)))
+ return NULL;
+
+ return pte_offset_huge(pmd, addr & gh_cont_mask(order));
+}
+
+/*
+ * A block is only as large as its own hint, so the walk can skip to the end of
+ * the containing last level page table but no further.
+ */
+unsigned long hugetlb_mask_last_page(struct hstate *h)
+{
+ return PMD_SIZE - huge_page_size(h);
+}
+
+pte_t huge_ptep_get(struct mm_struct *mm, unsigned long addr, pte_t *ptep)
+{
+ pte_t orig_pte = ptep_get(ptep);
+ unsigned long i, ncontig;
+
+ /* Swap, migration and marker entries carry no hint. */
+ if (!pte_huge(orig_pte))
+ return orig_pte;
+
+ ncontig = gh_pte_num(pte_gh_order(orig_pte));
+
+ for (i = 0; i < ncontig; i++, ptep++) {
+ pte_t pte = ptep_get(ptep);
+
+ if (pte_dirty(pte))
+ orig_pte = pte_mkdirty(orig_pte);
+ if (pte_young(pte))
+ orig_pte = pte_mkyoung(orig_pte);
+ }
+
+ return orig_pte;
+}
+
+static pte_t get_clear_contig(struct mm_struct *mm, unsigned long addr,
+ pte_t *ptep, unsigned long ncontig)
+{
+ pte_t pte = ptep_get_and_clear(mm, addr, ptep);
+ bool present = pte_present(pte);
+
+ while (--ncontig) {
+ pte_t tmp_pte;
+
+ ptep++;
+ addr += PAGE_SIZE;
+ tmp_pte = ptep_get_and_clear(mm, addr, ptep);
+ if (present) {
+ if (pte_dirty(tmp_pte))
+ pte = pte_mkdirty(pte);
+ if (pte_young(tmp_pte))
+ pte = pte_mkyoung(pte);
+ }
+ }
+
+ return pte;
+}
+
+static pte_t get_clear_contig_flush(struct mm_struct *mm, unsigned long addr,
+ pte_t *ptep, unsigned long ncontig)
+{
+ pte_t orig_pte = get_clear_contig(mm, addr, ptep, ncontig);
+ struct vm_area_struct vma = TLB_FLUSH_VMA(mm, 0);
+
+ if (!pte_none(orig_pte))
+ flush_tlb_range(&vma, addr, addr + ncontig * PAGE_SIZE);
+
+ return orig_pte;
+}
+
+pte_t arch_make_huge_pte(pte_t entry, unsigned int shift, vm_flags_t flags)
+{
+ unsigned int order = gh_order_from_size(1UL << shift);
+
+ /*
+ * Not every caller asks for a size Alpha can hint; mm/debug_vm_pgtable.c
+ * probes with PMD_SHIFT. Leave the entry alone rather than complaining.
+ */
+ if (order)
+ entry = pte_mkgh(entry, order);
+
+ return entry;
+}
+
+void set_huge_pte_at(struct mm_struct *mm, unsigned long addr, pte_t *ptep,
+ pte_t pte, unsigned long sz)
+{
+ unsigned long i, ncontig = num_contig_ptes(sz);
+
+ /* The hint requires the block to be naturally aligned in the VA too. */
+ VM_WARN_ON_ONCE(addr & (sz - 1));
+
+ if (!pte_present(pte)) {
+ /*
+ * Swap, migration and marker entries hold no frame number, so
+ * every slot gets the same value.
+ */
+ for (i = 0; i < ncontig; i++, ptep++, addr += PAGE_SIZE)
+ set_ptes(mm, addr, ptep, pte, 1);
+ return;
+ }
+
+ /*
+ * ARM 11.6.1 requires an already valid entry to be invalidated
+ * everywhere before it is rewritten, and the invalidate has to assume
+ * the hint is zero, so it must cover every page of the block.
+ */
+ if (pte_present(ptep_get(ptep)))
+ get_clear_contig_flush(mm, addr, ptep, ncontig);
+
+ /* set_ptes() advances the frame number for each entry. */
+ set_ptes(mm, addr, ptep, pte, ncontig);
+}
+
+pte_t huge_ptep_get_and_clear(struct mm_struct *mm, unsigned long addr,
+ pte_t *ptep, unsigned long sz)
+{
+ return get_clear_contig(mm, addr, ptep, num_contig_ptes(sz));
+}
+
+/*
+ * The block size comes from the hstate rather than from the entry, because
+ * this is also reached for non-present entries, which carry no hint.
+ */
+pte_t huge_ptep_clear_flush(struct vm_area_struct *vma, unsigned long addr,
+ pte_t *ptep)
+{
+ unsigned long sz = huge_page_size(hstate_vma(vma));
+
+ return get_clear_contig_flush(vma->vm_mm, addr, ptep,
+ num_contig_ptes(sz));
+}
+
+pte_t huge_ptep_modify_prot_start(struct vm_area_struct *vma, unsigned long addr,
+ pte_t *ptep)
+{
+ return huge_ptep_clear_flush(vma, addr, ptep);
+}
+
+/* True if any entry of the block differs from @pte outside the frame number. */
+static bool gh_access_flags_changed(pte_t *ptep, pte_t pte,
+ unsigned long ncontig)
+{
+ unsigned long i;
+
+ for (i = 0; i < ncontig; i++)
+ if ((pte_val(ptep_get(ptep + i)) ^ pte_val(pte)) & ~_PFN_MASK)
+ return true;
+
+ return false;
+}
+
+int huge_ptep_set_access_flags(struct vm_area_struct *vma, unsigned long addr,
+ pte_t *ptep, pte_t pte, int dirty)
+{
+ struct mm_struct *mm = vma->vm_mm;
+ unsigned long sz = huge_page_size(hstate_vma(vma));
+ unsigned long ncontig = num_contig_ptes(sz);
+ pte_t orig_pte = huge_ptep_get(mm, addr, ptep);
+
+ /* Keep the young and dirty state the block already has. */
+ if (pte_dirty(orig_pte))
+ pte = pte_mkdirty(pte);
+ if (pte_young(orig_pte))
+ pte = pte_mkyoung(pte);
+
+ /* Breaking an unchanged block only makes concurrent touchers refault. */
+ if (!gh_access_flags_changed(ptep, pte, ncontig))
+ return 0;
+
+ get_clear_contig_flush(mm, addr, ptep, ncontig);
+ set_ptes(mm, addr, ptep, pte, ncontig);
+
+ return 1;
+}
+
+void huge_ptep_set_wrprotect(struct mm_struct *mm, unsigned long addr,
+ pte_t *ptep)
+{
+ pte_t pte = ptep_get(ptep);
+ unsigned long ncontig = gh_pte_num(pte_gh_order(pte));
+
+ VM_WARN_ON_ONCE(!pte_huge(pte));
+
+ /*
+ * pte_wrprotect() sets _PAGE_FOW, which lives in the bits the whole
+ * block must agree on, so this takes the break before make path.
+ */
+ pte = pte_wrprotect(get_clear_contig_flush(mm, addr, ptep, ncontig));
+ set_ptes(mm, addr, ptep, pte, ncontig);
+}
+
+void huge_pte_clear(struct mm_struct *mm, unsigned long addr, pte_t *ptep,
+ unsigned long sz)
+{
+ unsigned long i, ncontig = num_contig_ptes(sz);
+
+ for (i = 0; i < ncontig; i++, addr += PAGE_SIZE, ptep++)
+ pte_clear(mm, addr, ptep);
+}
+
+bool __init arch_hugetlb_valid_size(unsigned long size)
+{
+ return gh_order_from_size(size) != 0;
+}
+
+static __init int alpha_hugetlbpage_init(void)
+{
+ unsigned int order;
+
+ for_each_gh_order(order)
+ hugetlb_add_hstate(gh_cont_shift(order) - PAGE_SHIFT);
+
+ return 0;
+}
+arch_initcall(alpha_hugetlbpage_init);
--
2.54.0
prev parent reply other threads:[~2026-10-06 14:04 UTC|newest]
Thread overview: 7+ messages / expand[flat|nested] mbox.gz Atom feed top
2026-10-06 14:04 [PATCH 0/6] alpha: hugetlb support using granularity hints Matt Turner
2026-10-06 14:04 ` [PATCH 1/6] alpha: add missing page_table_check_pte_clear() to ptep_get_and_clear() Matt Turner
2026-10-06 14:04 ` [PATCH 2/6] alpha: handle VM_FAULT_HWPOISON in do_page_fault() Matt Turner
2026-10-06 14:04 ` [PATCH 3/6] alpha: clarify that _PAGE_URE and _PAGE_UWE are the Executive bits Matt Turner
2026-10-06 14:04 ` [PATCH 4/6] alpha: define granularity hint PTE bits Matt Turner
2026-10-06 14:04 ` [PATCH 5/6] alpha: align hugetlb mappings in arch_get_unmapped_area() Matt Turner
2026-10-06 14:04 ` Matt Turner [this message]
Reply instructions:
You may reply publicly to this message via plain-text email
using any one of the following methods:
* Save the following mbox file, import it into your mail client,
and reply-to-all from there: mbox
Avoid top-posting and favor interleaved quoting:
https://en.wikipedia.org/wiki/Posting_style#Interleaved_style
* Reply using the --to, --cc, and --in-reply-to
switches of git-send-email(1):
git send-email \
--in-reply-to=20261006-alpha-hugepages-v1-6-a673a18aaa70@gmail.com \
--to=mattst88@gmail.com \
--cc=linmag7@gmail.com \
--cc=linux-alpha@vger.kernel.org \
--cc=linux-kernel@vger.kernel.org \
--cc=richard.henderson@linaro.org \
/path/to/YOUR_REPLY
https://kernel.org/pub/software/scm/git/docs/git-send-email.html
* If your mail client supports setting the In-Reply-To header
via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line
before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
all inboxes | Powered by JetHome®