mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Babu Moger <babu.moger@amd.com>
To: Tony Luck <tony.luck@intel.com>, Fenghua Yu <fenghuay@nvidia.com>,
	Reinette Chatre <reinette.chatre@intel.com>,
	Maciej Wieczor-Retman <maciej.wieczor-retman@intel.com>,
	Peter Newman <peternewman@google.com>,
	James Morse <james.morse@arm.com>,
	Drew Fustini <dfustini@baylibre.com>,
	Dave Martin <Dave.Martin@arm.com>, Chen Yu <yu.c.chen@intel.com>,
	David E Box <david.e.box@intel.com>,
	x86@kernel.org
Cc: Christoph Hellwig <hch@infradead.org>,
	linux-kernel@vger.kernel.org, patches@lists.linux.dev
Subject: Re: [PATCH v13 13/25] arm,x86,fs/resctrl: Allocate maximum needed rmid_ptrs[]
Date: Tue, 6 Oct 2026 14:53:49 -0500	[thread overview]
Message-ID: <327d9248-2e4f-41b8-b4ad-4f6d01c85ecb@amd.com> (raw)
In-Reply-To: <20260928221509.68002-14-tony.luck@intel.com>

Hi Tony,

On 9/28/26 17:14, Tony Luck wrote:
> resctrl keeps per-RMID state in rmid_ptrs[]. It is allocated on first mount,
> sized for the number of RMIDs available at that time, and reused by every
> subsequent mount.
> 
> Application Energy Telemetry (AET) requires the pmt_telemetry driver to be
> built in. Allowing it to be built as a module means the number of RMIDs can
> change from one mount to the next, so a later mount may need more entries
> than the first mount allocated. Reallocating per mount is not possible because
> the limbo handler continues to access rmid_ptrs[] after resctrl is unmounted.
> 
> Size rmid_ptrs[] for the maximum number of RMIDs the system can ever need,
> so that it is large enough for any future mount. Build the free list
> based on the number of RMIDs needed for the current mount cycle.
> 
> Signed-off-by: Tony Luck <tony.luck@intel.com>
> ---
> v13:
> 	Updated commit message using Reinette suggestion.
> 	Merged patch 14 "Rebuild free RMID list on each mount" into
> 	this patch as the split into separate patches didn't work
> 	as well as I hoped.
> ---
>   include/linux/resctrl.h            |  1 +
>   arch/x86/kernel/cpu/resctrl/core.c | 27 +++++++++++++
>   drivers/resctrl/mpam_resctrl.c     |  9 +++++
>   fs/resctrl/monitor.c               | 64 ++++++++++++++++++++----------
>   4 files changed, 79 insertions(+), 22 deletions(-)
> 
> diff --git a/include/linux/resctrl.h b/include/linux/resctrl.h
> index 764c45be2d59..7afeb3bd55d9 100644
> --- a/include/linux/resctrl.h
> +++ b/include/linux/resctrl.h
> @@ -447,6 +447,7 @@ static inline u32 resctrl_get_default_ctrl(struct rdt_resource *r)
>   /* The number of closid supported by this resource regardless of CDP */
>   u32 resctrl_arch_get_num_closid(struct rdt_resource *r);
>   u32 resctrl_arch_system_num_rmid_idx(void);
> +u32 resctrl_arch_system_max_rmid_idx(void);
>   int resctrl_arch_update_domains(struct rdt_resource *r, u32 closid);
>   
>   /**
> diff --git a/arch/x86/kernel/cpu/resctrl/core.c b/arch/x86/kernel/cpu/resctrl/core.c
> index 2e3b9c16cbda..8702a6da2374 100644
> --- a/arch/x86/kernel/cpu/resctrl/core.c
> +++ b/arch/x86/kernel/cpu/resctrl/core.c
> @@ -45,6 +45,9 @@ static DEFINE_MUTEX(domain_list_lock);
>    */
>   DEFINE_PER_CPU(struct resctrl_pqr_state, pqr_state);
>   
> +/* Number of unique RMID values that can be written to MSR_IA32_PQR_ASSOC.RMID */
> +static u32 pqr_assoc_num_rmid;
> +
>   static void mba_wrmsr_intel(struct msr_param *m);
>   static void cat_wrmsr(struct msr_param *m);
>   static void mba_wrmsr_amd(struct msr_param *m);
> @@ -124,6 +127,28 @@ u32 resctrl_arch_system_num_rmid_idx(void)
>   	return num_rmids == U32_MAX ? 0 : num_rmids;
>   }
>   
> +/**
> + * resctrl_arch_system_max_rmid_idx - Largest possible RMID index
> + *
> + * Return: Largest possible RMID index used for boot time allocations.
> + */
> +u32 resctrl_arch_system_max_rmid_idx(void)
> +{
> +	struct rdt_resource *r = &rdt_resources_all[RDT_RESOURCE_L3].r_resctrl;
> +	u32 num_rmid = pqr_assoc_num_rmid;
> +
> +	/*
> +	 * If the system is capable of L3 monitoring the maximum RMID value may
> +	 * be lower than the system maximum. Either because the L3 monitoring
> +	 * feature supports fewer RMIDs, or because SNC (Sub-NUMA Cluster)
> +	 * is enabled and divides RMIDs per cluster.
> +	 */
> +	if (r->mon_capable)
> +		num_rmid = r->mon.num_rmid;
> +
> +	return num_rmid;
> +}
> +
>   struct rdt_resource *resctrl_arch_get_resource(enum resctrl_res_level l)
>   {
>   	if (l >= RDT_NUM_RESOURCES)
> @@ -967,6 +992,8 @@ static __init bool get_rdt_mon_resources(void)
>   	if (!cpu_feature_enabled(X86_FEATURE_CQM))
>   		return false;
>   
> +	pqr_assoc_num_rmid = cpuid_ebx(0xf) + 1;
> +
>   	/* Any of the L3 monitoring features? */
>   	if (!cpu_feature_enabled(X86_FEATURE_CQM_LLC))
>   		return false;
> diff --git a/drivers/resctrl/mpam_resctrl.c b/drivers/resctrl/mpam_resctrl.c
> index 7ddee8f5162f..52726f7fd03f 100644
> --- a/drivers/resctrl/mpam_resctrl.c
> +++ b/drivers/resctrl/mpam_resctrl.c
> @@ -257,6 +257,15 @@ u32 resctrl_arch_system_num_rmid_idx(void)
>   	return (mpam_pmg_max + 1) * (mpam_partid_max + 1);
>   }
>   
> +/*
> + * File system calls this for one-time allocation of structures.
> + * Return the largest possible value.
> + */
> +u32 resctrl_arch_system_max_rmid_idx(void)
> +{
> +	return resctrl_arch_system_num_rmid_idx();
> +}
> +
>   u32 resctrl_arch_rmid_idx_encode(u32 closid, u32 rmid)
>   {
>   	return closid * (mpam_pmg_max + 1) + rmid;
> diff --git a/fs/resctrl/monitor.c b/fs/resctrl/monitor.c
> index 2a28fe04284b..ccfb1548fea1 100644
> --- a/fs/resctrl/monitor.c
> +++ b/fs/resctrl/monitor.c
> @@ -75,6 +75,11 @@ static unsigned int rmid_limbo_count;
>    */
>   static struct rmid_entry	*rmid_ptrs;
>   
> +/*
> + * @num_rmid_ptrs - The number of elements in rmid_ptrs[].
> + */
> +static u32 num_rmid_ptrs;
> +
>   /*
>    * This is the threshold cache occupancy in bytes at which we will consider an
>    * RMID available for re-allocation.
> @@ -961,45 +966,60 @@ void mbm_setup_overflow_handler(struct rdt_l3_mon_domain *dom, unsigned long del
>   
>   int setup_rmid_lru_list(void)
>   {
> -	struct rmid_entry *entry = NULL;
> -	u32 idx_limit;
> -	u32 idx;
> +	struct rmid_entry *entry;
> +	u32 cur_idx_limit;
> +	u32 rsvd_idx;
>   	int i;
>   
>   	if (!resctrl_mon_capable())
>   		return 0;
>   
>   	/*
> -	 * Called on every mount, but the number of RMIDs cannot change
> -	 * after the first mount, so keep using the same set of rmid_ptrs[]
> -	 * until resctrl_exit(). Note that the limbo handler continues to
> -	 * access rmid_ptrs[] after resctrl is unmounted.
> +	 * Allocate the largest number of RMIDs that this system will ever
> +	 * need. These cannot be freed until resctrl_exit() because the limbo
> +	 * handler continues to access rmid_ptrs[] after resctrl is unmounted.
>   	 */
> -	if (rmid_ptrs)
> -		return 0;
> +	if (!rmid_ptrs) {
> +		num_rmid_ptrs = resctrl_arch_system_max_rmid_idx();
> +		rmid_ptrs = kzalloc_objs(struct rmid_entry, num_rmid_ptrs);
> +		if (!rmid_ptrs) {
> +			num_rmid_ptrs = 0;
> +			return -ENOMEM;
> +		}
>   
> -	idx_limit = resctrl_arch_system_num_rmid_idx();
> -	rmid_ptrs = kzalloc_objs(struct rmid_entry, idx_limit);
> -	if (!rmid_ptrs)
> -		return -ENOMEM;
> +		for (i = 0; i < num_rmid_ptrs; i++) {
> +			entry = &rmid_ptrs[i];
> +			INIT_LIST_HEAD(&entry->list);

[1]. All the nodes are initialized. Basically, it is pointing to itself 
now.

>   
> -	for (i = 0; i < idx_limit; i++) {
> -		entry = &rmid_ptrs[i];
> -		INIT_LIST_HEAD(&entry->list);
> +			resctrl_arch_rmid_idx_decode(i, &entry->closid, &entry->rmid);
> +		}
> +	}
>   
> -		resctrl_arch_rmid_idx_decode(i, &entry->closid, &entry->rmid);
> -		list_add_tail(&entry->list, &rmid_free_lru);
> +	/* Find how many RMIDs are available for this mount */
> +	cur_idx_limit = resctrl_arch_system_num_rmid_idx();
> +	if (cur_idx_limit > num_rmid_ptrs) {
> +		pr_warn_once("RMID count %u exceeds allocation. Limit to %u\n",
> +			     cur_idx_limit, num_rmid_ptrs);
> +		cur_idx_limit = num_rmid_ptrs;
>   	}
>   
> +	INIT_LIST_HEAD(&rmid_free_lru);

Initializing the list head without first cleaning up existing entries 
could be problematic here.

The first mount works correctly, and INIT_LIST_HEAD() is not needed in 
that case because the list head is already statically initialized.

However, on a subsequent mount, the list entries remain linked from the 
previous mount through the setup performed in [2] below.

INIT_LIST_HEAD() only reinitializes the list head itself. It does not 
traverse the list and detach or reinitialize existing entries as was 
done in [1].

I think you need to replace

INIT_LIST_HEAD(&rmid_free_lru);

to

while (!list_empty(&rmid_free_lru))
         list_del_init(rmid_free_lru.next);


This ensures that any entries left over from a previous mount are 
properly detached before the list is reinitialized.


> +
>   	/*
>   	 * RESCTRL_RESERVED_CLOSID and RESCTRL_RESERVED_RMID are special and
>   	 * are always allocated. These are used for the rdtgroup_default
>   	 * control group, which was setup earlier in rdtgroup_setup_default().
>   	 */
> -	idx = resctrl_arch_rmid_idx_encode(RESCTRL_RESERVED_CLOSID,
> -					   RESCTRL_RESERVED_RMID);
> -	entry = __rmid_entry(idx);
> -	list_del(&entry->list);
> +	rsvd_idx = resctrl_arch_rmid_idx_encode(RESCTRL_RESERVED_CLOSID,
> +						RESCTRL_RESERVED_RMID);
> +
> +	for (i = 0; i < cur_idx_limit; i++) {
> +		entry = &rmid_ptrs[i];
> +		/* Don't add reserved or busy entries to free list */
> +		if (i == rsvd_idx || entry->busy)
> +			continue;
> +		list_add_tail(&entry->list, &rmid_free_lru);

[2]. Here nodes are linked to head.  Basically, Head <->A <->B <->HEAD


Thanks

Babu

  reply	other threads:[~2026-10-06 19:53 UTC|newest]

Thread overview: 34+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-09-28 22:14 [PATCH v13 00/25] Allow AET to use PMT as loadable module Tony Luck
2026-09-28 22:14 ` [PATCH v13 01/25] fs/resctrl: Ensure default group reports tasks on monitor-only systems Tony Luck
2026-09-28 22:14 ` [PATCH v13 02/25] x86/cpufeatures: Add missing CQM feature dependency Tony Luck
2026-09-28 22:14 ` [PATCH v13 03/25] x86/resctrl: Check if monitoring features are supported Tony Luck
2026-09-28 22:14 ` [PATCH v13 04/25] x86/resctrl: Centralize monitoring feature enumeration Tony Luck
2026-09-28 22:14 ` [PATCH v13 05/25] x86/resctrl: Apply Intel MBM quirk from rdt_get_l3_mon_config() Tony Luck
2026-09-28 22:14 ` [PATCH v13 06/25] x86/resctrl: Delete resctrl_cpu_detect() Tony Luck
2026-09-28 22:14 ` [PATCH v13 07/25] arm,x86,fs/resctrl: Replace architecture resctrl_arch_{alloc,mon}_capable() Tony Luck
2026-09-28 22:14 ` [PATCH v13 08/25] x86/resctrl: Update special case for Intel Haswell enumeration Tony Luck
2026-09-28 22:14 ` [PATCH v13 09/25] x86/resctrl: Delete rdt_alloc_capable and rdt_mon_capable Tony Luck
2026-09-28 22:14 ` [PATCH v13 10/25] fs/resctrl: Remove redundant calls to resctrl_mon_capable() Tony Luck
2026-09-28 22:14 ` [PATCH v13 11/25] x86/resctrl: Honor rdt={perf|energy} options to force enable AET events Tony Luck
2026-09-28 22:14 ` [PATCH v13 12/25] fs/resctrl: Add interface to disable a monitor event Tony Luck
2026-09-28 22:14 ` [PATCH v13 13/25] arm,x86,fs/resctrl: Allocate maximum needed rmid_ptrs[] Tony Luck
2026-10-06 19:53   ` Babu Moger [this message]
2026-10-06 22:07     ` Luck, Tony
2026-09-28 22:14 ` [PATCH v13 14/25] arm,x86,fs/resctrl: Use right size for L3 monitor data structures Tony Luck
2026-09-28 22:14 ` [PATCH v13 15/25] x86,fs/resctrl: Handle systems where AET is the only resource Tony Luck
2026-09-28 22:15 ` [PATCH v13 16/25] x86/resctrl: Add PMT registration API for AET enumeration callbacks Tony Luck
2026-09-28 22:15 ` [PATCH v13 17/25] platform/x86/intel/pmt: Register enumeration functions with resctrl Tony Luck
2026-09-28 22:15 ` [PATCH v13 18/25] x86/resctrl: Use registered function pointers for AET enumeration Tony Luck
2026-09-29 17:52   ` [PATCH v14.1 " Tony Luck
2026-09-28 22:15 ` [PATCH v13 19/25] arm,x86,fs/resctrl: Enumerate AET on every resctrl mount Tony Luck
2026-09-29 17:52   ` [PATCH v14.1 " Tony Luck
2026-09-28 22:15 ` [PATCH v13 20/25] x86/resctrl: Enforce system RMID limit on AET Tony Luck
2026-09-29 17:52   ` [PATCH v14.1 " Tony Luck
2026-09-28 22:15 ` [PATCH v13 21/25] x86/resctrl: Export interface to report telemetry unbind/remove Tony Luck
2026-09-29 17:52   ` [PATCH v14.1 " Tony Luck
2026-09-28 22:15 ` [PATCH v13 22/25] platform/x86/intel/pmt: Inform resctrl when MMIO maps are being removed Tony Luck
2026-09-28 22:15 ` [PATCH v13 23/25] x86/resctrl: Require 64-bit x86 for resctrl support Tony Luck
2026-09-28 22:15 ` [PATCH v13 24/25] x86/resctrl: Simplify Kconfig options for resctrl Tony Luck
2026-09-28 22:15 ` [PATCH v13 25/25] x86,fs/resctrl: Document telemetry mount timing caveat Tony Luck
2026-09-29  0:29 ` [PATCH v13 00/25] Allow AET to use PMT as loadable module Luck, Tony
2026-09-29 19:43   ` Luck, Tony

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=327d9248-2e4f-41b8-b4ad-4f6d01c85ecb@amd.com \
    --to=babu.moger@amd.com \
    --cc=Dave.Martin@arm.com \
    --cc=david.e.box@intel.com \
    --cc=dfustini@baylibre.com \
    --cc=fenghuay@nvidia.com \
    --cc=hch@infradead.org \
    --cc=james.morse@arm.com \
    --cc=linux-kernel@vger.kernel.org \
    --cc=maciej.wieczor-retman@intel.com \
    --cc=patches@lists.linux.dev \
    --cc=peternewman@google.com \
    --cc=reinette.chatre@intel.com \
    --cc=tony.luck@intel.com \
    --cc=x86@kernel.org \
    --cc=yu.c.chen@intel.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®