Merge branch 'for-next/mpam' into for-next/core

* for-next/mpam:
  arm_mpam: Disable driver unbind to avoid UAF
  arm_mpam: Fix a NULL pointer dereference on unbinding after an error interrupt
  arm_mpam: Apply T241-MPAM-6 to 63-bit counters
  arm64: mpam: Add memory bandwidth usage (MBWU) documentation
  arm_mpam: resctrl: Add resctrl_arch_cntr_read() & resctrl_arch_reset_cntr()
  arm_mpam: resctrl: Add resctrl_arch_config_cntr() for ABMC use
  arm_mpam: resctrl: Pre-allocate assignable monitors
  arm_mpam: resctrl: Pick classes for use as MBM counters
This commit is contained in:
Will Deacon
2026-08-14 10:16:09 +00:00
4 changed files with 329 additions and 27 deletions

View File

@@ -65,6 +65,28 @@ The supported features are:
there is at least one CSU monitor on each MSC that makes up the L3 group.
Exposing CSU counters from other caches or devices is not supported.
* Memory Bandwidth Usage (MBWU) on or after the L3 cache. resctrl uses the
L3 cache-id to identify where the memory bandwidth is measured. For this
reason the platform must have an L3 cache with cache-id's supplied by
firmware. (The platform doesn't need to support MPAM.)
Memory bandwidth monitoring makes use of MBWU monitors in each MSC that
makes up the L3 group. If the memory bandwidth monitoring is on the memory
rather than the L3 then there must be a single global L3 as otherwise it
is unknown which L3 the traffic came from.
To expose 'mbm_total_bytes', the topology of the group of MSC chosen must
match the topology of the L3 cache so that the cache-id's can be
repainted. For example: Platforms with Memory bandwidth monitors on
CPU-less NUMA nodes cannot expose 'mbm_total_bytes' as these nodes do not
have a corresponding L3 cache. 'mbm_local_bytes' is not exposed as MPAM
cannot distinguish local traffic from global traffic.
All these restrictions based on L3 cache are due to resctrl, currently, only
supporting monitoring at the L3 scope. It is expected that going forward more
MBWU monitors can be exposed to the user after support for more monitoring
scopes is added to resctrl.
Reporting Bugs
==============
If you are not seeing the counters or controls you expect please share the

View File

@@ -1196,8 +1196,7 @@ static u64 mpam_msmon_overflow_val(enum mpam_device_features type,
{
u64 overflow_val = __mpam_msmon_overflow_val(type);
if (mpam_has_quirk(T241_MBW_COUNTER_SCALE_64, msc) &&
type != mpam_feat_msmon_mbwu_63counter)
if (mpam_has_quirk(T241_MBW_COUNTER_SCALE_64, msc))
overflow_val *= 64;
return overflow_val;
@@ -1293,8 +1292,7 @@ static void __ris_msmon_read(void *arg)
now = FIELD_GET(MSMON___VALUE, now);
}
if (mpam_has_quirk(T241_MBW_COUNTER_SCALE_64, msc) &&
m->type != mpam_feat_msmon_mbwu_63counter)
if (mpam_has_quirk(T241_MBW_COUNTER_SCALE_64, msc))
now *= 64;
if (nrdy)
@@ -2019,6 +2017,9 @@ static void mpam_msc_drv_remove(struct platform_device *pdev)
{
struct mpam_msc *msc = platform_get_drvdata(pdev);
if (!msc)
return;
mutex_lock(&mpam_list_lock);
mpam_msc_destroy(msc);
mutex_unlock(&mpam_list_lock);
@@ -2132,6 +2133,7 @@ static int mpam_msc_drv_probe(struct platform_device *pdev)
static struct platform_driver mpam_msc_driver = {
.driver = {
.name = "mpam_msc",
.suppress_bind_attrs = true,
},
.probe = mpam_msc_drv_probe,
.remove = mpam_msc_drv_remove,

View File

@@ -409,7 +409,11 @@ struct mpam_resctrl_res {
struct mpam_resctrl_mon {
struct mpam_class *class;
/* per-class data that resctrl needs will live here */
/* Array of allocated MBWU monitors, indexed by (closid, rmid). */
int *mbwu_idx_to_mon;
/* Array of assigned MBWU monitors, indexed by resctrl's cntr_id. */
int *assigned_counters;
};
static inline int mpam_alloc_csu_mon(struct mpam_class *class)

View File

@@ -119,28 +119,9 @@ void resctrl_arch_reset_rmid(struct rdt_resource *r, struct rdt_l3_mon_domain *d
{
}
void resctrl_arch_reset_cntr(struct rdt_resource *r, struct rdt_l3_mon_domain *d,
u32 closid, u32 rmid, int cntr_id,
enum resctrl_event_id eventid)
{
}
void resctrl_arch_config_cntr(struct rdt_resource *r, struct rdt_l3_mon_domain *d,
enum resctrl_event_id evtid, u32 rmid, u32 closid,
u32 cntr_id, bool assign)
{
}
int resctrl_arch_cntr_read(struct rdt_resource *r, struct rdt_l3_mon_domain *d,
u32 unused, u32 rmid, int cntr_id,
enum resctrl_event_id eventid, u64 *val)
{
return -EOPNOTSUPP;
}
bool resctrl_arch_mbm_cntr_assign_enabled(struct rdt_resource *r)
{
return false;
return (r == &mpam_resctrl_controls[RDT_RESOURCE_L3].resctrl_res);
}
int resctrl_arch_mbm_cntr_assign_set(struct rdt_resource *r, bool enable)
@@ -185,6 +166,26 @@ static void resctrl_reset_task_closids(void)
read_unlock(&tasklist_lock);
}
static void mpam_resctrl_monitor_sync_abmc_vals(struct rdt_resource *l3)
{
struct mpam_resctrl_mon *mon = &mpam_resctrl_counters[QOS_L3_MBM_TOTAL_EVENT_ID];
if (!mon->class)
return;
if (!mon->assigned_counters)
return;
l3->mon.num_mbm_cntrs = mon->class->props.num_mbwu_mon;
if (cdp_enabled)
l3->mon.num_mbm_cntrs /= 2;
/*
* Continue as normal even if enabling cdp causes there to be
* zero counters. This avoids giving resctrl mixed messages.
*/
}
int resctrl_arch_set_cdp_enabled(enum resctrl_res_level rid, bool enable)
{
u32 partid_i = RESCTRL_RESERVED_CLOSID, partid_d = RESCTRL_RESERVED_CLOSID;
@@ -244,6 +245,7 @@ int resctrl_arch_set_cdp_enabled(enum resctrl_res_level rid, bool enable)
WRITE_ONCE(arm64_mpam_global_default, mpam_get_regval(current));
resctrl_reset_task_closids();
mpam_resctrl_monitor_sync_abmc_vals(l3);
for_each_possible_cpu(cpu)
mpam_set_cpu_defaults(cpu, partid_d, partid_i, 0, 0);
@@ -454,6 +456,14 @@ static int __read_mon(struct mpam_resctrl_mon *mon, struct mpam_component *mon_c
/* Shift closid to account for CDP */
closid = resctrl_get_config_index(closid, cdp_type);
if (mon_idx == USE_PRE_ALLOCATED) {
int mbwu_idx = resctrl_arch_rmid_idx_encode(closid, rmid);
mon_idx = mon->mbwu_idx_to_mon[mbwu_idx];
if (mon_idx == -1)
return -ENOENT;
}
if (irqs_disabled()) {
/* Check if we can access this domain without an IPI */
return -EIO;
@@ -526,6 +536,84 @@ int resctrl_arch_rmid_read(struct rdt_resource *r, struct rdt_domain_hdr *hdr,
closid, rmid, val);
}
/* MBWU counters when in ABMC mode */
int resctrl_arch_cntr_read(struct rdt_resource *r, struct rdt_l3_mon_domain *d,
u32 closid, u32 rmid, int mon_idx,
enum resctrl_event_id eventid, u64 *val)
{
struct mpam_resctrl_mon *mon = &mpam_resctrl_counters[eventid];
struct mpam_resctrl_dom *l3_dom;
struct mpam_component *mon_comp;
if (!mpam_is_enabled())
return -EINVAL;
if (eventid == QOS_L3_OCCUP_EVENT_ID || !mon->class)
return -EINVAL;
l3_dom = container_of(d, struct mpam_resctrl_dom, resctrl_mon_dom);
mon_comp = l3_dom->mon_comp[eventid];
return read_mon_cdp_safe(mon, mon_comp, mpam_feat_msmon_mbwu,
USE_PRE_ALLOCATED, closid, rmid, val);
}
static void __reset_mon(struct mpam_resctrl_mon *mon, struct mpam_component *mon_comp,
int mon_idx,
enum resctrl_conf_type cdp_type, u32 closid, u32 rmid)
{
struct mon_cfg cfg = { };
if (!mpam_is_enabled())
return;
/* Shift closid to account for CDP */
closid = resctrl_get_config_index(closid, cdp_type);
if (mon_idx == USE_PRE_ALLOCATED) {
int mbwu_idx = resctrl_arch_rmid_idx_encode(closid, rmid);
mon_idx = mon->mbwu_idx_to_mon[mbwu_idx];
}
if (mon_idx == -1)
return;
cfg.mon = mon_idx;
mpam_msmon_reset_mbwu(mon_comp, &cfg);
}
static void reset_mon_cdp_safe(struct mpam_resctrl_mon *mon, struct mpam_component *mon_comp,
int mon_idx, u32 closid, u32 rmid)
{
if (cdp_enabled) {
__reset_mon(mon, mon_comp, mon_idx, CDP_CODE, closid, rmid);
__reset_mon(mon, mon_comp, mon_idx, CDP_DATA, closid, rmid);
} else {
__reset_mon(mon, mon_comp, mon_idx, CDP_NONE, closid, rmid);
}
}
/* Reset an assigned counter */
void resctrl_arch_reset_cntr(struct rdt_resource *r, struct rdt_l3_mon_domain *d,
u32 closid, u32 rmid, int cntr_id,
enum resctrl_event_id eventid)
{
struct mpam_resctrl_mon *mon = &mpam_resctrl_counters[eventid];
struct mpam_resctrl_dom *l3_dom;
struct mpam_component *mon_comp;
if (!mpam_is_enabled())
return;
if (eventid == QOS_L3_OCCUP_EVENT_ID || !mon->class)
return;
l3_dom = container_of(d, struct mpam_resctrl_dom, resctrl_mon_dom);
mon_comp = l3_dom->mon_comp[eventid];
reset_mon_cdp_safe(mon, mon_comp, USE_PRE_ALLOCATED, closid, rmid);
}
/*
* The rmid realloc threshold should be for the smallest cache exposed to
* resctrl.
@@ -606,6 +694,19 @@ static bool cache_has_usable_csu(struct mpam_class *class)
return true;
}
static bool class_has_usable_mbwu(struct mpam_class *class)
{
struct mpam_props *cprops = &class->props;
if (!mpam_has_feature(mpam_feat_msmon_mbwu, cprops))
return false;
if (!cprops->num_mbwu_mon)
return false;
return true;
}
/*
* Calculate the worst-case percentage change from each implemented step
* in the control.
@@ -925,6 +1026,50 @@ static void mpam_resctrl_pick_mba(void)
}
}
static void __free_mbwu_mon(struct mpam_class *class, int *array,
u16 num_mbwu_mon)
{
for (int i = 0; i < num_mbwu_mon; i++) {
if (array[i] < 0)
continue;
mpam_free_mbwu_mon(class, array[i]);
array[i] = -1;
}
}
static int __alloc_mbwu_mon(struct mpam_class *class, int *array,
u16 num_mbwu_mon)
{
for (int i = 0; i < num_mbwu_mon; i++) {
int mbwu_mon = mpam_alloc_mbwu_mon(class);
if (mbwu_mon < 0) {
__free_mbwu_mon(class, array, num_mbwu_mon);
return mbwu_mon;
}
array[i] = mbwu_mon;
}
return 0;
}
static int *__alloc_mbwu_array(struct mpam_class *class, u16 num_mbwu_mon)
{
int err;
int *array __free(kvfree) = kvmalloc_objs(*array, num_mbwu_mon);
if (!array)
return ERR_PTR(-ENOMEM);
memset(array, -1, num_mbwu_mon * sizeof(*array));
err = __alloc_mbwu_mon(class, array, num_mbwu_mon);
if (err)
return ERR_PTR(err);
return_ptr(array);
}
static void counter_update_class(enum resctrl_event_id evt_id,
struct mpam_class *class)
{
@@ -983,9 +1128,69 @@ static void mpam_resctrl_pick_counters(void)
break;
}
}
if (class_has_usable_mbwu(class) &&
topology_matches_l3(class) &&
traffic_matches_l3(class)) {
pr_debug("class %u has usable MBWU, and matches L3 topology and traffic\n",
class->level);
/*
* An MSC measures bandwidth for a path determined by
* its location in hardware. We can't distinguish
* traffic by destination so we don't know if it's
* staying on the same NUMA node. Hence, we can't
* calculate mbm_local except when we only have one L3
* and it's equivalent to mbm_total and so always use
* mbm_total.
*/
counter_update_class(QOS_L3_MBM_TOTAL_EVENT_ID, class);
}
}
}
static void __config_cntr(struct mpam_resctrl_mon *mon, u32 cntr_id,
enum resctrl_conf_type cdp_type, u32 closid, u32 rmid,
bool assign)
{
/* Same CDP index remap as closid; maps cntr_id to assigned_counters[]. */
u32 mbwu_idx, mon_idx = resctrl_get_config_index(cntr_id, cdp_type);
closid = resctrl_get_config_index(closid, cdp_type);
mbwu_idx = resctrl_arch_rmid_idx_encode(closid, rmid);
if (assign)
mon->mbwu_idx_to_mon[mbwu_idx] = mon->assigned_counters[mon_idx];
else
mon->mbwu_idx_to_mon[mbwu_idx] = -1;
}
void resctrl_arch_config_cntr(struct rdt_resource *r, struct rdt_l3_mon_domain *d,
enum resctrl_event_id evtid, u32 rmid, u32 closid,
u32 cntr_id, bool assign)
{
struct mpam_resctrl_mon *mon = &mpam_resctrl_counters[evtid];
if (evtid != QOS_L3_MBM_TOTAL_EVENT_ID) {
pr_debug("unexpected event id\n");
return;
}
if (!mon->mbwu_idx_to_mon || !mon->assigned_counters) {
pr_debug("monitor arrays not allocated\n");
return;
}
if (cdp_enabled) {
__config_cntr(mon, cntr_id, CDP_CODE, closid, rmid, assign);
__config_cntr(mon, cntr_id, CDP_DATA, closid, rmid, assign);
} else {
__config_cntr(mon, cntr_id, CDP_NONE, closid, rmid, assign);
}
resctrl_arch_reset_cntr(r, d, closid, rmid, cntr_id, QOS_L3_MBM_TOTAL_EVENT_ID);
}
static int mpam_resctrl_control_init(struct mpam_resctrl_res *res)
{
struct mpam_class *class = res->class;
@@ -1063,6 +1268,43 @@ static int mpam_resctrl_pick_domain_id(int cpu, struct mpam_component *comp)
return comp->comp_id;
}
/*
* This must run after all event counters have been picked so that any free
* running counters have already been allocated.
*/
static int mpam_resctrl_monitor_init_abmc(struct mpam_resctrl_mon *mon)
{
struct mpam_resctrl_res *res = &mpam_resctrl_controls[RDT_RESOURCE_L3];
size_t num_rmid = resctrl_arch_system_num_rmid_idx();
struct rdt_resource *l3 = &res->resctrl_res;
struct mpam_class *class = mon->class;
u16 num_mbwu_mon;
int *cntrs;
int *rmid_array __free(kvfree) = kvmalloc_objs(*rmid_array, num_rmid);
if (!rmid_array) {
pr_debug("Failed to allocate RMID array\n");
return -ENOMEM;
}
memset(rmid_array, -1, num_rmid * sizeof(*rmid_array));
num_mbwu_mon = class->props.num_mbwu_mon;
cntrs = __alloc_mbwu_array(mon->class, num_mbwu_mon);
if (IS_ERR(cntrs))
return PTR_ERR(cntrs);
mon->assigned_counters = cntrs;
mon->mbwu_idx_to_mon = no_free_ptr(rmid_array);
l3->mon.mbm_cntr_assignable = true;
l3->mon.mbm_assign_on_mkdir = true;
l3->mon.mbm_cntr_configurable = false;
l3->mon.mbm_cntr_assign_fixed = true;
mpam_resctrl_monitor_sync_abmc_vals(l3);
return 0;
}
static int mpam_resctrl_monitor_init(struct mpam_resctrl_mon *mon,
enum resctrl_event_id type)
{
@@ -1107,8 +1349,21 @@ static int mpam_resctrl_monitor_init(struct mpam_resctrl_mon *mon,
*/
l3->mon.num_rmid = resctrl_arch_system_num_rmid_idx();
if (resctrl_enable_mon_event(type, false, 0, NULL))
l3->mon_capable = true;
if (type == QOS_L3_MBM_TOTAL_EVENT_ID) {
int err;
err = mpam_resctrl_monitor_init_abmc(mon);
if (err)
return err;
static_assert(MAX_EVT_CONFIG_BITS == 0x7f);
l3->mon.mbm_cfg_mask = MAX_EVT_CONFIG_BITS;
}
if (!resctrl_enable_mon_event(type, false, 0, NULL))
return -EINVAL;
l3->mon_capable = true;
return 0;
}
@@ -1671,6 +1926,23 @@ void mpam_resctrl_exit(void)
resctrl_exit();
}
static void mpam_resctrl_teardown_mon(struct mpam_resctrl_mon *mon, struct mpam_class *class)
{
u32 num_mbwu_mon = class->props.num_mbwu_mon;
if (!mon->mbwu_idx_to_mon)
return;
if (mon->assigned_counters) {
__free_mbwu_mon(class, mon->assigned_counters, num_mbwu_mon);
kvfree(mon->assigned_counters);
mon->assigned_counters = NULL;
}
kvfree(mon->mbwu_idx_to_mon);
mon->mbwu_idx_to_mon = NULL;
}
/*
* The driver is detaching an MSC from this class, if resctrl was using it,
* pull on resctrl_exit().
@@ -1693,6 +1965,8 @@ void mpam_resctrl_teardown_class(struct mpam_class *class)
for_each_mpam_resctrl_mon(mon, eventid) {
if (mon->class == class) {
mon->class = NULL;
mpam_resctrl_teardown_mon(mon, class);
break;
}
}