mirror of
https://git.kernel.org/pub/scm/linux/kernel/git/torvalds/linux.git
synced 2026-08-31 15:22:21 -04:00
fs/resctrl: Fix UAF from worker threads when domains are removed
The mbm_handle_overflow() and cqm_handle_limbo() workers read event counters
and may sleep while doing so. They are scheduled via delayed_work embedded in
struct rdt_l3_mon_domain. Architecture allocates and frees these domains from
CPU hotplug callbacks under cpus_write_lock(), and the workers acquire
cpus_read_lock() to keep the domain alive across their access.
A use-after-free can occur when a worker is blocked waiting for
cpus_read_lock() while the hotplug core holds cpus_write_lock(): the
architecture frees the rdt_l3_mon_domain that contains the worker's
work_struct. When the worker unblocks, the container_of() it performs on the
embedded work pointer dereferences freed memory.
Drop cpus_read_lock() from the workers and instead drain pending and in-flight
work synchronously before the architecture can free the domain. Since
architecture offlines the domain under cpus_write_lock() after it has been
unlinked from the RCU list and a grace period has elapsed, no new work can be
scheduled. The cancel only needs to wait out existing work. Drop
rdtgroup_mutex during CPU offline around cancel_delayed_work_sync() so that
a worker waiting on the mutex can complete before re-pinning the work on
a different CPU.
When offlining a CPU the architecture may iterate over resources in any order.
For example, the MBA control domain may be offlined before or after
a corresponding L3 monitor domain. Ensure that resctrl fs cancels the workers
no matter what order the architecture offlines the domains.
Fixes: 24247aeeab ("x86/intel_rdt/cqm: Improve limbo list processing")
Closes: https://sashiko.dev/#/patchset/20260429184858.36423-1-tony.luck%40intel.com # [1]
Reported-by: Sashiko <sashiko-bot@kernel.org>
Co-developed-by: Tony Luck <tony.luck@intel.com>
Signed-off-by: Tony Luck <tony.luck@intel.com>
Signed-off-by: Reinette Chatre <reinette.chatre@intel.com>
Signed-off-by: Borislav Petkov (AMD) <bp@alien8.de>
Link: https://patch.msgid.link/3f0e0752deb3421606dfc4600f0ab3a4ae098cd7.1783963505.git.reinette.chatre@intel.com
This commit is contained in:
committed by
Borislav Petkov (AMD)
parent
25afd838fb
commit
2566b5cd6a
@@ -633,14 +633,22 @@ void mon_event_count(void *info)
|
||||
rr->err = 0;
|
||||
}
|
||||
|
||||
static struct rdt_ctrl_domain *get_ctrl_domain_from_cpu(int cpu,
|
||||
struct rdt_resource *r)
|
||||
/*
|
||||
* Find the software controller's ctrl domain that contains @cpu on resource @r.
|
||||
*
|
||||
* Only called from the mbm_over worker via update_mba_bw() where the returned
|
||||
* domain is kept alive by cancel_delayed_work_sync() in
|
||||
* resctrl_offline_ctrl_domain(). This drains this worker and then waits on
|
||||
* rdtgroup_mutex held here before the architecture can free the ctrl domain.
|
||||
*
|
||||
* Context: Call from RCU read-side critical section.
|
||||
*/
|
||||
static struct rdt_ctrl_domain *get_sc_ctrl_domain_from_cpu(int cpu,
|
||||
struct rdt_resource *r)
|
||||
{
|
||||
struct rdt_ctrl_domain *d;
|
||||
|
||||
lockdep_assert_cpus_held();
|
||||
|
||||
list_for_each_entry(d, &r->ctrl_domains, hdr.list) {
|
||||
list_for_each_entry_rcu(d, &r->ctrl_domains, hdr.list) {
|
||||
/* Find the domain that contains this CPU */
|
||||
if (cpumask_test_cpu(cpu, &d->hdr.cpu_mask))
|
||||
return d;
|
||||
@@ -701,7 +709,8 @@ static void update_mba_bw(struct rdtgroup *rgrp, struct rdt_l3_mon_domain *dom_m
|
||||
if (WARN_ON_ONCE(!pmbm_data))
|
||||
return;
|
||||
|
||||
dom_mba = get_ctrl_domain_from_cpu(smp_processor_id(), r_mba);
|
||||
guard(rcu)();
|
||||
dom_mba = get_sc_ctrl_domain_from_cpu(smp_processor_id(), r_mba);
|
||||
if (!dom_mba) {
|
||||
pr_warn_once("Failure to get domain for MBA update\n");
|
||||
return;
|
||||
@@ -804,11 +813,25 @@ void cqm_handle_limbo(struct work_struct *work)
|
||||
unsigned long delay = msecs_to_jiffies(CQM_LIMBOCHECK_INTERVAL);
|
||||
struct rdt_l3_mon_domain *d;
|
||||
|
||||
cpus_read_lock();
|
||||
/*
|
||||
* Safe to run without CPU hotplug lock. Work is guaranteed to be
|
||||
* canceled before the domain structure is removed.
|
||||
*/
|
||||
mutex_lock(&rdtgroup_mutex);
|
||||
|
||||
/*
|
||||
* Ensure the worker is dedicated to a CPU as intended and not
|
||||
* relocated by workqueue subsystem as part of CPU going offline.
|
||||
*/
|
||||
if (!is_percpu_thread())
|
||||
goto out_unlock;
|
||||
|
||||
d = container_of(work, struct rdt_l3_mon_domain, cqm_limbo.work);
|
||||
|
||||
/* Domain is going offline */
|
||||
if (cpumask_empty(&d->hdr.cpu_mask))
|
||||
goto out_unlock;
|
||||
|
||||
__check_limbo(d, false);
|
||||
|
||||
if (has_busy_rmid(d)) {
|
||||
@@ -818,8 +841,8 @@ void cqm_handle_limbo(struct work_struct *work)
|
||||
delay);
|
||||
}
|
||||
|
||||
out_unlock:
|
||||
mutex_unlock(&rdtgroup_mutex);
|
||||
cpus_read_unlock();
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -851,7 +874,10 @@ void mbm_handle_overflow(struct work_struct *work)
|
||||
struct list_head *head;
|
||||
struct rdt_resource *r;
|
||||
|
||||
cpus_read_lock();
|
||||
/*
|
||||
* Safe to run without CPU hotplug lock. Work is guaranteed to be
|
||||
* canceled before the domain structure is removed.
|
||||
*/
|
||||
mutex_lock(&rdtgroup_mutex);
|
||||
|
||||
/*
|
||||
@@ -861,9 +887,24 @@ void mbm_handle_overflow(struct work_struct *work)
|
||||
if (!resctrl_mounted || !resctrl_arch_mon_capable())
|
||||
goto out_unlock;
|
||||
|
||||
/*
|
||||
* Ensure the worker is dedicated to a CPU and not relocated by
|
||||
* workqueue subsystem as part of CPU going offline since reading
|
||||
* events depend on smp_processor_id(). After passing this check
|
||||
* smp_processor_id() is valid for entire duration of this worker
|
||||
* since it runs with rdtgroup_mutex held and the offline handler needs
|
||||
* rdtgroup_mutex to offline the CPU being run on here.
|
||||
*/
|
||||
if (!is_percpu_thread())
|
||||
goto out_unlock;
|
||||
|
||||
r = resctrl_arch_get_resource(RDT_RESOURCE_L3);
|
||||
d = container_of(work, struct rdt_l3_mon_domain, mbm_over.work);
|
||||
|
||||
/* Domain is going offline */
|
||||
if (cpumask_empty(&d->hdr.cpu_mask))
|
||||
goto out_unlock;
|
||||
|
||||
list_for_each_entry(prgrp, &rdt_all_groups, rdtgroup_list) {
|
||||
mbm_update(r, d, prgrp);
|
||||
|
||||
@@ -885,7 +926,6 @@ void mbm_handle_overflow(struct work_struct *work)
|
||||
|
||||
out_unlock:
|
||||
mutex_unlock(&rdtgroup_mutex);
|
||||
cpus_read_unlock();
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -4489,6 +4489,29 @@ static void domain_destroy_l3_mon_state(struct rdt_l3_mon_domain *d)
|
||||
|
||||
void resctrl_offline_ctrl_domain(struct rdt_resource *r, struct rdt_ctrl_domain *d)
|
||||
{
|
||||
/*
|
||||
* mbm_handle_overflow() may dereference this ctrl domain via
|
||||
* update_mba_bw()->get_sc_ctrl_domain_from_cpu(). The architecture has
|
||||
* unlinked the domain from the RCU list and waited a grace period, so
|
||||
* no new worker iteration can find it; drain any worker that already
|
||||
* holds a pointer to it before the architecture frees the domain.
|
||||
*
|
||||
* Software controller is enabled/disabled on mount/unmount with
|
||||
* cpus_read_lock() held. Running here with cpus_write_lock() so
|
||||
* there are no concurrent changes to software controller status.
|
||||
*/
|
||||
if (r->rid == RDT_RESOURCE_MBA && is_mba_sc(r)) {
|
||||
struct rdt_resource *l3 = resctrl_arch_get_resource(RDT_RESOURCE_L3);
|
||||
struct rdt_l3_mon_domain *mon_d;
|
||||
|
||||
list_for_each_entry_rcu(mon_d, &l3->mon_domains, hdr.list, lockdep_is_cpus_held()) {
|
||||
if (mon_d->hdr.id == d->hdr.id) {
|
||||
cancel_delayed_work_sync(&mon_d->mbm_over);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
mutex_lock(&rdtgroup_mutex);
|
||||
|
||||
if (supports_mba_mbps() && r->rid == RDT_RESOURCE_MBA)
|
||||
@@ -4501,6 +4524,24 @@ void resctrl_offline_mon_domain(struct rdt_resource *r, struct rdt_domain_hdr *h
|
||||
{
|
||||
struct rdt_l3_mon_domain *d;
|
||||
|
||||
/*
|
||||
* Called by architecture under CPU hotplug lock as it prepares to remove
|
||||
* the domain which is guaranteed to be accessible here.
|
||||
* The domain has been unlinked from the RCU list and a grace period
|
||||
* has elapsed, so no new worker can be scheduled. Drain any worker that
|
||||
* is in flight or pending before letting architecture proceed to free
|
||||
* the domain that has the workers' struct delayed_work embedded.
|
||||
* Do so before taking rdtgroup_mutex since the workers also acquire it.
|
||||
*/
|
||||
if (r->rid == RDT_RESOURCE_L3 &&
|
||||
domain_header_is_valid(hdr, RESCTRL_MON_DOMAIN, RDT_RESOURCE_L3)) {
|
||||
d = container_of(hdr, struct rdt_l3_mon_domain, hdr);
|
||||
if (resctrl_is_mbm_enabled())
|
||||
cancel_delayed_work_sync(&d->mbm_over);
|
||||
if (resctrl_is_mon_event_enabled(QOS_L3_OCCUP_EVENT_ID))
|
||||
cancel_delayed_work_sync(&d->cqm_limbo);
|
||||
}
|
||||
|
||||
mutex_lock(&rdtgroup_mutex);
|
||||
|
||||
/*
|
||||
@@ -4517,8 +4558,6 @@ void resctrl_offline_mon_domain(struct rdt_resource *r, struct rdt_domain_hdr *h
|
||||
goto out_unlock;
|
||||
|
||||
d = container_of(hdr, struct rdt_l3_mon_domain, hdr);
|
||||
if (resctrl_is_mbm_enabled())
|
||||
cancel_delayed_work(&d->mbm_over);
|
||||
if (resctrl_is_mon_event_enabled(QOS_L3_OCCUP_EVENT_ID) && has_busy_rmid(d)) {
|
||||
/*
|
||||
* When a package is going down, forcefully
|
||||
@@ -4529,7 +4568,6 @@ void resctrl_offline_mon_domain(struct rdt_resource *r, struct rdt_domain_hdr *h
|
||||
* package never comes back.
|
||||
*/
|
||||
__check_limbo(d, true);
|
||||
cancel_delayed_work(&d->cqm_limbo);
|
||||
}
|
||||
|
||||
domain_destroy_l3_mon_state(d);
|
||||
@@ -4710,12 +4748,16 @@ void resctrl_offline_cpu(unsigned int cpu)
|
||||
d = get_mon_domain_from_cpu(cpu, l3);
|
||||
if (d) {
|
||||
if (resctrl_is_mbm_enabled() && cpu == d->mbm_work_cpu) {
|
||||
cancel_delayed_work(&d->mbm_over);
|
||||
mutex_unlock(&rdtgroup_mutex);
|
||||
cancel_delayed_work_sync(&d->mbm_over);
|
||||
mutex_lock(&rdtgroup_mutex);
|
||||
mbm_setup_overflow_handler(d, 0, cpu);
|
||||
}
|
||||
if (resctrl_is_mon_event_enabled(QOS_L3_OCCUP_EVENT_ID) &&
|
||||
cpu == d->cqm_work_cpu && has_busy_rmid(d)) {
|
||||
cancel_delayed_work(&d->cqm_limbo);
|
||||
mutex_unlock(&rdtgroup_mutex);
|
||||
cancel_delayed_work_sync(&d->cqm_limbo);
|
||||
mutex_lock(&rdtgroup_mutex);
|
||||
cqm_setup_limbo_handler(d, 0, cpu);
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user