diff options
Diffstat (limited to 'fs/resctrl/monitor.c')
| -rw-r--r-- | fs/resctrl/monitor.c | 274 |
1 files changed, 177 insertions, 97 deletions
diff --git a/fs/resctrl/monitor.c b/fs/resctrl/monitor.c index 9fd901c78dc6..73413cb128ea 100644 --- a/fs/resctrl/monitor.c +++ b/fs/resctrl/monitor.c @@ -135,17 +135,17 @@ void __check_limbo(struct rdt_l3_mon_domain *d, bool force_free) struct rdt_resource *r = resctrl_arch_get_resource(RDT_RESOURCE_L3); u32 idx_limit = resctrl_arch_system_num_rmid_idx(); struct rmid_entry *entry; + bool rmid_dirty = true; u32 idx, cur_idx = 1; void *arch_mon_ctx; void *arch_priv; - bool rmid_dirty; u64 val = 0; arch_priv = mon_event_all[QOS_L3_OCCUP_EVENT_ID].arch_priv; arch_mon_ctx = resctrl_arch_mon_ctx_alloc(r, QOS_L3_OCCUP_EVENT_ID); if (IS_ERR(arch_mon_ctx)) { - pr_warn_ratelimited("Failed to allocate monitor context: %ld", - PTR_ERR(arch_mon_ctx)); + pr_warn_ratelimited("Failed to allocate monitor context: %pe", + arch_mon_ctx); return; } @@ -161,22 +161,27 @@ void __check_limbo(struct rdt_l3_mon_domain *d, bool force_free) break; entry = __rmid_entry(idx); - if (resctrl_arch_rmid_read(r, &d->hdr, entry->closid, entry->rmid, - QOS_L3_OCCUP_EVENT_ID, arch_priv, &val, - arch_mon_ctx)) { - rmid_dirty = true; - } else { - rmid_dirty = (val >= resctrl_rmid_realloc_threshold); - - /* - * x86's CLOSID and RMID are independent numbers, so the entry's - * CLOSID is an empty CLOSID (X86_RESCTRL_EMPTY_CLOSID). On Arm the - * RMID (PMG) extends the CLOSID (PARTID) space with bits that aren't - * used to select the configuration. It is thus necessary to track both - * CLOSID and RMID because there may be dependencies between them - * on some architectures. - */ - trace_mon_llc_occupancy_limbo(entry->closid, entry->rmid, d->hdr.id, val); + if (!force_free) { + if (resctrl_arch_rmid_read(r, &d->hdr, entry->closid, + entry->rmid, QOS_L3_OCCUP_EVENT_ID, + arch_priv, &val, arch_mon_ctx)) { + rmid_dirty = true; + } else { + rmid_dirty = (val >= resctrl_rmid_realloc_threshold); + + /* + * x86's CLOSID and RMID are independent numbers, + * so the entry's CLOSID is an empty CLOSID + * (X86_RESCTRL_EMPTY_CLOSID). On Arm the RMID + * (PMG) extends the CLOSID (PARTID) space with + * bits that aren't used to select the configuration. + * It is thus necessary to track both CLOSID and + * RMID because there may be dependencies between + * them on some architectures. + */ + trace_mon_llc_occupancy_limbo(entry->closid, entry->rmid, + d->hdr.id, val); + } } if (force_free || !rmid_dirty) { @@ -304,7 +309,7 @@ static void add_rmid_to_limbo(struct rmid_entry *entry) idx = resctrl_arch_rmid_idx_encode(entry->closid, entry->rmid); entry->busy = 0; - list_for_each_entry(d, &r->mon_domains, hdr.list) { + list_for_each_entry_rcu(d, &r->mon_domains, hdr.list, lockdep_is_cpus_held()) { /* * For the first limbo RMID in the domain, * setup up the limbo worker. @@ -453,8 +458,10 @@ static int __l3_mon_event_count(struct rdtgroup *rdtgrp, struct rmid_read *rr) } /* Reading a single domain, must be on a CPU in that domain. */ - if (!cpumask_test_cpu(cpu, &d->hdr.cpu_mask)) + if (!cpumask_test_cpu(cpu, &d->hdr.cpu_mask)) { + rr->err = -EIO; return -EINVAL; + } if (rr->is_mbm_cntr) rr->err = resctrl_arch_cntr_read(rr->r, d, closid, rmid, cntr_id, rr->evt->evtid, &tval); @@ -491,8 +498,10 @@ static int __l3_mon_event_count_sum(struct rdtgroup *rdtgrp, struct rmid_read *r } /* Summing domains that share a cache, must be on a CPU for that cache. */ - if (!cpumask_test_cpu(cpu, &rr->ci->shared_cpu_map)) + if (!cpumask_test_cpu(cpu, &rr->ci->shared_cpu_map)) { + rr->err = -EIO; return -EINVAL; + } /* * Legacy files must report the sum of an event across all @@ -502,6 +511,11 @@ static int __l3_mon_event_count_sum(struct rdtgroup *rdtgrp, struct rmid_read *r * all domains fail for any reason. */ ret = -EINVAL; + /* + * RCU list being traversed with CPU hotplug lock held. lockdep + * unable to help prove this here since this work is scheduled via + * smp_call*(). Not called from MBM overflow handler. + */ list_for_each_entry(d, &rr->r->mon_domains, hdr.list) { if (d->ci_id != rr->ci->id) continue; @@ -623,14 +637,22 @@ void mon_event_count(void *info) rr->err = 0; } -static struct rdt_ctrl_domain *get_ctrl_domain_from_cpu(int cpu, - struct rdt_resource *r) +/* + * Find the software controller's ctrl domain that contains @cpu on resource @r. + * + * Only called from the mbm_over worker via update_mba_bw() where the returned + * domain is kept alive by cancel_delayed_work_sync() in + * resctrl_offline_ctrl_domain(). This drains this worker and then waits on + * rdtgroup_mutex held here before the architecture can free the ctrl domain. + * + * Context: Call from RCU read-side critical section. + */ +static struct rdt_ctrl_domain *get_sc_ctrl_domain_from_cpu(int cpu, + struct rdt_resource *r) { struct rdt_ctrl_domain *d; - lockdep_assert_cpus_held(); - - list_for_each_entry(d, &r->ctrl_domains, hdr.list) { + list_for_each_entry_rcu(d, &r->ctrl_domains, hdr.list) { /* Find the domain that contains this CPU */ if (cpumask_test_cpu(cpu, &d->hdr.cpu_mask)) return d; @@ -691,7 +713,8 @@ static void update_mba_bw(struct rdtgroup *rgrp, struct rdt_l3_mon_domain *dom_m if (WARN_ON_ONCE(!pmbm_data)) return; - dom_mba = get_ctrl_domain_from_cpu(smp_processor_id(), r_mba); + guard(rcu)(); + dom_mba = get_sc_ctrl_domain_from_cpu(smp_processor_id(), r_mba); if (!dom_mba) { pr_warn_once("Failure to get domain for MBA update\n"); return; @@ -752,8 +775,8 @@ static void mbm_update_one_event(struct rdt_resource *r, struct rdt_l3_mon_domai } else { rr.arch_mon_ctx = resctrl_arch_mon_ctx_alloc(rr.r, evtid); if (IS_ERR(rr.arch_mon_ctx)) { - pr_warn_ratelimited("Failed to allocate monitor context: %ld", - PTR_ERR(rr.arch_mon_ctx)); + pr_warn_ratelimited("Failed to allocate monitor context: %pe", + rr.arch_mon_ctx); return; } } @@ -794,11 +817,25 @@ void cqm_handle_limbo(struct work_struct *work) unsigned long delay = msecs_to_jiffies(CQM_LIMBOCHECK_INTERVAL); struct rdt_l3_mon_domain *d; - cpus_read_lock(); + /* + * Safe to run without CPU hotplug lock. Work is guaranteed to be + * canceled before the domain structure is removed. + */ mutex_lock(&rdtgroup_mutex); + /* + * Ensure the worker is dedicated to a CPU as intended and not + * relocated by workqueue subsystem as part of CPU going offline. + */ + if (!is_percpu_thread()) + goto out_unlock; + d = container_of(work, struct rdt_l3_mon_domain, cqm_limbo.work); + /* Domain is going offline */ + if (cpumask_empty(&d->hdr.cpu_mask)) + goto out_unlock; + __check_limbo(d, false); if (has_busy_rmid(d)) { @@ -808,8 +845,8 @@ void cqm_handle_limbo(struct work_struct *work) delay); } +out_unlock: mutex_unlock(&rdtgroup_mutex); - cpus_read_unlock(); } /** @@ -841,7 +878,10 @@ void mbm_handle_overflow(struct work_struct *work) struct list_head *head; struct rdt_resource *r; - cpus_read_lock(); + /* + * Safe to run without CPU hotplug lock. Work is guaranteed to be + * canceled before the domain structure is removed. + */ mutex_lock(&rdtgroup_mutex); /* @@ -851,9 +891,24 @@ void mbm_handle_overflow(struct work_struct *work) if (!resctrl_mounted || !resctrl_arch_mon_capable()) goto out_unlock; + /* + * Ensure the worker is dedicated to a CPU and not relocated by + * workqueue subsystem as part of CPU going offline since reading + * events depend on smp_processor_id(). After passing this check + * smp_processor_id() is valid for entire duration of this worker + * since it runs with rdtgroup_mutex held and the offline handler needs + * rdtgroup_mutex to offline the CPU being run on here. + */ + if (!is_percpu_thread()) + goto out_unlock; + r = resctrl_arch_get_resource(RDT_RESOURCE_L3); d = container_of(work, struct rdt_l3_mon_domain, mbm_over.work); + /* Domain is going offline */ + if (cpumask_empty(&d->hdr.cpu_mask)) + goto out_unlock; + list_for_each_entry(prgrp, &rdt_all_groups, rdtgroup_list) { mbm_update(r, d, prgrp); @@ -875,7 +930,6 @@ void mbm_handle_overflow(struct work_struct *work) out_unlock: mutex_unlock(&rdtgroup_mutex); - cpus_read_unlock(); } /** @@ -1052,7 +1106,8 @@ int event_filter_show(struct kernfs_open_file *of, struct seq_file *seq, void *v bool sep = false; int ret = 0, i; - mutex_lock(&rdtgroup_mutex); + if (!info_kn_lock(of->kn)) + return -ENOENT; rdt_last_cmd_clear(); r = resctrl_arch_get_resource(mevt->rid); @@ -1073,7 +1128,7 @@ int event_filter_show(struct kernfs_open_file *of, struct seq_file *seq, void *v seq_putc(seq, '\n'); out_unlock: - mutex_unlock(&rdtgroup_mutex); + info_kn_unlock(of->kn); return ret; } @@ -1084,7 +1139,8 @@ int resctrl_mbm_assign_on_mkdir_show(struct kernfs_open_file *of, struct seq_fil struct rdt_resource *r = rdt_kn_parent_priv(of->kn); int ret = 0; - mutex_lock(&rdtgroup_mutex); + if (!info_kn_lock(of->kn)) + return -ENOENT; rdt_last_cmd_clear(); if (!resctrl_arch_mbm_cntr_assign_enabled(r)) { @@ -1096,7 +1152,7 @@ int resctrl_mbm_assign_on_mkdir_show(struct kernfs_open_file *of, struct seq_fil seq_printf(s, "%u\n", r->mon.mbm_assign_on_mkdir); out_unlock: - mutex_unlock(&rdtgroup_mutex); + info_kn_unlock(of->kn); return ret; } @@ -1108,13 +1164,16 @@ ssize_t resctrl_mbm_assign_on_mkdir_write(struct kernfs_open_file *of, char *buf bool value; int ret; - ret = kstrtobool(buf, &value); - if (ret) - return ret; - - mutex_lock(&rdtgroup_mutex); + if (!info_kn_lock(of->kn)) + return -ENOENT; rdt_last_cmd_clear(); + ret = kstrtobool(buf, &value); + if (ret) { + rdt_last_cmd_puts("mbm_assign_on_mkdir: Invalid input\n"); + goto out_unlock; + } + if (!resctrl_arch_mbm_cntr_assign_enabled(r)) { rdt_last_cmd_puts("mbm_event counter assignment mode is not enabled\n"); ret = -EINVAL; @@ -1124,7 +1183,7 @@ ssize_t resctrl_mbm_assign_on_mkdir_write(struct kernfs_open_file *of, char *buf r->mon.mbm_assign_on_mkdir = value; out_unlock: - mutex_unlock(&rdtgroup_mutex); + info_kn_unlock(of->kn); return ret ?: nbytes; } @@ -1211,9 +1270,10 @@ static int rdtgroup_alloc_assign_cntr(struct rdt_resource *r, struct rdt_l3_mon_ * NULL; otherwise, assign the counter to the specified domain @d. * * If all counters in a domain are already in use, rdtgroup_alloc_assign_cntr() - * will fail. The assignment process will abort at the first failure encountered - * during domain traversal, which may result in the event being only partially - * assigned. + * will fail. When attempting to assign counters to all domains, carry on trying + * to assign counters after a failure since only some domains may have counters + * and the goal is to assign counters where possible. If any counter assignment + * fails, return the error from the last failing assignment. * * Return: * 0 on success, < 0 on failure. @@ -1225,10 +1285,12 @@ static int rdtgroup_assign_cntr_event(struct rdt_l3_mon_domain *d, struct rdtgro int ret = 0; if (!d) { - list_for_each_entry(d, &r->mon_domains, hdr.list) { - ret = rdtgroup_alloc_assign_cntr(r, d, rdtgrp, mevt); - if (ret) - return ret; + list_for_each_entry_rcu(d, &r->mon_domains, hdr.list, lockdep_is_cpus_held()) { + int err; + + err = rdtgroup_alloc_assign_cntr(r, d, rdtgrp, mevt); + if (err) + ret = err; } } else { ret = rdtgroup_alloc_assign_cntr(r, d, rdtgrp, mevt); @@ -1295,7 +1357,7 @@ static void rdtgroup_unassign_cntr_event(struct rdt_l3_mon_domain *d, struct rdt struct rdt_resource *r = resctrl_arch_get_resource(mevt->rid); if (!d) { - list_for_each_entry(d, &r->mon_domains, hdr.list) + list_for_each_entry_rcu(d, &r->mon_domains, hdr.list, lockdep_is_cpus_held()) rdtgroup_free_unassign_cntr(r, d, rdtgrp, mevt); } else { rdtgroup_free_unassign_cntr(r, d, rdtgrp, mevt); @@ -1367,7 +1429,7 @@ static void rdtgroup_update_cntr_event(struct rdt_resource *r, struct rdtgroup * struct rdt_l3_mon_domain *d; int cntr_id; - list_for_each_entry(d, &r->mon_domains, hdr.list) { + list_for_each_entry_rcu(d, &r->mon_domains, hdr.list, lockdep_is_cpus_held()) { cntr_id = mbm_cntr_get(r, d, rdtgrp, evtid); if (cntr_id >= 0) rdtgroup_assign_cntr(r, d, evtid, rdtgrp->mon.rmid, @@ -1405,16 +1467,19 @@ ssize_t event_filter_write(struct kernfs_open_file *of, char *buf, size_t nbytes u32 evt_cfg = 0; int ret = 0; - /* Valid input requires a trailing newline */ - if (nbytes == 0 || buf[nbytes - 1] != '\n') - return -EINVAL; + if (!info_kn_lock(of->kn)) + return -ENOENT; - buf[nbytes - 1] = '\0'; + rdt_last_cmd_clear(); - cpus_read_lock(); - mutex_lock(&rdtgroup_mutex); + /* Valid input requires a trailing newline */ + if (nbytes == 0 || buf[nbytes - 1] != '\n') { + rdt_last_cmd_puts("event_filter: Invalid input\n"); + ret = -EINVAL; + goto out_unlock; + } - rdt_last_cmd_clear(); + buf[nbytes - 1] = '\0'; r = resctrl_arch_get_resource(mevt->rid); if (!resctrl_arch_mbm_cntr_assign_enabled(r)) { @@ -1422,6 +1487,11 @@ ssize_t event_filter_write(struct kernfs_open_file *of, char *buf, size_t nbytes ret = -EINVAL; goto out_unlock; } + if (!r->mon.mbm_cntr_configurable) { + rdt_last_cmd_puts("event_filter is not configurable\n"); + ret = -EPERM; + goto out_unlock; + } ret = resctrl_parse_mem_transactions(buf, &evt_cfg); if (!ret && mevt->evt_cfg != evt_cfg) { @@ -1430,8 +1500,7 @@ ssize_t event_filter_write(struct kernfs_open_file *of, char *buf, size_t nbytes } out_unlock: - mutex_unlock(&rdtgroup_mutex); - cpus_read_unlock(); + info_kn_unlock(of->kn); return ret ?: nbytes; } @@ -1442,7 +1511,8 @@ int resctrl_mbm_assign_mode_show(struct kernfs_open_file *of, struct rdt_resource *r = rdt_kn_parent_priv(of->kn); bool enabled; - mutex_lock(&rdtgroup_mutex); + if (!info_kn_lock(of->kn)) + return -ENOENT; enabled = resctrl_arch_mbm_cntr_assign_enabled(r); if (r->mon.mbm_cntr_assignable) { @@ -1451,7 +1521,7 @@ int resctrl_mbm_assign_mode_show(struct kernfs_open_file *of, else seq_puts(s, "[default]\n"); - if (!IS_ENABLED(CONFIG_RESCTRL_ASSIGN_FIXED)) { + if (!r->mon.mbm_cntr_assign_fixed) { if (enabled) seq_puts(s, "default\n"); else @@ -1461,7 +1531,7 @@ int resctrl_mbm_assign_mode_show(struct kernfs_open_file *of, seq_puts(s, "[default]\n"); } - mutex_unlock(&rdtgroup_mutex); + info_kn_unlock(of->kn); return 0; } @@ -1474,16 +1544,19 @@ ssize_t resctrl_mbm_assign_mode_write(struct kernfs_open_file *of, char *buf, int ret = 0; bool enable; - /* Valid input requires a trailing newline */ - if (nbytes == 0 || buf[nbytes - 1] != '\n') - return -EINVAL; + if (!info_kn_lock(of->kn)) + return -ENOENT; - buf[nbytes - 1] = '\0'; + rdt_last_cmd_clear(); - cpus_read_lock(); - mutex_lock(&rdtgroup_mutex); + /* Valid input requires a trailing newline */ + if (nbytes == 0 || buf[nbytes - 1] != '\n') { + rdt_last_cmd_puts("mbm_assign_mode: Invalid input\n"); + ret = -EINVAL; + goto out_unlock; + } - rdt_last_cmd_clear(); + buf[nbytes - 1] = '\0'; if (!strcmp(buf, "default")) { enable = 0; @@ -1502,6 +1575,12 @@ ssize_t resctrl_mbm_assign_mode_write(struct kernfs_open_file *of, char *buf, } if (enable != resctrl_arch_mbm_cntr_assign_enabled(r)) { + if (r->mon.mbm_cntr_assign_fixed) { + ret = -EINVAL; + rdt_last_cmd_puts("Counter assignment mode is not configurable\n"); + goto out_unlock; + } + ret = resctrl_arch_mbm_cntr_assign_set(r, enable); if (ret) goto out_unlock; @@ -1526,15 +1605,14 @@ ssize_t resctrl_mbm_assign_mode_write(struct kernfs_open_file *of, char *buf, /* * Reset all the non-achitectural RMID state and assignable counters. */ - list_for_each_entry(d, &r->mon_domains, hdr.list) { + list_for_each_entry_rcu(d, &r->mon_domains, hdr.list, lockdep_is_cpus_held()) { mbm_cntr_free_all(r, d); resctrl_reset_rmid_all(r, d); } } out_unlock: - mutex_unlock(&rdtgroup_mutex); - cpus_read_unlock(); + info_kn_unlock(of->kn); return ret ?: nbytes; } @@ -1546,10 +1624,10 @@ int resctrl_num_mbm_cntrs_show(struct kernfs_open_file *of, struct rdt_l3_mon_domain *dom; bool sep = false; - cpus_read_lock(); - mutex_lock(&rdtgroup_mutex); + if (!info_kn_lock(of->kn)) + return -ENOENT; - list_for_each_entry(dom, &r->mon_domains, hdr.list) { + list_for_each_entry_rcu(dom, &r->mon_domains, hdr.list, lockdep_is_cpus_held()) { if (sep) seq_putc(s, ';'); @@ -1558,8 +1636,7 @@ int resctrl_num_mbm_cntrs_show(struct kernfs_open_file *of, } seq_putc(s, '\n'); - mutex_unlock(&rdtgroup_mutex); - cpus_read_unlock(); + info_kn_unlock(of->kn); return 0; } @@ -1572,8 +1649,8 @@ int resctrl_available_mbm_cntrs_show(struct kernfs_open_file *of, u32 cntrs, i; int ret = 0; - cpus_read_lock(); - mutex_lock(&rdtgroup_mutex); + if (!info_kn_lock(of->kn)) + return -ENOENT; rdt_last_cmd_clear(); @@ -1583,7 +1660,7 @@ int resctrl_available_mbm_cntrs_show(struct kernfs_open_file *of, goto out_unlock; } - list_for_each_entry(dom, &r->mon_domains, hdr.list) { + list_for_each_entry_rcu(dom, &r->mon_domains, hdr.list, lockdep_is_cpus_held()) { if (sep) seq_putc(s, ';'); @@ -1599,8 +1676,7 @@ int resctrl_available_mbm_cntrs_show(struct kernfs_open_file *of, seq_putc(s, '\n'); out_unlock: - mutex_unlock(&rdtgroup_mutex); - cpus_read_unlock(); + info_kn_unlock(of->kn); return ret; } @@ -1620,7 +1696,6 @@ int mbm_L3_assignments_show(struct kernfs_open_file *of, struct seq_file *s, voi goto out_unlock; } - rdt_last_cmd_clear(); if (!resctrl_arch_mbm_cntr_assign_enabled(r)) { rdt_last_cmd_puts("mbm_event counter assignment mode is not enabled\n"); ret = -EINVAL; @@ -1633,7 +1708,7 @@ int mbm_L3_assignments_show(struct kernfs_open_file *of, struct seq_file *s, voi sep = false; seq_printf(s, "%s:", mevt->name); - list_for_each_entry(d, &r->mon_domains, hdr.list) { + list_for_each_entry_rcu(d, &r->mon_domains, hdr.list, lockdep_is_cpus_held()) { if (sep) seq_putc(s, ';'); @@ -1731,7 +1806,7 @@ next: } /* Verify if the dom_id is valid */ - list_for_each_entry(d, &r->mon_domains, hdr.list) { + list_for_each_entry_rcu(d, &r->mon_domains, hdr.list, lockdep_is_cpus_held()) { if (d->hdr.id == dom_id) { ret = rdtgroup_modify_assign_state(dom_str, d, rdtgrp, mevt); if (ret) { @@ -1755,23 +1830,25 @@ ssize_t mbm_L3_assignments_write(struct kernfs_open_file *of, char *buf, char *token, *event; int ret = 0; - /* Valid input requires a trailing newline */ - if (nbytes == 0 || buf[nbytes - 1] != '\n') - return -EINVAL; - - buf[nbytes - 1] = '\0'; - rdtgrp = rdtgroup_kn_lock_live(of->kn); if (!rdtgrp) { rdtgroup_kn_unlock(of->kn); return -ENOENT; } - rdt_last_cmd_clear(); + + /* Valid input requires a trailing newline */ + if (nbytes == 0 || buf[nbytes - 1] != '\n') { + rdt_last_cmd_puts("mbm_L3_assignments: Invalid input\n"); + ret = -EINVAL; + goto out_unlock; + } + + buf[nbytes - 1] = '\0'; if (!resctrl_arch_mbm_cntr_assign_enabled(r)) { rdt_last_cmd_puts("mbm_event mode is not enabled\n"); - rdtgroup_kn_unlock(of->kn); - return -EINVAL; + ret = -EINVAL; + goto out_unlock; } while ((token = strsep(&buf, "\n")) != NULL) { @@ -1787,6 +1864,7 @@ ssize_t mbm_L3_assignments_write(struct kernfs_open_file *of, char *buf, break; } +out_unlock: rdtgroup_kn_unlock(of->kn); return ret ?: nbytes; @@ -1886,6 +1964,8 @@ int resctrl_l3_mon_resource_init(void) resctrl_file_fflags_init("available_mbm_cntrs", RFTYPE_MON_INFO | RFTYPE_RES_CACHE); resctrl_file_fflags_init("event_filter", RFTYPE_ASSIGN_CONFIG); + if (r->mon.mbm_cntr_configurable) + resctrl_file_mode_init("event_filter", 0644); resctrl_file_fflags_init("mbm_assign_on_mkdir", RFTYPE_MON_INFO | RFTYPE_RES_CACHE); resctrl_file_fflags_init("mbm_L3_assignments", RFTYPE_MON_BASE); |
