[External] [PATCH v8 2/5] riscv_cbqri: resctrl: Add cache allocation via capacity block mask
yunhui cui
cuiyunhui at bytedance.com
Wed Sep 23 02:13:15 PDT 2026
Hi Fustini,
On Fri, Sep 18, 2026 at 12:41 AM Drew Fustini <fustini at kernel.org> wrote:
>
> Wire CBQRI capacity controllers into resctrl as RDT_RESOURCE_L2 and
> RDT_RESOURCE_L3 schemata.
>
> resctrl exposes a single capacity mask per resource, so when more than
> one capacity controller backs a cache level they must report the same
> number of capacity blocks. The same holds for the other capabilities
> resctrl exposes once per resource. A level whose controllers disagree
> is left unexposed. Other levels are unaffected.
>
> Control domains are created in the cpuhp online callback and keyed on
> the onlining CPU's cache id from cacheinfo, so a CPU that is offline at
> probe time still joins its domain once it comes online. A domain is
> published on the ctrl_domains RCU list only after it is fully
> initialized and brought online, and removed from the list before it is
> taken offline, following the documented safe traversal rules.
>
> Assisted-by: LLM
> Tested-by: Chen Pei <cp0613 at linux.alibaba.com>
> Co-developed-by: Adrien Ricciardi <aricciardi at baylibre.com>
> Signed-off-by: Adrien Ricciardi <aricciardi at baylibre.com>
> Signed-off-by: Drew Fustini <fustini at kernel.org>
> ---
> MAINTAINERS | 2 +
> arch/riscv/include/asm/resctrl.h | 147 ++++++++
> drivers/resctrl/Kconfig | 4 +
> drivers/resctrl/Makefile | 1 +
> drivers/resctrl/cbqri_resctrl.c | 785 +++++++++++++++++++++++++++++++++++++++
> 5 files changed, 939 insertions(+)
>
> diff --git a/MAINTAINERS b/MAINTAINERS
> index 92b111cd25c0..db0799d8cbf8 100644
> --- a/MAINTAINERS
> +++ b/MAINTAINERS
> @@ -23570,9 +23570,11 @@ R: yunhui cui <cuiyunhui at bytedance.com>
> L: linux-riscv at lists.infradead.org
> S: Supported
> F: arch/riscv/include/asm/qos.h
> +F: arch/riscv/include/asm/resctrl.h
> F: arch/riscv/kernel/qos.c
> F: drivers/resctrl/cbqri_devices.c
> F: drivers/resctrl/cbqri_internal.h
> +F: drivers/resctrl/cbqri_resctrl.c
> F: include/linux/riscv_cbqri.h
>
> RISC-V RPMI AND MPXY DRIVERS
> diff --git a/arch/riscv/include/asm/resctrl.h b/arch/riscv/include/asm/resctrl.h
> new file mode 100644
> index 000000000000..b08f4e12f7aa
> --- /dev/null
> +++ b/arch/riscv/include/asm/resctrl.h
> @@ -0,0 +1,147 @@
> +/* SPDX-License-Identifier: GPL-2.0-only */
> +#ifndef _ASM_RISCV_RESCTRL_H
> +#define _ASM_RISCV_RESCTRL_H
> +
> +#include <linux/resctrl_types.h>
> +#include <linux/sched.h>
> +#include <linux/types.h>
> +
> +#include <asm/qos.h>
> +
> +struct rdt_resource;
> +
> +/*
> + * Sentinel "no CLOSID assigned" used by resctrl_arch_rmid_idx_decode().
> + * fs/resctrl treats this opaquely. CBQRI uses MCID directly as the linear
> + * rmid index, so closid is unused on decode.
> + */
> +#define RISCV_RESCTRL_EMPTY_CLOSID ((u32)~0)
> +
> +/*
> + * Terminology mapping between x86 (Intel RDT/AMD QoS) and RISC-V:
> + *
> + * CLOSID on x86 is RCID on RISC-V
> + * RMID on x86 is MCID on RISC-V
> + * CDP on x86 is AT (access type) on RISC-V
> + */
> +
> +/**
> + * resctrl_arch_alloc_capable() - any CBQRI controller exposes resctrl alloc
> + *
> + * Returns true once at least one CBQRI controller has successfully probed for
> + * a resctrl-exposed cache capacity allocation feature. Only meaningful after
> + * cbqri_resctrl_setup() runs at late_initcall.
> + */
> +bool resctrl_arch_alloc_capable(void);
> +
> +/**
> + * resctrl_arch_mon_capable() - any CBQRI controller exposes resctrl monitoring
> + *
> + * The CBQRI driver implements capacity allocation only and wires up no
> + * monitoring events, so this always returns false. fs/resctrl references it
> + * unconditionally, hence the stub.
> + */
> +bool resctrl_arch_mon_capable(void);
> +
> +/**
> + * resctrl_arch_rmid_idx_encode() - encode (RCID, MCID) into a linear index
> + * @closid: RCID (resource control id)
> + * @rmid: MCID (monitoring counter id)
> + *
> + * RISC-V uses MCID directly as the linear index into per-RMID arrays
> + * managed by fs/resctrl, since CBQRI controllers admit any MCID for any
> + * RCID. closid is unused here. CDP is encoded via the AT field on each
> + * CBQRI op rather than via the index.
> + */
> +u32 resctrl_arch_rmid_idx_encode(u32 closid, u32 rmid);
> +
> +/**
> + * resctrl_arch_rmid_idx_decode() - inverse of resctrl_arch_rmid_idx_encode()
> + * @idx: linear index
> + * @closid: out: always RISCV_RESCTRL_EMPTY_CLOSID
> + * @rmid: out: the MCID that @idx encodes
> + */
> +void resctrl_arch_rmid_idx_decode(u32 idx, u32 *closid, u32 *rmid);
> +
> +/**
> + * resctrl_arch_set_cpu_default_closid_rmid() - install per-CPU srmcfg default
> + * @cpu: CPU number
> + * @closid: RCID to use when no task is matched
> + * @rmid: MCID to use when no task is matched
> + *
> + * Sets the per-CPU cpu_srmcfg_default so __switch_to_srmcfg() can fall back
> + * to the CPU's default RCID/MCID for default-group tasks (those whose
> + * thread.srmcfg encodes to 0, i.e. closid == RESCTRL_RESERVED_CLOSID and
> + * rmid == RESCTRL_RESERVED_RMID). Implements resctrl allocation rule 2
> + * ("CPU default") on RISC-V.
> + */
> +void resctrl_arch_set_cpu_default_closid_rmid(int cpu, u32 closid, u32 rmid);
> +
> +/**
> + * resctrl_arch_sched_in() - context-switch hook to install task RCID/MCID
> + * @tsk: the task being scheduled in
> + *
> + * Called from finish_task_switch() to write tsk->thread.srmcfg into the
> + * srmcfg CSR. Tasks tagged with RISCV_RESCTRL_EMPTY_CLOSID inherit the
> + * per-CPU default set via resctrl_arch_set_cpu_default_closid_rmid().
> + */
> +void resctrl_arch_sched_in(struct task_struct *tsk);
> +
> +/**
> + * resctrl_arch_set_closid_rmid() - tag a task with an RCID/MCID
> + * @tsk: task to tag
> + * @closid: RCID to install
> + * @rmid: MCID to install
> + *
> + * Updates tsk->thread.srmcfg with the encoded (RCID, MCID) pair. The new
> + * value takes effect on the next resctrl_arch_sched_in() for this task.
> + */
> +void resctrl_arch_set_closid_rmid(struct task_struct *tsk, u32 closid, u32 rmid);
> +
> +/**
> + * resctrl_arch_match_closid() - test whether a task carries a given RCID
> + * @tsk: task
> + * @closid: RCID
> + */
> +bool resctrl_arch_match_closid(struct task_struct *tsk, u32 closid);
> +
> +/**
> + * resctrl_arch_match_rmid() - test whether a task carries a given (RCID, MCID)
> + * @tsk: task
> + * @closid: RCID
> + * @rmid: MCID
> + */
> +bool resctrl_arch_match_rmid(struct task_struct *tsk, u32 closid, u32 rmid);
> +
> +/**
> + * resctrl_arch_mon_ctx_alloc() - allocate per-monitor-event arch context
> + * @r: resctrl resource being monitored
> + * @evtid: which monitor event needs context
> + *
> + * The CBQRI driver implements no monitoring events, so there is no per-event
> + * context to allocate and the stub returns NULL. fs/resctrl references it
> + * unconditionally before checking resctrl_arch_mon_capable().
> + */
> +void *resctrl_arch_mon_ctx_alloc(struct rdt_resource *r, enum resctrl_event_id evtid);
> +
> +/**
> + * resctrl_arch_mon_ctx_free() - release context returned by mon_ctx_alloc()
> + * @r: resctrl resource
> + * @evtid: monitor event id
> + * @arch_mon_ctx: pointer returned by resctrl_arch_mon_ctx_alloc()
> + */
> +void resctrl_arch_mon_ctx_free(struct rdt_resource *r, enum resctrl_event_id evtid,
> + void *arch_mon_ctx);
> +
> +static inline unsigned int resctrl_arch_round_mon_val(unsigned int val)
> +{
> + return val;
> +}
> +
> +/* Not needed for RISC-V */
> +static inline void resctrl_arch_enable_mon(void) { }
> +static inline void resctrl_arch_disable_mon(void) { }
> +static inline void resctrl_arch_enable_alloc(void) { }
> +static inline void resctrl_arch_disable_alloc(void) { }
> +
> +#endif /* _ASM_RISCV_RESCTRL_H */
> diff --git a/drivers/resctrl/Kconfig b/drivers/resctrl/Kconfig
> index 1219fb29aa2e..cac23d20124b 100644
> --- a/drivers/resctrl/Kconfig
> +++ b/drivers/resctrl/Kconfig
> @@ -38,3 +38,7 @@ menuconfig RISCV_CBQRI
> RISC-V cache QoS resources. CBQRI exposes cache capacity
> allocation through the resctrl filesystem at /sys/fs/resctrl when
> RESCTRL_FS is also enabled.
> +
> +config RISCV_CBQRI_RESCTRL_FS
> + bool
> + default y if RISCV_CBQRI && RESCTRL_FS
> diff --git a/drivers/resctrl/Makefile b/drivers/resctrl/Makefile
> index 4d8a2c4b5627..a7631712dba9 100644
> --- a/drivers/resctrl/Makefile
> +++ b/drivers/resctrl/Makefile
> @@ -6,3 +6,4 @@ ccflags-$(CONFIG_ARM64_MPAM_DRIVER_DEBUG) += -DDEBUG
>
> obj-$(CONFIG_RISCV_CBQRI) += cbqri.o
> cbqri-y += cbqri_devices.o
> +cbqri-$(CONFIG_RISCV_CBQRI_RESCTRL_FS) += cbqri_resctrl.o
> diff --git a/drivers/resctrl/cbqri_resctrl.c b/drivers/resctrl/cbqri_resctrl.c
> new file mode 100644
> index 000000000000..0c4bfa2a7f43
> --- /dev/null
> +++ b/drivers/resctrl/cbqri_resctrl.c
> @@ -0,0 +1,785 @@
> +// SPDX-License-Identifier: GPL-2.0-only
> +
> +#define pr_fmt(fmt) "%s:%s: " fmt, KBUILD_MODNAME, __func__
> +
> +#include <linux/bitfield.h>
> +#include <linux/cacheinfo.h>
> +#include <linux/cleanup.h>
> +#include <linux/riscv_cbqri.h>
> +#include <linux/cpu.h>
> +#include <linux/cpufeature.h>
> +#include <linux/cpuhotplug.h>
> +#include <linux/err.h>
> +#include <linux/init.h>
> +#include <linux/rculist.h>
> +#include <linux/resctrl.h>
> +#include <linux/slab.h>
> +#include <linux/types.h>
> +
> +#include <asm/csr.h>
> +#include <asm/qos.h>
> +
> +#include "cbqri_internal.h"
> +
> +struct cbqri_resctrl_res {
> + struct cbqri_controller *ctrl;
> + struct rdt_resource resctrl_res;
> + bool cdp_enabled;
> +};
> +
> +struct cbqri_resctrl_dom {
> + struct rdt_ctrl_domain resctrl_ctrl_dom;
> + struct cbqri_controller *hw_ctrl;
> +};
> +
> +static struct cbqri_resctrl_res cbqri_resctrl_resources[RDT_NUM_RESOURCES];
> +
> +static bool exposed_alloc_capable;
> +
> +/* Protects ctrl_domain list mutations across CPU hotplug. */
> +static DEFINE_MUTEX(cbqri_domain_list_lock);
> +
> +static struct rdt_ctrl_domain *
> +cbqri_find_ctrl_domain(struct list_head *h, int id)
> +{
> + struct rdt_domain_hdr *hdr = resctrl_find_domain(h, id, NULL);
> +
> + return hdr ? container_of(hdr, struct rdt_ctrl_domain, hdr) : NULL;
> +}
> +
> +/* Map a hardware cache level to its resctrl resource id, or -ENODEV. */
> +static int cbqri_cache_level_to_rid(u32 cache_level)
> +{
> + switch (cache_level) {
> + case 2:
> + return RDT_RESOURCE_L2;
> + case 3:
> + return RDT_RESOURCE_L3;
> + default:
> + return -ENODEV;
> + }
> +}
> +
> +static int cbqri_apply_cache_config_dom(struct cbqri_resctrl_dom *hw_dom,
> + struct rdt_resource *r,
> + u32 closid, enum resctrl_conf_type t,
> + u64 cbm)
> +{
> + struct cbqri_resctrl_res *hw_res =
> + container_of(r, struct cbqri_resctrl_res, resctrl_res);
> + struct cbqri_cc_config cfg = {
> + .cbm = cbm,
> + .at = (t == CDP_CODE) ? CBQRI_CONTROL_REGISTERS_AT_CODE :
> + CBQRI_CONTROL_REGISTERS_AT_DATA,
> + .cdp_enabled = hw_res->cdp_enabled,
> + };
> +
> + return cbqri_apply_cache_config(hw_dom->hw_ctrl, closid, &cfg);
> +}
> +
> +bool resctrl_arch_alloc_capable(void)
> +{
> + return exposed_alloc_capable;
> +}
> +
> +bool resctrl_arch_mon_capable(void)
> +{
> + return false;
> +}
> +
> +bool resctrl_arch_get_cdp_enabled(enum resctrl_res_level rid)
> +{
> + if (rid != RDT_RESOURCE_L2 && rid != RDT_RESOURCE_L3)
> + return false;
> + return cbqri_resctrl_resources[rid].cdp_enabled;
> +}
> +
> +int resctrl_arch_set_cdp_enabled(enum resctrl_res_level rid, bool enable)
> +{
> + struct cbqri_resctrl_res *cbqri_res;
> +
> + if (rid != RDT_RESOURCE_L2 && rid != RDT_RESOURCE_L3)
> + return -ENODEV;
> +
> + cbqri_res = &cbqri_resctrl_resources[rid];
> + if (!cbqri_res->resctrl_res.cdp_capable)
> + return -ENODEV;
> +
> + cbqri_res->cdp_enabled = enable;
> + return 0;
> +}
> +
> +struct rdt_resource *resctrl_arch_get_resource(enum resctrl_res_level l)
> +{
> + if (l >= RDT_NUM_RESOURCES)
> + return NULL;
> +
> + return &cbqri_resctrl_resources[l].resctrl_res;
> +}
> +
> +bool resctrl_arch_is_evt_configurable(enum resctrl_event_id evt)
> +{
> + return false;
> +}
> +
> +void *resctrl_arch_mon_ctx_alloc(struct rdt_resource *r,
> + enum resctrl_event_id evtid)
> +{
> + return NULL;
> +}
> +
> +void resctrl_arch_mon_ctx_free(struct rdt_resource *r,
> + enum resctrl_event_id evtid, void *arch_mon_ctx)
> +{
> +}
> +
> +void resctrl_arch_config_cntr(struct rdt_resource *r, struct rdt_l3_mon_domain *d,
> + enum resctrl_event_id evtid, u32 rmid, u32 closid,
> + u32 cntr_id, bool assign)
> +{
> +}
> +
> +int resctrl_arch_cntr_read(struct rdt_resource *r, struct rdt_l3_mon_domain *d,
> + u32 unused, u32 rmid, int cntr_id,
> + enum resctrl_event_id eventid, u64 *val)
> +{
> + return -EOPNOTSUPP;
> +}
> +
> +bool resctrl_arch_mbm_cntr_assign_enabled(struct rdt_resource *r)
> +{
> + return false;
> +}
> +
> +int resctrl_arch_mbm_cntr_assign_set(struct rdt_resource *r, bool enable)
> +{
> + return -EOPNOTSUPP;
> +}
> +
> +void resctrl_arch_reset_cntr(struct rdt_resource *r, struct rdt_l3_mon_domain *d,
> + u32 unused, u32 rmid, int cntr_id,
> + enum resctrl_event_id eventid)
> +{
> +}
> +
> +bool resctrl_arch_get_io_alloc_enabled(struct rdt_resource *r)
> +{
> + return false;
> +}
> +
> +int resctrl_arch_io_alloc_enable(struct rdt_resource *r, bool enable)
> +{
> + return -EOPNOTSUPP;
> +}
> +
> +void resctrl_arch_mon_event_config_read(void *info)
> +{
> +}
> +
> +void resctrl_arch_mon_event_config_write(void *info)
> +{
> +}
> +
> +void resctrl_arch_reset_rmid_all(struct rdt_resource *r, struct rdt_l3_mon_domain *d)
> +{
> +}
> +
> +void resctrl_arch_reset_rmid(struct rdt_resource *r, struct rdt_l3_mon_domain *d,
> + u32 unused, u32 rmid, enum resctrl_event_id eventid)
> +{
> +}
> +
> +int resctrl_arch_rmid_read(struct rdt_resource *r, struct rdt_domain_hdr *hdr,
> + u32 closid, u32 rmid, enum resctrl_event_id eventid,
> + void *arch_priv, u64 *val, void *arch_mon_ctx)
> +{
> + return -ENODATA;
> +}
> +
> +/*
> + * Note about terminology between x86 (Intel RDT/AMD QoS) and RISC-V:
> + * CLOSID on x86 is RCID on RISC-V
> + * RMID on x86 is MCID on RISC-V
> + */
> +u32 resctrl_arch_get_num_closid(struct rdt_resource *res)
> +{
> + struct cbqri_resctrl_res *hw_res;
> +
> + hw_res = container_of(res, struct cbqri_resctrl_res, resctrl_res);
> +
> + if (!hw_res->ctrl)
> + return 0;
> +
> + return hw_res->ctrl->rcid_count;
> +}
When CDP is enabled, the resctrl core halves the value returned by
resctrl_arch_get_num_closid(). Is that correct for CBQRI?
CBQRI v1.0 defines allocation using an (RCID, AT) pair, and its example
uses the same RCID=5 for separate AT=0 and AT=1 allocations.
Thanks,
Yunhui
> +
> +u32 resctrl_arch_system_num_rmid_idx(void)
> +{
> + return 1;
> +}
> +
> +u32 resctrl_arch_rmid_idx_encode(u32 closid, u32 rmid)
> +{
> + return rmid;
> +}
> +
> +void resctrl_arch_rmid_idx_decode(u32 idx, u32 *closid, u32 *rmid)
> +{
> + *closid = RISCV_RESCTRL_EMPTY_CLOSID;
> + *rmid = idx;
> +}
> +
> +void resctrl_arch_set_cpu_default_closid_rmid(int cpu, u32 closid, u32 rmid)
> +{
> + u32 srmcfg = FIELD_PREP(SRMCFG_RCID_MASK, closid) |
> + FIELD_PREP(SRMCFG_MCID_MASK, rmid);
> +
> + WRITE_ONCE(per_cpu(cpu_srmcfg_default, cpu), srmcfg);
> +}
> +
> +void resctrl_arch_sched_in(struct task_struct *tsk)
> +{
> + __switch_to_srmcfg(tsk);
> +}
> +
> +void resctrl_arch_set_closid_rmid(struct task_struct *tsk, u32 closid, u32 rmid)
> +{
> + u32 srmcfg = FIELD_PREP(SRMCFG_RCID_MASK, closid) |
> + FIELD_PREP(SRMCFG_MCID_MASK, rmid);
> +
> + WRITE_ONCE(tsk->thread.srmcfg, srmcfg);
> +}
> +
> +void resctrl_arch_sync_cpu_closid_rmid(void *info)
> +{
> + struct resctrl_cpu_defaults *r = info;
> +
> + lockdep_assert_preemption_disabled();
> +
> + if (r) {
> + resctrl_arch_set_cpu_default_closid_rmid(smp_processor_id(),
> + r->closid, r->rmid);
> + }
> +
> + resctrl_arch_sched_in(current);
> +}
> +
> +bool resctrl_arch_match_closid(struct task_struct *tsk, u32 closid)
> +{
> + return FIELD_GET(SRMCFG_RCID_MASK, READ_ONCE(tsk->thread.srmcfg)) == closid;
> +}
> +
> +bool resctrl_arch_match_rmid(struct task_struct *tsk, u32 closid, u32 rmid)
> +{
> + return FIELD_GET(SRMCFG_MCID_MASK, READ_ONCE(tsk->thread.srmcfg)) == rmid;
> +}
> +
> +void resctrl_arch_pre_mount(void)
> +{
> + /* All controllers discovered at boot via late_initcall. Nothing to do. */
> +}
> +
> +int resctrl_arch_update_one(struct rdt_resource *r, struct rdt_ctrl_domain *d,
> + u32 closid, enum resctrl_conf_type t, u32 cfg_val)
> +{
> + struct cbqri_resctrl_dom *dom;
> +
> + dom = container_of(d, struct cbqri_resctrl_dom, resctrl_ctrl_dom);
> +
> + if (!r->alloc_capable)
> + return -EINVAL;
> +
> + switch (r->rid) {
> + case RDT_RESOURCE_L2:
> + case RDT_RESOURCE_L3:
> + return cbqri_apply_cache_config_dom(dom, r, closid, t, cfg_val);
> + default:
> + return -EINVAL;
> + }
> +}
> +
> +int resctrl_arch_update_domains(struct rdt_resource *r, u32 closid)
> +{
> + struct resctrl_staged_config *cfg;
> + enum resctrl_conf_type t;
> + struct rdt_ctrl_domain *d;
> + int err = 0;
> +
> + /* Walking r->ctrl_domains, ensure it can't race with cpuhp */
> + lockdep_assert_cpus_held();
> +
> + list_for_each_entry(d, &r->ctrl_domains, hdr.list) {
> + for (t = 0; t < CDP_NUM_TYPES; t++) {
> + cfg = &d->staged_config[t];
> + if (!cfg->have_new_ctrl)
> + continue;
> + err = resctrl_arch_update_one(r, d, closid, t, cfg->new_ctrl);
> + if (err)
> + return err;
> + }
> + }
> + return err;
> +}
> +
> +u32 resctrl_arch_get_config(struct rdt_resource *r, struct rdt_ctrl_domain *d,
> + u32 closid, enum resctrl_conf_type type)
> +{
> + struct cbqri_resctrl_dom *hw_dom;
> + struct cbqri_controller *ctrl;
> + u32 at;
> + u32 val;
> + int err;
> +
> + hw_dom = container_of(d, struct cbqri_resctrl_dom, resctrl_ctrl_dom);
> + ctrl = hw_dom->hw_ctrl;
> + val = resctrl_get_default_ctrl(r);
> +
> + if (!r->alloc_capable)
> + return val;
> +
> + switch (r->rid) {
> + case RDT_RESOURCE_L2:
> + case RDT_RESOURCE_L3:
> + at = (type == CDP_CODE) ? CBQRI_CONTROL_REGISTERS_AT_CODE :
> + CBQRI_CONTROL_REGISTERS_AT_DATA;
> + err = cbqri_read_cache_config(ctrl, closid, at, &val);
> + if (err < 0)
> + val = resctrl_get_default_ctrl(r);
> + break;
> + default:
> + break;
> + }
> +
> + return val;
> +}
> +
> +void resctrl_arch_reset_all_ctrls(struct rdt_resource *r)
> +{
> + struct cbqri_resctrl_res *hw_res;
> + struct rdt_ctrl_domain *d;
> + enum resctrl_conf_type t;
> + u32 default_ctrl;
> + int i;
> +
> + lockdep_assert_cpus_held();
> +
> + hw_res = container_of(r, struct cbqri_resctrl_res, resctrl_res);
> + default_ctrl = resctrl_get_default_ctrl(r);
> +
> + if (!hw_res->ctrl)
> + return;
> +
> + list_for_each_entry(d, &r->ctrl_domains, hdr.list) {
> + for (i = 0; i < hw_res->ctrl->rcid_count; i++) {
> + for (t = 0; t < CDP_NUM_TYPES; t++) {
> + int rerr;
> +
> + rerr = resctrl_arch_update_one(r, d, i, t, default_ctrl);
> + if (rerr)
> + pr_err_ratelimited("rid=%d reset RCID %u type %u failed (%d)\n",
> + r->rid, i, t, rerr);
> + }
> + }
> + }
> +}
> +
> +static struct rdt_ctrl_domain *cbqri_new_domain(struct cbqri_controller *ctrl)
> +{
> + struct cbqri_resctrl_dom *hw_dom;
> + struct rdt_ctrl_domain *domain;
> +
> + hw_dom = kzalloc_obj(*hw_dom, GFP_KERNEL);
> + if (!hw_dom)
> + return NULL;
> +
> + hw_dom->hw_ctrl = ctrl;
> + domain = &hw_dom->resctrl_ctrl_dom;
> +
> + INIT_LIST_HEAD(&domain->hdr.list);
> +
> + return domain;
> +}
> +
> +static int cbqri_init_domain_ctrlval(struct rdt_resource *r, struct rdt_ctrl_domain *d)
> +{
> + struct cbqri_resctrl_dom *hw_dom;
> + enum resctrl_conf_type t;
> + int err = 0;
> + int i;
> +
> + hw_dom = container_of(d, struct cbqri_resctrl_dom, resctrl_ctrl_dom);
> +
> + for (i = 0; i < hw_dom->hw_ctrl->rcid_count; i++) {
> + /*
> + * Seed both DATA and CODE staged slots so a later mount
> + * with -o cdp does not see stale CODE values.
> + * On non-AT controllers cbqri_cc_alloc_op() masks AT to 0
> + * so all three iterations land on the same hardware state.
> + * The redundant writes are harmless.
> + */
> + for (t = 0; t < CDP_NUM_TYPES; t++) {
> + err = resctrl_arch_update_one(r, d, i, t,
> + resctrl_get_default_ctrl(r));
> + if (err)
> + return err;
> + }
> + }
> + return 0;
> +}
> +
> +/*
> + * Two capacity controllers at the same cache level are interchangeable to
> + * resctrl only if they expose identical caps, since resctrl publishes one
> + * set of caps per rid but any of the level's controllers may service a
> + * given RCID.
> + */
> +static bool cbqri_cc_caps_agree(const struct cbqri_controller *a,
> + const struct cbqri_controller *b)
> +{
> + return a->rcid_count == b->rcid_count &&
> + a->cc.ncblks == b->cc.ncblks &&
> + a->cc.supports_alloc_at_code == b->cc.supports_alloc_at_code &&
> + a->alloc_capable == b->alloc_capable;
> +}
> +
> +/*
> + * Walk cbqri_controllers and pick one capacity controller (CC) per cache
> + * level (L2/L3) to back the corresponding RDT_RESOURCE_L*. When more than
> + * one CC sits at the same level (e.g. one per socket), they must agree on
> + * rcid_count / ncblks / supports_alloc_at_code / alloc_capable. A level
> + * whose controllers disagree is dropped, since resctrl exposes a single
> + * set of caps per rid, but the other level is still picked. The first
> + * matching controller wins.
> + */
> +static void cbqri_resctrl_pick_caches(void)
> +{
> + struct cbqri_controller *ctrl, *pick;
> + int rid;
> +
> + guard(mutex)(&cbqri_controllers_lock);
> +
> + for (rid = 0; rid < RDT_NUM_RESOURCES; rid++) {
> + pick = NULL;
> +
> + list_for_each_entry(ctrl, &cbqri_controllers, list) {
> + if (ctrl->type != CBQRI_CONTROLLER_TYPE_CAPACITY ||
> + !ctrl->alloc_capable ||
> + cbqri_cache_level_to_rid(ctrl->cache.cache_level) != rid)
> + continue;
> +
> + if (!pick) {
> + pick = ctrl;
> + } else if (!cbqri_cc_caps_agree(pick, ctrl)) {
> + pr_err("L%u controllers have mismatched capabilities, skipping this level\n",
> + ctrl->cache.cache_level);
> + pick = NULL;
> + break;
> + }
> + }
> +
> + cbqri_resctrl_resources[rid].ctrl = pick;
> + }
> +}
> +
> +/*
> + * Fill the rdt_resource fields for one picked rid. An rid with no picked
> + * controller is left untouched so it stays out of resctrl_arch_get_resource().
> + */
> +static void cbqri_resctrl_control_init(struct cbqri_resctrl_res *cbqri_res)
> +{
> + struct cbqri_controller *ctrl = cbqri_res->ctrl;
> + struct rdt_resource *res = &cbqri_res->resctrl_res;
> +
> + if (!ctrl)
> + return;
> +
> + switch (res->rid) {
> + case RDT_RESOURCE_L2:
> + case RDT_RESOURCE_L3:
> + res->name = (res->rid == RDT_RESOURCE_L2) ? "L2" : "L3";
> + res->schema_fmt = RESCTRL_SCHEMA_BITMAP;
> + res->ctrl_scope = (res->rid == RDT_RESOURCE_L2) ?
> + RESCTRL_L2_CACHE : RESCTRL_L3_CACHE;
> + res->cache.cbm_len = ctrl->cc.ncblks;
> + res->cache.shareable_bits = 0;
> + res->cache.min_cbm_bits = 1;
> + res->cache.arch_has_sparse_bitmasks = false;
> + res->cdp_capable = ctrl->cc.supports_alloc_at_code;
> + res->alloc_capable = ctrl->alloc_capable;
> + INIT_LIST_HEAD(&res->ctrl_domains);
> + INIT_LIST_HEAD(&res->mon_domains);
> + break;
> + default:
> + break;
> + }
> +}
> +
> +static void cbqri_resctrl_accumulate_caps(void)
> +{
> + int rid;
> +
> + for (rid = 0; rid < RDT_NUM_RESOURCES; rid++) {
> + struct cbqri_resctrl_res *hw_res = &cbqri_resctrl_resources[rid];
> +
> + if (!hw_res->ctrl)
> + continue;
> + if (hw_res->ctrl->alloc_capable)
> + exposed_alloc_capable = true;
> + }
> +}
> +
> +/*
> + * Create, list-insert, and online a fresh ctrl_domain backing ctrl on
> + * resource res, seeded with cpu and identified by dom_id. Caller must
> + * hold cbqri_domain_list_lock and must have already verified that no
> + * existing ctrl_domain on res carries this id.
> + */
> +static struct rdt_ctrl_domain *cbqri_create_ctrl_domain(struct cbqri_controller *ctrl,
> + struct rdt_resource *res,
> + unsigned int cpu, int dom_id)
> +{
> + struct rdt_ctrl_domain *domain;
> + struct list_head *pos = NULL;
> + int err;
> +
> + domain = cbqri_new_domain(ctrl);
> + if (!domain)
> + return ERR_PTR(-ENOMEM);
> +
> + cpumask_set_cpu(cpu, &domain->hdr.cpu_mask);
> + domain->hdr.id = dom_id;
> + domain->hdr.type = RESCTRL_CTRL_DOMAIN;
> + domain->hdr.rid = res->rid;
> +
> + err = cbqri_init_domain_ctrlval(res, domain);
> + if (err)
> + goto free;
> +
> + err = resctrl_online_ctrl_domain(res, domain);
> + if (err)
> + goto free;
> +
> + /*
> + * Publish only after the domain is fully initialized and online, so a
> + * reader walking the RCU list never sees a half-built domain.
> + */
> + resctrl_find_domain(&res->ctrl_domains, dom_id, &pos);
> + list_add_tail_rcu(&domain->hdr.list, pos);
> +
> + return domain;
> +free:
> + kfree(container_of(domain, struct cbqri_resctrl_dom, resctrl_ctrl_dom));
> + return ERR_PTR(err);
> +}
> +
> +/*
> + * Remove a CPU from the domain at each level whose cache it shares. The
> + * domain is looked up by the CPU's own cache id and the lookup returns NULL
> + * if it is already gone, so this is idempotent and undoes a partial online
> + * attach as well as a full offline. Caller holds cbqri_domain_list_lock.
> + */
> +static void cbqri_detach_cpu_from_all_ctrls(unsigned int cpu)
> +{
> + static const u32 levels[] = { 2, 3 };
> + struct cbqri_resctrl_res *hw_res;
> + struct rdt_ctrl_domain *d;
> + struct rdt_resource *res;
> + struct cacheinfo *ci;
> + int i, rid;
> +
> + lockdep_assert_held(&cbqri_domain_list_lock);
> +
> + for (i = 0; i < ARRAY_SIZE(levels); i++) {
> + ci = get_cpu_cacheinfo_level(cpu, levels[i]);
> + if (!ci)
> + continue;
> +
> + rid = cbqri_cache_level_to_rid(levels[i]);
> + hw_res = &cbqri_resctrl_resources[rid];
> + if (!hw_res->ctrl)
> + continue;
> +
> + res = &hw_res->resctrl_res;
> + d = cbqri_find_ctrl_domain(&res->ctrl_domains, ci->id);
> + if (!d)
> + continue;
> +
> + cpumask_clear_cpu(cpu, &d->hdr.cpu_mask);
> + if (cpumask_empty(&d->hdr.cpu_mask)) {
> + list_del_rcu(&d->hdr.list);
> + synchronize_rcu();
> + resctrl_offline_ctrl_domain(res, d);
> + kfree(container_of(d, struct cbqri_resctrl_dom,
> + resctrl_ctrl_dom));
> + }
> + }
> +}
> +
> +/*
> + * Attach a CPU to the capacity controller at each cache level whose cache
> + * the CPU shares. On failure, detach the CPU from everything attached so
> + * far: the cpuhp core does not run this state's offline teardown when its
> + * startup fails, so a partial attach would otherwise leak into the domain
> + * cpu_masks. Caller holds cbqri_domain_list_lock.
> + */
> +static int cbqri_attach_cpu_to_all_ctrls(unsigned int cpu)
> +{
> + static const u32 levels[] = { 2, 3 };
> + struct cbqri_controller *ctrl, *c;
> + struct cbqri_resctrl_res *hw_res;
> + struct rdt_ctrl_domain *d;
> + struct cacheinfo *ci;
> + int i, rid;
> +
> + lockdep_assert_held(&cbqri_domain_list_lock);
> +
> + /*
> + * Hold cbqri_controllers_lock across the walk so a controller
> + * registered after boot cannot corrupt it. The register path takes
> + * it as a leaf and never cbqri_domain_list_lock, so this nesting
> + * cannot invert.
> + */
> + guard(mutex)(&cbqri_controllers_lock);
> +
> + for (i = 0; i < ARRAY_SIZE(levels); i++) {
> + ci = get_cpu_cacheinfo_level(cpu, levels[i]);
> + if (!ci)
> + continue;
> +
> + rid = cbqri_cache_level_to_rid(levels[i]);
> + hw_res = &cbqri_resctrl_resources[rid];
> + if (!hw_res->ctrl)
> + continue;
> +
> + /* The controller backing this CPU's cache at this level. */
> + ctrl = NULL;
> + list_for_each_entry(c, &cbqri_controllers, list) {
> + if (c->type == CBQRI_CONTROLLER_TYPE_CAPACITY &&
> + c->alloc_capable &&
> + c->cache.cache_level == levels[i] &&
> + c->cache.cache_id == ci->id) {
> + ctrl = c;
> + break;
> + }
> + }
> + if (!ctrl)
> + continue;
> +
> + d = cbqri_find_ctrl_domain(&hw_res->resctrl_res.ctrl_domains,
> + ci->id);
> + if (d) {
> + cpumask_set_cpu(cpu, &d->hdr.cpu_mask);
> + continue;
> + }
> +
> + d = cbqri_create_ctrl_domain(ctrl, &hw_res->resctrl_res, cpu,
> + ci->id);
> + if (IS_ERR(d)) {
> + cbqri_detach_cpu_from_all_ctrls(cpu);
> + return PTR_ERR(d);
> + }
> + }
> +
> + return 0;
> +}
> +
> +static bool cbqri_resctrl_inited;
> +
> +static void cbqri_resctrl_teardown(void)
> +{
> + int rid;
> +
> + if (!cbqri_resctrl_inited)
> + return;
> +
> + resctrl_exit();
> +
> + for (rid = 0; rid < RDT_NUM_RESOURCES; rid++) {
> + struct cbqri_resctrl_res *hw_res = &cbqri_resctrl_resources[rid];
> +
> + hw_res->ctrl = NULL;
> + hw_res->cdp_enabled = false;
> + }
> + exposed_alloc_capable = false;
> + cbqri_resctrl_inited = false;
> +}
> +
> +static int cbqri_resctrl_setup(void)
> +{
> + int rid;
> + int err;
> +
> + for (rid = 0; rid < RDT_NUM_RESOURCES; rid++)
> + cbqri_resctrl_resources[rid].resctrl_res.rid = rid;
> +
> + cbqri_resctrl_pick_caches();
> +
> + for (rid = 0; rid < RDT_NUM_RESOURCES; rid++)
> + cbqri_resctrl_control_init(&cbqri_resctrl_resources[rid]);
> +
> + cbqri_resctrl_accumulate_caps();
> +
> + if (!exposed_alloc_capable) {
> + pr_debug("no resctrl-capable CBQRI controllers found\n");
> + return -ENODEV;
> + }
> +
> + err = resctrl_init();
> + if (err)
> + return err;
> +
> + cbqri_resctrl_inited = true;
> + return 0;
> +}
> +
> +static int cbqri_resctrl_online_cpu(unsigned int cpu)
> +{
> + int err;
> +
> + mutex_lock(&cbqri_domain_list_lock);
> + err = cbqri_attach_cpu_to_all_ctrls(cpu);
> + mutex_unlock(&cbqri_domain_list_lock);
> + if (err)
> + return err;
> +
> + /*
> + * Seed the per-CPU default RCID/MCID to the reserved (0, 0) pair and
> + * notify the resctrl core so it tracks this CPU in the default group.
> + */
> + resctrl_arch_set_cpu_default_closid_rmid(cpu, 0, 0);
> + resctrl_online_cpu(cpu);
> + return 0;
> +}
> +
> +static int cbqri_resctrl_offline_cpu(unsigned int cpu)
> +{
> + resctrl_offline_cpu(cpu);
> +
> + mutex_lock(&cbqri_domain_list_lock);
> + cbqri_detach_cpu_from_all_ctrls(cpu);
> + mutex_unlock(&cbqri_domain_list_lock);
> + return 0;
> +}
> +
> +static int __init cbqri_arch_late_init(void)
> +{
> + int err;
> +
> + if (!riscv_isa_extension_available(NULL, SSQOSID))
> + return -ENODEV;
> +
> + err = cbqri_resctrl_setup();
> + if (err)
> + return err;
> +
> + err = cpuhp_setup_state(CPUHP_AP_ONLINE_DYN, "cbqri:online",
> + cbqri_resctrl_online_cpu,
> + cbqri_resctrl_offline_cpu);
> + if (err < 0) {
> + cbqri_resctrl_teardown();
> + return err;
> + }
> +
> + return 0;
> +}
> +late_initcall(cbqri_arch_late_init);
>
> --
> 2.43.0
>
More information about the linux-riscv
mailing list