[PATCH v9 2/5] riscv_cbqri: resctrl: Add cache allocation via capacity block mask

Drew Fustini fustini at kernel.org
Sat Oct 3 00:12:27 PDT 2026


Wire CBQRI capacity controllers into resctrl as RDT_RESOURCE_L2 and
RDT_RESOURCE_L3 schemata.

resctrl exposes a single capacity mask per resource, so when more than
one capacity controller backs a cache level they must report the same
number of capacity blocks. The same holds for the other capabilities
resctrl exposes once per resource. A level whose controllers disagree
is left unexposed. Other levels are unaffected.

The set of controllers backing each level is fixed when resctrl is set
up at late_initcall. Control domains are created in the cpuhp online
callback and keyed on the onlining CPU's cache id from cacheinfo, so a
CPU that is offline at probe time still joins its domain once it comes
online, provided a controller for its cache is in that set. A CPU whose
cache has no controller is left out of that level and a warning is
logged once. A domain is published on the ctrl_domains RCU list only
after it is fully initialized and brought online, and removed from the
list before it is taken offline, following the documented safe
traversal rules.

Assisted-by: LLM
Tested-by: Chen Pei <cp0613 at linux.alibaba.com>
Co-developed-by: Adrien Ricciardi <aricciardi at baylibre.com>
Signed-off-by: Adrien Ricciardi <aricciardi at baylibre.com>
Signed-off-by: Drew Fustini <fustini at kernel.org>
---
 MAINTAINERS                      |   2 +
 arch/riscv/include/asm/resctrl.h | 147 +++++++
 drivers/resctrl/Kconfig          |   4 +
 drivers/resctrl/Makefile         |   1 +
 drivers/resctrl/cbqri_resctrl.c  | 831 +++++++++++++++++++++++++++++++++++++++
 5 files changed, 985 insertions(+)

diff --git a/MAINTAINERS b/MAINTAINERS
index 92b111cd25c0..db0799d8cbf8 100644
--- a/MAINTAINERS
+++ b/MAINTAINERS
@@ -23570,9 +23570,11 @@ R:	yunhui cui <cuiyunhui at bytedance.com>
 L:	linux-riscv at lists.infradead.org
 S:	Supported
 F:	arch/riscv/include/asm/qos.h
+F:	arch/riscv/include/asm/resctrl.h
 F:	arch/riscv/kernel/qos.c
 F:	drivers/resctrl/cbqri_devices.c
 F:	drivers/resctrl/cbqri_internal.h
+F:	drivers/resctrl/cbqri_resctrl.c
 F:	include/linux/riscv_cbqri.h
 
 RISC-V RPMI AND MPXY DRIVERS
diff --git a/arch/riscv/include/asm/resctrl.h b/arch/riscv/include/asm/resctrl.h
new file mode 100644
index 000000000000..867db663d958
--- /dev/null
+++ b/arch/riscv/include/asm/resctrl.h
@@ -0,0 +1,147 @@
+/* SPDX-License-Identifier: GPL-2.0-only */
+#ifndef _ASM_RISCV_RESCTRL_H
+#define _ASM_RISCV_RESCTRL_H
+
+#include <linux/resctrl_types.h>
+#include <linux/sched.h>
+#include <linux/types.h>
+
+#include <asm/qos.h>
+
+struct rdt_resource;
+
+/*
+ * Sentinel "no CLOSID assigned" used by resctrl_arch_rmid_idx_decode().
+ * fs/resctrl treats this opaquely. CBQRI uses MCID directly as the linear
+ * rmid index, so closid is unused on decode.
+ */
+#define RISCV_RESCTRL_EMPTY_CLOSID	((u32)~0)
+
+/*
+ * Terminology mapping between x86 (Intel RDT/AMD QoS) and RISC-V:
+ *
+ *  CLOSID on x86 is RCID on RISC-V
+ *    RMID on x86 is MCID on RISC-V
+ *     CDP on x86 is AT (access type) on RISC-V
+ */
+
+/**
+ * resctrl_arch_alloc_capable() - any CBQRI controller exposes resctrl alloc
+ *
+ * True while at least one resource set up by cbqri_resctrl_setup() has
+ * rdt_resource::alloc_capable set, so it gives the same answer as walking
+ * the resources. Only meaningful after late_initcall.
+ */
+bool resctrl_arch_alloc_capable(void);
+
+/**
+ * resctrl_arch_mon_capable() - any CBQRI controller exposes resctrl monitoring
+ *
+ * The CBQRI driver implements capacity allocation only and wires up no
+ * monitoring events, so this always returns false. fs/resctrl references it
+ * unconditionally, hence the stub.
+ */
+bool resctrl_arch_mon_capable(void);
+
+/**
+ * resctrl_arch_rmid_idx_encode() - encode (RCID, MCID) into a linear index
+ * @closid: RCID (resource control id)
+ * @rmid:   MCID (monitoring counter id)
+ *
+ * RISC-V uses MCID directly as the linear index into per-RMID arrays
+ * managed by fs/resctrl, since CBQRI controllers admit any MCID for any
+ * RCID. closid is unused here. CDP is encoded via the AT field on each
+ * CBQRI op rather than via the index.
+ */
+u32  resctrl_arch_rmid_idx_encode(u32 closid, u32 rmid);
+
+/**
+ * resctrl_arch_rmid_idx_decode() - inverse of resctrl_arch_rmid_idx_encode()
+ * @idx:    linear index
+ * @closid: out: always RISCV_RESCTRL_EMPTY_CLOSID
+ * @rmid:   out: the MCID that @idx encodes
+ */
+void resctrl_arch_rmid_idx_decode(u32 idx, u32 *closid, u32 *rmid);
+
+/**
+ * resctrl_arch_set_cpu_default_closid_rmid() - install per-CPU srmcfg default
+ * @cpu:    CPU number
+ * @closid: RCID to use when no task is matched
+ * @rmid:   MCID to use when no task is matched
+ *
+ * Sets the per-CPU cpu_srmcfg_default so __switch_to_srmcfg() can fall back
+ * to the CPU's default RCID/MCID for default-group tasks (those whose
+ * thread.srmcfg encodes to 0, i.e. closid == RESCTRL_RESERVED_CLOSID and
+ * rmid == RESCTRL_RESERVED_RMID). Implements resctrl allocation rule 2
+ * ("CPU default") on RISC-V.
+ */
+void resctrl_arch_set_cpu_default_closid_rmid(int cpu, u32 closid, u32 rmid);
+
+/**
+ * resctrl_arch_sched_in() - context-switch hook to install task RCID/MCID
+ * @tsk: the task being scheduled in
+ *
+ * Called from finish_task_switch() to write tsk->thread.srmcfg into the
+ * srmcfg CSR. Tasks tagged with RISCV_RESCTRL_EMPTY_CLOSID inherit the
+ * per-CPU default set via resctrl_arch_set_cpu_default_closid_rmid().
+ */
+void resctrl_arch_sched_in(struct task_struct *tsk);
+
+/**
+ * resctrl_arch_set_closid_rmid() - tag a task with an RCID/MCID
+ * @tsk:    task to tag
+ * @closid: RCID to install
+ * @rmid:   MCID to install
+ *
+ * Updates tsk->thread.srmcfg with the encoded (RCID, MCID) pair. The new
+ * value takes effect on the next resctrl_arch_sched_in() for this task.
+ */
+void resctrl_arch_set_closid_rmid(struct task_struct *tsk, u32 closid, u32 rmid);
+
+/**
+ * resctrl_arch_match_closid() - test whether a task carries a given RCID
+ * @tsk:    task
+ * @closid: RCID
+ */
+bool resctrl_arch_match_closid(struct task_struct *tsk, u32 closid);
+
+/**
+ * resctrl_arch_match_rmid() - test whether a task carries a given (RCID, MCID)
+ * @tsk:    task
+ * @closid: RCID
+ * @rmid:   MCID
+ */
+bool resctrl_arch_match_rmid(struct task_struct *tsk, u32 closid, u32 rmid);
+
+/**
+ * resctrl_arch_mon_ctx_alloc() - allocate per-monitor-event arch context
+ * @r:     resctrl resource being monitored
+ * @evtid: which monitor event needs context
+ *
+ * The CBQRI driver implements no monitoring events, so there is no per-event
+ * context to allocate and the stub returns NULL. fs/resctrl references it
+ * unconditionally before checking resctrl_arch_mon_capable().
+ */
+void *resctrl_arch_mon_ctx_alloc(struct rdt_resource *r, enum resctrl_event_id evtid);
+
+/**
+ * resctrl_arch_mon_ctx_free() - release context returned by mon_ctx_alloc()
+ * @r:            resctrl resource
+ * @evtid:        monitor event id
+ * @arch_mon_ctx: pointer returned by resctrl_arch_mon_ctx_alloc()
+ */
+void resctrl_arch_mon_ctx_free(struct rdt_resource *r, enum resctrl_event_id evtid,
+			       void *arch_mon_ctx);
+
+static inline unsigned int resctrl_arch_round_mon_val(unsigned int val)
+{
+	return val;
+}
+
+/* Not needed for RISC-V */
+static inline void resctrl_arch_enable_mon(void) { }
+static inline void resctrl_arch_disable_mon(void) { }
+static inline void resctrl_arch_enable_alloc(void) { }
+static inline void resctrl_arch_disable_alloc(void) { }
+
+#endif /* _ASM_RISCV_RESCTRL_H */
diff --git a/drivers/resctrl/Kconfig b/drivers/resctrl/Kconfig
index 1219fb29aa2e..cac23d20124b 100644
--- a/drivers/resctrl/Kconfig
+++ b/drivers/resctrl/Kconfig
@@ -38,3 +38,7 @@ menuconfig RISCV_CBQRI
 	  RISC-V cache QoS resources. CBQRI exposes cache capacity
 	  allocation through the resctrl filesystem at /sys/fs/resctrl when
 	  RESCTRL_FS is also enabled.
+
+config RISCV_CBQRI_RESCTRL_FS
+	bool
+	default y if RISCV_CBQRI && RESCTRL_FS
diff --git a/drivers/resctrl/Makefile b/drivers/resctrl/Makefile
index 4d8a2c4b5627..a7631712dba9 100644
--- a/drivers/resctrl/Makefile
+++ b/drivers/resctrl/Makefile
@@ -6,3 +6,4 @@ ccflags-$(CONFIG_ARM64_MPAM_DRIVER_DEBUG)	+= -DDEBUG
 
 obj-$(CONFIG_RISCV_CBQRI)			+= cbqri.o
 cbqri-y						+= cbqri_devices.o
+cbqri-$(CONFIG_RISCV_CBQRI_RESCTRL_FS)		+= cbqri_resctrl.o
diff --git a/drivers/resctrl/cbqri_resctrl.c b/drivers/resctrl/cbqri_resctrl.c
new file mode 100644
index 000000000000..f10a24761b25
--- /dev/null
+++ b/drivers/resctrl/cbqri_resctrl.c
@@ -0,0 +1,831 @@
+// SPDX-License-Identifier: GPL-2.0-only
+
+#define pr_fmt(fmt) "%s:%s: " fmt, KBUILD_MODNAME, __func__
+
+#include <linux/bitfield.h>
+#include <linux/cacheinfo.h>
+#include <linux/cleanup.h>
+#include <linux/cpu.h>
+#include <linux/cpufeature.h>
+#include <linux/cpuhotplug.h>
+#include <linux/err.h>
+#include <linux/init.h>
+#include <linux/rculist.h>
+#include <linux/resctrl.h>
+#include <linux/riscv_cbqri.h>
+#include <linux/slab.h>
+#include <linux/types.h>
+
+#include <asm/csr.h>
+#include <asm/qos.h>
+
+#include "cbqri_internal.h"
+
+struct cbqri_resctrl_res {
+	/*
+	 * Capacity controllers picked for this resource at late_initcall,
+	 * one per cache instance at its level. cbqri_resctrl_pick_caches()
+	 * only accepts a set that agrees on the capabilities resctrl exposes
+	 * once per resource, so ctrls[0] speaks for all of them.
+	 */
+	struct cbqri_controller **ctrls;
+	unsigned int            nr_ctrls;
+	struct rdt_resource     resctrl_res;
+	bool                    cdp_enabled;
+};
+
+struct cbqri_resctrl_dom {
+	struct rdt_ctrl_domain  resctrl_ctrl_dom;
+	struct cbqri_controller *hw_ctrl;
+};
+
+static struct cbqri_resctrl_res cbqri_resctrl_resources[RDT_NUM_RESOURCES];
+
+/* Protects ctrl_domain list mutations across CPU hotplug. */
+static DEFINE_MUTEX(cbqri_domain_list_lock);
+
+/*
+ * Look up the ctrl_domain with this id. When there is none and pos is
+ * given, pos receives the list position a new domain with this id has to
+ * be inserted before, so the caller can add it without a second walk.
+ */
+static struct rdt_ctrl_domain *
+cbqri_find_ctrl_domain(struct list_head *h, int id, struct list_head **pos)
+{
+	struct rdt_domain_hdr *hdr = resctrl_find_domain(h, id, pos);
+
+	return hdr ? container_of(hdr, struct rdt_ctrl_domain, hdr) : NULL;
+}
+
+/* The picked controller backing cache instance cache_id, or NULL. */
+static struct cbqri_controller *
+cbqri_resctrl_find_ctrl(const struct cbqri_resctrl_res *hw_res, u32 cache_id)
+{
+	unsigned int i;
+
+	for (i = 0; i < hw_res->nr_ctrls; i++) {
+		if (hw_res->ctrls[i]->cache.cache_id == cache_id)
+			return hw_res->ctrls[i];
+	}
+
+	return NULL;
+}
+
+/* Map a hardware cache level to its resctrl resource id, or -ENODEV. */
+static int cbqri_cache_level_to_rid(u32 cache_level)
+{
+	switch (cache_level) {
+	case 2:
+		return RDT_RESOURCE_L2;
+	case 3:
+		return RDT_RESOURCE_L3;
+	default:
+		return -ENODEV;
+	}
+}
+
+static int cbqri_apply_cache_config_dom(struct cbqri_resctrl_dom *hw_dom,
+					struct rdt_resource *r,
+					u32 closid, enum resctrl_conf_type t,
+					u64 cbm)
+{
+	struct cbqri_resctrl_res *hw_res =
+		container_of(r, struct cbqri_resctrl_res, resctrl_res);
+	struct cbqri_cc_config cfg = {
+		.cbm = cbm,
+		.at = (t == CDP_CODE) ? CBQRI_CONTROL_REGISTERS_AT_CODE :
+					CBQRI_CONTROL_REGISTERS_AT_DATA,
+		.cdp_enabled = hw_res->cdp_enabled,
+	};
+
+	return cbqri_apply_cache_config(hw_dom->hw_ctrl, closid, &cfg);
+}
+
+/* True while at least one picked resource offers capacity allocation. */
+static bool cbqri_resctrl_any_alloc_capable(void)
+{
+	int rid;
+
+	for (rid = 0; rid < RDT_NUM_RESOURCES; rid++) {
+		if (cbqri_resctrl_resources[rid].resctrl_res.alloc_capable)
+			return true;
+	}
+
+	return false;
+}
+
+bool resctrl_arch_alloc_capable(void)
+{
+	return cbqri_resctrl_any_alloc_capable();
+}
+
+bool resctrl_arch_mon_capable(void)
+{
+	return false;
+}
+
+bool resctrl_arch_get_cdp_enabled(enum resctrl_res_level rid)
+{
+	if (rid != RDT_RESOURCE_L2 && rid != RDT_RESOURCE_L3)
+		return false;
+	return cbqri_resctrl_resources[rid].cdp_enabled;
+}
+
+int resctrl_arch_set_cdp_enabled(enum resctrl_res_level rid, bool enable)
+{
+	struct cbqri_resctrl_res *cbqri_res;
+
+	if (rid != RDT_RESOURCE_L2 && rid != RDT_RESOURCE_L3)
+		return -ENODEV;
+
+	cbqri_res = &cbqri_resctrl_resources[rid];
+	if (!cbqri_res->resctrl_res.cdp_capable)
+		return -ENODEV;
+
+	cbqri_res->cdp_enabled = enable;
+	return 0;
+}
+
+struct rdt_resource *resctrl_arch_get_resource(enum resctrl_res_level l)
+{
+	if (l >= RDT_NUM_RESOURCES)
+		return NULL;
+
+	return &cbqri_resctrl_resources[l].resctrl_res;
+}
+
+bool resctrl_arch_is_evt_configurable(enum resctrl_event_id evt)
+{
+	return false;
+}
+
+void *resctrl_arch_mon_ctx_alloc(struct rdt_resource *r,
+				 enum resctrl_event_id evtid)
+{
+	return NULL;
+}
+
+void resctrl_arch_mon_ctx_free(struct rdt_resource *r,
+			       enum resctrl_event_id evtid, void *arch_mon_ctx)
+{
+}
+
+void resctrl_arch_config_cntr(struct rdt_resource *r, struct rdt_l3_mon_domain *d,
+			      enum resctrl_event_id evtid, u32 rmid, u32 closid,
+			      u32 cntr_id, bool assign)
+{
+}
+
+int resctrl_arch_cntr_read(struct rdt_resource *r, struct rdt_l3_mon_domain *d,
+			   u32 unused, u32 rmid, int cntr_id,
+			   enum resctrl_event_id eventid, u64 *val)
+{
+	return -EOPNOTSUPP;
+}
+
+bool resctrl_arch_mbm_cntr_assign_enabled(struct rdt_resource *r)
+{
+	return false;
+}
+
+int resctrl_arch_mbm_cntr_assign_set(struct rdt_resource *r, bool enable)
+{
+	return -EOPNOTSUPP;
+}
+
+void resctrl_arch_reset_cntr(struct rdt_resource *r, struct rdt_l3_mon_domain *d,
+			     u32 unused, u32 rmid, int cntr_id,
+			     enum resctrl_event_id eventid)
+{
+}
+
+bool resctrl_arch_get_io_alloc_enabled(struct rdt_resource *r)
+{
+	return false;
+}
+
+int resctrl_arch_io_alloc_enable(struct rdt_resource *r, bool enable)
+{
+	return -EOPNOTSUPP;
+}
+
+void resctrl_arch_mon_event_config_read(void *info)
+{
+}
+
+void resctrl_arch_mon_event_config_write(void *info)
+{
+}
+
+void resctrl_arch_reset_rmid_all(struct rdt_resource *r, struct rdt_l3_mon_domain *d)
+{
+}
+
+void resctrl_arch_reset_rmid(struct rdt_resource *r, struct rdt_l3_mon_domain *d,
+			     u32 unused, u32 rmid, enum resctrl_event_id eventid)
+{
+}
+
+int resctrl_arch_rmid_read(struct rdt_resource *r, struct rdt_domain_hdr *hdr,
+			   u32 closid, u32 rmid, enum resctrl_event_id eventid,
+			   void *arch_priv, u64 *val, void *arch_mon_ctx)
+{
+	return -ENODATA;
+}
+
+/*
+ * Note about terminology between x86 (Intel RDT/AMD QoS) and RISC-V:
+ *   CLOSID on x86 is RCID on RISC-V
+ *     RMID on x86 is MCID on RISC-V
+ *
+ * schemata_list_add() halves this count when CDP is enabled because a
+ * CDP CLOSID takes two hardware IDs on x86. CBQRI selects the code or
+ * data allocation with the AT field next to the RCID, so every CLOSID
+ * resctrl hands out under CDP is already a distinct RCID and the RCIDs
+ * from rcid_count/2 upward sit idle. Skipping the halving would need an
+ * fs/resctrl change, so it is left as is.
+ */
+u32 resctrl_arch_get_num_closid(struct rdt_resource *res)
+{
+	struct cbqri_resctrl_res *hw_res;
+
+	hw_res = container_of(res, struct cbqri_resctrl_res, resctrl_res);
+
+	if (!hw_res->nr_ctrls)
+		return 0;
+
+	return hw_res->ctrls[0]->rcid_count;
+}
+
+u32 resctrl_arch_system_num_rmid_idx(void)
+{
+	return 1;
+}
+
+u32 resctrl_arch_rmid_idx_encode(u32 closid, u32 rmid)
+{
+	return rmid;
+}
+
+void resctrl_arch_rmid_idx_decode(u32 idx, u32 *closid, u32 *rmid)
+{
+	*closid = RISCV_RESCTRL_EMPTY_CLOSID;
+	*rmid = idx;
+}
+
+void resctrl_arch_set_cpu_default_closid_rmid(int cpu, u32 closid, u32 rmid)
+{
+	u32 srmcfg = FIELD_PREP(SRMCFG_RCID_MASK, closid) |
+		     FIELD_PREP(SRMCFG_MCID_MASK, rmid);
+
+	WRITE_ONCE(per_cpu(cpu_srmcfg_default, cpu), srmcfg);
+}
+
+void resctrl_arch_sched_in(struct task_struct *tsk)
+{
+	__switch_to_srmcfg(tsk);
+}
+
+void resctrl_arch_set_closid_rmid(struct task_struct *tsk, u32 closid, u32 rmid)
+{
+	u32 srmcfg = FIELD_PREP(SRMCFG_RCID_MASK, closid) |
+		     FIELD_PREP(SRMCFG_MCID_MASK, rmid);
+
+	WRITE_ONCE(tsk->thread.srmcfg, srmcfg);
+}
+
+void resctrl_arch_sync_cpu_closid_rmid(void *info)
+{
+	struct resctrl_cpu_defaults *r = info;
+
+	lockdep_assert_preemption_disabled();
+
+	if (r) {
+		resctrl_arch_set_cpu_default_closid_rmid(smp_processor_id(),
+							 r->closid, r->rmid);
+	}
+
+	resctrl_arch_sched_in(current);
+}
+
+bool resctrl_arch_match_closid(struct task_struct *tsk, u32 closid)
+{
+	return FIELD_GET(SRMCFG_RCID_MASK, READ_ONCE(tsk->thread.srmcfg)) == closid;
+}
+
+bool resctrl_arch_match_rmid(struct task_struct *tsk, u32 closid, u32 rmid)
+{
+	return FIELD_GET(SRMCFG_MCID_MASK, READ_ONCE(tsk->thread.srmcfg)) == rmid;
+}
+
+void resctrl_arch_pre_mount(void)
+{
+	/* All controllers discovered at boot via late_initcall. Nothing to do. */
+}
+
+int resctrl_arch_update_one(struct rdt_resource *r, struct rdt_ctrl_domain *d,
+			    u32 closid, enum resctrl_conf_type t, u32 cfg_val)
+{
+	struct cbqri_resctrl_dom *dom;
+
+	dom = container_of(d, struct cbqri_resctrl_dom, resctrl_ctrl_dom);
+
+	if (!r->alloc_capable)
+		return -EINVAL;
+
+	switch (r->rid) {
+	case RDT_RESOURCE_L2:
+	case RDT_RESOURCE_L3:
+		return cbqri_apply_cache_config_dom(dom, r, closid, t, cfg_val);
+	default:
+		return -EINVAL;
+	}
+}
+
+int resctrl_arch_update_domains(struct rdt_resource *r, u32 closid)
+{
+	struct resctrl_staged_config *cfg;
+	enum resctrl_conf_type t;
+	struct rdt_ctrl_domain *d;
+	int err = 0;
+
+	/* Walking r->ctrl_domains, ensure it can't race with cpuhp */
+	lockdep_assert_cpus_held();
+
+	list_for_each_entry(d, &r->ctrl_domains, hdr.list) {
+		for (t = 0; t < CDP_NUM_TYPES; t++) {
+			cfg = &d->staged_config[t];
+			if (!cfg->have_new_ctrl)
+				continue;
+			err = resctrl_arch_update_one(r, d, closid, t, cfg->new_ctrl);
+			if (err)
+				return err;
+		}
+	}
+	return err;
+}
+
+u32 resctrl_arch_get_config(struct rdt_resource *r, struct rdt_ctrl_domain *d,
+			    u32 closid, enum resctrl_conf_type type)
+{
+	struct cbqri_resctrl_dom *hw_dom;
+	struct cbqri_controller *ctrl;
+	u32 at;
+	u32 val;
+	int err;
+
+	hw_dom = container_of(d, struct cbqri_resctrl_dom, resctrl_ctrl_dom);
+	ctrl = hw_dom->hw_ctrl;
+	val = resctrl_get_default_ctrl(r);
+
+	if (!r->alloc_capable)
+		return val;
+
+	switch (r->rid) {
+	case RDT_RESOURCE_L2:
+	case RDT_RESOURCE_L3:
+		at = (type == CDP_CODE) ? CBQRI_CONTROL_REGISTERS_AT_CODE :
+					  CBQRI_CONTROL_REGISTERS_AT_DATA;
+		err = cbqri_read_cache_config(ctrl, closid, at, &val);
+		if (err < 0)
+			val = resctrl_get_default_ctrl(r);
+		break;
+	default:
+		break;
+	}
+
+	return val;
+}
+
+void resctrl_arch_reset_all_ctrls(struct rdt_resource *r)
+{
+	u32 num_closid = resctrl_arch_get_num_closid(r);
+	struct rdt_ctrl_domain *d;
+	enum resctrl_conf_type t;
+	u32 default_ctrl;
+	u32 i;
+
+	lockdep_assert_cpus_held();
+
+	if (!num_closid)
+		return;
+
+	default_ctrl = resctrl_get_default_ctrl(r);
+
+	list_for_each_entry(d, &r->ctrl_domains, hdr.list) {
+		for (i = 0; i < num_closid; i++) {
+			for (t = 0; t < CDP_NUM_TYPES; t++) {
+				int rerr;
+
+				rerr = resctrl_arch_update_one(r, d, i, t, default_ctrl);
+				if (rerr)
+					pr_err_ratelimited("rid=%d reset RCID %u type %u failed (%d)\n",
+							   r->rid, i, t, rerr);
+			}
+		}
+	}
+}
+
+static struct rdt_ctrl_domain *cbqri_new_domain(struct cbqri_controller *ctrl)
+{
+	struct cbqri_resctrl_dom *hw_dom;
+	struct rdt_ctrl_domain *domain;
+
+	hw_dom = kzalloc_obj(*hw_dom, GFP_KERNEL);
+	if (!hw_dom)
+		return NULL;
+
+	hw_dom->hw_ctrl = ctrl;
+	domain = &hw_dom->resctrl_ctrl_dom;
+
+	INIT_LIST_HEAD(&domain->hdr.list);
+
+	return domain;
+}
+
+static int cbqri_init_domain_ctrlval(struct rdt_resource *r, struct rdt_ctrl_domain *d)
+{
+	u32 num_closid = resctrl_arch_get_num_closid(r);
+	enum resctrl_conf_type t;
+	int err = 0;
+	u32 i;
+
+	for (i = 0; i < num_closid; i++) {
+		/*
+		 * Seed both DATA and CODE staged slots so a later mount
+		 * with -o cdp does not see stale CODE values.
+		 * On non-AT controllers cbqri_cc_alloc_op() masks AT to 0
+		 * so all three iterations land on the same hardware state.
+		 * The redundant writes are harmless.
+		 */
+		for (t = 0; t < CDP_NUM_TYPES; t++) {
+			err = resctrl_arch_update_one(r, d, i, t,
+						      resctrl_get_default_ctrl(r));
+			if (err)
+				return err;
+		}
+	}
+	return 0;
+}
+
+/*
+ * Two capacity controllers at the same cache level are interchangeable to
+ * resctrl only if they expose identical caps, since resctrl publishes one
+ * set of caps per rid but any of the level's controllers may service a
+ * given RCID.
+ */
+static bool cbqri_cc_caps_agree(const struct cbqri_controller *a,
+				const struct cbqri_controller *b)
+{
+	return a->rcid_count == b->rcid_count &&
+	       a->cc.ncblks == b->cc.ncblks &&
+	       a->cc.supports_alloc_at_code == b->cc.supports_alloc_at_code &&
+	       a->alloc_capable == b->alloc_capable;
+}
+
+/* An alloc-capable capacity controller at the cache level rid stands for. */
+static bool cbqri_ctrl_backs_rid(const struct cbqri_controller *ctrl, int rid)
+{
+	return ctrl->type == CBQRI_CONTROLLER_TYPE_CAPACITY &&
+	       ctrl->alloc_capable &&
+	       cbqri_cache_level_to_rid(ctrl->cache.cache_level) == rid;
+}
+
+/*
+ * Walk cbqri_controllers once and keep, per cache level (L2/L3), every
+ * capacity controller (CC) that backs the corresponding RDT_RESOURCE_L*.
+ * When more than one CC sits at the same level (e.g. one per socket) they
+ * must agree on rcid_count / ncblks / supports_alloc_at_code, since resctrl
+ * exposes a single set of caps per rid. A level whose controllers disagree
+ * is dropped, but the other level is still picked. Controllers registered
+ * after this walk are not part of any set and resctrl never sees them.
+ */
+static int cbqri_resctrl_pick_caches(void)
+{
+	struct cbqri_controller *ctrl;
+	int rid;
+
+	guard(mutex)(&cbqri_controllers_lock);
+
+	for (rid = 0; rid < RDT_NUM_RESOURCES; rid++) {
+		struct cbqri_resctrl_res *hw_res = &cbqri_resctrl_resources[rid];
+		unsigned int n = 0;
+
+		list_for_each_entry(ctrl, &cbqri_controllers, list) {
+			if (cbqri_ctrl_backs_rid(ctrl, rid))
+				n++;
+		}
+		if (!n)
+			continue;
+
+		hw_res->ctrls = kcalloc(n, sizeof(*hw_res->ctrls), GFP_KERNEL);
+		if (!hw_res->ctrls)
+			return -ENOMEM;
+
+		list_for_each_entry(ctrl, &cbqri_controllers, list) {
+			if (!cbqri_ctrl_backs_rid(ctrl, rid))
+				continue;
+
+			if (hw_res->nr_ctrls &&
+			    !cbqri_cc_caps_agree(hw_res->ctrls[0], ctrl)) {
+				pr_err("L%u controllers have mismatched capabilities, skipping this level\n",
+				       ctrl->cache.cache_level);
+				kfree(hw_res->ctrls);
+				hw_res->ctrls = NULL;
+				hw_res->nr_ctrls = 0;
+				break;
+			}
+
+			hw_res->ctrls[hw_res->nr_ctrls++] = ctrl;
+		}
+	}
+
+	return 0;
+}
+
+/*
+ * Fill the rdt_resource fields for one picked rid. An rid with no picked
+ * controller is left untouched so it stays out of resctrl_arch_get_resource().
+ */
+static void cbqri_resctrl_control_init(struct cbqri_resctrl_res *cbqri_res)
+{
+	struct rdt_resource *res = &cbqri_res->resctrl_res;
+	struct cbqri_controller *ctrl;
+
+	if (!cbqri_res->nr_ctrls)
+		return;
+	ctrl = cbqri_res->ctrls[0];
+
+	switch (res->rid) {
+	case RDT_RESOURCE_L2:
+	case RDT_RESOURCE_L3:
+		res->name = (res->rid == RDT_RESOURCE_L2) ? "L2" : "L3";
+		res->schema_fmt = RESCTRL_SCHEMA_BITMAP;
+		res->ctrl_scope = (res->rid == RDT_RESOURCE_L2) ?
+				    RESCTRL_L2_CACHE : RESCTRL_L3_CACHE;
+		res->cache.cbm_len = ctrl->cc.ncblks;
+		res->cache.shareable_bits = 0;
+		res->cache.min_cbm_bits = 1;
+		res->cache.arch_has_sparse_bitmasks = false;
+		res->cdp_capable = ctrl->cc.supports_alloc_at_code;
+		res->alloc_capable = ctrl->alloc_capable;
+		INIT_LIST_HEAD(&res->ctrl_domains);
+		INIT_LIST_HEAD(&res->mon_domains);
+		break;
+	default:
+		break;
+	}
+}
+
+/*
+ * Create, list-insert, and online a fresh ctrl_domain backing ctrl on
+ * resource res, seeded with cpu and identified by dom_id. Caller must
+ * hold cbqri_domain_list_lock, must have already verified that no
+ * existing ctrl_domain on res carries this id, and passes the list
+ * position cbqri_find_ctrl_domain() returned for it as pos.
+ */
+static struct rdt_ctrl_domain *cbqri_create_ctrl_domain(struct cbqri_controller *ctrl,
+							struct rdt_resource *res,
+							unsigned int cpu, int dom_id,
+							struct list_head *pos)
+{
+	struct rdt_ctrl_domain *domain;
+	int err;
+
+	domain = cbqri_new_domain(ctrl);
+	if (!domain)
+		return ERR_PTR(-ENOMEM);
+
+	cpumask_set_cpu(cpu, &domain->hdr.cpu_mask);
+	domain->hdr.id = dom_id;
+	domain->hdr.type = RESCTRL_CTRL_DOMAIN;
+	domain->hdr.rid = res->rid;
+
+	err = cbqri_init_domain_ctrlval(res, domain);
+	if (err)
+		goto free;
+
+	err = resctrl_online_ctrl_domain(res, domain);
+	if (err)
+		goto free;
+
+	/*
+	 * Publish only after the domain is fully initialized and online, so a
+	 * reader walking the RCU list never sees a half-built domain.
+	 */
+	list_add_tail_rcu(&domain->hdr.list, pos);
+
+	return domain;
+free:
+	kfree(container_of(domain, struct cbqri_resctrl_dom, resctrl_ctrl_dom));
+	return ERR_PTR(err);
+}
+
+/*
+ * Remove a CPU from the domain at each level whose cache it shares. The
+ * domain is looked up by the CPU's own cache id and the lookup returns NULL
+ * if it is already gone, so this is idempotent and undoes a partial online
+ * attach as well as a full offline. Caller holds cbqri_domain_list_lock.
+ */
+static void cbqri_detach_cpu_from_all_ctrls(unsigned int cpu)
+{
+	static const u32 levels[] = { 2, 3 };
+	struct cbqri_resctrl_res *hw_res;
+	struct rdt_ctrl_domain *d;
+	struct rdt_resource *res;
+	struct cacheinfo *ci;
+	int i, rid;
+
+	lockdep_assert_held(&cbqri_domain_list_lock);
+
+	for (i = 0; i < ARRAY_SIZE(levels); i++) {
+		ci = get_cpu_cacheinfo_level(cpu, levels[i]);
+		if (!ci)
+			continue;
+
+		rid = cbqri_cache_level_to_rid(levels[i]);
+		hw_res = &cbqri_resctrl_resources[rid];
+		if (!hw_res->nr_ctrls)
+			continue;
+
+		res = &hw_res->resctrl_res;
+		d = cbqri_find_ctrl_domain(&res->ctrl_domains, ci->id, NULL);
+		if (!d)
+			continue;
+
+		cpumask_clear_cpu(cpu, &d->hdr.cpu_mask);
+		if (cpumask_empty(&d->hdr.cpu_mask)) {
+			list_del_rcu(&d->hdr.list);
+			synchronize_rcu();
+			resctrl_offline_ctrl_domain(res, d);
+			kfree(container_of(d, struct cbqri_resctrl_dom,
+					   resctrl_ctrl_dom));
+		}
+	}
+}
+
+/*
+ * Attach a CPU to the picked capacity controller at each cache level whose
+ * cache the CPU shares. On failure, detach the CPU from everything attached
+ * so far: the cpuhp core does not run this state's offline teardown when
+ * its startup fails, so a partial attach would otherwise leak into the
+ * domain cpu_masks. Caller holds cbqri_domain_list_lock.
+ */
+static int cbqri_attach_cpu_to_all_ctrls(unsigned int cpu)
+{
+	static const u32 levels[] = { 2, 3 };
+	struct cbqri_resctrl_res *hw_res;
+	struct cbqri_controller *ctrl;
+	struct rdt_ctrl_domain *d;
+	struct list_head *pos;
+	struct cacheinfo *ci;
+	int i, rid;
+
+	lockdep_assert_held(&cbqri_domain_list_lock);
+
+	for (i = 0; i < ARRAY_SIZE(levels); i++) {
+		ci = get_cpu_cacheinfo_level(cpu, levels[i]);
+		if (!ci)
+			continue;
+
+		rid = cbqri_cache_level_to_rid(levels[i]);
+		hw_res = &cbqri_resctrl_resources[rid];
+		if (!hw_res->nr_ctrls)
+			continue;
+
+		ctrl = cbqri_resctrl_find_ctrl(hw_res, ci->id);
+		if (!ctrl) {
+			pr_warn_once("no L%u controller for cache id %u, CPU %u is not controlled at this level\n",
+				     levels[i], ci->id, cpu);
+			continue;
+		}
+
+		d = cbqri_find_ctrl_domain(&hw_res->resctrl_res.ctrl_domains,
+					   ci->id, &pos);
+		if (d) {
+			cpumask_set_cpu(cpu, &d->hdr.cpu_mask);
+			continue;
+		}
+
+		d = cbqri_create_ctrl_domain(ctrl, &hw_res->resctrl_res, cpu,
+					     ci->id, pos);
+		if (IS_ERR(d)) {
+			cbqri_detach_cpu_from_all_ctrls(cpu);
+			return PTR_ERR(d);
+		}
+	}
+
+	return 0;
+}
+
+/* Undo cbqri_resctrl_setup(). Only reached before resctrl_init() succeeded. */
+static void cbqri_resctrl_teardown(void)
+{
+	int rid;
+
+	for (rid = 0; rid < RDT_NUM_RESOURCES; rid++) {
+		struct cbqri_resctrl_res *hw_res = &cbqri_resctrl_resources[rid];
+
+		kfree(hw_res->ctrls);
+		hw_res->ctrls = NULL;
+		hw_res->nr_ctrls = 0;
+		hw_res->cdp_enabled = false;
+		hw_res->resctrl_res.alloc_capable = false;
+		hw_res->resctrl_res.cdp_capable = false;
+	}
+}
+
+static int cbqri_resctrl_setup(void)
+{
+	int rid;
+	int err;
+
+	for (rid = 0; rid < RDT_NUM_RESOURCES; rid++)
+		cbqri_resctrl_resources[rid].resctrl_res.rid = rid;
+
+	err = cbqri_resctrl_pick_caches();
+	if (err)
+		goto teardown;
+
+	for (rid = 0; rid < RDT_NUM_RESOURCES; rid++)
+		cbqri_resctrl_control_init(&cbqri_resctrl_resources[rid]);
+
+	if (!cbqri_resctrl_any_alloc_capable()) {
+		pr_debug("no resctrl-capable CBQRI controllers found\n");
+		err = -ENODEV;
+		goto teardown;
+	}
+
+	return 0;
+
+teardown:
+	cbqri_resctrl_teardown();
+	return err;
+}
+
+static int cbqri_resctrl_online_cpu(unsigned int cpu)
+{
+	int err;
+
+	mutex_lock(&cbqri_domain_list_lock);
+	err = cbqri_attach_cpu_to_all_ctrls(cpu);
+	mutex_unlock(&cbqri_domain_list_lock);
+	if (err)
+		return err;
+
+	/*
+	 * Seed the per-CPU default RCID/MCID to the reserved (0, 0) pair and
+	 * notify the resctrl core so it tracks this CPU in the default group.
+	 */
+	resctrl_arch_set_cpu_default_closid_rmid(cpu, 0, 0);
+	resctrl_online_cpu(cpu);
+	return 0;
+}
+
+static int cbqri_resctrl_offline_cpu(unsigned int cpu)
+{
+	resctrl_offline_cpu(cpu);
+
+	mutex_lock(&cbqri_domain_list_lock);
+	cbqri_detach_cpu_from_all_ctrls(cpu);
+	mutex_unlock(&cbqri_domain_list_lock);
+	return 0;
+}
+
+/*
+ * Same order as x86: the cpuhp callbacks build the domains first and
+ * resctrl_init() runs last, so a failure here has no filesystem state
+ * to unwind and cpuhp_remove_state() takes the domains down again.
+ */
+static int __init cbqri_arch_late_init(void)
+{
+	int err, state;
+
+	if (!riscv_isa_extension_available(NULL, SSQOSID))
+		return -ENODEV;
+
+	err = cbqri_resctrl_setup();
+	if (err)
+		return err;
+
+	state = cpuhp_setup_state(CPUHP_AP_ONLINE_DYN, "cbqri:online",
+				  cbqri_resctrl_online_cpu,
+				  cbqri_resctrl_offline_cpu);
+	if (state < 0) {
+		err = state;
+		goto teardown;
+	}
+
+	err = resctrl_init();
+	if (err) {
+		cpuhp_remove_state(state);
+		goto teardown;
+	}
+
+	return 0;
+
+teardown:
+	cbqri_resctrl_teardown();
+	return err;
+}
+late_initcall(cbqri_arch_late_init);

-- 
2.43.0




More information about the linux-riscv mailing list