[PATCH v8 14/22] RISC-V: perf: Implement supervisor counter delegation support
Atish Patra
atish.patra at linux.dev
Thu Aug 6 14:45:03 PDT 2026
On 8/6/26 5:16 AM, Yicong Yang wrote:
> On 8/6/26 10:00 AM, Atish Patra wrote:
>> On 7/7/26 1:24 AM, Yicong Yang wrote:
>>> On 7/1/26 4:47 PM, Atish Patra wrote:
>>>> From: Atish Patra <atishp at rivosinc.com>
>>>>
>>>> There are few new RISC-V ISA exensions (ssccfg, sscsrind, smcntrpmf) which
>>>> allows the hpmcounter/hpmevents to be programmed directly from S-mode. The
>>>> implementation detects the ISA extension at runtime and uses them if
>>>> available instead of SBI PMU extension. SBI PMU extension will still be
>>>> used for firmware counters if the user requests it.
>>>>
>>>> The current linux driver relies on event encoding defined by SBI PMU
>>>> specification for standard perf events. However, there are no standard
>>>> event encoding available in the ISA. In the future, we may want to
>>>> decouple the counter delegation and SBI PMU completely. In that case,
>>>> counter delegation supported platforms must rely on the event encoding
>>>> defined in the perf json file or in the pmu driver.
>>>>
>>>> For firmware events, it will continue to use the SBI PMU encoding as
>>>> one can not support firmware event without SBI PMU.
>>>>
>>>> Signed-off-by: Atish Patra <atishp at rivosinc.com>
>>>> ---
>>>> arch/riscv/include/asm/csr.h | 1 +
>>>> drivers/perf/riscv_pmu_sbi.c | 578 +++++++++++++++++++++++++++++++++--------
>>>> include/linux/perf/riscv_pmu.h | 3 +
>>>> 3 files changed, 478 insertions(+), 104 deletions(-)
>>>>
>>>> diff --git a/arch/riscv/include/asm/csr.h b/arch/riscv/include/asm/csr.h
>>>> index a3b24b88e401..cd22b5168689 100644
>>>> --- a/arch/riscv/include/asm/csr.h
>>>> +++ b/arch/riscv/include/asm/csr.h
>>>> @@ -258,6 +258,7 @@
>>>> #endif
>>>> #define SISELECT_SSCCFG_BASE 0x40
>>>> +#define HPMEVENT_MASK GENMASK_ULL(63, 56)
>>>> /* mseccfg bits */
>>>> #define MSECCFG_PMM ENVCFG_PMM
>>>> diff --git a/drivers/perf/riscv_pmu_sbi.c b/drivers/perf/riscv_pmu_sbi.c
>>>> index 2568c6808f5d..7995da4a98a1 100644
>>>> --- a/drivers/perf/riscv_pmu_sbi.c
>>>> +++ b/drivers/perf/riscv_pmu_sbi.c
>>>> @@ -28,6 +28,8 @@
>>>> #include <asm/cpufeature.h>
>>>> #include <asm/vendor_extensions.h>
>>>> #include <asm/vendor_extensions/andes.h>
>>>> +#include <asm/hwcap.h>
>>>> +#include <asm/csr_ind.h>
>>>> #define ALT_SBI_PMU_OVERFLOW(__ovl) \
>>>> asm volatile(ALTERNATIVE_2( \
>>>> @@ -60,7 +62,20 @@ asm volatile(ALTERNATIVE( \
>>>> #define PERF_EVENT_FLAG_USER_ACCESS BIT(SYSCTL_USER_ACCESS)
>>>> #define PERF_EVENT_FLAG_LEGACY BIT(SYSCTL_LEGACY)
>>>> -PMU_FORMAT_ATTR(event, "config:0-55");
>>>> +#define RVPMU_SBI_PMU_FORMAT_ATTR "config:0-47"
>>>> +#define RVPMU_CDELEG_PMU_FORMAT_ATTR "config:0-55"
>>>> +
>>>> +static ssize_t __maybe_unused rvpmu_format_show(struct device *dev, struct device_attribute *attr,
>>>> + char *buf);
>>>> +
>>>> +#define RVPMU_ATTR_ENTRY(_name, _func, _config) ( \
>>>> + &((struct dev_ext_attribute[]) { \
>>>> + { __ATTR(_name, 0444, _func, NULL), (void *)_config } \
>>>> + })[0].attr.attr)
>>>> +
>>>> +#define RVPMU_FORMAT_ATTR_ENTRY(_name, _config) \
>>>> + RVPMU_ATTR_ENTRY(_name, rvpmu_format_show, (char *)_config)
>>>> +
>>>> PMU_FORMAT_ATTR(firmware, "config:62-63");
>>>> static bool sbi_v2_available;
>>>> @@ -68,7 +83,11 @@ static bool sbi_v3_available;
>>>> static DEFINE_STATIC_KEY_FALSE(sbi_pmu_snapshot_available);
>>>> #define sbi_pmu_snapshot_available() \
>>>> static_branch_unlikely(&sbi_pmu_snapshot_available)
>>>> +
>>>> static DEFINE_STATIC_KEY_FALSE(riscv_pmu_sbi_available);
>>>> +#define riscv_pmu_sbi_available() \
>>>> + static_branch_likely(&riscv_pmu_sbi_available)
>>>> +
>>>> static DEFINE_STATIC_KEY_FALSE(riscv_pmu_cdeleg_available);
>>>> /* Avoid unnecessary code patching in the one time booting path*/
>>>> @@ -83,19 +102,35 @@ static DEFINE_STATIC_KEY_FALSE(riscv_pmu_cdeleg_available);
>>>> #define riscv_pmu_sbi_available() \
>>>> static_branch_likely(&riscv_pmu_sbi_available)
>>>> -static struct attribute *riscv_arch_formats_attr[] = {
>>>> - &format_attr_event.attr,
>>>> +static struct attribute *riscv_sbi_pmu_formats_attr[] = {
>>>> + RVPMU_FORMAT_ATTR_ENTRY(event, RVPMU_SBI_PMU_FORMAT_ATTR),
>>>> &format_attr_firmware.attr,
>>>> NULL,
>>>> };
>>>> -static struct attribute_group riscv_pmu_format_group = {
>>>> +static struct attribute_group riscv_sbi_pmu_format_group = {
>>>> .name = "format",
>>>> - .attrs = riscv_arch_formats_attr,
>>>> + .attrs = riscv_sbi_pmu_formats_attr,
>>>> };
>>>> -static const struct attribute_group *riscv_pmu_attr_groups[] = {
>>>> - &riscv_pmu_format_group,
>>>> +static const struct attribute_group *riscv_sbi_pmu_attr_groups[] = {
>>>> + &riscv_sbi_pmu_format_group,
>>>> + NULL,
>>>> +};
>>>> +
>>>> +static struct attribute *riscv_cdeleg_pmu_formats_attr[] = {
>>>> + RVPMU_FORMAT_ATTR_ENTRY(event, RVPMU_CDELEG_PMU_FORMAT_ATTR),
>>>> + &format_attr_firmware.attr,
>>>> + NULL,
>>>> +};
>>>> +
>>>> +static struct attribute_group riscv_cdeleg_pmu_format_group = {
>>>> + .name = "format",
>>>> + .attrs = riscv_cdeleg_pmu_formats_attr,
>>>> +};
>>>> +
>>>> +static const struct attribute_group *riscv_cdeleg_pmu_attr_groups[] = {
>>>> + &riscv_cdeleg_pmu_format_group,
>>>> NULL,
>>>> };
>>>> @@ -482,6 +517,14 @@ static void rvpmu_sbi_check_std_events(struct work_struct *work)
>>>> static DECLARE_WORK(check_std_events_work, rvpmu_sbi_check_std_events);
>>>> +static ssize_t rvpmu_format_show(struct device *dev,
>>>> + struct device_attribute *attr, char *buf)
>>>> +{
>>>> + struct dev_ext_attribute *eattr = container_of(attr,
>>>> + struct dev_ext_attribute, attr);
>>>> + return sysfs_emit(buf, "%s\n", (char *)eattr->var);
>>>> +}
>>>> +
>>>> static int rvpmu_ctr_get_width(int idx)
>>>> {
>>>> return pmu_ctr_list[idx].width;
>>>> @@ -599,6 +642,38 @@ static uint8_t rvpmu_csr_index(struct perf_event *event)
>>>> return pmu_ctr_list[event->hw.idx].csr - CSR_CYCLE;
>>>> }
>>>> +static uint64_t get_deleg_priv_filter_bits(struct perf_event *event)
>>>> +{
>>>> + u64 priv_filter_bits = 0;
>>> could we explicitly initialize the priv_filter_bits to MINH? though this bit
>>> is S-mode read-only and won't have any effects, but considering the semantic
>>> is to inhibit counting at certain privilege mode, initialize it to MINH will
>>> make it clear.
>> Wouldn't that be confusing ? I feel we need a comment there as well to explain why
>> are setting MINH in S-mode if we need that.
>>
> sounds reasonable. let's keep it as is.
>
>>>> + bool guest_events = false;
>>>> +
>>>> + if (event->attr.config1 & RISCV_PMU_CONFIG1_GUEST_EVENTS)
>>>> + guest_events = true;
>>> use attr::config1 to distinguish the KVM events looks incorrect. it's a user
>>> visible interface and could be set explicitly by the user and break the
>>> logic here.. e.g. perf stat -e cycles/config1=0x1/
>> Reusing config1 for kvm guest events are already in upstream. As KVM is just another user of host perf driver, it is used. The config is event specific so a user specific event vs kvm initiated event would be separate anyways.
>>
>> But we can harden the interface with additional check using this which was exclusive to perf_event_create_kernel_counter which
>> kvm invokes[1].
>>
>> event->owner = TASK_TOMBSTONE;
>>
>> [1] https://github.com/torvalds/linux/blob/master/kernel/events/core.c#L14324
>>
> makes sense. there's a is_kernel_event() wrapper but currently is local
> there in the event core (so is TASK_TOMBSTONE).
Correct.
> but yes it's not introduced by this patch.
>
>>>> + if (event->attr.exclude_kernel)
>>>> + priv_filter_bits |= guest_events ? HPMEVENT_VSINH : HPMEVENT_SINH;
>>>> + if (event->attr.exclude_user)
>>>> + priv_filter_bits |= guest_events ? HPMEVENT_VUINH : HPMEVENT_UINH;
>>>> + if (guest_events && event->attr.exclude_hv)
>>>> + priv_filter_bits |= HPMEVENT_SINH;
>>>> + if (event->attr.exclude_host)
>>>> + priv_filter_bits |= HPMEVENT_UINH | HPMEVENT_SINH;
>>>> + if (event->attr.exclude_guest)
>>>> + priv_filter_bits |= HPMEVENT_VSINH | HPMEVENT_VUINH;
>>>> +
>>>> + return priv_filter_bits;
>>>> +}
>>>> +
> [...]
>
>>>> +static void rvpmu_deleg_ctr_enable_irq(struct perf_event *event)
>>>> +{
>>>> + unsigned long hpmevent_curr;
>>>> + unsigned long of_mask;
>>>> + struct hw_perf_event *hwc = &event->hw;
>>>> + int counter_idx = hwc->idx;
>>>> + unsigned long sip_val = csr_read(CSR_SIP);
>>>> +
>>>> + if (!is_sampling_event(event) || (sip_val & SIP_LCOFIP))
>>> we should also consider overflow for counters of non-sampling events..
>> This is intentional. As per the sscofpmf, OF bits as a interrupt disable and overflow indicator.
>> So we need to clear the bit if you need the interrupt for the sampled events.
>>
>> For non sampled events, that is not necessary. But If it overflows, it shouldn't raise an interrupt.
>>
> okay I see the existing interrupt handler skip the non-sample events as well
> so it's the current policy of the driver. but generally we should handle
> the overflow of non sampled events as well to avoid count lost if the counter
> is implemened less than 64bit. in such case it's possible to overflow
> several times for a long counting period.
The wrap is already accounted for: riscv_pmu_event_update() recovers the
delta with (new - prev) & cmask, which is correct for any advance short of a full lap,
and the half-width preload in riscv_pmu_event_init() keeps prev_count away from the
wrap point. This common design across architectures. If a platform implement too narrow
counter width such that overflow several times for a long counting period without a single
context switch, that's a platform problem.
> it's also another thing not introduced by this patch so we may keep the handling
> here as is.
>
> thanks.
More information about the linux-arm-kernel
mailing list