[PATCH bpf-next v4 2/3] riscv, bpf: Add BPF stack arguments support for RV64 JIT

Pu Lehui pulehui at huaweicloud.com
Sat Sep 5 01:15:54 PDT 2026



On 2026/9/5 16:00, bot+bpf-ci at kernel.org wrote:
>> diff --git a/arch/riscv/net/bpf_jit.h b/arch/riscv/net/bpf_jit.h
>> index 419b9d795f2a..039877f286fc 100644
>> --- a/arch/riscv/net/bpf_jit.h
>> +++ b/arch/riscv/net/bpf_jit.h
>> @@ -82,6 +82,7 @@ struct rv_jit_context {
>>   	unsigned long flags;
>>   	int stack_size;
>>   	int tcc_offset;
>> +	int stack_arg_sz;
>>   	u64 arena_vm_start;
>>   	u64 user_vm_start;
>>   };
>> diff --git a/arch/riscv/net/bpf_jit_comp64.c b/arch/riscv/net/bpf_jit_comp64.c
>> index aac128b9f0a4..cda99c1f9ffe 100644
>> --- a/arch/riscv/net/bpf_jit_comp64.c
>> +++ b/arch/riscv/net/bpf_jit_comp64.c
>> @@ -498,6 +498,18 @@ static void emit_ldx(u8 rd, s16 off, u8 rs, u8 size, bool sign_ext,
>>   	ctx->ex_jmp_off = ctx->ninsns;
>>   }
>>
>> +static void emit_stack_arg_ldx(u8 rd, s16 off, struct rv_jit_context *ctx)
>> +{
>> +	int idx = off / 8 - 1;
>> +
>> +	if (idx < 3) {
>> +		emit_mv(rd, RV_REG_A5 + idx, ctx);
>> +		return;
>> +	}
>> +
>> +	emit_ldx_insn(rd, (idx - 3) * 8, RV_REG_FP, BPF_DW, false, ctx);
>> +}
>> +
> 
> This isn't a bug, but the literal 3 appears seven times across this file
> and bpf_jit_core.c (lines 505, 510, 535, 540, 563, 568, and bpf_jit_core.c
> line 78). The value really represents RV_MAX_REG_ARGS - MAX_BPF_FUNC_REG_ARGS
> (8 - 5). The file already defines RV_MAX_REG_ARGS symbolically and uses it in
> the kfunc fix-up loop (if (idx < RV_MAX_REG_ARGS) / (idx - RV_MAX_REG_ARGS)),
> but the stack argument helpers use the literal instead.
> 
> Would a named constant (e.g. NR_STACK_ARG_REGS = RV_MAX_REG_ARGS -
> MAX_BPF_FUNC_REG_ARGS) placed in bpf_jit.h be worth it here, so the same
> value in bpf_jit_core.c and the three emit helpers cannot drift apart? arm64
> uses NR_STACK_ARG_REGS with an explicit stack_arg_reg[] table.

ok

> 
>>   static void emit_st(u8 rd, s16 off, s32 imm, u8 size, struct rv_jit_context *ctx)
>>   {
>>   	emit_imm(RV_REG_T1, imm, ctx);
>> @@ -515,6 +527,19 @@ static void emit_st(u8 rd, s16 off, s32 imm, u8 size, struct rv_jit_context *ctx
>>   	ctx->ex_jmp_off = ctx->ninsns;
>>   }
>>
>> +static void emit_stack_arg_st(s16 off, s32 imm, struct rv_jit_context *ctx)
>> +{
>> +	int idx = -off / 8 - 1;
>> +
>> +	emit_imm(RV_REG_T1, imm, ctx);
>> +	if (idx < 3) {
>> +		emit_mv(RV_REG_A5 + idx, RV_REG_T1, ctx);
>> +		return;
>> +	}
>> +
>> +	emit_stx_insn(RV_REG_SP, (idx - 3) * 8, RV_REG_T1, BPF_DW, ctx);
>> +}
>> +
> 
> This isn't a bug, but for idx < 3 (the common case, covering BPF args 6-8),
> could this be emit_imm(RV_REG_A5 + idx, imm, ctx) directly, avoiding the
> RV_REG_T1 round trip and one instruction per store? The emit_imm then emit_mv
> sequence is carried over from emit_st() where materialising into a scratch
> register is unavoidable because the destination is memory. Here in the
> register case the immediate could be materialised straight into the
> destination. The two sibling helpers (emit_stack_arg_ldx, emit_stack_arg_stx)
> have no scratch use at all, and arm64's emit_stack_arg_store() writes the
> immediate path directly into the argument register.

very nice

> 
>> +static void emit_stx(u8 rd, s16 off, u8 rs, u8 size, struct rv_jit_context *ctx)
>> +{
>> +	if (is_12b_int(off)) {
>> @@ -531,6 +556,18 @@ static void emit_stx(u8 rd, s16 off, u8 rs, u8 size, struct rv_jit_context *ctx)
>>   	ctx->ex_jmp_off = ctx->ninsns;
>>   }
>>
>> +static void emit_stack_arg_stx(s16 off, u8 rs, struct rv_jit_context *ctx)
>> +{
>> +	int idx = -off / 8 - 1;
>> +
>> +	if (idx < 3) {
>> +		emit_mv(RV_REG_A5 + idx, rs, ctx);
>> +		return;
>> +	}
>> +
>> +	emit_stx_insn(RV_REG_SP, (idx - 3) * 8, rs, BPF_DW, ctx);
>> +}
>> +
> 
> [ ... ]
> 
>> diff --git a/arch/riscv/net/bpf_jit_core.c b/arch/riscv/net/bpf_jit_core.c
>> index cbfcd287ea16..ed48005568c8 100644
>> --- a/arch/riscv/net/bpf_jit_core.c
>> +++ b/arch/riscv/net/bpf_jit_core.c
>> @@ -48,6 +48,7 @@ struct bpf_prog *bpf_int_jit_compile(struct bpf_verifier_env *env, struct bpf_pr
>>   	int pass = 0, prev_ninsns = 0, i;
>>   	struct rv_jit_data *jit_data;
>>   	struct rv_jit_context *ctx;
>> +	u16 stack_arg_cnt;
>>
>>   	if (!prog->jit_requested)
>>   		return prog;
>> @@ -71,6 +72,11 @@ struct bpf_prog *bpf_int_jit_compile(struct bpf_verifier_env *env, struct bpf_pr
>>
>>   	ctx->arena_vm_start = bpf_arena_get_kern_vm_start(prog->aux->arena);
>>   	ctx->user_vm_start = bpf_arena_get_user_vm_start(prog->aux->arena);
>> +
>> +	stack_arg_cnt = bpf_out_stack_arg_cnt(env, prog);
>> +	/* First 3 stack args in regs, rest on stack */
>> +	ctx->stack_arg_sz = stack_arg_cnt > 3 ? round_up((stack_arg_cnt - 3) * 8, STACK_ALIGN) : 0;
>> +
>>   	ctx->prog = prog;
>>   	ctx->offset = kvzalloc_objs(int, prog->len);
>>   	if (!ctx->offset)
> 
> 
> ---
> AI reviewed your patch. Please fix the bug or email reply why it's not a bug.
> See: https://github.com/kernel-patches/vmtest/blob/master/ci/claude/README.md
> 
> CI run summary: https://github.com/kernel-patches/bpf/actions/runs/33952202002




More information about the linux-riscv mailing list