summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
-rw-r--r--arch/loongarch/include/asm/inst.h25
-rw-r--r--arch/loongarch/include/asm/pgtable.h9
-rw-r--r--arch/loongarch/include/asm/ptrace.h6
-rw-r--r--arch/loongarch/kernel/acpi.c3
-rw-r--r--arch/loongarch/kernel/module-sections.c6
-rw-r--r--arch/loongarch/mm/init.c17
-rw-r--r--arch/loongarch/net/Makefile2
-rw-r--r--arch/loongarch/net/bpf_jit.c197
-rw-r--r--arch/loongarch/net/bpf_jit.h2
-rw-r--r--arch/loongarch/net/bpf_timed_may_goto.S47
-rw-r--r--tools/testing/selftests/bpf/prog_tests/kptr_xchg_inline.c3
11 files changed, 225 insertions, 92 deletions
diff --git a/arch/loongarch/include/asm/inst.h b/arch/loongarch/include/asm/inst.h
index 76b723590023..585667e361c2 100644
--- a/arch/loongarch/include/asm/inst.h
+++ b/arch/loongarch/include/asm/inst.h
@@ -97,6 +97,7 @@ enum reg2i6_op {
};
enum reg2i12_op {
+ sltui_op = 0x09,
addiw_op = 0x0a,
addid_op = 0x0b,
lu52id_op = 0x0c,
@@ -153,6 +154,8 @@ enum reg3_op {
addd_op = 0x21,
subw_op = 0x22,
subd_op = 0x23,
+ maskeqz_op = 0x26,
+ masknez_op = 0x27,
nor_op = 0x28,
and_op = 0x29,
or_op = 0x2a,
@@ -199,6 +202,10 @@ enum reg3_op {
amswaph_op = 0x70b9,
amaddb_op = 0x70ba,
amaddh_op = 0x70bb,
+ amswapdbb_op = 0x70bc,
+ amswapdbh_op = 0x70bd,
+ amadddbb_op = 0x70be,
+ amadddbh_op = 0x70bf,
amswapw_op = 0x70c0,
amswapd_op = 0x70c1,
amaddw_op = 0x70c2,
@@ -404,6 +411,7 @@ enum loongarch_gpr {
LOONGARCH_GPR_T6,
LOONGARCH_GPR_T7,
LOONGARCH_GPR_T8,
+ LOONGARCH_GPR_U0 = 21, /* Kernel per-CPU base register ($r21) */
LOONGARCH_GPR_FP = 22,
LOONGARCH_GPR_S0 = 23,
LOONGARCH_GPR_S1,
@@ -644,6 +652,7 @@ static inline void emit_##NAME(union loongarch_instruction *insn, \
insn->reg2i12_format.rj = rj; \
}
+DEF_EMIT_REG2I12_FORMAT(sltui, sltui_op)
DEF_EMIT_REG2I12_FORMAT(addiw, addiw_op)
DEF_EMIT_REG2I12_FORMAT(addid, addid_op)
DEF_EMIT_REG2I12_FORMAT(lu52id, lu52id_op)
@@ -749,6 +758,8 @@ DEF_EMIT_REG3_FORMAT(divd, divd_op)
DEF_EMIT_REG3_FORMAT(modd, modd_op)
DEF_EMIT_REG3_FORMAT(divdu, divdu_op)
DEF_EMIT_REG3_FORMAT(moddu, moddu_op)
+DEF_EMIT_REG3_FORMAT(maskeqz, maskeqz_op)
+DEF_EMIT_REG3_FORMAT(masknez, masknez_op)
DEF_EMIT_REG3_FORMAT(and, and_op)
DEF_EMIT_REG3_FORMAT(or, or_op)
DEF_EMIT_REG3_FORMAT(xor, xor_op)
@@ -783,6 +794,20 @@ DEF_EMIT_REG3_FORMAT(amswapb, amswapb_op)
DEF_EMIT_REG3_FORMAT(amswaph, amswaph_op)
DEF_EMIT_REG3_FORMAT(amswapw, amswapw_op)
DEF_EMIT_REG3_FORMAT(amswapd, amswapd_op)
+DEF_EMIT_REG3_FORMAT(amswapdbb, amswapdbb_op)
+DEF_EMIT_REG3_FORMAT(amswapdbh, amswapdbh_op)
+DEF_EMIT_REG3_FORMAT(amadddbb, amadddbb_op)
+DEF_EMIT_REG3_FORMAT(amadddbh, amadddbh_op)
+DEF_EMIT_REG3_FORMAT(amadddbw, amadddbw_op)
+DEF_EMIT_REG3_FORMAT(amadddbd, amadddbd_op)
+DEF_EMIT_REG3_FORMAT(amanddbw, amanddbw_op)
+DEF_EMIT_REG3_FORMAT(amanddbd, amanddbd_op)
+DEF_EMIT_REG3_FORMAT(amordbw, amordbw_op)
+DEF_EMIT_REG3_FORMAT(amordbd, amordbd_op)
+DEF_EMIT_REG3_FORMAT(amxordbw, amxordbw_op)
+DEF_EMIT_REG3_FORMAT(amxordbd, amxordbd_op)
+DEF_EMIT_REG3_FORMAT(amswapdbw, amswapdbw_op)
+DEF_EMIT_REG3_FORMAT(amswapdbd, amswapdbd_op)
#define DEF_EMIT_REG3SA2_FORMAT(NAME, OP) \
static inline void emit_##NAME(union loongarch_instruction *insn, \
diff --git a/arch/loongarch/include/asm/pgtable.h b/arch/loongarch/include/asm/pgtable.h
index 223528c04d73..a2191044435d 100644
--- a/arch/loongarch/include/asm/pgtable.h
+++ b/arch/loongarch/include/asm/pgtable.h
@@ -96,7 +96,7 @@ struct vm_area_struct;
#ifdef CONFIG_64BIT
#define MODULES_VADDR (vm_map_base + PCI_IOSIZE + (2 * PAGE_SIZE))
-#define MODULES_END (MODULES_VADDR + SZ_256M)
+#define MODULES_END (MODULES_VADDR + SZ_2G) /* 256MB for text, rest for data */
#ifdef CONFIG_KFENCE
#define KFENCE_AREA_SIZE (((CONFIG_KFENCE_NUM_OBJECTS + 1) * 2 + 2) * PAGE_SIZE)
@@ -125,6 +125,13 @@ struct vm_area_struct;
#endif
+/* Needed to limit get_free_mem_region() */
+#ifndef CONFIG_SPARSEMEM
+#define DIRECT_MAP_PHYSMEM_END ((1ULL << (cpu_pabits + 1)) - 1)
+#else
+#define DIRECT_MAP_PHYSMEM_END min((1ULL << (cpu_pabits + 1)) - 1, (1ULL << MAX_PHYSMEM_BITS) - 1)
+#endif
+
#define ptep_get(ptep) READ_ONCE(*(ptep))
#define pmdp_get(pmdp) READ_ONCE(*(pmdp))
diff --git a/arch/loongarch/include/asm/ptrace.h b/arch/loongarch/include/asm/ptrace.h
index e5d21e836d99..2a7ed442d905 100644
--- a/arch/loongarch/include/asm/ptrace.h
+++ b/arch/loongarch/include/asm/ptrace.h
@@ -170,11 +170,7 @@ static inline void die_if_kernel(const char *str, struct pt_regs *regs)
die(str, regs);
}
-#define current_pt_regs() \
-({ \
- unsigned long sp = (unsigned long)__builtin_frame_address(0); \
- (struct pt_regs *)((sp | (THREAD_SIZE - 1)) + 1) - 1; \
-})
+#define current_pt_regs() ((struct pt_regs *)((current_stack_pointer | (THREAD_SIZE - 1)) + 1) - 1)
/* Helpers for working with the user stack pointer */
diff --git a/arch/loongarch/kernel/acpi.c b/arch/loongarch/kernel/acpi.c
index 873e90990771..cb454ee92b20 100644
--- a/arch/loongarch/kernel/acpi.c
+++ b/arch/loongarch/kernel/acpi.c
@@ -13,6 +13,7 @@
#include <linux/export.h>
#include <linux/irq.h>
#include <linux/irqdomain.h>
+#include <linux/kvm_host.h>
#include <linux/memblock.h>
#include <linux/of_fdt.h>
#include <linux/serial_core.h>
@@ -202,7 +203,7 @@ static void __init acpi_process_madt(void)
int pptt_enabled;
static int acpi_nr_packages;
-static int acpi_package_ids[MAX_PACKAGES];
+static int acpi_package_ids[MAX(MAX_PACKAGES, KVM_MAX_VCPUS)];
int __init parse_acpi_topology(void)
{
diff --git a/arch/loongarch/kernel/module-sections.c b/arch/loongarch/kernel/module-sections.c
index 9fa1c9814fcc..0259d9c7ea30 100644
--- a/arch/loongarch/kernel/module-sections.c
+++ b/arch/loongarch/kernel/module-sections.c
@@ -62,16 +62,14 @@ Elf_Addr module_emit_plt_entry(struct module *mod, Elf_Shdr *sechdrs, Elf_Addr v
return (Elf_Addr)&plt[nr];
}
-#define cmp_3way(a, b) ((a) < (b) ? -1 : (a) > (b))
-
static int compare_rela(const void *x, const void *y)
{
int ret;
const Elf_Rela *rela_x = x, *rela_y = y;
- ret = cmp_3way(rela_x->r_info, rela_y->r_info);
+ ret = cmp_int(rela_x->r_info, rela_y->r_info);
if (ret == 0)
- ret = cmp_3way(rela_x->r_addend, rela_y->r_addend);
+ ret = cmp_int(rela_x->r_addend, rela_y->r_addend);
return ret;
}
diff --git a/arch/loongarch/mm/init.c b/arch/loongarch/mm/init.c
index 3407030f3e7a..4b46c5d30708 100644
--- a/arch/loongarch/mm/init.c
+++ b/arch/loongarch/mm/init.c
@@ -237,15 +237,26 @@ pte_t invalid_pte_table[PTRS_PER_PTE] __page_aligned_bss;
EXPORT_SYMBOL(invalid_pte_table);
#if defined(CONFIG_EXECMEM) && defined(MODULES_VADDR)
+#define MODULES_TEXT_START (MODULES_VADDR)
+#define MODULES_TEXT_END (MODULES_VADDR + SZ_256M)
+#define MODULES_DATA_START (MODULES_VADDR + SZ_256M)
+#define MODULES_DATA_END (MODULES_END)
+
static struct execmem_info execmem_info __ro_after_init;
struct execmem_info __init *execmem_arch_setup(void)
{
execmem_info = (struct execmem_info){
.ranges = {
- [EXECMEM_DEFAULT] = {
- .start = MODULES_VADDR,
- .end = MODULES_END,
+ [EXECMEM_MODULE_TEXT] = {
+ .start = MODULES_TEXT_START,
+ .end = MODULES_TEXT_END,
+ .pgprot = PAGE_KERNEL,
+ .alignment = 1,
+ },
+ [EXECMEM_MODULE_DATA] = {
+ .start = MODULES_DATA_START,
+ .end = MODULES_DATA_END,
.pgprot = PAGE_KERNEL,
.alignment = 1,
},
diff --git a/arch/loongarch/net/Makefile b/arch/loongarch/net/Makefile
index 1ec12a0c324a..8d9ddb48f9ea 100644
--- a/arch/loongarch/net/Makefile
+++ b/arch/loongarch/net/Makefile
@@ -4,4 +4,4 @@
#
# Copyright (C) 2022 Loongson Technology Corporation Limited
#
-obj-$(CONFIG_BPF_JIT) += bpf_jit.o
+obj-$(CONFIG_BPF_JIT) += bpf_jit.o bpf_timed_may_goto.o
diff --git a/arch/loongarch/net/bpf_jit.c b/arch/loongarch/net/bpf_jit.c
index 29c281bef28e..1eb588e443c9 100644
--- a/arch/loongarch/net/bpf_jit.c
+++ b/arch/loongarch/net/bpf_jit.c
@@ -8,6 +8,9 @@
#include <asm/asm-offsets.h>
#include "bpf_jit.h"
+/* DBAR hint for LL/SC completion ordering, see __WEAK_LLSC_MB */
+#define DBAR_LLSC_MB 0x700
+
#define LOONGARCH_MAX_REG_ARGS 8
#define LOONGARCH_SAVE_RA_NINSNS 1
@@ -52,50 +55,29 @@ static void prepare_bpf_tail_call_cnt(struct jit_ctx *ctx, int *store_offset)
const struct bpf_prog *prog = ctx->prog;
const bool is_main_prog = !bpf_is_subprog(prog);
+ *store_offset -= sizeof(long);
if (is_main_prog) {
- /*
- * LOONGARCH_GPR_T3 = MAX_TAIL_CALL_CNT
- * if (REG_TCC > T3 )
- * std REG_TCC -> LOONGARCH_GPR_SP + store_offset
- * else
- * std REG_TCC -> LOONGARCH_GPR_SP + store_offset
- * REG_TCC = LOONGARCH_GPR_SP + store_offset
- *
- * std REG_TCC -> LOONGARCH_GPR_SP + store_offset
- *
- * The purpose of this code is to first push the TCC into stack,
- * and then push the address of TCC into stack.
- * In cases where bpf2bpf and tailcall are used in combination,
- * the value in REG_TCC may be a count or an address,
- * these two cases need to be judged and handled separately.
- */
- emit_insn(ctx, addid, LOONGARCH_GPR_T3, LOONGARCH_GPR_ZERO, MAX_TAIL_CALL_CNT);
- *store_offset -= sizeof(long);
-
- emit_cond_jmp(ctx, BPF_JGT, REG_TCC, LOONGARCH_GPR_T3, 4);
-
- /*
- * If REG_TCC < MAX_TAIL_CALL_CNT, the value in REG_TCC is a count,
- * push tcc into stack
- */
+ /* Save entrance TCC state (scalar count or kernel pointer) to local 'tcc' slot */
emit_insn(ctx, std, REG_TCC, LOONGARCH_GPR_SP, *store_offset);
- /* Push the address of TCC into the REG_TCC */
- emit_insn(ctx, addid, REG_TCC, LOONGARCH_GPR_SP, *store_offset);
-
- emit_uncond_jmp(ctx, 2);
+ /* Compute the absolute pointer to the local 'tcc' slot */
+ emit_insn(ctx, addid, LOONGARCH_GPR_T7, LOONGARCH_GPR_SP, *store_offset);
/*
- * If REG_TCC > MAX_TAIL_CALL_CNT, the value in REG_TCC is an address,
- * push tcc_ptr into stack
+ * Branchless classification and blending:
+ * Combine interleaved inputs between a scalar count (0 to 33)
+ * and a kernel pointer address without runtime branching.
*/
- emit_insn(ctx, std, REG_TCC, LOONGARCH_GPR_SP, *store_offset);
+ emit_insn(ctx, sltui, LOONGARCH_GPR_T8, REG_TCC, MAX_TAIL_CALL_CNT + 1);
+ emit_insn(ctx, maskeqz, LOONGARCH_GPR_T7, LOONGARCH_GPR_T7, LOONGARCH_GPR_T8);
+ emit_insn(ctx, masknez, REG_TCC, REG_TCC, LOONGARCH_GPR_T8);
+ emit_insn(ctx, or, REG_TCC, REG_TCC, LOONGARCH_GPR_T7);
} else {
- *store_offset -= sizeof(long);
+ /* Subprograms: backup the verified TCC pointer inherited via REG_TCC */
emit_insn(ctx, std, REG_TCC, LOONGARCH_GPR_SP, *store_offset);
}
- /* Push tcc_ptr into stack */
+ /* Store the finalized TCC pointer value securely into the local 'tcc_ptr' slot */
*store_offset -= sizeof(long);
emit_insn(ctx, std, REG_TCC, LOONGARCH_GPR_SP, *store_offset);
}
@@ -124,6 +106,9 @@ static void prepare_bpf_tail_call_cnt(struct jit_ctx *ctx, int *store_offset)
* | tcc |
* +-------------------------+
* | tcc_ptr |
+ * +-------------------------+
+ * | arena |
+ * | (optional) |
* +-------------------------+ <--BPF_REG_FP
* | prog->aux->stack_depth |
* | (optional) |
@@ -145,7 +130,7 @@ static void build_prologue(struct jit_ctx *ctx)
stack_adjust += sizeof(long) * 2;
if (ctx->arena_vm_start)
- stack_adjust += 8;
+ stack_adjust += sizeof(long);
stack_adjust = round_up(stack_adjust, 16);
stack_adjust += bpf_stack_adjust;
@@ -194,13 +179,13 @@ static void build_prologue(struct jit_ctx *ctx)
store_offset -= sizeof(long);
emit_insn(ctx, std, LOONGARCH_GPR_S5, LOONGARCH_GPR_SP, store_offset);
+ prepare_bpf_tail_call_cnt(ctx, &store_offset);
+
if (ctx->arena_vm_start) {
store_offset -= sizeof(long);
emit_insn(ctx, std, REG_ARENA, LOONGARCH_GPR_SP, store_offset);
}
- prepare_bpf_tail_call_cnt(ctx, &store_offset);
-
emit_insn(ctx, addid, LOONGARCH_GPR_FP, LOONGARCH_GPR_SP, stack_adjust);
if (bpf_stack_adjust)
@@ -241,21 +226,18 @@ static void __build_epilogue(struct jit_ctx *ctx, bool is_tail_call)
load_offset -= sizeof(long);
emit_insn(ctx, ldd, LOONGARCH_GPR_S5, LOONGARCH_GPR_SP, load_offset);
+ /* Only restore the TCC state into REG_TCC from the higher slot */
+ load_offset -= sizeof(long);
+ emit_insn(ctx, ldd, REG_TCC, LOONGARCH_GPR_SP, load_offset);
+
+ /* Skip the unused local 'tcc_ptr' slot to align with arena */
+ load_offset -= sizeof(long);
+
if (ctx->arena_vm_start) {
load_offset -= sizeof(long);
emit_insn(ctx, ldd, REG_ARENA, LOONGARCH_GPR_SP, load_offset);
}
- /*
- * When push into the stack, follow the order of tcc then tcc_ptr.
- * When pop from the stack, first pop tcc_ptr then followed by tcc.
- */
- load_offset -= 2 * sizeof(long);
- emit_insn(ctx, ldd, REG_TCC, LOONGARCH_GPR_SP, load_offset);
-
- load_offset += sizeof(long);
- emit_insn(ctx, ldd, REG_TCC, LOONGARCH_GPR_SP, load_offset);
-
emit_insn(ctx, addid, LOONGARCH_GPR_SP, LOONGARCH_GPR_SP, stack_adjust);
if (!is_tail_call) {
@@ -290,17 +272,13 @@ bool bpf_jit_supports_far_kfunc_call(void)
static int emit_bpf_tail_call(struct jit_ctx *ctx, int insn)
{
- int off, tc_ninsn = 0;
+ int off, jmp_offset;
int tcc_ptr_off = BPF_TAIL_CALL_CNT_PTR_STACK_OFF(ctx->stack_size);
u8 a1 = LOONGARCH_GPR_A1;
u8 a2 = LOONGARCH_GPR_A2;
u8 t1 = LOONGARCH_GPR_T1;
u8 t2 = LOONGARCH_GPR_T2;
u8 t3 = LOONGARCH_GPR_T3;
- const int idx0 = ctx->idx;
-
-#define cur_offset (ctx->idx - idx0)
-#define jmp_offset (tc_ninsn - (cur_offset))
/*
* a0: &ctx
@@ -310,12 +288,12 @@ static int emit_bpf_tail_call(struct jit_ctx *ctx, int insn)
* if (index >= array->map.max_entries)
* goto out;
*/
- tc_ninsn = insn ? ctx->offset[insn+1] - ctx->offset[insn] : ctx->offset[0];
emit_zext_32(ctx, a2, true);
off = offsetof(struct bpf_array, map.max_entries);
emit_insn(ctx, ldwu, t1, a1, off);
/* bgeu $a2, $t1, jmp_offset */
+ jmp_offset = ctx->image ? (ctx->offset[insn + 1] - ctx->idx) : 0;
if (emit_tailcall_jmp(ctx, BPF_JGE, a2, t1, jmp_offset) < 0)
goto toofar;
@@ -326,6 +304,7 @@ static int emit_bpf_tail_call(struct jit_ctx *ctx, int insn)
emit_insn(ctx, ldd, REG_TCC, LOONGARCH_GPR_SP, tcc_ptr_off);
emit_insn(ctx, ldd, t3, REG_TCC, 0);
emit_insn(ctx, addid, t2, LOONGARCH_GPR_ZERO, MAX_TAIL_CALL_CNT);
+ jmp_offset = ctx->image ? (ctx->offset[insn + 1] - ctx->idx) : 0;
if (emit_tailcall_jmp(ctx, BPF_JSGE, t3, t2, jmp_offset) < 0)
goto toofar;
@@ -340,6 +319,7 @@ static int emit_bpf_tail_call(struct jit_ctx *ctx, int insn)
off = offsetof(struct bpf_array, ptrs);
emit_insn(ctx, ldd, t2, t2, off);
/* beq $t2, $zero, jmp_offset */
+ jmp_offset = ctx->image ? (ctx->offset[insn + 1] - ctx->idx) : 0;
if (emit_tailcall_jmp(ctx, BPF_JEQ, t2, LOONGARCH_GPR_ZERO, jmp_offset) < 0)
goto toofar;
@@ -355,8 +335,6 @@ static int emit_bpf_tail_call(struct jit_ctx *ctx, int insn)
toofar:
pr_info_once("tail_call: jump too far\n");
return -1;
-#undef cur_offset
-#undef jmp_offset
}
static void emit_store_stack_imm64(struct jit_ctx *ctx, int reg, int stack_off, u64 imm64)
@@ -433,7 +411,7 @@ static int emit_atomic_rmw(const struct bpf_insn *insn, struct jit_ctx *ctx)
pr_err_once("bpf-jit: amadd.b instruction is not supported\n");
return -EINVAL;
}
- emit_insn(ctx, amaddb, src, t1, t3);
+ emit_insn(ctx, amadddbb, src, t1, t3);
emit_zext_32(ctx, src, true);
break;
case BPF_H:
@@ -441,39 +419,39 @@ static int emit_atomic_rmw(const struct bpf_insn *insn, struct jit_ctx *ctx)
pr_err_once("bpf-jit: amadd.h instruction is not supported\n");
return -EINVAL;
}
- emit_insn(ctx, amaddh, src, t1, t3);
+ emit_insn(ctx, amadddbh, src, t1, t3);
emit_zext_32(ctx, src, true);
break;
case BPF_W:
- emit_insn(ctx, amaddw, src, t1, t3);
+ emit_insn(ctx, amadddbw, src, t1, t3);
emit_zext_32(ctx, src, true);
break;
case BPF_DW:
- emit_insn(ctx, amaddd, src, t1, t3);
+ emit_insn(ctx, amadddbd, src, t1, t3);
break;
}
break;
case BPF_AND | BPF_FETCH:
if (isdw) {
- emit_insn(ctx, amandd, src, t1, t3);
+ emit_insn(ctx, amanddbd, src, t1, t3);
} else {
- emit_insn(ctx, amandw, src, t1, t3);
+ emit_insn(ctx, amanddbw, src, t1, t3);
emit_zext_32(ctx, src, true);
}
break;
case BPF_OR | BPF_FETCH:
if (isdw) {
- emit_insn(ctx, amord, src, t1, t3);
+ emit_insn(ctx, amordbd, src, t1, t3);
} else {
- emit_insn(ctx, amorw, src, t1, t3);
+ emit_insn(ctx, amordbw, src, t1, t3);
emit_zext_32(ctx, src, true);
}
break;
case BPF_XOR | BPF_FETCH:
if (isdw) {
- emit_insn(ctx, amxord, src, t1, t3);
+ emit_insn(ctx, amxordbd, src, t1, t3);
} else {
- emit_insn(ctx, amxorw, src, t1, t3);
+ emit_insn(ctx, amxordbw, src, t1, t3);
emit_zext_32(ctx, src, true);
}
break;
@@ -485,7 +463,7 @@ static int emit_atomic_rmw(const struct bpf_insn *insn, struct jit_ctx *ctx)
pr_err_once("bpf-jit: amswap.b instruction is not supported\n");
return -EINVAL;
}
- emit_insn(ctx, amswapb, src, t1, t3);
+ emit_insn(ctx, amswapdbb, src, t1, t3);
emit_zext_32(ctx, src, true);
break;
case BPF_H:
@@ -493,15 +471,15 @@ static int emit_atomic_rmw(const struct bpf_insn *insn, struct jit_ctx *ctx)
pr_err_once("bpf-jit: amswap.h instruction is not supported\n");
return -EINVAL;
}
- emit_insn(ctx, amswaph, src, t1, t3);
+ emit_insn(ctx, amswapdbh, src, t1, t3);
emit_zext_32(ctx, src, true);
break;
case BPF_W:
- emit_insn(ctx, amswapw, src, t1, t3);
+ emit_insn(ctx, amswapdbw, src, t1, t3);
emit_zext_32(ctx, src, true);
break;
case BPF_DW:
- emit_insn(ctx, amswapd, src, t1, t3);
+ emit_insn(ctx, amswapdbd, src, t1, t3);
break;
}
break;
@@ -524,6 +502,7 @@ static int emit_atomic_rmw(const struct bpf_insn *insn, struct jit_ctx *ctx)
emit_insn(ctx, beq, t3, LOONGARCH_GPR_ZERO, -6);
emit_zext_32(ctx, r0, true);
}
+ emit_insn(ctx, dbar, DBAR_LLSC_MB);
break;
default:
pr_err_once("bpf-jit: invalid atomic read-modify-write opcode %02x\n", imm);
@@ -743,6 +722,15 @@ static int build_insn(const struct bpf_insn *insn, struct jit_ctx *ctx, bool ext
move_reg(ctx, dst, t1);
break;
}
+ if (insn_is_mov_percpu_addr(insn)) {
+ if (dst != src)
+ move_reg(ctx, dst, src);
+#ifdef CONFIG_SMP
+ /* dst += __my_cpu_offset, held in $r21 */
+ emit_insn(ctx, addd, dst, dst, LOONGARCH_GPR_U0);
+#endif
+ break;
+ }
switch (off) {
case 0:
move_reg(ctx, dst, src);
@@ -894,7 +882,6 @@ static int build_insn(const struct bpf_insn *insn, struct jit_ctx *ctx, bool ext
/* dst = -dst */
case BPF_ALU | BPF_NEG:
case BPF_ALU64 | BPF_NEG:
- move_imm(ctx, t1, imm, is32);
emit_insn(ctx, subd, dst, LOONGARCH_GPR_ZERO, dst);
emit_zext_32(ctx, dst, is32);
break;
@@ -1151,11 +1138,12 @@ static int build_insn(const struct bpf_insn *insn, struct jit_ctx *ctx, bool ext
/* PC += off */
case BPF_JMP | BPF_JA:
+ jmp_offset = bpf2la_offset(i, off, ctx);
+ if (emit_uncond_jmp(ctx, jmp_offset) < 0)
+ goto toofar;
+ break;
case BPF_JMP32 | BPF_JA:
- if (BPF_CLASS(code) == BPF_JMP)
- jmp_offset = bpf2la_offset(i, off, ctx);
- else
- jmp_offset = bpf2la_offset(i, imm, ctx);
+ jmp_offset = bpf2la_offset(i, imm, ctx);
if (emit_uncond_jmp(ctx, jmp_offset) < 0)
goto toofar;
break;
@@ -1204,7 +1192,13 @@ static int build_insn(const struct bpf_insn *insn, struct jit_ctx *ctx, bool ext
move_addr(ctx, t1, func_addr);
emit_insn(ctx, jirl, LOONGARCH_GPR_RA, t1, 0);
- if (insn->src_reg != BPF_PSEUDO_CALL)
+ /*
+ * Call to arch_bpf_timed_may_goto() uses a custom calling
+ * convention with the argument and return value in BPF_REG_AX,
+ * so skip moving the C return value into BPF_REG_0.
+ */
+ if (insn->src_reg != BPF_PSEUDO_CALL &&
+ func_addr != (u64)arch_bpf_timed_may_goto)
move_reg(ctx, regmap[BPF_REG_0], LOONGARCH_GPR_A0);
break;
@@ -2372,6 +2366,44 @@ void bpf_jit_free(struct bpf_prog *prog)
bpf_prog_unlock_free(prog);
}
+#if defined(CONFIG_UNWINDER_ORC)
+#include <asm/unwind.h>
+
+static noinline void walk_bpf_stackframe(bool (*consume_fn)(void *cookie, u64 ip, u64 sp, u64 bp),
+ void *cookie, unsigned long fp)
+{
+ unsigned long addr;
+ struct unwind_state state;
+ struct pt_regs dummyregs;
+ struct pt_regs *regs = &dummyregs;
+
+ regs->regs[1] = 0;
+ regs->regs[22] = fp;
+ regs->regs[3] = (unsigned long)__builtin_frame_address(0);
+ regs->csr_era = (unsigned long)__builtin_return_address(0);
+
+ for (unwind_start(&state, current, regs);
+ !unwind_done(&state); unwind_next_frame(&state)) {
+ addr = unwind_get_return_address(&state);
+ if (!addr || !consume_fn(cookie, (u64)addr, (u64)state.sp, (u64)state.fp))
+ break;
+ }
+}
+
+void arch_bpf_stack_walk(bool (*consume_fn)(void *cookie, u64 ip, u64 sp, u64 bp), void *cookie)
+{
+ unsigned long fp;
+
+ /*
+ * Capture the live frame pointer ($r22) at the very front-line before
+ * any kernel C code clobbers it. This must be a thin wrapper with no
+ * large stack locals to prevent the compiler from reusing $r22 early.
+ */
+ asm volatile("move %0, $r22" : "=r"(fp));
+ walk_bpf_stackframe(consume_fn, cookie, fp);
+}
+#endif /* CONFIG_UNWINDER_ORC */
+
bool bpf_jit_bypass_spec_v1(void)
{
return true;
@@ -2392,12 +2424,27 @@ bool bpf_jit_supports_fsession(void)
return true;
}
+bool bpf_jit_supports_percpu_insn(void)
+{
+ return true;
+}
+
+bool bpf_jit_supports_ptr_xchg(void)
+{
+ return true;
+}
+
/* Indicate the JIT backend supports mixing bpf2bpf and tailcalls. */
bool bpf_jit_supports_subprog_tailcalls(void)
{
return true;
}
+bool bpf_jit_supports_timed_may_goto(void)
+{
+ return true;
+}
+
bool bpf_jit_inlines_helper_call(s32 imm)
{
switch (imm) {
diff --git a/arch/loongarch/net/bpf_jit.h b/arch/loongarch/net/bpf_jit.h
index a8e29be35fa8..bb58c42c2f2a 100644
--- a/arch/loongarch/net/bpf_jit.h
+++ b/arch/loongarch/net/bpf_jit.h
@@ -156,7 +156,7 @@ static inline void move_imm(struct jit_ctx *ctx, enum loongarch_gpr rd, long imm
/* ori rd, $zero, imm_11_0 */
if (is_unsigned_imm12(imm)) {
emit_insn(ctx, ori, rd, LOONGARCH_GPR_ZERO, imm);
- goto zext;
+ return;
}
/* lu52id rd, $zero, imm_63_52 */
diff --git a/arch/loongarch/net/bpf_timed_may_goto.S b/arch/loongarch/net/bpf_timed_may_goto.S
new file mode 100644
index 000000000000..fa128acdb108
--- /dev/null
+++ b/arch/loongarch/net/bpf_timed_may_goto.S
@@ -0,0 +1,47 @@
+/* SPDX-License-Identifier: GPL-2.0 */
+/*
+ * Author: George Guo <guodongtai@kylinos.cn>
+ * Copyright (C) 2026 KylinSoft Corporation.
+ */
+
+#include <asm/asmmacro.h>
+#include <asm/regdef.h>
+#include <linux/export.h>
+#include <linux/linkage.h>
+
+SYM_FUNC_START(arch_bpf_timed_may_goto)
+ addi.d sp, sp, -64
+ st.d ra, sp, 56
+
+ /* Save BPF registers R0 - R5 (a5, a0 - a4) */
+ st.d a5, sp, 8
+ st.d a0, sp, 16
+ st.d a1, sp, 24
+ st.d a2, sp, 32
+ st.d a3, sp, 40
+ st.d a4, sp, 48
+
+ /*
+ * BPF_REG_AX (t0) holds the offset passed in by the verifier;
+ * add it to BPF_REG_FP (s4) to get the pointer to the count and
+ * timestamp, then pass it as the first argument in a0.
+ *
+ * The verifier emits a load using FP right before this call,
+ * so BPF_REG_FP (s4) is always set up by the JIT in this case.
+ */
+ add.d a0, t0, s4
+ bl bpf_check_timed_may_goto
+ /* BPF_REG_AX (t0) will be stored into count, so move the return value to it. */
+ move t0, a0
+
+ ld.d ra, sp, 56
+ ld.d a5, sp, 8
+ ld.d a0, sp, 16
+ ld.d a1, sp, 24
+ ld.d a2, sp, 32
+ ld.d a3, sp, 40
+ ld.d a4, sp, 48
+ addi.d sp, sp, 64
+
+ jr ra
+SYM_FUNC_END(arch_bpf_timed_may_goto)
diff --git a/tools/testing/selftests/bpf/prog_tests/kptr_xchg_inline.c b/tools/testing/selftests/bpf/prog_tests/kptr_xchg_inline.c
index 7def158da9eb..1215d6edd590 100644
--- a/tools/testing/selftests/bpf/prog_tests/kptr_xchg_inline.c
+++ b/tools/testing/selftests/bpf/prog_tests/kptr_xchg_inline.c
@@ -14,7 +14,8 @@ void test_kptr_xchg_inline(void)
int err;
#if !(defined(__x86_64__) || defined(__aarch64__) || \
- (defined(__riscv) && __riscv_xlen == 64))
+ (defined(__riscv) && __riscv_xlen == 64) || \
+ (defined(__loongarch__) && __loongarch_grlen == 64))
test__skip();
return;
#endif