Re: [PATCH dovetail v7 11/11] riscv: dovetail: add core support
Tobias Schaffner <[email protected]>
| Newsgroups | dev.linux.lists.xenomai |
|---|---|
| Message-ID | <[email protected]> |
Hi Florian, On 5/28/26 08:39, Florian Bezdeka wrote: > On Thu, 2026-04-30 at 13:11 +0200, Tobias Schaffner wrote: >> Add Dovetail co-kernel support for RISC-V, including, irq pipeline >> integration, out-of-band aware trap handling and memory management. >> >> Signed-off-by: Tobias Schaffner <[email protected]> >> --- >> arch/riscv/Kconfig | 2 + >> arch/riscv/include/asm/dovetail.h | 23 ++++++ >> arch/riscv/include/asm/mmu_context.h | 2 + >> arch/riscv/include/asm/syscall.h | 6 ++ >> arch/riscv/include/asm/thread_info.h | 8 +++ >> arch/riscv/kernel/traps.c | 102 ++++++++++++++++++++++----- >> arch/riscv/mm/context.c | 20 +++++- >> arch/riscv/mm/fault.c | 28 +++++--- >> 8 files changed, 162 insertions(+), 29 deletions(-) >> create mode 100644 arch/riscv/include/asm/dovetail.h >> >> diff --git a/arch/riscv/Kconfig b/arch/riscv/Kconfig >> index 0d8291a6c4da5..70ea8457e1f66 100644 >> --- a/arch/riscv/Kconfig >> +++ b/arch/riscv/Kconfig >> @@ -151,6 +151,7 @@ config RISCV >> select HAVE_ARCH_USERFAULTFD_WP if 64BIT && MMU && USERFAULTFD && RISCV_ISA_SVRSW60T59B >> select HAVE_ARCH_VMAP_STACK if MMU && 64BIT >> select HAVE_IRQ_PIPELINE >> + select HAVE_DOVETAIL >> select HAVE_ASM_MODVERSIONS >> select HAVE_CONTEXT_TRACKING_USER >> select HAVE_DEBUG_KMEMLEAK >> @@ -379,6 +380,7 @@ config AS_HAS_OPTION_ARCH >> >> source "arch/riscv/Kconfig.socs" >> source "arch/riscv/Kconfig.errata" >> +source "kernel/Kconfig.dovetail" >> >> menu "Platform type" >> >> diff --git a/arch/riscv/include/asm/dovetail.h b/arch/riscv/include/asm/dovetail.h >> new file mode 100644 >> index 0000000000000..ac08db99f4e92 >> --- /dev/null >> +++ b/arch/riscv/include/asm/dovetail.h >> @@ -0,0 +1,23 @@ >> +/* >> +* SPDX-License-Identifier: GPL-2.0 >> +* >> +* Copyright (C) 2024-2026 Tobias Schaffner >> +*/ >> +#ifndef _ASM_RISCV_DOVETAIL_H >> +#define _ASM_RISCV_DOVETAIL_H >> + >> +#if !defined(__ASSEMBLY__) >> +#ifdef CONFIG_DOVETAIL >> + >> +static inline void arch_dovetail_exec_prepare(void) >> +{ } >> + >> +static inline void arch_dovetail_switch_prepare(bool leave_inband) >> +{ } >> + >> +static inline void arch_dovetail_switch_finish(bool enter_inband) >> +{ } >> + >> +#endif /* CONFIG_DOVETAIL */ >> +#endif /* !__ASSEMBLY__ */ >> +#endif /* _ASM_RISCV_DOVETAIL_H */ >> diff --git a/arch/riscv/include/asm/mmu_context.h b/arch/riscv/include/asm/mmu_context.h >> index dbf27a78df6c8..a1ac9192d51c8 100644 >> --- a/arch/riscv/include/asm/mmu_context.h >> +++ b/arch/riscv/include/asm/mmu_context.h >> @@ -15,6 +15,8 @@ >> >> void switch_mm(struct mm_struct *prev, struct mm_struct *next, >> struct task_struct *task); >> +void switch_oob_mm(struct mm_struct *prev, struct mm_struct *next, >> + struct task_struct *task); >> >> #define activate_mm activate_mm >> static inline void activate_mm(struct mm_struct *prev, >> diff --git a/arch/riscv/include/asm/syscall.h b/arch/riscv/include/asm/syscall.h >> index 8067e666a4ca6..12376b54268e0 100644 >> --- a/arch/riscv/include/asm/syscall.h >> +++ b/arch/riscv/include/asm/syscall.h >> @@ -117,6 +117,12 @@ static inline bool arch_syscall_is_vdso_sigreturn(struct pt_regs *regs) >> return false; >> } >> >> +static inline unsigned long syscall_get_arg0(struct task_struct *task, >> + struct pt_regs *regs) >> +{ >> + return regs->orig_a0; >> +} >> + >> asmlinkage long sys_riscv_flush_icache(uintptr_t, uintptr_t, uintptr_t); >> >> asmlinkage long sys_riscv_hwprobe(struct riscv_hwprobe *, size_t, size_t, >> diff --git a/arch/riscv/include/asm/thread_info.h b/arch/riscv/include/asm/thread_info.h >> index 8a9e85c6e6316..8153a4f263638 100644 >> --- a/arch/riscv/include/asm/thread_info.h >> +++ b/arch/riscv/include/asm/thread_info.h >> @@ -41,6 +41,7 @@ >> >> #include <asm/processor.h> >> #include <asm/csr.h> >> +#include <dovetail/thread_info.h> >> >> /* >> * low level task data that entry.S needs immediate access to >> @@ -80,6 +81,7 @@ struct thread_info { >> #ifdef CONFIG_RISCV_USER_CFI >> struct cfi_state user_cfi_state; >> #endif >> + struct oob_thread_state oob_state; /* co-kernel thread state */ >> }; >> >> #ifdef CONFIG_SHADOW_CALL_STACK >> @@ -123,6 +125,9 @@ int arch_dup_task_struct(struct task_struct *dst, struct task_struct *src); >> >> #include <asm-generic/thread_info_tif.h> >> >> +#define TIF_MAYDAY 14 /* emergency trap pending */ >> +#define _TIF_MAYDAY (1 << TIF_MAYDAY) >> + >> #define TIF_32BIT 16 /* compat-mode 32bit process */ >> #define TIF_RISCV_V_DEFER_RESTORE 17 /* restore Vector before returing to user */ >> >> @@ -132,5 +137,8 @@ int arch_dup_task_struct(struct task_struct *dst, struct task_struct *src); >> * Local (synchronous) thread flags. >> */ >> #define _TLF_OOB 0x0001 >> +#define _TLF_DOVETAIL 0x0002 >> +#define _TLF_OFFSTAGE 0x0004 >> +#define _TLF_OOBTRAP 0x0008 >> >> #endif /* _ASM_RISCV_THREAD_INFO_H */ >> diff --git a/arch/riscv/kernel/traps.c b/arch/riscv/kernel/traps.c >> index b37bb5da87c5c..1f7d471457892 100644 >> --- a/arch/riscv/kernel/traps.c >> +++ b/arch/riscv/kernel/traps.c >> @@ -115,26 +115,47 @@ void die(struct pt_regs *regs, const char *str) >> static __always_inline >> bool mark_trap_entry(int signo, struct pt_regs *regs) >> { >> + oob_trap_notify(signo, regs); >> + >> if (likely(running_inband())) { >> hard_cond_local_irq_enable(); >> return true; >> } >> >> + oob_trap_unwind(signo, regs); >> + >> return false; >> } >> >> static __always_inline >> void mark_trap_exit(int signo, struct pt_regs *regs) >> { >> + oob_trap_unwind(signo, regs); >> hard_cond_local_irq_disable(); >> } >> >> -void do_trap(struct pt_regs *regs, int signo, int code, unsigned long addr) >> +static __always_inline >> +bool mark_trap_entry_raw(int trapnr, struct pt_regs *regs) >> { >> - struct task_struct *tsk = current; >> + oob_trap_notify(trapnr, regs); >> >> - if (!mark_trap_entry(signo, regs)) >> - return; >> + if (running_oob()) { >> + oob_trap_unwind(trapnr, regs); >> + return false; >> + } >> + >> + return true; >> +} > > I think that it should be possible to merge mark_trap_{entry,exit}_raw() > with the non-raw variant, so that only mark_trap_{entry,exit} would > stay. > > The _raw variant is used for traps over kernel-, while non-raw is used > for traps over user-space. > > We could check for user_mode() and handle that within one function. > > But: See do_trap_break(), break point handling. I think you're right. Now that we do the hard irq enable/disablement anyways when comming from user mode, we should be able to merge this entirely. I'll do some testing and create a v8. >> + >> +static __always_inline >> +void mark_trap_exit_raw(int trapnr, struct pt_regs *regs) >> +{ >> + oob_trap_unwind(trapnr, regs); >> +} >> + >> +static void do_trap_raw(struct pt_regs *regs, int signo, int code, unsigned long addr) >> +{ >> + struct task_struct *tsk = current; >> >> if (show_unhandled_signals && unhandled_signal(tsk, signo) >> && printk_ratelimit()) { >> @@ -147,6 +168,14 @@ void do_trap(struct pt_regs *regs, int signo, int code, unsigned long addr) >> } >> >> force_sig_fault(signo, code, (void __user *)addr); >> +} >> + >> +void do_trap(struct pt_regs *regs, int signo, int code, unsigned long addr) >> +{ >> + if(!mark_trap_entry(signo, regs)) >> + return; >> + >> + do_trap_raw(regs, signo, code, addr); >> >> mark_trap_exit(signo, regs); >> } >> @@ -157,7 +186,7 @@ static void do_trap_error(struct pt_regs *regs, int signo, int code, >> current->thread.bad_cause = regs->cause; >> >> if (user_mode(regs)) { >> - do_trap(regs, signo, code, addr); >> + do_trap_raw(regs, signo, code, addr); >> } else { >> /* >> * Dovetail: If we trapped from kernel space, either >> @@ -175,37 +204,46 @@ static void do_trap_error(struct pt_regs *regs, int signo, int code, >> #else >> #define __trap_section noinstr >> #endif >> -#define DO_ERROR_INFO(name, signo, code, str) \ >> +#define DO_ERROR_INFO(name, signo, code, str, trapnr) \ >> asmlinkage __visible __trap_section void name(struct pt_regs *regs) \ >> { \ >> if (user_mode(regs)) { \ >> + if(!mark_trap_entry(trapnr, regs)) \ >> + return; \ >> irqentry_enter_from_user_mode(regs); \ >> local_irq_enable(); \ >> do_trap_error(regs, signo, code, regs->epc, "Oops - " str); \ >> local_irq_disable(); \ >> irqentry_exit_to_user_mode(regs); \ >> + mark_trap_exit(trapnr, regs); \ >> } else { \ >> + if(!mark_trap_entry_raw(trapnr, regs)) \ >> + return; \ >> int stalled = test_and_stall_inband_nocheck(); \ >> irqentry_state_t state = irqentry_nmi_enter(regs); \ >> do_trap_error(regs, signo, code, regs->epc, "Oops - " str); \ >> irqentry_nmi_exit(regs, state); \ >> if (!stalled) \ >> unstall_inband_nocheck(); \ >> + mark_trap_exit_raw(trapnr, regs); \ >> } \ >> } >> >> DO_ERROR_INFO(do_trap_unknown, >> - SIGILL, ILL_ILLTRP, "unknown exception"); >> + SIGILL, ILL_ILLTRP, "unknown exception", regs->cause); >> DO_ERROR_INFO(do_trap_insn_misaligned, >> - SIGBUS, BUS_ADRALN, "instruction address misaligned"); >> + SIGBUS, BUS_ADRALN, "instruction address misaligned", EXC_INST_MISALIGNED); >> DO_ERROR_INFO(do_trap_insn_fault, >> - SIGSEGV, SEGV_ACCERR, "instruction access fault"); >> + SIGSEGV, SEGV_ACCERR, "instruction access fault", EXC_INST_ACCESS); >> >> asmlinkage __visible __trap_section void do_trap_insn_illegal(struct pt_regs *regs) >> { >> bool handled; >> >> if (user_mode(regs)) { >> + if(!mark_trap_entry(EXC_INST_ILLEGAL, regs)) >> + return; >> + >> irqentry_enter_from_user_mode(regs); >> unstall_inband_nocheck(); >> hard_local_irq_enable(); >> @@ -220,7 +258,12 @@ asmlinkage __visible __trap_section void do_trap_insn_illegal(struct pt_regs *re >> "Oops - illegal instruction"); >> >> irqentry_exit_to_user_mode(regs); >> + >> + mark_trap_exit(EXC_INST_ILLEGAL, regs); >> } else { >> + if(!mark_trap_entry_raw(EXC_INST_ILLEGAL, regs)) >> + return; >> + >> int stalled = test_and_stall_inband_nocheck(); >> irqentry_state_t state = irqentry_nmi_enter(regs); >> >> @@ -230,11 +273,13 @@ asmlinkage __visible __trap_section void do_trap_insn_illegal(struct pt_regs *re >> irqentry_nmi_exit(regs, state); >> if (!stalled) >> unstall_inband_nocheck(); >> + >> + mark_trap_exit_raw(EXC_INST_ILLEGAL, regs); >> } >> } >> >> DO_ERROR_INFO(do_trap_load_fault, >> - SIGSEGV, SEGV_ACCERR, "load access fault"); >> + SIGSEGV, SEGV_ACCERR, "load access fault", EXC_LOAD_ACCESS); >> >> enum misaligned_access_type { >> MISALIGNED_STORE, >> @@ -254,15 +299,22 @@ static const struct { >> }, >> }; >> >> -static void do_trap_misaligned(struct pt_regs *regs, enum misaligned_access_type type) >> +static void do_trap_misaligned(struct pt_regs *regs, enum misaligned_access_type type, >> + int trapnr) >> { >> irqentry_state_t state; >> int stalled; >> >> if (user_mode(regs)) { >> + if(!mark_trap_entry(trapnr, regs)) >> + return; >> + >> irqentry_enter_from_user_mode(regs); >> local_irq_enable(); >> } else { >> + if(!mark_trap_entry_raw(trapnr, regs)) >> + return; >> + >> stalled = test_and_stall_inband_nocheck(); >> state = irqentry_nmi_enter(regs); >> } >> @@ -274,29 +326,32 @@ static void do_trap_misaligned(struct pt_regs *regs, enum misaligned_access_type >> if (user_mode(regs)) { >> local_irq_disable(); >> irqentry_exit_to_user_mode(regs); >> + mark_trap_exit(trapnr, regs); >> } else { >> irqentry_nmi_exit(regs, state); >> if (!stalled) >> unstall_inband_nocheck(); >> + >> + mark_trap_exit_raw(trapnr, regs); >> } >> } >> >> asmlinkage __visible __trap_section void do_trap_load_misaligned(struct pt_regs *regs) >> { >> - do_trap_misaligned(regs, MISALIGNED_LOAD); >> + do_trap_misaligned(regs, MISALIGNED_LOAD, EXC_LOAD_MISALIGNED); >> } >> >> asmlinkage __visible __trap_section void do_trap_store_misaligned(struct pt_regs *regs) >> { >> - do_trap_misaligned(regs, MISALIGNED_STORE); >> + do_trap_misaligned(regs, MISALIGNED_STORE, EXC_STORE_MISALIGNED); >> } >> >> DO_ERROR_INFO(do_trap_store_fault, >> - SIGSEGV, SEGV_ACCERR, "store (or AMO) access fault"); >> + SIGSEGV, SEGV_ACCERR, "store (or AMO) access fault", EXC_STORE_ACCESS); >> DO_ERROR_INFO(do_trap_ecall_s, >> - SIGILL, ILL_ILLTRP, "environment call from S-mode"); >> + SIGILL, ILL_ILLTRP, "environment call from S-mode", EXC_SYSCALL); >> DO_ERROR_INFO(do_trap_ecall_m, >> - SIGILL, ILL_ILLTRP, "environment call from M-mode"); >> + SIGILL, ILL_ILLTRP, "environment call from M-mode", EXC_SUPERVISOR_SYSCALL); >> >> static inline unsigned long get_break_insn_length(unsigned long pc) >> { >> @@ -348,6 +403,9 @@ void handle_break(struct pt_regs *regs) >> >> asmlinkage __visible __trap_section void do_trap_break(struct pt_regs *regs) >> { >> + if(!mark_trap_entry_raw(EXC_BREAKPOINT, regs)) >> + return; >> + >> if (user_mode(regs)) { >> irqentry_enter_from_user_mode(regs); >> unstall_inband_nocheck(); >> @@ -369,6 +427,8 @@ asmlinkage __visible __trap_section void do_trap_break(struct pt_regs *regs) >> if (!stalled) >> unstall_inband_nocheck(); >> } >> + >> + mark_trap_exit_raw(EXC_BREAKPOINT, regs); >> } > > This is the only trap - or more specific: caller of _raw() - that is > different. Why? > > All other traps are using _raw() for the kernel- and non-raw for user- > space part. > >> >> asmlinkage __visible __trap_section __no_stack_protector >> @@ -385,6 +445,15 @@ void do_trap_ecall_u(struct pt_regs *regs) >> >> syscall = syscall_enter_from_user_mode(regs, syscall); >> >> + if(dovetailing()) { >> + if (syscall == EXIT_SYSCALL_OOB) { >> + hard_local_irq_disable(); >> + return; >> + } >> + if (syscall == EXIT_SYSCALL_TAIL) >> + goto done_inband; >> + } >> + >> add_random_kstack_offset(); >> >> if (syscall >= 0 && syscall < NR_syscalls) { >> @@ -404,6 +473,7 @@ void do_trap_ecall_u(struct pt_regs *regs) >> */ >> choose_random_kstack_offset(get_random_u16()); >> >> +done_inband: >> syscall_exit_to_user_mode(regs); >> } else { >> int stalled = test_and_stall_inband_nocheck(); >> diff --git a/arch/riscv/mm/context.c b/arch/riscv/mm/context.c >> index 55c20ad1f7444..8f38b4a5eb262 100644 >> --- a/arch/riscv/mm/context.c >> +++ b/arch/riscv/mm/context.c >> @@ -25,7 +25,7 @@ static unsigned long num_asids; >> >> static atomic_long_t current_version; >> >> -static DEFINE_RAW_SPINLOCK(context_lock); >> +static DEFINE_HARD_SPINLOCK(context_lock); >> static cpumask_t context_tlb_flush_pending; >> static unsigned long *context_asid_map; >> >> @@ -315,7 +315,7 @@ static inline void flush_icache_deferred(struct mm_struct *mm, unsigned int cpu, >> #endif >> } >> >> -void switch_mm(struct mm_struct *prev, struct mm_struct *next, >> +static void do_switch_mm(struct mm_struct *prev, struct mm_struct *next, >> struct task_struct *task) >> { >> unsigned int cpu; >> @@ -336,3 +336,19 @@ void switch_mm(struct mm_struct *prev, struct mm_struct *next, >> >> flush_icache_deferred(next, cpu, task); >> } >> + >> +void switch_mm(struct mm_struct *prev, struct mm_struct *next, >> + struct task_struct *task) >> +{ >> + unsigned long flags; >> + >> + protect_inband_mm(flags); >> + do_switch_mm(prev, next, task); >> + unprotect_inband_mm(flags); >> +} >> + >> +void switch_oob_mm(struct mm_struct *prev, struct mm_struct *next, >> + struct task_struct *task) >> +{ >> + do_switch_mm(prev, next, task); >> +} >> diff --git a/arch/riscv/mm/fault.c b/arch/riscv/mm/fault.c >> index 885e2f66515e3..32cb7285d6db8 100644 >> --- a/arch/riscv/mm/fault.c >> +++ b/arch/riscv/mm/fault.c >> @@ -7,6 +7,7 @@ >> */ >> >> >> +#include <linux/preempt.h> >> #include <linux/mm.h> >> #include <linux/kernel.h> >> #include <linux/interrupt.h> >> @@ -315,8 +316,12 @@ void handle_page_fault(struct pt_regs *regs) >> return; >> } >> >> + oob_trap_notify(cause, regs); >> + if (!running_inband()) >> + goto out; >> + >> /* Enable interrupts if they were enabled in the parent context. */ >> - if (!regs_irqs_disabled(regs) && running_inband()) >> + if (!regs_irqs_disabled(regs)) >> local_irq_enable_full(); >> >> /* >> @@ -326,7 +331,7 @@ void handle_page_fault(struct pt_regs *regs) >> if (unlikely(faulthandler_disabled() || !mm)) { >> tsk->thread.bad_cause = cause; >> no_context(regs, addr); >> - return; >> + goto out; >> } >> >> if (user_mode(regs)) >> @@ -334,7 +339,7 @@ void handle_page_fault(struct pt_regs *regs) >> >> if (!user_mode(regs) && addr < TASK_SIZE && unlikely(!(regs->status & SR_SUM))) { >> if (fixup_exception(regs)) >> - return; >> + goto out; >> >> die_kernel_fault("access to user memory without uaccess routines", addr, regs); >> } >> @@ -357,7 +362,7 @@ void handle_page_fault(struct pt_regs *regs) >> count_vm_vma_lock_event(VMA_LOCK_SUCCESS); >> tsk->thread.bad_cause = cause; >> bad_area_nosemaphore(regs, SEGV_ACCERR, addr); >> - return; >> + goto out; >> } >> >> fault = handle_mm_fault(vma, addr, flags | FAULT_FLAG_VMA_LOCK, regs); >> @@ -375,7 +380,7 @@ void handle_page_fault(struct pt_regs *regs) >> if (fault_signal_pending(fault, regs)) { >> if (!user_mode(regs)) >> no_context(regs, addr); >> - return; >> + goto out; >> } >> lock_mmap: >> >> @@ -384,7 +389,7 @@ void handle_page_fault(struct pt_regs *regs) >> if (unlikely(!vma)) { >> tsk->thread.bad_cause = cause; >> bad_area_nosemaphore(regs, code, addr); >> - return; >> + goto out; >> } >> >> /* >> @@ -396,7 +401,7 @@ void handle_page_fault(struct pt_regs *regs) >> if (unlikely(access_error(cause, vma))) { >> tsk->thread.bad_cause = cause; >> bad_area(regs, mm, code, addr); >> - return; >> + goto out; >> } >> >> /* >> @@ -414,12 +419,12 @@ void handle_page_fault(struct pt_regs *regs) >> if (fault_signal_pending(fault, regs)) { >> if (!user_mode(regs)) >> no_context(regs, addr); >> - return; >> + goto out; >> } >> >> /* The fault is fully completed (including releasing mmap lock) */ >> if (fault & VM_FAULT_COMPLETED) >> - return; >> + goto out; >> >> if (unlikely(fault & VM_FAULT_RETRY)) { >> flags |= FAULT_FLAG_TRIED; >> @@ -438,7 +443,8 @@ void handle_page_fault(struct pt_regs *regs) >> if (unlikely(fault & VM_FAULT_ERROR)) { >> tsk->thread.bad_cause = cause; >> mm_fault_error(regs, addr, fault); >> - return; >> } >> - return; >> + >> +out: >> + oob_trap_unwind(cause, regs); >> } >> -- >> 2.43.0