mirror of
https://git.kernel.org/pub/scm/linux/kernel/git/stable/linux.git
synced 2026-08-09 06:14:34 +02:00
Like other architectures such as x86, arm64, riscv, powerpc and s390, select THREAD_INFO_IN_TASK for LoongArch to move thread_info off the stack into task_struct. This follows modern kernel standards and also makes the system more secure. With this patch, thread_info is included in task_struct at an offset of 0 instead of being placed at the bottom of the kernel stack. Thus, the $tp register points to both thread_info and task_struct. To support this, introduce a per-CPU variable cpu_tasks to store the pointer to the current task_struct. This decouples the recovery of the $tp register from the stack pointer during exception entry. Then initialize cpu_tasks for the primary and secondary CPUs during arch-specific setup and SMP boot paths. To eliminate the dangerous windows during the early initialization where the cpu_tasks remains uninitialized, set_current() is invoked as early as possible in both setup_arch() and start_secondary(). This ensures the $tp recovery barrier is armed in case any early boot exceptions or kernel panics occur. Modify SAVE_SOME and handle_syscall to restore the $tp register from cpu_tasks, and also use the la_abs absolute addressing for cpu_tasks access in assembly to bypass the relocation limits within exception handling sections. By advancing the preservation of u0 in SAVE_SOME, we reuse the PERCPU_BASE_KS value in u0 for the cpu_tasks calculation, effectively eliminating a duplicate csrrd instruction execution on SMP platforms. Update <asm/switch_to.h> and <kernel/switch.S> to fully support the CONFIG_THREAD_INFO_IN_TASK feature. Remove the obsolete next_ti argument from __switch_to(), which shifts the remaining arguments ahead in the calling convention (sched_ra from a3 to a2, and sched_cfa from a4 to a3). Under the new configuration, __switch_to() now directly derives the thread pointer ($tp) from the next task_struct pointer in a1. To preserve the optimal and clean "move tp, a1" path for 64-bit kernels, the thread pointer ($tp) is assigned directly from a1 in the core path. For 32-bit kernels, where a1 carries a 2000-byte structural pointer bias at entry, an explicit adjustment "PTR_ADDI tp, tp, -TASK_STRUCT_OFFSET" is introduced at the function exit. In the context of __switch_to(), local interrupts are disabled, and the kernel is in a critical switching phase where handling any synchronous exception is practically impossible and prohibited. If any synchronous exception or watchpoint does trigger in this narrow window, it constitutes a fatal double fault and the kernel is expected to die/panic immediately anyway. Therefore, the temporary biased value in $tp is safe and acceptable here. Additionally, evaluate the stack lookup as a single load instruction "LONG_LPTR t0, a1, (TASK_STACK - TASK_STRUCT_OFFSET)", this perfectly satisfies both 32-bit and 64-bit kernels. Using the "next" pointer in a1 as the base register, rather than $tp, effectively unchains the data dependency (RAW hazard) from the preceding move instruction, maximizing the instruction-level parallelism and superscalar execution efficiency while naturally adapting the structural shift. With CONFIG_THREAD_INFO_IN_TASK enabled, the kernel stack life cycle is decoupled from task_struct and can be freed concurrently. Currently, show_stacktrace() reads raw stack data via __get_addr() and subsequently calls show_backtrace() to unwind the frame, without holding any reference to the target task's stack. If show_stacktrace() is called on a concurrently exiting task, it could attempt to read from a freed or reallocated kernel stack. This introduces a severe use-after-free (UAF) read risk or kernel panics. Wrap the entire stack inspection process inside show_stacktrace() with a try_get_task_stack() and put_task_stack() pair. This ensures the task stack remains pinned safely during both the raw stack data dump loop and the subsequent stack unwinding phase. Also, ensure that the task pointer is initialized to "current" early if it is NULL, so that try_get_task_stack() always operates on a valid task reference. Signed-off-by: Tiezhu Yang <yangtiezhu@loongson.cn> Signed-off-by: Huacai Chen <chenhuacai@loongson.cn>
411 lines
10 KiB
C
411 lines
10 KiB
C
// SPDX-License-Identifier: GPL-2.0
|
|
/*
|
|
* Author: Huacai Chen <chenhuacai@loongson.cn>
|
|
* Copyright (C) 2020-2022 Loongson Technology Corporation Limited
|
|
*
|
|
* Derived from MIPS:
|
|
* Copyright (C) 1994 - 1999, 2000 by Ralf Baechle and others.
|
|
* Copyright (C) 2005, 2006 by Ralf Baechle (ralf@linux-mips.org)
|
|
* Copyright (C) 1999, 2000 Silicon Graphics, Inc.
|
|
* Copyright (C) 2004 Thiemo Seufer
|
|
* Copyright (C) 2013 Imagination Technologies Ltd.
|
|
*/
|
|
#include <linux/cpu.h>
|
|
#include <linux/init.h>
|
|
#include <linux/kernel.h>
|
|
#include <linux/entry-common.h>
|
|
#include <linux/errno.h>
|
|
#include <linux/sched.h>
|
|
#include <linux/sched/debug.h>
|
|
#include <linux/sched/task.h>
|
|
#include <linux/sched/task_stack.h>
|
|
#include <linux/hw_breakpoint.h>
|
|
#include <linux/mm.h>
|
|
#include <linux/stddef.h>
|
|
#include <linux/unistd.h>
|
|
#include <linux/export.h>
|
|
#include <linux/ptrace.h>
|
|
#include <linux/mman.h>
|
|
#include <linux/personality.h>
|
|
#include <linux/sys.h>
|
|
#include <linux/completion.h>
|
|
#include <linux/kallsyms.h>
|
|
#include <linux/random.h>
|
|
#include <linux/prctl.h>
|
|
#include <linux/nmi.h>
|
|
|
|
#include <asm/asm.h>
|
|
#include <asm/asm-prototypes.h>
|
|
#include <asm/bootinfo.h>
|
|
#include <asm/cpu.h>
|
|
#include <asm/elf.h>
|
|
#include <asm/exec.h>
|
|
#include <asm/fpu.h>
|
|
#include <asm/lbt.h>
|
|
#include <asm/io.h>
|
|
#include <asm/irq.h>
|
|
#include <asm/irq_regs.h>
|
|
#include <asm/loongarch.h>
|
|
#include <asm/pgtable.h>
|
|
#include <asm/processor.h>
|
|
#include <asm/reg.h>
|
|
#include <asm/switch_to.h>
|
|
#include <asm/unwind.h>
|
|
#include <asm/vdso.h>
|
|
#include <asm/vdso/vdso.h>
|
|
|
|
#ifdef CONFIG_STACKPROTECTOR
|
|
#include <linux/stackprotector.h>
|
|
unsigned long __stack_chk_guard __read_mostly;
|
|
EXPORT_SYMBOL(__stack_chk_guard);
|
|
#endif
|
|
|
|
DEFINE_PER_CPU(struct task_struct *, cpu_tasks);
|
|
|
|
/*
|
|
* Idle related variables and functions
|
|
*/
|
|
|
|
unsigned long boot_option_idle_override = IDLE_NO_OVERRIDE;
|
|
EXPORT_SYMBOL(boot_option_idle_override);
|
|
|
|
asmlinkage void restore_and_ret(void);
|
|
asmlinkage void ret_from_fork_asm(void);
|
|
asmlinkage void ret_from_kernel_thread_asm(void);
|
|
|
|
void start_thread(struct pt_regs *regs, unsigned long pc, unsigned long sp)
|
|
{
|
|
unsigned long crmd;
|
|
unsigned long prmd;
|
|
unsigned long euen;
|
|
|
|
/* New thread loses kernel privileges. */
|
|
crmd = regs->csr_crmd & ~(PLV_MASK);
|
|
crmd |= PLV_USER;
|
|
regs->csr_crmd = crmd;
|
|
|
|
prmd = regs->csr_prmd & ~(PLV_MASK);
|
|
prmd |= PLV_USER;
|
|
regs->csr_prmd = prmd;
|
|
|
|
euen = regs->csr_euen & ~(CSR_EUEN_FPEN);
|
|
regs->csr_euen = euen;
|
|
lose_fpu(0);
|
|
lose_lbt(0);
|
|
current->thread.fpu.fcsr = boot_cpu_data.fpu_csr0;
|
|
|
|
clear_thread_flag(TIF_LSX_CTX_LIVE);
|
|
clear_thread_flag(TIF_LASX_CTX_LIVE);
|
|
clear_thread_flag(TIF_LBT_CTX_LIVE);
|
|
clear_used_math();
|
|
regs->csr_era = pc;
|
|
regs->regs[3] = sp;
|
|
}
|
|
|
|
void flush_thread(void)
|
|
{
|
|
flush_ptrace_hw_breakpoint(current);
|
|
}
|
|
|
|
void exit_thread(struct task_struct *tsk)
|
|
{
|
|
}
|
|
|
|
int arch_dup_task_struct(struct task_struct *dst, struct task_struct *src)
|
|
{
|
|
/*
|
|
* Save any process state which is live in hardware registers to the
|
|
* parent context prior to duplication. This prevents the new child
|
|
* state becoming stale if the parent is preempted before copy_thread()
|
|
* gets a chance to save the parent's live hardware registers to the
|
|
* child context.
|
|
*/
|
|
preempt_disable();
|
|
|
|
if (is_fpu_owner()) {
|
|
if (is_lasx_enabled())
|
|
save_lasx(current);
|
|
else if (is_lsx_enabled())
|
|
save_lsx(current);
|
|
else
|
|
save_fp(current);
|
|
}
|
|
|
|
preempt_enable();
|
|
|
|
if (IS_ENABLED(CONFIG_RANDSTRUCT)) {
|
|
memcpy(dst, src, sizeof(struct task_struct));
|
|
return 0;
|
|
}
|
|
|
|
dst->thread.fpu.fcsr = src->thread.fpu.fcsr;
|
|
|
|
if (!used_math())
|
|
memcpy(dst, src, offsetof(struct task_struct, thread.fpu.fpr));
|
|
else
|
|
memcpy(dst, src, offsetof(struct task_struct, thread.lbt.scr0));
|
|
|
|
#ifdef CONFIG_CPU_HAS_LBT
|
|
memcpy(&dst->thread.lbt, &src->thread.lbt, sizeof(struct loongarch_lbt));
|
|
#endif
|
|
|
|
return 0;
|
|
}
|
|
|
|
asmlinkage void noinstr __no_stack_protector ret_from_fork(struct task_struct *prev,
|
|
struct pt_regs *regs)
|
|
{
|
|
schedule_tail(prev);
|
|
syscall_exit_to_user_mode(regs);
|
|
}
|
|
|
|
asmlinkage void noinstr __no_stack_protector ret_from_kernel_thread(struct task_struct *prev,
|
|
struct pt_regs *regs,
|
|
int (*fn)(void *),
|
|
void *fn_arg)
|
|
{
|
|
schedule_tail(prev);
|
|
fn(fn_arg);
|
|
syscall_exit_to_user_mode(regs);
|
|
}
|
|
|
|
/*
|
|
* Copy architecture-specific thread state
|
|
*/
|
|
int copy_thread(struct task_struct *p, const struct kernel_clone_args *args)
|
|
{
|
|
unsigned long childksp;
|
|
unsigned long tls = args->tls;
|
|
unsigned long usp = args->stack;
|
|
u64 clone_flags = args->flags;
|
|
struct pt_regs *childregs, *regs = current_pt_regs();
|
|
|
|
childksp = (unsigned long)task_stack_page(p) + THREAD_SIZE;
|
|
|
|
/* set up new TSS. */
|
|
childregs = (struct pt_regs *) childksp - 1;
|
|
/* Put the stack after the struct pt_regs. */
|
|
childksp = (unsigned long) childregs;
|
|
p->thread.sched_cfa = 0;
|
|
p->thread.csr_euen = 0;
|
|
p->thread.csr_crmd = csr_read32(LOONGARCH_CSR_CRMD);
|
|
p->thread.csr_prmd = csr_read32(LOONGARCH_CSR_PRMD);
|
|
p->thread.csr_ecfg = csr_read32(LOONGARCH_CSR_ECFG);
|
|
if (unlikely(args->fn)) {
|
|
/* kernel thread */
|
|
p->thread.reg03 = childksp;
|
|
p->thread.reg23 = (unsigned long)args->fn;
|
|
p->thread.reg24 = (unsigned long)args->fn_arg;
|
|
p->thread.reg01 = (unsigned long)ret_from_kernel_thread_asm;
|
|
p->thread.sched_ra = (unsigned long)ret_from_kernel_thread_asm;
|
|
memset(childregs, 0, sizeof(struct pt_regs));
|
|
childregs->csr_euen = p->thread.csr_euen;
|
|
childregs->csr_crmd = p->thread.csr_crmd;
|
|
childregs->csr_prmd = p->thread.csr_prmd;
|
|
childregs->csr_ecfg = p->thread.csr_ecfg;
|
|
goto out;
|
|
}
|
|
|
|
/* user thread */
|
|
*childregs = *regs;
|
|
childregs->regs[4] = 0; /* Child gets zero as return value */
|
|
if (usp)
|
|
childregs->regs[3] = usp;
|
|
|
|
p->thread.reg03 = (unsigned long) childregs;
|
|
p->thread.reg01 = (unsigned long) ret_from_fork_asm;
|
|
p->thread.sched_ra = (unsigned long) ret_from_fork_asm;
|
|
|
|
/*
|
|
* New tasks lose permission to use the fpu. This accelerates context
|
|
* switching for most programs since they don't use the fpu.
|
|
*/
|
|
childregs->csr_euen = 0;
|
|
|
|
if (clone_flags & CLONE_SETTLS)
|
|
childregs->regs[2] = tls;
|
|
|
|
out:
|
|
ptrace_hw_copy_thread(p);
|
|
clear_tsk_thread_flag(p, TIF_USEDFPU);
|
|
clear_tsk_thread_flag(p, TIF_USEDSIMD);
|
|
clear_tsk_thread_flag(p, TIF_USEDLBT);
|
|
clear_tsk_thread_flag(p, TIF_LSX_CTX_LIVE);
|
|
clear_tsk_thread_flag(p, TIF_LASX_CTX_LIVE);
|
|
clear_tsk_thread_flag(p, TIF_LBT_CTX_LIVE);
|
|
|
|
return 0;
|
|
}
|
|
|
|
unsigned long __get_wchan(struct task_struct *task)
|
|
{
|
|
unsigned long pc = 0;
|
|
struct unwind_state state;
|
|
|
|
if (!try_get_task_stack(task))
|
|
return 0;
|
|
|
|
for (unwind_start(&state, task, NULL);
|
|
!unwind_done(&state); unwind_next_frame(&state)) {
|
|
pc = unwind_get_return_address(&state);
|
|
if (!pc)
|
|
break;
|
|
if (in_sched_functions(pc))
|
|
continue;
|
|
break;
|
|
}
|
|
|
|
put_task_stack(task);
|
|
|
|
return pc;
|
|
}
|
|
|
|
bool in_irq_stack(unsigned long stack, struct stack_info *info)
|
|
{
|
|
unsigned long nextsp;
|
|
unsigned long begin = (unsigned long)this_cpu_read(irq_stack);
|
|
unsigned long end = begin + IRQ_STACK_START;
|
|
|
|
if (stack < begin || stack >= end)
|
|
return false;
|
|
|
|
nextsp = *(unsigned long *)end;
|
|
if (nextsp & (SZREG - 1))
|
|
return false;
|
|
|
|
info->begin = begin;
|
|
info->end = end;
|
|
info->next_sp = nextsp;
|
|
info->type = STACK_TYPE_IRQ;
|
|
|
|
return true;
|
|
}
|
|
|
|
bool in_task_stack(unsigned long stack, struct task_struct *task,
|
|
struct stack_info *info)
|
|
{
|
|
unsigned long begin = (unsigned long)task_stack_page(task);
|
|
unsigned long end = begin + THREAD_SIZE;
|
|
|
|
if (stack < begin || stack >= end)
|
|
return false;
|
|
|
|
info->begin = begin;
|
|
info->end = end;
|
|
info->next_sp = 0;
|
|
info->type = STACK_TYPE_TASK;
|
|
|
|
return true;
|
|
}
|
|
|
|
int get_stack_info(unsigned long stack, struct task_struct *task,
|
|
struct stack_info *info)
|
|
{
|
|
task = task ? : current;
|
|
|
|
if (!stack || stack & (SZREG - 1))
|
|
goto unknown;
|
|
|
|
if (in_task_stack(stack, task, info))
|
|
return 0;
|
|
|
|
if (task != current)
|
|
goto unknown;
|
|
|
|
if (in_irq_stack(stack, info))
|
|
return 0;
|
|
|
|
unknown:
|
|
info->type = STACK_TYPE_UNKNOWN;
|
|
return -EINVAL;
|
|
}
|
|
|
|
unsigned long stack_top(void)
|
|
{
|
|
unsigned long top = TASK_SIZE & PAGE_MASK;
|
|
|
|
if (current->thread.vdso) {
|
|
/* Space for the VDSO & data page */
|
|
top -= PAGE_ALIGN(current->thread.vdso->size);
|
|
top -= VVAR_SIZE;
|
|
|
|
/* Space to randomize the VDSO base */
|
|
if (current->flags & PF_RANDOMIZE)
|
|
top -= VDSO_RANDOMIZE_SIZE;
|
|
}
|
|
|
|
return top;
|
|
}
|
|
|
|
/*
|
|
* Don't forget that the stack pointer must be aligned on a 8 bytes
|
|
* boundary for 32-bits ABI and 16 bytes for 64-bits ABI.
|
|
*/
|
|
unsigned long arch_align_stack(unsigned long sp)
|
|
{
|
|
if (!(current->personality & ADDR_NO_RANDOMIZE) && randomize_va_space)
|
|
sp -= get_random_u32_below(PAGE_SIZE);
|
|
|
|
return sp & STACK_ALIGN;
|
|
}
|
|
|
|
static DEFINE_PER_CPU(call_single_data_t, backtrace_csd);
|
|
static struct cpumask backtrace_csd_busy;
|
|
|
|
static void handle_backtrace(void *info)
|
|
{
|
|
nmi_cpu_backtrace(get_irq_regs());
|
|
cpumask_clear_cpu(smp_processor_id(), &backtrace_csd_busy);
|
|
}
|
|
|
|
static void raise_backtrace(cpumask_t *mask)
|
|
{
|
|
call_single_data_t *csd;
|
|
int cpu;
|
|
|
|
for_each_cpu(cpu, mask) {
|
|
/*
|
|
* If we previously sent an IPI to the target CPU & it hasn't
|
|
* cleared its bit in the busy cpumask then it didn't handle
|
|
* our previous IPI & it's not safe for us to reuse the
|
|
* call_single_data_t.
|
|
*/
|
|
if (cpumask_test_and_set_cpu(cpu, &backtrace_csd_busy)) {
|
|
pr_warn("Unable to send backtrace IPI to CPU%u - perhaps it hung?\n",
|
|
cpu);
|
|
continue;
|
|
}
|
|
|
|
csd = &per_cpu(backtrace_csd, cpu);
|
|
csd->func = handle_backtrace;
|
|
smp_call_function_single_async(cpu, csd);
|
|
}
|
|
}
|
|
|
|
void arch_trigger_cpumask_backtrace(const cpumask_t *mask, int exclude_cpu)
|
|
{
|
|
nmi_trigger_cpumask_backtrace(mask, exclude_cpu, raise_backtrace);
|
|
}
|
|
|
|
#ifdef CONFIG_32BIT
|
|
void loongarch_dump_regs32(u32 *uregs, const struct pt_regs *regs)
|
|
#else
|
|
void loongarch_dump_regs64(u64 *uregs, const struct pt_regs *regs)
|
|
#endif
|
|
{
|
|
unsigned int i;
|
|
|
|
for (i = LOONGARCH_EF_R1; i <= LOONGARCH_EF_R31; i++) {
|
|
uregs[i] = regs->regs[i - LOONGARCH_EF_R0];
|
|
}
|
|
|
|
uregs[LOONGARCH_EF_ORIG_A0] = regs->orig_a0;
|
|
uregs[LOONGARCH_EF_CSR_ERA] = regs->csr_era;
|
|
uregs[LOONGARCH_EF_CSR_BADV] = regs->csr_badvaddr;
|
|
uregs[LOONGARCH_EF_CSR_CRMD] = regs->csr_crmd;
|
|
uregs[LOONGARCH_EF_CSR_PRMD] = regs->csr_prmd;
|
|
uregs[LOONGARCH_EF_CSR_EUEN] = regs->csr_euen;
|
|
uregs[LOONGARCH_EF_CSR_ECFG] = regs->csr_ecfg;
|
|
uregs[LOONGARCH_EF_CSR_ESTAT] = regs->csr_estat;
|
|
}
|