mirror of
https://git.kernel.org/pub/scm/linux/kernel/git/stable/linux.git
synced 2026-08-09 06:14:34 +02:00
Hypervisors invoke resume_user_mode_work() before entering the guest, which clears TIF_NOTIFY_RESUME. The @regs argument is NULL as there is no user space context available to them, so the rseq notify handler skips inspecting the critical section, but updates the CPU/MM CID values unconditionally so that the eventual pending rseq event is not lost on the way to user space. This is a pointless exercise as the task might be rescheduled before actually returning to user space and it creates unnecessary work in the vcpu_run() loops. It's way more efficient to ignore that invocation based on @regs == NULL and let the hypervisors re-raise TIF_NOTIFY_RESUME after returning from the vcpu_run() loop before returning from the ioctl(). This ensures that a pending RSEQ update is not lost and the IDs are updated before returning to user space. Once the RSEQ handling is decoupled from TIF_NOTIFY_RESUME, this turns into a NOOP. Signed-off-by: Thomas Gleixner <tglx@linutronix.de> Signed-off-by: Peter Zijlstra (Intel) <peterz@infradead.org> Signed-off-by: Ingo Molnar <mingo@kernel.org> Reviewed-by: Mathieu Desnoyers <mathieu.desnoyers@efficios.com> Acked-by: Sean Christopherson <seanjc@google.com> Link: https://patch.msgid.link/20251027084306.399495855@linutronix.de
100 lines
2.8 KiB
C
100 lines
2.8 KiB
C
/* SPDX-License-Identifier: GPL-2.0+ WITH Linux-syscall-note */
|
|
#ifndef _LINUX_RSEQ_H
|
|
#define _LINUX_RSEQ_H
|
|
|
|
#ifdef CONFIG_RSEQ
|
|
#include <linux/sched.h>
|
|
|
|
void __rseq_handle_notify_resume(struct ksignal *sig, struct pt_regs *regs);
|
|
|
|
static inline void rseq_handle_notify_resume(struct pt_regs *regs)
|
|
{
|
|
if (current->rseq)
|
|
__rseq_handle_notify_resume(NULL, regs);
|
|
}
|
|
|
|
static inline void rseq_signal_deliver(struct ksignal *ksig, struct pt_regs *regs)
|
|
{
|
|
if (current->rseq) {
|
|
current->rseq_event_pending = true;
|
|
__rseq_handle_notify_resume(ksig, regs);
|
|
}
|
|
}
|
|
|
|
static inline void rseq_sched_switch_event(struct task_struct *t)
|
|
{
|
|
if (t->rseq) {
|
|
t->rseq_event_pending = true;
|
|
set_tsk_thread_flag(t, TIF_NOTIFY_RESUME);
|
|
}
|
|
}
|
|
|
|
static __always_inline void rseq_exit_to_user_mode(void)
|
|
{
|
|
if (IS_ENABLED(CONFIG_DEBUG_RSEQ)) {
|
|
if (WARN_ON_ONCE(current->rseq && current->rseq_event_pending))
|
|
current->rseq_event_pending = false;
|
|
}
|
|
}
|
|
|
|
/*
|
|
* KVM/HYPERV invoke resume_user_mode_work() before entering guest mode,
|
|
* which clears TIF_NOTIFY_RESUME. To avoid updating user space RSEQ in
|
|
* that case just to do it eventually again before returning to user space,
|
|
* the entry resume_user_mode_work() invocation is ignored as the register
|
|
* argument is NULL.
|
|
*
|
|
* After returning from guest mode, they have to invoke this function to
|
|
* re-raise TIF_NOTIFY_RESUME if necessary.
|
|
*/
|
|
static inline void rseq_virt_userspace_exit(void)
|
|
{
|
|
if (current->rseq_event_pending)
|
|
set_tsk_thread_flag(current, TIF_NOTIFY_RESUME);
|
|
}
|
|
|
|
/*
|
|
* If parent process has a registered restartable sequences area, the
|
|
* child inherits. Unregister rseq for a clone with CLONE_VM set.
|
|
*/
|
|
static inline void rseq_fork(struct task_struct *t, u64 clone_flags)
|
|
{
|
|
if (clone_flags & CLONE_VM) {
|
|
t->rseq = NULL;
|
|
t->rseq_len = 0;
|
|
t->rseq_sig = 0;
|
|
t->rseq_event_pending = false;
|
|
} else {
|
|
t->rseq = current->rseq;
|
|
t->rseq_len = current->rseq_len;
|
|
t->rseq_sig = current->rseq_sig;
|
|
t->rseq_event_pending = current->rseq_event_pending;
|
|
}
|
|
}
|
|
|
|
static inline void rseq_execve(struct task_struct *t)
|
|
{
|
|
t->rseq = NULL;
|
|
t->rseq_len = 0;
|
|
t->rseq_sig = 0;
|
|
t->rseq_event_pending = false;
|
|
}
|
|
|
|
#else /* CONFIG_RSEQ */
|
|
static inline void rseq_handle_notify_resume(struct pt_regs *regs) { }
|
|
static inline void rseq_signal_deliver(struct ksignal *ksig, struct pt_regs *regs) { }
|
|
static inline void rseq_sched_switch_event(struct task_struct *t) { }
|
|
static inline void rseq_virt_userspace_exit(void) { }
|
|
static inline void rseq_fork(struct task_struct *t, u64 clone_flags) { }
|
|
static inline void rseq_execve(struct task_struct *t) { }
|
|
static inline void rseq_exit_to_user_mode(void) { }
|
|
#endif /* !CONFIG_RSEQ */
|
|
|
|
#ifdef CONFIG_DEBUG_RSEQ
|
|
void rseq_syscall(struct pt_regs *regs);
|
|
#else /* CONFIG_DEBUG_RSEQ */
|
|
static inline void rseq_syscall(struct pt_regs *regs) { }
|
|
#endif /* !CONFIG_DEBUG_RSEQ */
|
|
|
|
#endif /* _LINUX_RSEQ_H */
|