ct_cpu_track_user() and the context_tracking_key static key are currently
restricted to boot-time use: the key is __ro_after_init and the function
is __init with __initdata state.  This prevents enabling nohz_full context
tracking for CPUs isolated at runtime via cpuset partitions.

Split ct_cpu_track_user() into three functions:

  ct_cpu_track_user(cpu)      - sets per_cpu(context_tracking.active) and
                                increments context_tracking_key; callable
                                at runtime with the CPU offline.

  ct_cpu_untrack_user(cpu)    - reverses the above; for de-isolation.

  ct_cpu_track_user_init(cpu) - __init wrapper; calls ct_cpu_track_user()
                                and handles TIF_NOHZ / tasklist setup.

Change context_tracking_key from DEFINE_STATIC_KEY_FALSE_RO to
DEFINE_STATIC_KEY_FALSE so that static_branch_inc/dec() can be called
after the __ro_after_init window closes.

Update tick_nohz_init() to call ct_cpu_track_user_init() so boot
behaviour is unchanged.

This is a prerequisite for DHM (Dynamic Housekeeping Management) runtime
CPU noise isolation without boot parameters.

context_tracking_key is a single systemwide static branch, not a
per-CPU gate: once live, __ct_user_enter()/__ct_user_exit() run
unconditionally on every CPU, regardless of that CPU's own
context_tracking.active.  Going live happens through code patching,
and other CPUs only observe the patched code some time after
static_branch_inc() returns.  A CPU whose kernel<->user transition
lands in that window sees context_tracking_enabled() as still false
and silently skips recording it, leaving context_tracking.state
stuck, so the next traced kernel entry on that CPU wrongly trips
CT_WARN_ON(__ct_state() != CT_STATE_USER).  Reordering the enable and
a fixup sweep around each other cannot close this: whichever runs
last still has its own propagation delay to every other CPU.

Add context_tracking_activating, a plain per-CPU bool with no
code-patching delay of its own, and
context_tracking_enabled_or_activating() to test it alongside the
static key.  ct_cpu_track_user() sets it on every CPU via IPI
strictly before static_branch_inc(), and clears it via another IPI
only after static_branch_inc() returns, so every relevant call site
sees it in place for the whole window during which the static key
might not have propagated yet.  Route user_enter_irqoff(),
user_exit_irqoff(), the guest variants, CT_WARN_ON() and ct_state()
through it instead of the raw static key.  Also directly bootstrap
CT_STATE_USER for a CPU caught sitting in user mode by its
interrupted pt_regs, rather than leaving it to self-correct on its
own next transition.

Signed-off-by: Qiliang Yuan <[email protected]>
---
 include/linux/context_tracking.h       | 16 +++---
 include/linux/context_tracking_state.h | 25 +++++++++-
 kernel/context_tracking.c              | 91 ++++++++++++++++++++++++++++++++--
 kernel/time/tick-sched.c               |  2 +-
 4 files changed, 122 insertions(+), 12 deletions(-)

diff --git a/include/linux/context_tracking.h b/include/linux/context_tracking.h
index af9fe87a09225..a83a2f1f9f9a9 100644
--- a/include/linux/context_tracking.h
+++ b/include/linux/context_tracking.h
@@ -12,6 +12,8 @@
 
 #ifdef CONFIG_CONTEXT_TRACKING_USER
 extern void ct_cpu_track_user(int cpu);
+extern void ct_cpu_untrack_user(int cpu);
+extern void __init ct_cpu_track_user_init(int cpu);
 
 /* Called with interrupts disabled.  */
 extern void __ct_user_enter(enum ctx_state state);
@@ -25,26 +27,26 @@ extern void user_exit_callable(void);
 
 static inline void user_enter(void)
 {
-       if (context_tracking_enabled())
+       if (context_tracking_enabled_or_activating())
                ct_user_enter(CT_STATE_USER);
 
 }
 static inline void user_exit(void)
 {
-       if (context_tracking_enabled())
+       if (context_tracking_enabled_or_activating())
                ct_user_exit(CT_STATE_USER);
 }
 
 /* Called with interrupts disabled.  */
 static __always_inline void user_enter_irqoff(void)
 {
-       if (context_tracking_enabled())
+       if (context_tracking_enabled_or_activating())
                __ct_user_enter(CT_STATE_USER);
 
 }
 static __always_inline void user_exit_irqoff(void)
 {
-       if (context_tracking_enabled())
+       if (context_tracking_enabled_or_activating())
                __ct_user_exit(CT_STATE_USER);
 }
 
@@ -74,7 +76,7 @@ static inline void exception_exit(enum ctx_state prev_ctx)
 
 static __always_inline bool context_tracking_guest_enter(void)
 {
-       if (context_tracking_enabled())
+       if (context_tracking_enabled_or_activating())
                __ct_user_enter(CT_STATE_GUEST);
 
        return context_tracking_enabled_this_cpu();
@@ -82,13 +84,13 @@ static __always_inline bool 
context_tracking_guest_enter(void)
 
 static __always_inline bool context_tracking_guest_exit(void)
 {
-       if (context_tracking_enabled())
+       if (context_tracking_enabled_or_activating())
                __ct_user_exit(CT_STATE_GUEST);
 
        return context_tracking_enabled_this_cpu();
 }
 
-#define CT_WARN_ON(cond) WARN_ON(context_tracking_enabled() && (cond))
+#define CT_WARN_ON(cond) WARN_ON(context_tracking_enabled_or_activating() && 
(cond))
 
 #else
 static inline void user_enter(void) { }
diff --git a/include/linux/context_tracking_state.h 
b/include/linux/context_tracking_state.h
index 0b81248aa03e2..25f87a9763313 100644
--- a/include/linux/context_tracking_state.h
+++ b/include/linux/context_tracking_state.h
@@ -138,6 +138,28 @@ static __always_inline bool context_tracking_enabled(void)
        return static_branch_unlikely(&context_tracking_key);
 }
 
+/*
+ * context_tracking_key goes live via code patching, which other CPUs only
+ * observe some time after ct_cpu_track_user() calls static_branch_inc().
+ * A CPU whose kernel<->user transition lands in that window would see
+ * context_tracking_enabled() as still false and silently skip recording
+ * it, leaving context_tracking.state stale.  ct_cpu_track_user() sets
+ * context_tracking_activating on every CPU with an IPI strictly before
+ * calling static_branch_inc(), and clears it again with another IPI only
+ * after static_branch_inc() returns (so only once every CPU is
+ * guaranteed to already observe the branch as enabled).  Checking it
+ * here closes that window: every transition in between is recorded via
+ * the normal __ct_user_enter()/__ct_user_exit() path instead of being
+ * silently dropped.
+ */
+DECLARE_PER_CPU(bool, context_tracking_activating);
+
+static __always_inline bool context_tracking_enabled_or_activating(void)
+{
+       return context_tracking_enabled() ||
+              unlikely(__this_cpu_read(context_tracking_activating));
+}
+
 static __always_inline bool context_tracking_enabled_cpu(int cpu)
 {
        return context_tracking_enabled() && per_cpu(context_tracking.active, 
cpu);
@@ -159,7 +181,7 @@ static __always_inline int ct_state(void)
 {
        int ret;
 
-       if (!context_tracking_enabled())
+       if (!context_tracking_enabled_or_activating())
                return CT_STATE_DISABLED;
 
        preempt_disable();
@@ -171,6 +193,7 @@ static __always_inline int ct_state(void)
 
 #else
 static __always_inline bool context_tracking_enabled(void) { return false; }
+static __always_inline bool context_tracking_enabled_or_activating(void) { 
return false; }
 static __always_inline bool context_tracking_enabled_cpu(int cpu) { return 
false; }
 static __always_inline bool context_tracking_enabled_this_cpu(void) { return 
false; }
 #endif /* CONFIG_CONTEXT_TRACKING_USER */
diff --git a/kernel/context_tracking.c b/kernel/context_tracking.c
index a743e7ffa6c00..326862a2679e3 100644
--- a/kernel/context_tracking.c
+++ b/kernel/context_tracking.c
@@ -23,6 +23,9 @@
 #include <linux/hardirq.h>
 #include <linux/export.h>
 #include <linux/kprobes.h>
+#include <linux/smp.h>
+#include <linux/ptrace.h>
+#include <asm/irq_regs.h>
 #include <trace/events/rcu.h>
 
 
@@ -411,9 +414,12 @@ static __always_inline void ct_kernel_enter(bool user, int 
offset) { }
 #define CREATE_TRACE_POINTS
 #include <trace/events/context_tracking.h>
 
-DEFINE_STATIC_KEY_FALSE_RO(context_tracking_key);
+DEFINE_STATIC_KEY_FALSE(context_tracking_key);
 EXPORT_SYMBOL_GPL(context_tracking_key);
 
+DEFINE_PER_CPU(bool, context_tracking_activating);
+EXPORT_SYMBOL_GPL(context_tracking_activating);
+
 static noinstr bool context_tracking_recursion_enter(void)
 {
        int recursion;
@@ -674,14 +680,93 @@ void user_exit_callable(void)
 }
 NOKPROBE_SYMBOL(user_exit_callable);
 
-void __init ct_cpu_track_user(int cpu)
+/*
+ * context_tracking_key is a single systemwide static branch, not a per-CPU
+ * gate: once live, __ct_user_enter()/__ct_user_exit() run unconditionally
+ * on every CPU (so that a task migrating between a tracked and an
+ * untracked CPU always sees consistent state), regardless of that CPU's
+ * own context_tracking.active.  Going live happens through code patching,
+ * though, and other CPUs only observe the patched code some time after
+ * static_branch_inc() is called on this one.  A CPU whose kernel<->user
+ * transition lands in that window would see context_tracking_enabled()
+ * as still false and silently skip recording it, leaving
+ * context_tracking.state stuck at whatever it was, so the first traced
+ * kernel entry on that CPU afterwards would wrongly trip
+ * CT_WARN_ON(__ct_state() != CT_STATE_USER).  Reordering the two calls
+ * below cannot close this: whichever runs last still has its own
+ * propagation delay to every other CPU.
+ *
+ * context_tracking_enabled_or_activating() closes the window instead:
+ * every user_enter_irqoff()/user_exit_irqoff()/CT_WARN_ON() site treats
+ * a CPU as tracking once context_tracking_activating is set on it, with
+ * no code-patching delay of its own, since it is a plain per-CPU bool
+ * set directly by the interrupting IPI handler rather than inferred
+ * from a jump label.  Set it on every CPU before static_branch_inc(),
+ * and only clear it once static_branch_inc() has returned, so there is
+ * no gap during which a CPU observes neither signal: every transition
+ * in between is recorded through the normal path instead of being
+ * silently dropped.  Also directly bootstrap CT_STATE_USER for a CPU
+ * caught sitting in user mode (via its interrupted pt_regs), rather
+ * than leaving it to self-correct on its own next transition.
+ */
+static void ct_activate_set_pending_ipi(void *unused)
 {
-       static __initdata bool initialized = false;
+       struct pt_regs *regs = get_irq_regs();
 
+       __this_cpu_write(context_tracking_activating, true);
+       if (regs && user_mode(regs))
+               __ct_user_enter(CT_STATE_USER);
+}
+
+static void ct_activate_clear_pending_ipi(void *unused)
+{
+       __this_cpu_write(context_tracking_activating, false);
+}
+
+/**
+ * ct_cpu_track_user - enable context tracking for a CPU
+ * @cpu: target CPU (must be offline when called at runtime)
+ *
+ * Marks @cpu as actively tracking user/kernel transitions and increments
+ * the context_tracking_key refcount.  Safe to call at runtime provided
+ * the CPU is offline so no context-tracking readers are active on it.
+ */
+void ct_cpu_track_user(int cpu)
+{
        if (!per_cpu(context_tracking.active, cpu)) {
+               bool first_activation = !context_tracking_enabled();
+
                per_cpu(context_tracking.active, cpu) = true;
+               if (first_activation)
+                       on_each_cpu(ct_activate_set_pending_ipi, NULL, 1);
                static_branch_inc(&context_tracking_key);
+               if (first_activation)
+                       on_each_cpu(ct_activate_clear_pending_ipi, NULL, 1);
        }
+}
+EXPORT_SYMBOL_GPL(ct_cpu_track_user);
+
+/**
+ * ct_cpu_untrack_user - disable context tracking for a CPU
+ * @cpu: target CPU (must be offline when called)
+ *
+ * Reverses ct_cpu_track_user().  The CPU must be offline so that no
+ * context-tracking readers are active on it.
+ */
+void ct_cpu_untrack_user(int cpu)
+{
+       if (per_cpu(context_tracking.active, cpu)) {
+               per_cpu(context_tracking.active, cpu) = false;
+               static_branch_dec(&context_tracking_key);
+       }
+}
+EXPORT_SYMBOL_GPL(ct_cpu_untrack_user);
+
+void __init ct_cpu_track_user_init(int cpu)
+{
+       static __initdata bool initialized = false;
+
+       ct_cpu_track_user(cpu);
 
        if (initialized)
                return;
diff --git a/kernel/time/tick-sched.c b/kernel/time/tick-sched.c
index 6c3fea3867139..8c53754c4ac4a 100644
--- a/kernel/time/tick-sched.c
+++ b/kernel/time/tick-sched.c
@@ -675,7 +675,7 @@ void __init tick_nohz_init(void)
        }
 
        for_each_cpu(cpu, tick_nohz_full_mask)
-               ct_cpu_track_user(cpu);
+               ct_cpu_track_user_init(cpu);
 
        ret = cpuhp_setup_state_nocalls(CPUHP_AP_ONLINE_DYN,
                                        "kernel/nohz:predown", NULL,

-- 
2.43.0


Reply via email to