On 9/1/26 5:51 PM, Dietmar Eggemann wrote:
On 01.09.26 08:21, Vincent Guittot wrote:
On Mon, 31 Aug 2026 at 18:54, Yury Norov <[email protected]> wrote:
On Mon, Aug 31, 2026 at 05:09:41PM +0200, Vincent Guittot wrote:
Dietmar/Vincent,
Do you think it makes sense to enable the driver on ARM64 now?
Or you think it is better to delay it and once the feature is stable
ARM ecosystem can enable it?
It's always better to support all arch by default, unless something is
missing which is not the case here.
ARM64 testing is obviously missed.
But the cpumask is already available not like if you need to create a new one
IMHO, when testing the steal_governor on arm64 w/o
`allow_mismatched_32bit_el0`, I wouldn't expect much difference in this
respect compared to the architectures already tested.
Also, AFAIK, `allow_mismatched_32bit_el0` was mainly relevant for
Android devices up to Android 13 that supported 32-bit userspace, so I
wouldn't expect it to be commonly enabled on recent devices.
Ok. So it is a narrow window in a relatively older versions.
As Vincent pointed out, `task_cpu_possible_mask(p)` (which defaults to
`cpu_possible_mask`) is already used in `kernel/sched/core.c` and
`kernel/cgroup/cpuset.c` to support `allow_mismatched_32bit_el0`.
Based on the discussion so far, I think we can keep the feature generic
for now. If a concrete architecture- or hypervisor-specific limitation is
identified and cannot be addressed in the generic code, then we can add
Kconfig gating. So far, there is no such case being identified and current
generic version is in good shape IMHO.
I have added cpumask_intersects_and and used that for valid cpu check.
Please see the v11->v12 diff attached at the end.
cpumask_intersects_and could also be used in other places such as
drivers/cpuidle/coupled.c:cpuidle_coupled_any_pokes_pending.
Patch for that will be sent separately.
If any of the above doesn't make sense, Please let me know.
I am planning to send v12 post running the workloads and sanity checks.
I have run cpumask_intersects vs cpumask_intersects_and with the workloads
I am currently running and did not observe a measurable difference.
---
include/linux/bitmap.h | 14 ++++++++++++++
include/linux/cpumask.h | 18 ++++++++++++++++++
kernel/sched/core.c | 16 ++--------------
lib/bitmap.c | 17 +++++++++++++++++
4 files changed, 51 insertions(+), 14 deletions(-)
diff --git a/include/linux/bitmap.h b/include/linux/bitmap.h
index b007d54a9036..e595d189047b 100644
--- a/include/linux/bitmap.h
+++ b/include/linux/bitmap.h
@@ -52,6 +52,7 @@ struct device;
* bitmap_complement(dst, src, nbits) *dst = ~(*src)
* bitmap_equal(src1, src2, nbits) Are *src1 and *src2 equal?
* bitmap_intersects(src1, src2, nbits) Do *src1 and *src2 overlap?
+ * bitmap_intersects_and(src1, src2, src3, nbits) Do *src1, *src2 and *src3
overlap?
* bitmap_subset(src1, src2, nbits) Is *src1 a subset of *src2?
* bitmap_empty(src, nbits) Are all bits zero in *src?
* bitmap_full(src, nbits) Are all bits set in *src?
@@ -181,6 +182,9 @@ void __bitmap_replace(unsigned long *dst,
const unsigned long *mask, unsigned int nbits);
bool __bitmap_intersects(const unsigned long *bitmap1,
const unsigned long *bitmap2, unsigned int nbits);
+bool __bitmap_intersects_and(const unsigned long *bitmap1,
+ const unsigned long *bitmap2,
+ const unsigned long *bitmap3, unsigned int nbits);
bool __bitmap_subset(const unsigned long *bitmap1,
const unsigned long *bitmap2, unsigned int nbits);
unsigned int __bitmap_weight(const unsigned long *bitmap, unsigned int nbits);
@@ -442,6 +446,16 @@ bool bitmap_intersects(const unsigned long *src1, const
unsigned long *src2, uns
return __bitmap_intersects(src1, src2, nbits);
}
+static __always_inline
+bool bitmap_intersects_and(const unsigned long *src1, const unsigned long
*src2,
+ const unsigned long *src3, unsigned int nbits)
+{
+ if (small_const_nbits(nbits))
+ return ((*src1 & *src2 & *src3) & BITMAP_LAST_WORD_MASK(nbits))
!= 0;
+ else
+ return __bitmap_intersects_and(src1, src2, src3, nbits);
+}
+
static __always_inline
bool bitmap_subset(const unsigned long *src1, const unsigned long *src2,
unsigned int nbits)
{
diff --git a/include/linux/cpumask.h b/include/linux/cpumask.h
index 34d08a3d80e1..d7f59bf32d32 100644
--- a/include/linux/cpumask.h
+++ b/include/linux/cpumask.h
@@ -833,6 +833,24 @@ bool cpumask_intersects(const struct cpumask *src1p, const
struct cpumask *src2p
small_cpumask_bits);
}
+/**
+ * cpumask_intersects_and - (*src1p & *src2p & *src3p) != 0
+ * @src1p: the first input
+ * @src2p: the second input
+ * @src3p: the third input
+ *
+ * Return: true if AND of the three cpumasks is non-empty,
+ * otherwise false
+ */
+static __always_inline
+bool cpumask_intersects_and(const struct cpumask *src1p,
+ const struct cpumask *src2p,
+ const struct cpumask *src3p)
+{
+ return bitmap_intersects_and(cpumask_bits(src1p), cpumask_bits(src2p),
+ cpumask_bits(src3p), small_cpumask_bits);
+}
+
/**
* cpumask_subset - (*src1p & ~*src2p) == 0
* @src1p: the first input
diff --git a/kernel/sched/core.c b/kernel/sched/core.c
index e73d2dd997b3..0dd1d92a23db 100644
--- a/kernel/sched/core.c
+++ b/kernel/sched/core.c
@@ -2496,9 +2496,6 @@ static inline bool rq_has_pinned_tasks(struct rq *rq)
static inline bool task_can_sched_on_preferred(int cpu, struct task_struct *p)
{
- const struct cpumask *valid_mask;
- int i;
-
if (cpu_preferred(cpu))
return false;
@@ -2510,17 +2507,8 @@ static inline bool task_can_sched_on_preferred(int cpu, struct task_struct *p)
if (unlikely(!cpumask_test_cpu(task_cpu(p), p->cpus_ptr)))
return false;
- valid_mask = task_cpu_possible_mask(p);
- if (likely(valid_mask == cpu_possible_mask))
- return cpumask_intersects(p->cpus_ptr, cpu_preferred_mask);
-
- /* Tasks with arch-specific CPU masks. e.g. 32-bit tasks on arm64. */
- for_each_cpu_and(i, p->cpus_ptr, cpu_preferred_mask) {
- if (cpumask_test_cpu(i, valid_mask))
- return true;
- }
-
- return false;
+ return cpumask_intersects_and(p->cpus_ptr, cpu_preferred_mask,
+ task_cpu_possible_mask(p));
}
/*
diff --git a/lib/bitmap.c b/lib/bitmap.c
index b9bfa157e095..34e202800c5d 100644
--- a/lib/bitmap.c
+++ b/lib/bitmap.c
@@ -308,6 +308,23 @@ bool __bitmap_intersects(const unsigned long *bitmap1,
}
EXPORT_SYMBOL(__bitmap_intersects);
+bool __bitmap_intersects_and(const unsigned long *bitmap1,
+ const unsigned long *bitmap2,
+ const unsigned long *bitmap3, unsigned int bits)
+{
+ unsigned int k, lim = bits / BITS_PER_LONG;
+
+ for (k = 0; k < lim; ++k)
+ if (bitmap1[k] & bitmap2[k] & bitmap3[k])
+ return true;
+
+ if (bits % BITS_PER_LONG)
+ if ((bitmap1[k] & bitmap2[k] & bitmap3[k]) &
BITMAP_LAST_WORD_MASK(bits))
+ return true;
+ return false;
+}
+EXPORT_SYMBOL(__bitmap_intersects_and);
+
bool __bitmap_subset(const unsigned long *bitmap1,
const unsigned long *bitmap2, unsigned int bits)
{
--
2.52.0