[RFC v6 2/3] workqueue: Add support for real-time workers

From: Tvrtko Ursulin

Date: Thu Oct 01 2026 - 12:14:22 EST


For use cases such as the DRM scheduler submitting work to the GPU on
behalf of low latency userspace applications, where latter have sufficient
privileges to have had successfully obtained realtime Vulkan global
priority, competing with random background CPU load can create large
latency spikes which gets in the way of a smooth user experience.

For these situations the existing WQ_HIGHPRI does not bring a noticeable
improvement and a stronger hint is needed.

Lets add WQ_RT which creates workers with a SCHED_FIFO scheduling class to
improve this.

We use a minimum priority level since we only care about winning the
contest against normal background CPU load.

Signed-off-by: Tvrtko Ursulin <tvrtko.ursulin@xxxxxxxxxx>
Cc: Boris Brezillon <boris.brezillon@xxxxxxxxxxxxx>
Cc: Bradley Morgan <include@xxxxxxxxx>
Cc: Chia-I Wu <olv@xxxxxxxxxx>
Cc: Liviu Dudau <liviu.dudau@xxxxxxx>
Cc: Matthew Brost <matthew.brost@xxxxxxxxx>
Cc: Steven Price <steven.price@xxxxxxx>
Cc: Tejun Heo <tj@xxxxxxxxxx>
---
v2:
* Limit WQ_RTPRI to unbound workqueues and make it have strict CPU
affinitity. (Tejun)
* Fixed commit message typos. (AI)
* Fixed sysfs handling, max_active setting and user modified nice
application. (AI)

v3:
* Fix worker->pool null pointer dereference race by moving the
global decrement to detach_dying_workers().
* Rebase for upstream changes.

v4:
* Fixed onion unwind.
* Moved affinity setting to default attributes.

v5:
* Dropped global and local limits.
* Documented in workqueue.rst.
* Added NR_WQ_ATTRIBUTES.
* Reverted BH handling changes.

v6:
* Dropped separate attr->prio in favour of RTPRI_NICE_LEVEL checks. (Tejun)
* Reworked on top of tj/for-7.4.

v7:
* Convert to attrs->prio encoded analoguous to task_struct->prio.
* Rename flag to WQ_PRIO and do not re-order enums.
* Forbid WQ_RT affinity modifications via sysfs.
* Added wq_dump.py support.
---
Documentation/core-api/workqueue.rst | 8 +++
include/linux/workqueue.h | 5 +-
kernel/workqueue.c | 101 +++++++++++++++++++--------
tools/workqueue/wq_dump.py | 9 ++-
4 files changed, 90 insertions(+), 33 deletions(-)

diff --git a/Documentation/core-api/workqueue.rst b/Documentation/core-api/workqueue.rst
index bb770f556568..d699c3832b19 100644
--- a/Documentation/core-api/workqueue.rst
+++ b/Documentation/core-api/workqueue.rst
@@ -225,6 +225,14 @@ resources, scheduled and executed.
each other. Each maintains its separate pool of workers and
implements concurrency management among its workers.

+``WQ_RT``
+ Real-time priority workqueues must be created as unbound and will be
+ configured with the strict CPU affinity set. Their worker threads use the FIFO
+ scheduling policy with the lowest applicable priority.
+
+ To be used sparingly for use cases such as the real-time GPU rendering
+ contexts accessible to privileged clients.
+
``WQ_CPU_INTENSIVE``
Work items of a CPU intensive wq do not contribute to the
concurrency level. In other words, runnable CPU intensive
diff --git a/include/linux/workqueue.h b/include/linux/workqueue.h
index a283766a192a..ebec9dcc9e5f 100644
--- a/include/linux/workqueue.h
+++ b/include/linux/workqueue.h
@@ -147,9 +147,9 @@ enum wq_affn_scope {
*/
struct workqueue_attrs {
/**
- * @nice: nice level
+ * @prio: priority encoded analoguous to task_struct->prio.
*/
- int nice;
+ int prio;

/**
* @cpumask: allowed CPUs
@@ -404,6 +404,7 @@ enum wq_flags {
*/
WQ_POWER_EFFICIENT = 1 << 7,
WQ_PERCPU = 1 << 8, /* bound to a specific cpu */
+ WQ_RT = 1 << 9, /* real-time priority, valid only with WQ_UNBOUND */

__WQ_DESTROYING = 1 << 15, /* internal: workqueue is destroying */
__WQ_DRAINING = 1 << 16, /* internal: workqueue is draining */
diff --git a/kernel/workqueue.c b/kernel/workqueue.c
index c83d68d7d0ee..afe39a18ad9e 100644
--- a/kernel/workqueue.c
+++ b/kernel/workqueue.c
@@ -47,6 +47,7 @@
#include <linux/jhash.h>
#include <linux/hashtable.h>
#include <linux/rculist.h>
+#include <linux/sched/rt.h>
#include <linux/nodemask.h>
#include <linux/moduleparam.h>
#include <linux/uaccess.h>
@@ -126,7 +127,8 @@ enum wq_internal_consts {
* all cpus. Give MIN_NICE.
*/
RESCUER_NICE_LEVEL = MIN_NICE,
- HIGHPRI_NICE_LEVEL = MIN_NICE,
+ HIGHPRI_PRIORITY = NICE_TO_PRIO(MIN_NICE),
+ RT_PRIORITY = MAX_PRIO,

WQ_NAME_LEN = 32,
WORKER_ID_LEN = 10 + WQ_NAME_LEN, /* "kworker/R-" + WQ_NAME_LEN */
@@ -1275,7 +1277,7 @@ static bool assign_work(struct work_struct *work, struct worker *worker,

static struct irq_work *bh_pool_irq_work(struct worker_pool *pool)
{
- int high = pool->attrs->nice == HIGHPRI_NICE_LEVEL ? 1 : 0;
+ int high = pool->attrs->prio == HIGHPRI_PRIORITY ? 1 : 0;

return &per_cpu(bh_pool_irq_works, pool->cpu)[high];
}
@@ -1290,7 +1292,7 @@ static void kick_bh_pool(struct worker_pool *pool)
return;
}
#endif
- if (pool->attrs->nice == HIGHPRI_NICE_LEVEL)
+ if (pool->attrs->prio == HIGHPRI_PRIORITY)
raise_softirq_irqoff(HI_SOFTIRQ);
else
raise_softirq_irqoff(TASKLET_SOFTIRQ);
@@ -2959,7 +2961,8 @@ static int format_worker_id(char *buf, size_t size, struct worker *worker,
if (pool->cpu >= 0)
return scnprintf(buf, size, "kworker/%d:%d%s",
pool->cpu, worker->id,
- pool->attrs->nice < 0 ? "H" : "");
+ pool->attrs->prio < NICE_TO_PRIO(0) ?
+ "H" : "");
else
return scnprintf(buf, size, "kworker/u%d:%d",
pool->id, worker->id);
@@ -3018,7 +3021,12 @@ static struct worker *create_worker(struct worker_pool *pool)
goto fail;
}

- set_user_nice(worker->task, pool->attrs->nice);
+ if (rt_prio(pool->attrs->prio))
+ sched_set_fifo_low(worker->task);
+ else
+ set_user_nice(worker->task,
+ PRIO_TO_NICE(pool->attrs->prio));
+
kthread_bind_mask(worker->task, pool_allowed_cpus(pool));
}

@@ -3910,7 +3918,7 @@ static void bh_worker(struct worker *worker)

if (budget_exhausted)
trace_workqueue_bh_budget_yield(pool, restarts, timeout,
- pool->attrs->nice == HIGHPRI_NICE_LEVEL);
+ pool->attrs->prio == HIGHPRI_PRIORITY);
}

/*
@@ -3969,7 +3977,7 @@ static void drain_dead_softirq_workfn(struct work_struct *work)
* don't hog this CPU's BH.
*/
if (repeat) {
- if (pool->attrs->nice == HIGHPRI_NICE_LEVEL)
+ if (pool->attrs->prio == HIGHPRI_PRIORITY)
queue_work(system_bh_highpri_wq, work);
else
queue_work(system_bh_wq, work);
@@ -4001,7 +4009,7 @@ void workqueue_softirq_dead(unsigned int cpu)
dead_work.pool = pool;
init_completion(&dead_work.done);

- if (pool->attrs->nice == HIGHPRI_NICE_LEVEL)
+ if (pool->attrs->prio == HIGHPRI_PRIORITY)
queue_work(system_bh_highpri_wq, &dead_work.work);
else
queue_work(system_bh_wq, &dead_work.work);
@@ -5015,7 +5023,7 @@ struct workqueue_attrs *alloc_workqueue_attrs_noprof(void)
static void copy_workqueue_attrs(struct workqueue_attrs *to,
const struct workqueue_attrs *from)
{
- to->nice = from->nice;
+ to->prio = from->prio;
cpumask_copy(to->cpumask, from->cpumask);
cpumask_copy(to->__pod_cpumask, from->__pod_cpumask);
to->affn_strict = from->affn_strict;
@@ -5046,7 +5054,7 @@ static u32 wqattrs_hash(const struct workqueue_attrs *attrs)
{
u32 hash = 0;

- hash = jhash_1word(attrs->nice, hash);
+ hash = jhash_1word(attrs->prio, hash);
hash = jhash_1word(attrs->affn_strict, hash);
hash = jhash(cpumask_bits(attrs->__pod_cpumask),
BITS_TO_LONGS(nr_cpumask_bits) * sizeof(long), hash);
@@ -5060,7 +5068,7 @@ static u32 wqattrs_hash(const struct workqueue_attrs *attrs)
static bool wqattrs_equal(const struct workqueue_attrs *a,
const struct workqueue_attrs *b)
{
- if (a->nice != b->nice)
+ if (a->prio != b->prio)
return false;
if (a->affn_strict != b->affn_strict)
return false;
@@ -5928,8 +5936,19 @@ static struct workqueue_attrs *alloc_wq_std_attrs(struct workqueue_struct *wq)
if (!attrs)
return NULL;

- if (wq->flags & WQ_HIGHPRI)
- attrs->nice = HIGHPRI_NICE_LEVEL;
+ if (wq->flags & WQ_RT) {
+ attrs->prio = RT_PRIORITY;
+ /*
+ * RT workqueues have strict CPU affinity for low
+ * latency execution.
+ */
+ attrs->affn_scope = WQ_AFFN_CPU;
+ attrs->affn_strict = true;
+ } else if (wq->flags & WQ_HIGHPRI) {
+ attrs->prio = HIGHPRI_PRIORITY;
+ } else {
+ attrs->prio = DEFAULT_PRIO;
+ }

if (wq->flags & __WQ_ORDERED)
attrs->ordered = true;
@@ -6115,6 +6134,12 @@ static struct workqueue_struct *__alloc_workqueue(const char *fmt,
return NULL;
}

+ if (flags & WQ_RT) {
+ if (WARN_ON_ONCE((flags & (WQ_HIGHPRI | WQ_UNBOUND)) !=
+ WQ_UNBOUND))
+ return NULL;
+ }
+
/* see the comment above the definition of WQ_POWER_EFFICIENT */
if ((flags & WQ_POWER_EFFICIENT) && wq_power_efficient)
flags = (flags & ~WQ_PERCPU) | WQ_UNBOUND;
@@ -6671,9 +6696,9 @@ static void pr_cont_pool_info(struct worker_pool *pool)
pr_cont(" flags=0x%x", pool->flags);
if (pool->flags & POOL_BH)
pr_cont(" bh%s",
- pool->attrs->nice == HIGHPRI_NICE_LEVEL ? "-hi" : "");
+ pool->attrs->prio == HIGHPRI_PRIORITY ? "-hi" : "");
else
- pr_cont(" nice=%d", pool->attrs->nice);
+ pr_cont(" nice=%d", PRIO_TO_NICE(pool->attrs->prio));
}

static void pr_cont_worker_id(struct worker *worker)
@@ -6682,7 +6707,7 @@ static void pr_cont_worker_id(struct worker *worker)

if (pool->flags & POOL_BH)
pr_cont("bh%s",
- pool->attrs->nice == HIGHPRI_NICE_LEVEL ? "-hi" : "");
+ pool->attrs->prio == HIGHPRI_PRIORITY ? "-hi" : "");
else
pr_cont("%d%s", task_pid_nr(worker->task),
worker->rescue_wq ? "(RESCUER)" : "");
@@ -7606,7 +7631,11 @@ static ssize_t nice_show(struct device *dev, struct device_attribute *attr,
int written;

mutex_lock(&wq->mutex);
- written = scnprintf(buf, PAGE_SIZE, "%d\n", wq->attrs->nice);
+ if (wq->attrs->prio == RT_PRIORITY)
+ written = scnprintf(buf, PAGE_SIZE, "rt\n");
+ else
+ written = scnprintf(buf, PAGE_SIZE, "%d\n",
+ PRIO_TO_NICE(wq->attrs->prio));
mutex_unlock(&wq->mutex);

return written;
@@ -7632,19 +7661,21 @@ static ssize_t nice_store(struct device *dev, struct device_attribute *attr,
{
struct workqueue_struct *wq = dev_to_wq(dev);
struct workqueue_attrs *attrs;
- int ret = -ENOMEM;
+ int ret, nice = 0;
+
+ if (sscanf(buf, "%d", &nice) != 1 || nice < MIN_NICE || nice > MAX_NICE)
+ return -EINVAL;

mutex_lock(&wq_pool_mutex);

attrs = wq_sysfs_prep_attrs(wq);
- if (!attrs)
+ if (!attrs) {
+ ret = -ENOMEM;
goto out_unlock;
+ }

- if (sscanf(buf, "%d", &attrs->nice) == 1 &&
- attrs->nice >= MIN_NICE && attrs->nice <= MAX_NICE)
- ret = apply_workqueue_attrs_locked(wq, attrs);
- else
- ret = -EINVAL;
+ attrs->prio = NICE_TO_PRIO(nice);
+ ret = apply_workqueue_attrs_locked(wq, attrs);

out_unlock:
mutex_unlock(&wq_pool_mutex);
@@ -7716,6 +7747,10 @@ static ssize_t affinity_scope_store(struct device *dev,
struct workqueue_attrs *attrs;
int affn, ret = -ENOMEM;

+ /* Do not allow affinity changes for RT workers. */
+ if (wq->flags & WQ_RT)
+ return -EINVAL;
+
affn = parse_affn_scope(buf);
if (affn < 0)
return affn;
@@ -7748,6 +7783,10 @@ static ssize_t affinity_strict_store(struct device *dev,
struct workqueue_attrs *attrs;
int v, ret = -ENOMEM;

+ /* Do not allow affinity changes for RT workers. */
+ if (wq->flags & WQ_RT)
+ return -EINVAL;
+
if (sscanf(buf, "%d", &v) != 1)
return -EINVAL;

@@ -7786,6 +7825,10 @@ static umode_t wq_sysfs_unbound_group_visible(struct kobject *kobj,
if (!(wq->flags & WQ_UNBOUND))
return SYSFS_GROUP_INVISIBLE;

+ /* Do not allow priority changes for RT workers. */
+ if ((wq->flags & WQ_RT) && !strcmp(attr->name, "nice"))
+ return 0444;
+
return attr->mode;
}

@@ -8310,13 +8353,13 @@ static void __init restrict_unbound_cpumask(const char *name, const struct cpuma
cpumask_and(wq_unbound_cpumask, wq_unbound_cpumask, mask);
}

-static void __init init_cpu_worker_pool(struct worker_pool *pool, int cpu, int nice)
+static void __init init_cpu_worker_pool(struct worker_pool *pool, int cpu, int prio)
{
BUG_ON(init_worker_pool(pool));
pool->cpu = cpu;
cpumask_copy(pool->attrs->cpumask, cpumask_of(cpu));
cpumask_copy(pool->attrs->__pod_cpumask, cpumask_of(cpu));
- pool->attrs->nice = nice;
+ pool->attrs->prio = prio;
pool->attrs->affn_strict = true;
pool->node = cpu_to_node(cpu);

@@ -8339,7 +8382,7 @@ static void __init init_cpu_worker_pool(struct worker_pool *pool, int cpu, int n
void __init workqueue_init_early(void)
{
struct wq_pod_type *pt = &wq_pod_types[WQ_AFFN_SYSTEM];
- int std_nice[NR_STD_WORKER_POOLS] = { 0, HIGHPRI_NICE_LEVEL };
+ int std_prio[NR_STD_WORKER_POOLS] = { DEFAULT_PRIO, HIGHPRI_PRIORITY };
void (*irq_work_fns[NR_STD_WORKER_POOLS])(struct irq_work *) =
{ bh_pool_kick_normal, bh_pool_kick_highpri };
int i, cpu;
@@ -8391,7 +8434,7 @@ void __init workqueue_init_early(void)

i = 0;
for_each_bh_worker_pool(pool, cpu) {
- init_cpu_worker_pool(pool, cpu, std_nice[i]);
+ init_cpu_worker_pool(pool, cpu, std_prio[i]);
pool->flags |= POOL_BH;
init_irq_work(bh_pool_irq_work(pool), irq_work_fns[i]);
i++;
@@ -8399,7 +8442,7 @@ void __init workqueue_init_early(void)

i = 0;
for_each_cpu_worker_pool(pool, cpu)
- init_cpu_worker_pool(pool, cpu, std_nice[i++]);
+ init_cpu_worker_pool(pool, cpu, std_prio[i++]);
}

system_wq = alloc_workqueue("events", WQ_PERCPU | __WQ_DEPRECATED, 0);
diff --git a/tools/workqueue/wq_dump.py b/tools/workqueue/wq_dump.py
index 9313ebe0c525..371601b086ca 100644
--- a/tools/workqueue/wq_dump.py
+++ b/tools/workqueue/wq_dump.py
@@ -24,7 +24,7 @@ Worker Pools
Lists all worker pools indexed by their ID. For each pool:

ref number of pool_workqueue's associated with this pool
- nice nice value of the worker threads in the pool
+ prio priority of the worker threads in the pool
idle number of idle workers
workers number of all workers
cpu CPU the pool is associated with (per-cpu pool)
@@ -122,6 +122,8 @@ POOL_BH = prog['POOL_BH']
WQ_NAME_LEN = prog['WQ_NAME_LEN'].value_()
cpumask_str_len = len(cpumask_str(wq_unbound_cpumask))

+rt_prio = prog.constant('RT_PRIORITY', filename='kernel/workqueue.c')
+
print('Affinity Scopes')
print('===============')

@@ -163,7 +165,10 @@ for pi, pool in idr_for_each(worker_pool_idr):

for pi, pool in idr_for_each(worker_pool_idr):
pool = drgn.Object(prog, 'struct worker_pool', address=pool)
- print(f'pool[{pi:0{max_pool_id_len}}] flags=0x{pool.flags.value_():02x} ref={pool.refcnt.value_():{max_ref_len}} nice={pool.attrs.nice.value_():3} ', end='')
+ prio = pool.attrs.prio.value_()
+ if prio == rt_prio:
+ prio = 'rt'
+ print(f'pool[{pi:0{max_pool_id_len}}] flags=0x{pool.flags.value_():02x} ref={pool.refcnt.value_():{max_ref_len}} prio={prio:3} ', end='')
print(f'idle/workers={pool.nr_idle.value_():3}/{pool.nr_workers.value_():3} ', end='')
if pool.cpu >= 0:
print(f'cpu={pool.cpu.value_():3}', end='')
--
2.55.0