* [RFC PATCH 1/3] workqueue: Add interface for user-defined workqueue lockdep map
2024-07-30 22:17 [RFC PATCH 0/3] Use user-defined workqueue lockdep map for drm sched Matthew Brost
@ 2024-07-30 22:17 ` Matthew Brost
2024-07-30 22:34 ` Tejun Heo
2024-07-30 22:17 ` [RFC PATCH 2/3] drm/sched: Use drm sched lockdep map for submit_wq Matthew Brost
2024-07-30 22:17 ` [RFC PATCH 3/3] drm/xe: Drop GuC submit_wq pool Matthew Brost
2 siblings, 1 reply; 8+ messages in thread
From: Matthew Brost @ 2024-07-30 22:17 UTC (permalink / raw)
To: intel-xe, dri-devel, linux-kernel
Cc: tj, jiangshanlai, christian.koenig, ltuikov89, daniel
Add an interface for a user-defined workqueue lockdep map, which is
helpful when multiple workqueues are created for the same purpose. This
also helps avoid leaking lockdep maps on each workqueue creation.
Implement a new workqueue flag, WQ_USER_OWNED_LOCKDEP, to indicate that
the user will set up the workqueue lockdep map using the new function
wq_init_user_lockdep_map.
Cc: Tejun Heo <tj@kernel.org>
Cc: Lai Jiangshan <jiangshanlai@gmail.com>
Signed-off-by: Matthew Brost <matthew.brost@intel.com>
---
include/linux/workqueue.h | 3 +++
kernel/workqueue.c | 44 ++++++++++++++++++++++++++++++++-------
2 files changed, 40 insertions(+), 7 deletions(-)
diff --git a/include/linux/workqueue.h b/include/linux/workqueue.h
index d9968bfc8eac..3e6db0889e2b 100644
--- a/include/linux/workqueue.h
+++ b/include/linux/workqueue.h
@@ -223,6 +223,8 @@ struct execute_work {
};
#ifdef CONFIG_LOCKDEP
+void wq_init_user_lockdep_map(struct workqueue_struct *wq,
+ struct lockdep_map *lockdep_map);
/*
* NB: because we have to copy the lockdep_map, setting _key
* here is required, otherwise it could get initialised to the
@@ -401,6 +403,7 @@ enum wq_flags {
* http://thread.gmane.org/gmane.linux.kernel/1480396
*/
WQ_POWER_EFFICIENT = 1 << 7,
+ WQ_USER_OWNED_LOCKDEP = 1 << 8, /* allow users to define lockdep map */
__WQ_DESTROYING = 1 << 15, /* internal: workqueue is destroying */
__WQ_DRAINING = 1 << 16, /* internal: workqueue is draining */
diff --git a/kernel/workqueue.c b/kernel/workqueue.c
index 3fbaecfc88c2..228b52b8d7c4 100644
--- a/kernel/workqueue.c
+++ b/kernel/workqueue.c
@@ -366,7 +366,8 @@ struct workqueue_struct {
#ifdef CONFIG_LOCKDEP
char *lock_name;
struct lock_class_key key;
- struct lockdep_map lockdep_map;
+ struct lockdep_map __lockdep_map;
+ struct lockdep_map *lockdep_map;
#endif
char name[WQ_NAME_LEN]; /* I: workqueue name */
@@ -3220,7 +3221,7 @@ __acquires(&pool->lock)
lockdep_start_depth = lockdep_depth(current);
/* see drain_dead_softirq_workfn() */
if (!bh_draining)
- lock_map_acquire(&pwq->wq->lockdep_map);
+ lock_map_acquire(pwq->wq->lockdep_map);
lock_map_acquire(&lockdep_map);
/*
* Strictly speaking we should mark the invariant state without holding
@@ -3254,7 +3255,7 @@ __acquires(&pool->lock)
pwq->stats[PWQ_STAT_COMPLETED]++;
lock_map_release(&lockdep_map);
if (!bh_draining)
- lock_map_release(&pwq->wq->lockdep_map);
+ lock_map_release(pwq->wq->lockdep_map);
if (unlikely((worker->task && in_atomic()) ||
lockdep_depth(current) != lockdep_start_depth ||
@@ -3892,8 +3893,8 @@ static void touch_wq_lockdep_map(struct workqueue_struct *wq)
if (wq->flags & WQ_BH)
local_bh_disable();
- lock_map_acquire(&wq->lockdep_map);
- lock_map_release(&wq->lockdep_map);
+ lock_map_acquire(wq->lockdep_map);
+ lock_map_release(wq->lockdep_map);
if (wq->flags & WQ_BH)
local_bh_enable();
@@ -3927,7 +3928,8 @@ void __flush_workqueue(struct workqueue_struct *wq)
struct wq_flusher this_flusher = {
.list = LIST_HEAD_INIT(this_flusher.list),
.flush_color = -1,
- .done = COMPLETION_INITIALIZER_ONSTACK_MAP(this_flusher.done, wq->lockdep_map),
+ .done = COMPLETION_INITIALIZER_ONSTACK_MAP(this_flusher.done,
+ (*wq->lockdep_map)),
};
int next_color;
@@ -4778,26 +4780,54 @@ static int init_worker_pool(struct worker_pool *pool)
}
#ifdef CONFIG_LOCKDEP
+/**
+ * wq_init_user_lockdep_map - init user lockdep map for workqueue
+ * @wq: workqueue to init lockdep map for
+ * @lockdep_map: lockdep map to use for workqueue
+ *
+ * Initialize workqueue with a user defined lockdep map. WQ_USER_OWNED_LOCKDEP
+ * must be set for workqueue.
+ */
+void wq_init_user_lockdep_map(struct workqueue_struct *wq,
+ struct lockdep_map *lockdep_map)
+{
+ if (WARN_ON_ONCE(!(wq->flags & WQ_USER_OWNED_LOCKDEP)))
+ return;
+
+ wq->lockdep_map = lockdep_map;
+}
+EXPORT_SYMBOL_GPL(wq_init_user_lockdep_map);
+
static void wq_init_lockdep(struct workqueue_struct *wq)
{
char *lock_name;
+ if (wq->flags & WQ_USER_OWNED_LOCKDEP)
+ return;
+
lockdep_register_key(&wq->key);
lock_name = kasprintf(GFP_KERNEL, "%s%s", "(wq_completion)", wq->name);
if (!lock_name)
lock_name = wq->name;
wq->lock_name = lock_name;
- lockdep_init_map(&wq->lockdep_map, lock_name, &wq->key, 0);
+ wq->lockdep_map = &wq->__lockdep_map;
+ lockdep_init_map(wq->lockdep_map, lock_name, &wq->key, 0);
}
static void wq_unregister_lockdep(struct workqueue_struct *wq)
{
+ if (wq->flags & WQ_USER_OWNED_LOCKDEP)
+ return;
+
lockdep_unregister_key(&wq->key);
}
static void wq_free_lockdep(struct workqueue_struct *wq)
{
+ if (wq->flags & WQ_USER_OWNED_LOCKDEP)
+ return;
+
if (wq->lock_name != wq->name)
kfree(wq->lock_name);
}
--
2.34.1
^ permalink raw reply [flat|nested] 8+ messages in thread* Re: [RFC PATCH 1/3] workqueue: Add interface for user-defined workqueue lockdep map
2024-07-30 22:17 ` [RFC PATCH 1/3] workqueue: Add interface for user-defined workqueue lockdep map Matthew Brost
@ 2024-07-30 22:34 ` Tejun Heo
2024-07-30 22:53 ` Matthew Brost
0 siblings, 1 reply; 8+ messages in thread
From: Tejun Heo @ 2024-07-30 22:34 UTC (permalink / raw)
To: Matthew Brost
Cc: intel-xe, dri-devel, linux-kernel, jiangshanlai,
christian.koenig, ltuikov89, daniel
Hello, Matthew.
On Tue, Jul 30, 2024 at 03:17:40PM -0700, Matthew Brost wrote:
> +/**
> + * wq_init_user_lockdep_map - init user lockdep map for workqueue
> + * @wq: workqueue to init lockdep map for
> + * @lockdep_map: lockdep map to use for workqueue
> + *
> + * Initialize workqueue with a user defined lockdep map. WQ_USER_OWNED_LOCKDEP
> + * must be set for workqueue.
> + */
> +void wq_init_user_lockdep_map(struct workqueue_struct *wq,
> + struct lockdep_map *lockdep_map)
> +{
> + if (WARN_ON_ONCE(!(wq->flags & WQ_USER_OWNED_LOCKDEP)))
> + return;
> +
> + wq->lockdep_map = lockdep_map;
> +}
> +EXPORT_SYMBOL_GPL(wq_init_user_lockdep_map);
Would it be possible to make it a one-piece interface - ie. add
alloc_workqueue_lockdep_map() which takes an external lockdep map rather
than splitting it over two calls?
Thanks.
--
tejun
^ permalink raw reply [flat|nested] 8+ messages in thread* Re: [RFC PATCH 1/3] workqueue: Add interface for user-defined workqueue lockdep map
2024-07-30 22:34 ` Tejun Heo
@ 2024-07-30 22:53 ` Matthew Brost
2024-07-30 22:56 ` Tejun Heo
0 siblings, 1 reply; 8+ messages in thread
From: Matthew Brost @ 2024-07-30 22:53 UTC (permalink / raw)
To: Tejun Heo
Cc: intel-xe, dri-devel, linux-kernel, jiangshanlai,
christian.koenig, ltuikov89, daniel
On Tue, Jul 30, 2024 at 12:34:08PM -1000, Tejun Heo wrote:
> Hello, Matthew.
>
> On Tue, Jul 30, 2024 at 03:17:40PM -0700, Matthew Brost wrote:
> > +/**
> > + * wq_init_user_lockdep_map - init user lockdep map for workqueue
> > + * @wq: workqueue to init lockdep map for
> > + * @lockdep_map: lockdep map to use for workqueue
> > + *
> > + * Initialize workqueue with a user defined lockdep map. WQ_USER_OWNED_LOCKDEP
> > + * must be set for workqueue.
> > + */
> > +void wq_init_user_lockdep_map(struct workqueue_struct *wq,
> > + struct lockdep_map *lockdep_map)
> > +{
> > + if (WARN_ON_ONCE(!(wq->flags & WQ_USER_OWNED_LOCKDEP)))
> > + return;
> > +
> > + wq->lockdep_map = lockdep_map;
> > +}
> > +EXPORT_SYMBOL_GPL(wq_init_user_lockdep_map);
>
> Would it be possible to make it a one-piece interface - ie. add
> alloc_workqueue_lockdep_map() which takes an external lockdep map rather
> than splitting it over two calls?
>
I didn't want to change the export alloc_workqueue() arguments so I went
with this approach. Are you suggesting export a new function
alloc_workqueue_lockdep_map() which will share an internal
implementation with the existing alloc_workqueue() but passes in a
lockdep map? That could work.
Matt
> Thanks.
>
> --
> tejun
^ permalink raw reply [flat|nested] 8+ messages in thread* Re: [RFC PATCH 1/3] workqueue: Add interface for user-defined workqueue lockdep map
2024-07-30 22:53 ` Matthew Brost
@ 2024-07-30 22:56 ` Tejun Heo
2024-07-30 22:56 ` Matthew Brost
0 siblings, 1 reply; 8+ messages in thread
From: Tejun Heo @ 2024-07-30 22:56 UTC (permalink / raw)
To: Matthew Brost
Cc: intel-xe, dri-devel, linux-kernel, jiangshanlai,
christian.koenig, ltuikov89, daniel
On Tue, Jul 30, 2024 at 10:53:38PM +0000, Matthew Brost wrote:
> I didn't want to change the export alloc_workqueue() arguments so I went
> with this approach. Are you suggesting export a new function
> alloc_workqueue_lockdep_map() which will share an internal
> implementation with the existing alloc_workqueue() but passes in a
> lockdep map? That could work.
Yeah, add a new exported function which takes lockdep_map and make
alloc_workqueue() to call that with the embedded map. No need to make the
latter inline either.
Thanks.
--
tejun
^ permalink raw reply [flat|nested] 8+ messages in thread
* Re: [RFC PATCH 1/3] workqueue: Add interface for user-defined workqueue lockdep map
2024-07-30 22:56 ` Tejun Heo
@ 2024-07-30 22:56 ` Matthew Brost
0 siblings, 0 replies; 8+ messages in thread
From: Matthew Brost @ 2024-07-30 22:56 UTC (permalink / raw)
To: Tejun Heo
Cc: intel-xe, dri-devel, linux-kernel, jiangshanlai,
christian.koenig, ltuikov89, daniel
On Tue, Jul 30, 2024 at 12:56:17PM -1000, Tejun Heo wrote:
> On Tue, Jul 30, 2024 at 10:53:38PM +0000, Matthew Brost wrote:
> > I didn't want to change the export alloc_workqueue() arguments so I went
> > with this approach. Are you suggesting export a new function
> > alloc_workqueue_lockdep_map() which will share an internal
> > implementation with the existing alloc_workqueue() but passes in a
> > lockdep map? That could work.
>
> Yeah, add a new exported function which takes lockdep_map and make
> alloc_workqueue() to call that with the embedded map. No need to make the
> latter inline either.
>
Sure, let me do that.
Matt
> Thanks.
>
> --
> tejun
^ permalink raw reply [flat|nested] 8+ messages in thread
* [RFC PATCH 2/3] drm/sched: Use drm sched lockdep map for submit_wq
2024-07-30 22:17 [RFC PATCH 0/3] Use user-defined workqueue lockdep map for drm sched Matthew Brost
2024-07-30 22:17 ` [RFC PATCH 1/3] workqueue: Add interface for user-defined workqueue lockdep map Matthew Brost
@ 2024-07-30 22:17 ` Matthew Brost
2024-07-30 22:17 ` [RFC PATCH 3/3] drm/xe: Drop GuC submit_wq pool Matthew Brost
2 siblings, 0 replies; 8+ messages in thread
From: Matthew Brost @ 2024-07-30 22:17 UTC (permalink / raw)
To: intel-xe, dri-devel, linux-kernel
Cc: tj, jiangshanlai, christian.koenig, ltuikov89, daniel
Avoid leaking a lockdep map on each drm sched creation and destruction
by using a single lockdep map for all drm sched allocated submit_wq.
Cc: Luben Tuikov <ltuikov89@gmail.com>
Cc: Christian König <christian.koenig@amd.com>
Signed-off-by: Matthew Brost <matthew.brost@intel.com>
---
drivers/gpu/drm/scheduler/sched_main.c | 12 +++++++++++-
1 file changed, 11 insertions(+), 1 deletion(-)
diff --git a/drivers/gpu/drm/scheduler/sched_main.c b/drivers/gpu/drm/scheduler/sched_main.c
index ab53ab486fe6..9849fd64aff9 100644
--- a/drivers/gpu/drm/scheduler/sched_main.c
+++ b/drivers/gpu/drm/scheduler/sched_main.c
@@ -87,6 +87,12 @@
#define CREATE_TRACE_POINTS
#include "gpu_scheduler_trace.h"
+#ifdef CONFIG_LOCKDEP
+static struct lockdep_map drm_sched_lockdep_map = {
+ .name = "drm_sched_lockdep_map"
+};
+#endif
+
#define to_drm_sched_job(sched_job) \
container_of((sched_job), struct drm_sched_job, queue_node)
@@ -1272,9 +1278,13 @@ int drm_sched_init(struct drm_gpu_scheduler *sched,
sched->submit_wq = submit_wq;
sched->own_submit_wq = false;
} else {
- sched->submit_wq = alloc_ordered_workqueue(name, 0);
+ sched->submit_wq = alloc_ordered_workqueue(name, WQ_USER_OWNED_LOCKDEP);
if (!sched->submit_wq)
return -ENOMEM;
+#ifdef CONFIG_LOCKDEP
+ wq_init_user_lockdep_map(sched->submit_wq,
+ &drm_sched_lockdep_map);
+#endif
sched->own_submit_wq = true;
}
--
2.34.1
^ permalink raw reply [flat|nested] 8+ messages in thread* [RFC PATCH 3/3] drm/xe: Drop GuC submit_wq pool
2024-07-30 22:17 [RFC PATCH 0/3] Use user-defined workqueue lockdep map for drm sched Matthew Brost
2024-07-30 22:17 ` [RFC PATCH 1/3] workqueue: Add interface for user-defined workqueue lockdep map Matthew Brost
2024-07-30 22:17 ` [RFC PATCH 2/3] drm/sched: Use drm sched lockdep map for submit_wq Matthew Brost
@ 2024-07-30 22:17 ` Matthew Brost
2 siblings, 0 replies; 8+ messages in thread
From: Matthew Brost @ 2024-07-30 22:17 UTC (permalink / raw)
To: intel-xe, dri-devel, linux-kernel
Cc: tj, jiangshanlai, christian.koenig, ltuikov89, daniel
Now that drm sched uses a single lockdep map for all submit_wq, drop the
GuC submit_wq pool hack.
Signed-off-by: Matthew Brost <matthew.brost@intel.com>
---
drivers/gpu/drm/xe/xe_guc_submit.c | 60 +-----------------------------
drivers/gpu/drm/xe/xe_guc_types.h | 7 ----
2 files changed, 1 insertion(+), 66 deletions(-)
diff --git a/drivers/gpu/drm/xe/xe_guc_submit.c b/drivers/gpu/drm/xe/xe_guc_submit.c
index 460808507947..882cef3a10dc 100644
--- a/drivers/gpu/drm/xe/xe_guc_submit.c
+++ b/drivers/gpu/drm/xe/xe_guc_submit.c
@@ -224,64 +224,11 @@ static bool exec_queue_killed_or_banned_or_wedged(struct xe_exec_queue *q)
EXEC_QUEUE_STATE_BANNED));
}
-#ifdef CONFIG_PROVE_LOCKING
-static int alloc_submit_wq(struct xe_guc *guc)
-{
- int i;
-
- for (i = 0; i < NUM_SUBMIT_WQ; ++i) {
- guc->submission_state.submit_wq_pool[i] =
- alloc_ordered_workqueue("submit_wq", 0);
- if (!guc->submission_state.submit_wq_pool[i])
- goto err_free;
- }
-
- return 0;
-
-err_free:
- while (i)
- destroy_workqueue(guc->submission_state.submit_wq_pool[--i]);
-
- return -ENOMEM;
-}
-
-static void free_submit_wq(struct xe_guc *guc)
-{
- int i;
-
- for (i = 0; i < NUM_SUBMIT_WQ; ++i)
- destroy_workqueue(guc->submission_state.submit_wq_pool[i]);
-}
-
-static struct workqueue_struct *get_submit_wq(struct xe_guc *guc)
-{
- int idx = guc->submission_state.submit_wq_idx++ % NUM_SUBMIT_WQ;
-
- return guc->submission_state.submit_wq_pool[idx];
-}
-#else
-static int alloc_submit_wq(struct xe_guc *guc)
-{
- return 0;
-}
-
-static void free_submit_wq(struct xe_guc *guc)
-{
-
-}
-
-static struct workqueue_struct *get_submit_wq(struct xe_guc *guc)
-{
- return NULL;
-}
-#endif
-
static void guc_submit_fini(struct drm_device *drm, void *arg)
{
struct xe_guc *guc = arg;
xa_destroy(&guc->submission_state.exec_queue_lookup);
- free_submit_wq(guc);
}
static void guc_submit_wedged_fini(struct drm_device *drm, void *arg)
@@ -337,10 +284,6 @@ int xe_guc_submit_init(struct xe_guc *guc, unsigned int num_ids)
if (err)
return err;
- err = alloc_submit_wq(guc);
- if (err)
- return err;
-
gt->exec_queue_ops = &guc_exec_queue_ops;
xa_init(&guc->submission_state.exec_queue_lookup);
@@ -1445,8 +1388,7 @@ static int guc_exec_queue_init(struct xe_exec_queue *q)
timeout = (q->vm && xe_vm_in_lr_mode(q->vm)) ? MAX_SCHEDULE_TIMEOUT :
msecs_to_jiffies(q->sched_props.job_timeout_ms);
err = xe_sched_init(&ge->sched, &drm_sched_ops, &xe_sched_ops,
- get_submit_wq(guc),
- q->lrc[0]->ring.size / MAX_JOB_SIZE_BYTES, 64,
+ NULL, q->lrc[0]->ring.size / MAX_JOB_SIZE_BYTES, 64,
timeout, guc_to_gt(guc)->ordered_wq, NULL,
q->name, gt_to_xe(q->gt)->drm.dev);
if (err)
diff --git a/drivers/gpu/drm/xe/xe_guc_types.h b/drivers/gpu/drm/xe/xe_guc_types.h
index 546ac6350a31..585f5c274f09 100644
--- a/drivers/gpu/drm/xe/xe_guc_types.h
+++ b/drivers/gpu/drm/xe/xe_guc_types.h
@@ -72,13 +72,6 @@ struct xe_guc {
atomic_t stopped;
/** @submission_state.lock: protects submission state */
struct mutex lock;
-#ifdef CONFIG_PROVE_LOCKING
-#define NUM_SUBMIT_WQ 256
- /** @submission_state.submit_wq_pool: submission ordered workqueues pool */
- struct workqueue_struct *submit_wq_pool[NUM_SUBMIT_WQ];
- /** @submission_state.submit_wq_idx: submission ordered workqueue index */
- int submit_wq_idx;
-#endif
/** @submission_state.enabled: submission is enabled */
bool enabled;
} submission_state;
--
2.34.1
^ permalink raw reply [flat|nested] 8+ messages in thread