spinlock_t queue_lock;
};
-/* Performance monitor object. The perform lifetime is controlled by userspace
- * using perfmon related ioctls. A perfmon can be attached to a submit_cl
- * request, and when this is the case, HW perf counters will be activated just
- * before the submit_cl is submitted to the GPU and disabled when the job is
- * done. This way, only events related to a specific job will be counted.
+/* Performance monitor object
+ *
+ * The performance monitor (perfmon) lifetime is controlled by userspace using
+ * perfmon related ioctls. A perfmon can be attached to a CL or CSD submission
+ * request, and when it is, HW performance counters will be activated just
+ * before the job is submitted to the GPU and disabled when the job is done.
+ * This way, only events related to a specific submission will be counted.
*/
struct v3d_perfmon {
/* Tracks the number of users of the perfmon, when this counter reaches
struct v3d_queue_state queue[V3D_MAX_QUEUES];
- /* Tracks the performance monitor state. */
+ /*
+ * Tracks the performance monitor state and consistency.
+ *
+ * When a non-global perfmon is attached to a job, the scheduler must
+ * not run any other job on the HW concurrently (otherwise, the
+ * counters would be polluted by unrelated work).
+ */
struct {
/* Protects @active. */
spinlock_t lock;
/* Perfmon currently programmed in HW (or NULL if none). */
struct v3d_perfmon *active;
+
+ /* Finished fence of the most recently submitted job that
+ * opened a serialization window (i.e. a job with a non-global
+ * perfmon attached).
+ */
+ struct dma_fence *fence;
+
+ /* Finished fence of the most recently submitted job on each HW
+ * queue. Used so that a new perfmon-carrying job can depend on
+ * every job currently in-flight across all queues.
+ */
+ struct dma_fence *last_hw_fence[V3D_MAX_QUEUES];
} perfmon_state;
/* Protects bo_stats */
* Copyright (C) 2023 Raspberry Pi
*/
+#include <linux/dma-fence-unwrap.h>
+
#include <drm/drm_print.h>
#include <drm/drm_syncobj.h>
return 0;
}
+/*
+ * Prepare fences to enforce job serialization when a perfmon is active. A job
+ * that carries a non-global perfmon must wait for every job currently in-flight
+ * across all HW queues to finish, otherwise concurrent unrelated work on the
+ * same core would pollute the performance counters. Symmetrically, while such a
+ * job is still in-flight, all subsequently submitted jobs must wait for it.
+ *
+ * We don't serialize the jobs when using a global perfmon as it's expected to
+ * track concurrent activity from all jobs.
+ */
+static int
+v3d_serialize_for_perfmon(struct v3d_job *job)
+{
+ struct v3d_dev *v3d = job->v3d;
+ struct dma_fence *merged;
+ bool is_global_perfmon;
+ int ret;
+
+ lockdep_assert_held(&v3d->sched_lock);
+
+ scoped_guard(spinlock_irqsave, &v3d->perfmon_state.lock)
+ is_global_perfmon = !!v3d->global_perfmon;
+
+ if (is_global_perfmon)
+ goto publish;
+
+ if (job->perfmon) {
+ for (enum v3d_queue q = 0; q < V3D_MAX_QUEUES; q++) {
+ struct dma_fence *f = v3d->perfmon_state.last_hw_fence[q];
+
+ if (!f || dma_fence_is_signaled(f))
+ continue;
+
+ ret = drm_sched_job_add_dependency(&job->base, dma_fence_get(f));
+ if (ret)
+ return ret;
+ }
+ } else if (v3d->perfmon_state.fence &&
+ !dma_fence_is_signaled(v3d->perfmon_state.fence)) {
+ ret = drm_sched_job_add_dependency(&job->base,
+ dma_fence_get(v3d->perfmon_state.fence));
+ if (ret)
+ return ret;
+ }
+
+publish:
+ /*
+ * Accumulate every in-flight job on this queue into one merged fence.
+ * A HW queue is fed by several scheduler entities (one per-fd), so jobs
+ * on it can complete out of order.
+ */
+ merged = dma_fence_unwrap_merge(v3d->perfmon_state.last_hw_fence[job->queue],
+ job->done_fence);
+ if (!merged)
+ return -ENOMEM;
+
+ dma_fence_put(v3d->perfmon_state.last_hw_fence[job->queue]);
+ v3d->perfmon_state.last_hw_fence[job->queue] = merged;
+
+ if (job->perfmon && !is_global_perfmon) {
+ dma_fence_put(v3d->perfmon_state.fence);
+ v3d->perfmon_state.fence = dma_fence_get(job->done_fence);
+ }
+
+ return 0;
+}
+
static void
v3d_submit_attach_object_fences(struct v3d_submit *submit)
{
goto err;
}
+ for (int i = 0; i < submit->job_count; i++) {
+ ret = v3d_serialize_for_perfmon(submit->jobs[i]);
+ if (ret)
+ goto err;
+ }
+
for (int i = 0; i < submit->job_count; i++)
drm_sched_entity_push_job(&submit->jobs[i]->base);