El lun, 21-09-2026 a las 18:24 -0300, Maíra Canal escribió:
> Every job merges its finished fence into the per-queue accumulator so
> that a job carrying a perfmon can later depend on everything still in
> flight. This is a (relatively high) cost that every job pays before
> it is submitted.
>
> Perfmons only exist while userspace runs a performance query, so the
> common case is a client paying that on every job for a dependency
> that is
> never actually added, increasing the submission latency.
>
> Address this situation by counting the number of perfmons alive on
> the
> device and returning early while that count is zero. This means that
> if
> no perfmon exists, v3d_serialize_for_perfmon() will bail out
> immediately.
>
> Jobs submitted while no perfmon is alive stay unaccounted and may
> overlap
> the first measured job. The count returns to zero whenever the last
> perfmon is destroyed, so that window reopens on every measurement
> cycle.
> Closing it would mean merging a fence on every submission for the
> lifetime of the device, so it is a reasonable compromise for the
> average
> use case.
>
> Signed-off-by: Maíra Canal <[email protected]>
> ---
> drivers/gpu/drm/v3d/v3d_drv.h | 7 +++++++
> drivers/gpu/drm/v3d/v3d_perfmon.c | 8 ++++++--
> drivers/gpu/drm/v3d/v3d_submit.c | 7 +++++++
> 3 files changed, 20 insertions(+), 2 deletions(-)
>
> diff --git a/drivers/gpu/drm/v3d/v3d_drv.h
> b/drivers/gpu/drm/v3d/v3d_drv.h
> index 79ef20c89ff2..cd544af8d8c8 100644
> --- a/drivers/gpu/drm/v3d/v3d_drv.h
> +++ b/drivers/gpu/drm/v3d/v3d_drv.h
> @@ -84,6 +84,8 @@ struct v3d_queue_state {
> * This way, only events related to a specific submission will be
> counted.
> */
> struct v3d_perfmon {
> + struct v3d_dev *v3d;
> +
> /* Tracks the number of users of the perfmon, when this
> counter reaches
> * zero the perfmon is destroyed.
> */
> @@ -184,6 +186,11 @@ struct v3d_dev {
> /* Perfmon currently programmed in HW (or NULL if
> none). */
> struct v3d_perfmon *active;
>
> + /* Number of perfmons alive on this device. Jobs are
> not
> + * serialized if the number is zero.
> + */
> + atomic_t nperfmons;
> +
Maybe we should also amend the comment for the last_hw_fence field, to
make it explicit that last fence tracking is only accurate while
nperfmons > 0?
> /* Finished fence of the most recently submitted job
> that
> * opened a serialization window (i.e. a job with a
> non-global
> * perfmon attached).
> diff --git a/drivers/gpu/drm/v3d/v3d_perfmon.c
> b/drivers/gpu/drm/v3d/v3d_perfmon.c
> index 07dab7fb3060..359334616bd8 100644
> --- a/drivers/gpu/drm/v3d/v3d_perfmon.c
> +++ b/drivers/gpu/drm/v3d/v3d_perfmon.c
> @@ -217,8 +217,10 @@ void v3d_perfmon_get(struct v3d_perfmon
> *perfmon)
>
> void v3d_perfmon_put(struct v3d_perfmon *perfmon)
> {
> - if (perfmon && refcount_dec_and_test(&perfmon->refcnt))
> + if (perfmon && refcount_dec_and_test(&perfmon->refcnt)) {
> + atomic_dec(&perfmon->v3d->perfmon_state.nperfmons);
> kfree(perfmon);
> + }
> }
>
> static void v3d_perfmon_hw_start(struct v3d_dev *v3d, struct
> v3d_perfmon *perfmon)
> @@ -434,13 +436,15 @@ int v3d_perfmon_create_ioctl(struct drm_device
> *dev, void *data,
> perfmon->counters[i] = req->counters[i];
>
> perfmon->ncounters = req->ncounters;
> + perfmon->v3d = v3d;
>
> refcount_set(&perfmon->refcnt, 1);
> + atomic_inc(&v3d->perfmon_state.nperfmons);
>
> ret = xa_alloc(&v3d_priv->perfmons, &id, perfmon,
> xa_limit_32b,
> GFP_KERNEL);
> if (ret < 0) {
> - kfree(perfmon);
> + v3d_perfmon_put(perfmon);
> return ret;
> }
>
> diff --git a/drivers/gpu/drm/v3d/v3d_submit.c
> b/drivers/gpu/drm/v3d/v3d_submit.c
> index 834d52030979..bc3c43fd4fd9 100644
> --- a/drivers/gpu/drm/v3d/v3d_submit.c
> +++ b/drivers/gpu/drm/v3d/v3d_submit.c
> @@ -357,6 +357,10 @@ v3d_attach_perfmon_to_jobs(struct v3d_submit
> *submit, u32 perfmon_id)
> *
> * We don't serialize the jobs when using a global perfmon as it's
> expected to
> * track concurrent activity from all jobs.
> + *
> + * Keeping track of the in-flight jobs costs a fence merge per job,
> so it is
> + * only done while at least one perfmon is alive. Jobs submitted
> while no
> + * perfmon exists go untracked and may overlap the first measured
> job.
> */
> static int
> v3d_serialize_for_perfmon(struct v3d_job *job)
> @@ -368,6 +372,9 @@ v3d_serialize_for_perfmon(struct v3d_job *job)
>
> lockdep_assert_held(&v3d->sched_lock);
>
> + if (!atomic_read(&v3d->perfmon_state.nperfmons))
> + return 0;
> +
> scoped_guard(spinlock_irqsave, &v3d->perfmon_state.lock)
> is_global_perfmon = !!v3d->global_perfmon;
>
>