El lun, 21-09-2026 a las 18:24 -0300, Maíra Canal escribió:
> Every job merges its finished fence into the per-queue accumulator so
> that a job carrying a perfmon can later depend on everything still in
> flight. This is a (relatively high) cost that every job pays before
> it is submitted.
> 
> Perfmons only exist while userspace runs a performance query, so the
> common case is a client paying that on every job for a dependency
> that is
> never actually added, increasing the submission latency.
> 
> Address this situation by counting the number of perfmons alive on
> the
> device and returning early while that count is zero. This means that
> if
> no perfmon exists, v3d_serialize_for_perfmon() will bail out
> immediately.
> 
> Jobs submitted while no perfmon is alive stay unaccounted and may
> overlap
> the first measured job. The count returns to zero whenever the last
> perfmon is destroyed, so that window reopens on every measurement
> cycle.
> Closing it would mean merging a fence on every submission for the
> lifetime of the device, so it is a reasonable compromise for the
> average
> use case.
> 
> Signed-off-by: Maíra Canal <[email protected]>
> ---
>  drivers/gpu/drm/v3d/v3d_drv.h     | 7 +++++++
>  drivers/gpu/drm/v3d/v3d_perfmon.c | 8 ++++++--
>  drivers/gpu/drm/v3d/v3d_submit.c  | 7 +++++++
>  3 files changed, 20 insertions(+), 2 deletions(-)
> 
> diff --git a/drivers/gpu/drm/v3d/v3d_drv.h
> b/drivers/gpu/drm/v3d/v3d_drv.h
> index 79ef20c89ff2..cd544af8d8c8 100644
> --- a/drivers/gpu/drm/v3d/v3d_drv.h
> +++ b/drivers/gpu/drm/v3d/v3d_drv.h
> @@ -84,6 +84,8 @@ struct v3d_queue_state {
>   * This way, only events related to a specific submission will be
> counted.
>   */
>  struct v3d_perfmon {
> +     struct v3d_dev *v3d;
> +
>       /* Tracks the number of users of the perfmon, when this
> counter reaches
>        * zero the perfmon is destroyed.
>        */
> @@ -184,6 +186,11 @@ struct v3d_dev {
>               /* Perfmon currently programmed in HW (or NULL if
> none). */
>               struct v3d_perfmon *active;
>  
> +             /* Number of perfmons alive on this device. Jobs are
> not
> +              * serialized if the number is zero.
> +              */
> +             atomic_t nperfmons;
> +

Maybe we should also amend the comment for the last_hw_fence field, to
make it explicit that last fence tracking is only accurate while
nperfmons > 0?

>               /* Finished fence of the most recently submitted job
> that
>                * opened a serialization window (i.e. a job with a
> non-global
>                * perfmon attached).
> diff --git a/drivers/gpu/drm/v3d/v3d_perfmon.c
> b/drivers/gpu/drm/v3d/v3d_perfmon.c
> index 07dab7fb3060..359334616bd8 100644
> --- a/drivers/gpu/drm/v3d/v3d_perfmon.c
> +++ b/drivers/gpu/drm/v3d/v3d_perfmon.c
> @@ -217,8 +217,10 @@ void v3d_perfmon_get(struct v3d_perfmon
> *perfmon)
>  
>  void v3d_perfmon_put(struct v3d_perfmon *perfmon)
>  {
> -     if (perfmon && refcount_dec_and_test(&perfmon->refcnt))
> +     if (perfmon && refcount_dec_and_test(&perfmon->refcnt)) {
> +             atomic_dec(&perfmon->v3d->perfmon_state.nperfmons);
>               kfree(perfmon);
> +     }
>  }
>  
>  static void v3d_perfmon_hw_start(struct v3d_dev *v3d, struct
> v3d_perfmon *perfmon)
> @@ -434,13 +436,15 @@ int v3d_perfmon_create_ioctl(struct drm_device
> *dev, void *data,
>               perfmon->counters[i] = req->counters[i];
>  
>       perfmon->ncounters = req->ncounters;
> +     perfmon->v3d = v3d;
>  
>       refcount_set(&perfmon->refcnt, 1);
> +     atomic_inc(&v3d->perfmon_state.nperfmons);
>  
>       ret = xa_alloc(&v3d_priv->perfmons, &id, perfmon,
> xa_limit_32b,
>                      GFP_KERNEL);
>       if (ret < 0) {
> -             kfree(perfmon);
> +             v3d_perfmon_put(perfmon);
>               return ret;
>       }
>  
> diff --git a/drivers/gpu/drm/v3d/v3d_submit.c
> b/drivers/gpu/drm/v3d/v3d_submit.c
> index 834d52030979..bc3c43fd4fd9 100644
> --- a/drivers/gpu/drm/v3d/v3d_submit.c
> +++ b/drivers/gpu/drm/v3d/v3d_submit.c
> @@ -357,6 +357,10 @@ v3d_attach_perfmon_to_jobs(struct v3d_submit
> *submit, u32 perfmon_id)
>   *
>   * We don't serialize the jobs when using a global perfmon as it's
> expected to
>   * track concurrent activity from all jobs.
> + *
> + * Keeping track of the in-flight jobs costs a fence merge per job,
> so it is
> + * only done while at least one perfmon is alive. Jobs submitted
> while no
> + * perfmon exists go untracked and may overlap the first measured
> job.
>   */
>  static int
>  v3d_serialize_for_perfmon(struct v3d_job *job)
> @@ -368,6 +372,9 @@ v3d_serialize_for_perfmon(struct v3d_job *job)
>  
>       lockdep_assert_held(&v3d->sched_lock);
>  
> +     if (!atomic_read(&v3d->perfmon_state.nperfmons))
> +             return 0;
> +
>       scoped_guard(spinlock_irqsave, &v3d->perfmon_state.lock)
>               is_global_perfmon = !!v3d->global_perfmon;
>  
> 

Reply via email to