Implement kernel-mode command submission and hardware queue packet assembly for AIE4: - Add aie4_cmd_submit() to validate incoming command buffers, reserve GEM fences, serialize submissions via the context pending list, and dispatch to the hardware queue. - Lock all job BOs and attach job->fence directly to their reservation objects. - In amdxdna_fence_create(), allocate a unique timeline context per job. - In amdxdna_fence_get_timeline_name(), use dev_name() backed by the struct device rather than hwctx->name. - Implement packet encoders: fill_direct_pkt() and fill_indirect_pkt(). - Advance the hardware write index, ring the doorbell, and register in-flight jobs on the running list for completion tracking. - Add aie4_hwctx_wait_for_running() with timeout to safely quiesce worker threads.
Co-developed-by: Max Zhen <[email protected]> Signed-off-by: Max Zhen <[email protected]> Co-developed-by: Wendy Liang <[email protected]> Signed-off-by: Wendy Liang <[email protected]> Signed-off-by: David Zhang <[email protected]> --- drivers/accel/amdxdna/aie4_ctx.c | 718 +++++++++++++++++++++++++++- drivers/accel/amdxdna/aie4_pci.c | 2 + drivers/accel/amdxdna/aie4_pci.h | 3 + drivers/accel/amdxdna/amdxdna_ctx.c | 20 +- 4 files changed, 733 insertions(+), 10 deletions(-) diff --git a/drivers/accel/amdxdna/aie4_ctx.c b/drivers/accel/amdxdna/aie4_ctx.c index 59d37bd5c5a0..e9ba1ed93997 100644 --- a/drivers/accel/amdxdna/aie4_ctx.c +++ b/drivers/accel/amdxdna/aie4_ctx.c @@ -25,9 +25,7 @@ #define CTX_INVALID_ID (~0U) #define CTX_INVALID_DOORBELL AMDXDNA_INVALID_DOORBELL_OFFSET -static void job_worker(struct work_struct *work) -{ -} +static void job_worker(struct work_struct *work); static struct cert_comp *aie4_lookup_cert_comp(struct amdxdna_dev_hdl *ndev, u32 msix_idx) { @@ -209,6 +207,7 @@ int aie4_hwctx_create(struct amdxdna_hwctx *hwctx) hwctx->fw_ctx_id = -1; return ret; } + WRITE_ONCE(priv->has_reset, false); WRITE_ONCE(priv->cert_comp, cert_comp); mutex_unlock(&priv->io_lock); hwctx->doorbell_offset = CTX_INVALID_DOORBELL; @@ -217,6 +216,23 @@ int aie4_hwctx_create(struct amdxdna_hwctx *hwctx) return 0; } +/* + * The connected sentinel for the submit wait gate: a linked cert_comp means the + * ctx is created and (for kernel submission) its doorbell is set up. Read + * locklessly (a pointer null-check, never a dereference); create publishes it as + * the last store under io_lock and destroy clears it first, so "connected" + * implies a valid doorbell. + */ +static bool aie4_hwctx_connected(struct amdxdna_hwctx *hwctx) +{ + return !!READ_ONCE(hwctx->priv->cert_comp); +} + +static bool aie4_hwctx_has_reset(struct amdxdna_hwctx *hwctx) +{ + return READ_ONCE(hwctx->priv->has_reset); +} + void aie4_hwctx_destroy(struct amdxdna_hwctx *hwctx, enum aie4_hwctx_flags flags) { struct amdxdna_client *client = hwctx->client; @@ -224,10 +240,16 @@ void aie4_hwctx_destroy(struct amdxdna_hwctx *hwctx, enum aie4_hwctx_flags flags struct amdxdna_dev *xdna = client->xdna; struct amdxdna_dev_hdl *ndev = xdna->dev_handle; struct cert_comp *cert_comp; + bool has_reset = false; drm_WARN_ON(&xdna->ddev, !mutex_is_locked(&xdna->dev_lock)); + if (flags == AIE4_HWCTX_DISCONNECT || flags == AIE4_HWCTX_ERROR) + has_reset = true; + mutex_lock(&priv->io_lock); + if (has_reset) + WRITE_ONCE(priv->has_reset, true); cert_comp = priv->cert_comp; WRITE_ONCE(priv->cert_comp, NULL); mutex_unlock(&priv->io_lock); @@ -237,6 +259,9 @@ void aie4_hwctx_destroy(struct amdxdna_hwctx *hwctx, enum aie4_hwctx_flags flags aie4_put_cert_comp(cert_comp); } + if (has_reset) + wake_up_all(&priv->job_list_wq); + if (flags != AIE4_HWCTX_DISCONNECT) aie4_msg_destroy_context(ndev, priv->hw_ctx_id); @@ -244,7 +269,15 @@ void aie4_hwctx_destroy(struct amdxdna_hwctx *hwctx, enum aie4_hwctx_flags flags hwctx->fw_ctx_id = -1; hwctx->doorbell_offset = CTX_INVALID_DOORBELL; - cancel_work_sync(&priv->job_work); + /* + * When has_reset is true (AIE4_HWCTX_DISCONNECT or ERROR), cancel_work_sync() + * is skipped so the worker can wake up and abort in-flight jobs. Callers + * that re-create the context after DISCONNECT (e.g. reset recovery) must + * synchronize the worker (via aie4_hwctx_wait_for_running()) before calling + * aie4_hwctx_create(). + */ + if (!has_reset) + cancel_work_sync(&priv->job_work); } static void aie4_hwctx_umq_fini(struct amdxdna_hwctx *hwctx) @@ -407,10 +440,18 @@ void aie4_hwctx_fini(struct amdxdna_hwctx *hwctx) { struct amdxdna_hwctx_priv *priv = hwctx->priv; + /* + * ERROR sets has_reset (like a TDR reset) so the worker drains the running + * list - the ctx is gone, so in-flight jobs are reaped - and stays live + * for aie4_hwctx_wait_for_running() to wait on. Submitters are already gone + * (amdxdna_hwctx_destroy_rcu() synchronize_srcu'd them out), then the + * queue is torn down. + */ aie4_hwctx_destroy(hwctx, AIE4_HWCTX_ERROR); - cancel_work_sync(&priv->job_work); - if (priv->job_work_q) + if (priv->job_work_q) { + aie4_hwctx_wait_for_running(hwctx); destroy_workqueue(priv->job_work_q); + } aie4_hwctx_umq_fini(hwctx); mutex_destroy(&priv->io_lock); kfree(hwctx->priv); @@ -537,3 +578,668 @@ int aie4_cmd_wait(struct amdxdna_hwctx *hwctx, u64 seq, u32 timeout) return ret <= 0 ? ret : 0; } + +/* ---- kernel-mode submission (driver fills the queue and rings doorbell) ---- */ + +/* Publish a command to CERT and return the assigned command sequence (slot). */ +static u64 publish_cmd(struct amdxdna_hwctx *hwctx) +{ + struct amdxdna_hwctx_priv *priv = hwctx->priv; + u64 wi = priv->write_index; + + /* Paired with the lockless READ_ONCE() readers of write_index. */ + WRITE_ONCE(priv->write_index, wi + 1); + /* Order the packet-slot writes before CERT sees the new write_index. */ + wmb(); + WRITE_ONCE(*priv->umq_write_index, wi + 1); + return wi; +} + +static int wait_till_seq_completed(struct amdxdna_hwctx *hwctx, u64 seq) +{ + struct cert_comp *cert_comp; + int ret; + + /* + * Freezable + interruptible: the submit path (wait_till_connected_hsa_not_full) + * reaches here while holding hwctx_srcu, and ctx teardown blocks on + * synchronize_srcu(), so a signal (e.g. the app being killed) must be + * able to unwind the wait - otherwise a full queue with a silent CERT + * would hang the submitter in D state and stall teardown forever. + * TASK_FREEZABLE lets the freezer suspend this wait in place during + * S3/S4 instead of aborting the suspend. Harmless for the job worker + * kthread (never gets a signal; simply freezes/thaws around it). + */ + cert_comp = aie4_get_cert_comp(hwctx); + if (!cert_comp) + return -EAGAIN; + + ret = wait_event_freezable(cert_comp->waitq, + check_cmd_done(hwctx, seq, cert_comp)); + if (ret) { + aie4_put_cert_comp(cert_comp); + return ret; /* -ERESTARTSYS: signal on the submit path */ + } + + if (check_cert_comp_linked(hwctx, cert_comp)) + ret = 0; /* real completion */ + else + ret = -EAGAIN; /* disconnect (suspend or TDR) */ + + aie4_put_cert_comp(cert_comp); + return ret; +} + +static int wait_till_connected_hsa_not_full(struct amdxdna_hwctx *hwctx, + bool wait_through_reset) +{ + struct amdxdna_dev *xdna = hwctx->client->xdna; + struct amdxdna_hwctx_priv *priv = hwctx->priv; + u64 wi = READ_ONCE(priv->write_index); + bool hsa_not_full = !!(wi < CTX_MAX_CMDS); + int ret; + + do { + mutex_unlock(&priv->io_lock); + if (!hsa_not_full) { + ret = wait_till_seq_completed(hwctx, wi - CTX_MAX_CMDS); + if (ret && ret != -EAGAIN) { + mutex_lock(&priv->io_lock); + return ret; + } + if (!ret) + hsa_not_full = true; + } + ret = wait_event_freezable(priv->job_list_wq, + aie4_hwctx_connected(hwctx) || + (!wait_through_reset && + aie4_hwctx_has_reset(hwctx))); + mutex_lock(&priv->io_lock); + if (ret) + return ret; + if (!wait_through_reset && aie4_hwctx_has_reset(hwctx)) { + XDNA_DBG(xdna, "ctx %s reset while submitting; unwinding -ECONNRESET", + hwctx->name); + return -ECONNRESET; + } + } while (!hsa_not_full || !aie4_hwctx_connected(hwctx)); + + return 0; +} + +static int fill_indirect_pkt(struct amdxdna_hwctx_priv *priv, u64 slot_idx, + u32 total_slots, struct amdxdna_cmd_start_dpu *dpu, + u16 entries) +{ + struct host_queue_packet *pkt = &priv->umq_pkts[slot_idx]; + struct host_indirect_packet_entry *hipe = + (struct host_indirect_packet_entry *)(pkt->data); + u16 i; + + for (i = 0; i < entries; i++, dpu++, hipe++) { + struct host_indirect_packet_data *hipd; + u64 indirect_pkt_dev_addr; + u32 uci = dpu->uc_index; + u32 idx; + + /* + * dpu is the user-shared cmd_abo payload, so uc_index is read at + * use time here and indexes priv->umq_indirect_pkts[]. Reject an + * out-of-range value: the slot is reused, so skipping the entry + * would leave a stale one that count still advertises to CERT. + * Abort before the packet is published. + */ + if (uci >= HSA_MAX_LEVEL1_INDIRECT_ENTRIES) { + XDNA_ERR(priv->hwctx->client->xdna, "Invalid uc index %d", uci); + return -EINVAL; + } + idx = uci * total_slots + slot_idx; + hipd = &priv->umq_indirect_pkts[idx]; + indirect_pkt_dev_addr = priv->umq_indirect_pkts_dev_addr + + sizeof(struct host_indirect_packet_data) * idx; + + /* Point the indirect entry at the indirect packet. */ + hipe->host_addr_low = lower_32_bits(indirect_pkt_dev_addr); + hipe_set_host_addr_high(&hipe->host_addr_high_uc_index, + upper_32_bits(indirect_pkt_dev_addr)); + hipe_set_uc_index(&hipe->host_addr_high_uc_index, uci); + + /* Fill in the indirect packet. */ + hipd->payload.dpu_control_code_host_addr_low = + lower_32_bits(dpu->instruction_buffer); + hipd->payload.dpu_control_code_host_addr_high = + upper_32_bits(dpu->instruction_buffer); + hipd->payload.dtrace_buf_host_addr_low = + lower_32_bits(dpu->dtrace_buffer); + hipd->payload.dtrace_buf_host_addr_high = + lower_16_bits(upper_32_bits(dpu->dtrace_buffer)); + } + pkt->pkt_header.common_header.distribute = 1; + pkt->pkt_header.common_header.indirect = 1; + pkt->pkt_header.common_header.count = entries * sizeof(*hipe); + return 0; +} + +static void fill_direct_pkt(struct amdxdna_hwctx_priv *priv, u64 slot_idx, + struct amdxdna_cmd_start_dpu *dpu) +{ + struct host_queue_packet *pkt = &priv->umq_pkts[slot_idx]; + struct exec_buf *ebuf = (struct exec_buf *)(pkt->data); + + memset(pkt->data, 0, sizeof(pkt->data)); + ebuf->dpu_control_code_host_addr_low = lower_32_bits(dpu->instruction_buffer); + ebuf->dpu_control_code_host_addr_high = upper_32_bits(dpu->instruction_buffer); + ebuf->dtrace_buf_host_addr_low = lower_32_bits(dpu->dtrace_buffer); + ebuf->dtrace_buf_host_addr_high = lower_16_bits(upper_32_bits(dpu->dtrace_buffer)); + pkt->pkt_header.common_header.distribute = 0; + pkt->pkt_header.common_header.indirect = 0; + pkt->pkt_header.common_header.count = sizeof(*ebuf); +} + +/* + * Build and submit one HSA command for @cmd_abo into the user host queue and + * ring the doorbell. Called with io_lock held. + * + * Security: cmd_abo is shared with user space; cache and validate its fields + * before use and never trust the queue content (only read_index is read back). + */ +static int submit_one_cmd(struct amdxdna_hwctx *hwctx, + struct amdxdna_gem_obj *cmd_abo, bool last_of_chain, + bool first_cmd, u64 *seq) +{ + struct amdxdna_hwctx_priv *priv = hwctx->priv; + struct amdxdna_dev *xdna = hwctx->client->xdna; + struct amdxdna_cmd_start_dpu *dpu; + struct host_queue_packet *pkt; + u32 payload_size; + u64 slot_idx; + u16 chained; + int ret; + u32 op; + + op = amdxdna_cmd_get_op(cmd_abo); + if (op != ERT_START_DPU) { + XDNA_ERR(xdna, "Invalid exec buf op, %d", op); + return -EINVAL; + } + + dpu = amdxdna_cmd_get_payload(cmd_abo, &payload_size); + if (!dpu) { + XDNA_ERR(xdna, "Invalid DPU payload"); + return -EINVAL; + } + /* + * cmd_abo is shared with user space; validate the cached chained count + * against the actual payload size before dereferencing chained+1 DPU + * entries, so a bogus count cannot drive an out-of-bounds read. + */ + chained = dpu->chained; + if (chained >= HSA_MAX_LEVEL1_INDIRECT_ENTRIES) { + XDNA_ERR(xdna, "Invalid DPU data"); + return -EINVAL; + } + if (payload_size < (u32)(chained + 1) * sizeof(*dpu)) { + XDNA_ERR(xdna, "DPU payload %u too small for %u entries", + payload_size, chained + 1); + return -EINVAL; + } + + /* + * Block until a queue slot is free and the ctx is connected (io_lock is + * dropped across the sleeps inside and re-acquired). The only failure is a + * signal interrupting the wait (-ERESTARTSYS), e.g. the app being killed. + */ + ret = wait_till_connected_hsa_not_full(hwctx, first_cmd); + if (ret) { + XDNA_DBG(xdna, "Wait for queue slot / ctx reconnect interrupted, ret %d", ret); + return ret; + } + + slot_idx = priv->write_index & (CTX_MAX_CMDS - 1); + if (chained) { + ret = fill_indirect_pkt(priv, slot_idx, CTX_MAX_CMDS, dpu, chained + 1); + if (ret) + return ret; + } else { + fill_direct_pkt(priv, slot_idx, dpu); + } + + pkt = &priv->umq_pkts[slot_idx]; + pkt->pkt_header.common_header.opcode = OPCODE_EXEC_BUF; + pkt->pkt_header.common_header.chain_flag = + last_of_chain ? CHAIN_FLG_LAST_CMD : CHAIN_FLG_NOT_LAST_CMD; + pkt->pkt_header.common_header.reserved = 0x0; + pkt->pkt_header.completion_signal = amdxdna_gem_dev_addr(cmd_abo) + + offsetof(struct amdxdna_cmd, header); + *seq = publish_cmd(hwctx); + aie4_doorbell_ring(hwctx); + XDNA_DBG(xdna, "Submitted one cmd, %s seq %lld", hwctx->name, *seq); + return 0; +} + +/* + * Return the head running job without removing it. The job worker keeps the + * in-flight job on the list while it waits so that a disconnect (suspend) can + * just leave it there for resume - no dequeue/requeue - and running_job_list is + * never transiently empty while a job is in flight. + */ +static struct amdxdna_sched_job *peek_running_job(struct amdxdna_hwctx *hwctx) +{ + struct amdxdna_hwctx_priv *priv = hwctx->priv; + struct amdxdna_sched_job *job; + + mutex_lock(&priv->io_lock); + job = list_first_entry_or_null(&priv->running_job_list, + struct amdxdna_sched_job, aie4_job_list); + mutex_unlock(&priv->io_lock); + return job; +} + +/* Remove a job from the running list once it is completed or reaped. */ +static void dequeue_running_job(struct amdxdna_hwctx *hwctx, struct amdxdna_sched_job *job) +{ + struct amdxdna_hwctx_priv *priv = hwctx->priv; + + mutex_lock(&priv->io_lock); + list_del(&job->aie4_job_list); + mutex_unlock(&priv->io_lock); +} + +static void aie4_job_release(struct kref *ref) +{ + struct amdxdna_sched_job *job = + container_of(ref, struct amdxdna_sched_job, refcnt); + + amdxdna_sched_job_cleanup(job); + if (job->out_fence) + dma_fence_put(job->out_fence); + kfree(job); +} + +static void job_done(struct amdxdna_sched_job *job) +{ + job->aie4_job_state = AIE4_JOB_STATE_DONE; + dma_fence_signal(job->fence); + /* + * Release the address-space reference taken at submit. On SVA/IOMMU + * platforms the device walks the submitter's page tables while the job + * runs, so its mm must stay alive until completion. + */ + mmput_async(job->mm); + kref_put(&job->refcnt, aie4_job_release); +} + +static void job_complete(struct amdxdna_sched_job *job) +{ + job_done(job); +} + +/* + * When CERT cannot complete a command (context teardown), the driver advances + * read_index so any waiter observes the command as finished. Only valid while + * the context is disconnected -- never race CERT's own read_index updates. + */ +static void update_read_index(struct amdxdna_hwctx *hwctx, u64 idx) +{ + struct amdxdna_hwctx_priv *priv = hwctx->priv; + + /* Order cmd-bo state write before the waiter observes completion. */ + wmb(); + WRITE_ONCE(*priv->umq_read_index, idx); +} + +static void job_abort(struct amdxdna_sched_job *job) +{ + struct amdxdna_hwctx *hwctx = job->hwctx; + + XDNA_WARN(hwctx->client->xdna, "aborting %s job %lld", hwctx->name, job->seq); + amdxdna_cmd_set_state(job->cmd_bo, ERT_CMD_STATE_ABORT); + dma_fence_set_error(job->fence, -ECANCELED); + /* + * Only force read_index forward when CERT has not already moved it past this + * job. On the reset drain read_index is still <= job->seq (CERT stopped), so + * advance it here to release waiters. For a partial chain closed by a later + * command's LAST_CMD, CERT has already advanced read_index past this job - + * do not clobber it. + */ + if (get_read_index(hwctx) <= job->seq) + update_read_index(hwctx, job->seq + 1); + job_done(job); +} + +static void job_worker(struct work_struct *work) +{ + struct amdxdna_hwctx_priv *priv = + container_of(work, struct amdxdna_hwctx_priv, job_work); + struct amdxdna_hwctx *hwctx = priv->hwctx; + struct amdxdna_sched_job *job; + + while ((job = peek_running_job(hwctx))) { + wait_till_seq_completed(hwctx, job->seq); + if (get_read_index(hwctx) > job->seq) { + dequeue_running_job(hwctx, job); + /* + * read_index advanced past this job. A fully published + * job (SUBMITTED) ran to completion. A partial chain + * (SUBMITTING: a later sub-command failed to publish, so + * the chain never got CHAIN_FLG_LAST_CMD) only reaches + * here once a *later* command's LAST_CMD closes the + * dangling runlist and advances read_index past it - so + * report it ABORT, not a false completion. If no such + * command follows, read_index never advances and we stay + * parked in wait_till_seq_completed() above until the + * user's wait_command() times out and breaks the wait + * (or ctx teardown reaps it). + */ + if (job->aie4_job_state != AIE4_JOB_STATE_SUBMITTED) + job_abort(job); + else + job_complete(job); + } else if (aie4_hwctx_has_reset(hwctx)) { + dequeue_running_job(hwctx, job); + job_abort(job); + } else { + /* suspend/resume */ + break; + } + } +} + +int aie4_hwctx_wait_for_running(struct amdxdna_hwctx *hwctx) +{ + struct amdxdna_hwctx_priv *priv = hwctx->priv; + struct amdxdna_dev *xdna = hwctx->client->xdna; + struct amdxdna_sched_job *job; + long error; + int ret = 0; + + mutex_lock(&priv->io_lock); + job = READ_ONCE(priv->pending_head); + if (job && job->aie4_job_state == AIE4_JOB_STATE_SUBMITTING) { + mutex_unlock(&priv->io_lock); + error = wait_event_timeout(priv->job_list_wq, + READ_ONCE(priv->pending_head) != job, + msecs_to_jiffies(2000)); + if (!error) { + XDNA_WARN(xdna, "hwctx %s wait for submitting job timed out", + hwctx->name); + ret = -ETIMEDOUT; + } + } else { + mutex_unlock(&priv->io_lock); + } + + queue_work(priv->job_work_q, &priv->job_work); + flush_work(&priv->job_work); + return ret; +} + +/* + * Submit the command(s) carried by @job into the host queue. Called with + * io_lock held. A single ERT_START_DPU maps to one queue entry; an + * ERT_CMD_CHAIN expands to one entry per sub-command, only the last of which + * carries CHAIN_FLG_LAST_CMD so CERT runs the whole chain back to back. + * + * job->seq tracks the last published sequence; the worker waits on it to reap + * the entire chain. job->aie4_job_state advances past PENDING as soon as any + * sub-command is published, so the caller knows whether in-flight commands must + * still be reaped even when a later sub-command fails to enqueue. + * + * Security: the chain payload and its BO handles come from user space; cache + * command_count and validate it against the payload size before walking the + * handle array so a bogus count cannot drive an out-of-bounds read. + */ +static int submit_job_cmds(struct amdxdna_hwctx *hwctx, + struct amdxdna_sched_job *job, u32 op) +{ + struct amdxdna_gem_obj *cmd_abo = job->cmd_bo; + struct amdxdna_dev *xdna = hwctx->client->xdna; + struct amdxdna_cmd_chain *payload; + u32 payload_len, ccnt; + int ret; + u32 i; + + /* Single cmd. */ + if (op == ERT_START_DPU) { + ret = submit_one_cmd(hwctx, cmd_abo, true, true, &job->seq); + if (!ret) + job->aie4_job_state = AIE4_JOB_STATE_SUBMITTED; + return ret; + } + + /* Cmd chain. */ + payload = amdxdna_cmd_get_payload(cmd_abo, &payload_len); + if (!payload) { + XDNA_ERR(xdna, "Invalid cmd payload for chained cmd"); + return -EINVAL; + } + ccnt = payload->command_count; + /* + * A chain (runlist) must fit within the queue. CERT advances the host-visible + * read_index only once per runlist - at the last sub-command (CHAIN_FLG_LAST_CMD), + * not per sub-command - so a chain's own entries never free a queue slot until + * the whole chain has been published and run. A chain longer than the queue + * could therefore never publish its tail: it would block forever in + * wait_till_connected_hsa_not_full() waiting for a slot that only frees at chain end. + * Reject ccnt > CTX_MAX_CMDS. Also validate against the payload size before + * walking the handle array so a bogus count cannot drive an out-of-bounds read. + */ + if (!ccnt || ccnt > CTX_MAX_CMDS || + payload_len < struct_size(payload, data, ccnt)) { + XDNA_ERR(xdna, "Invalid command count %u", ccnt); + return -EINVAL; + } + + for (i = 0; i < ccnt; i++) { + u32 boh = (u32)(payload->data[i]); + struct amdxdna_gem_obj *abo; + + abo = amdxdna_gem_get_obj(hwctx->client, boh, AMDXDNA_BO_SHARE); + if (!abo) { + XDNA_ERR(xdna, "Failed to find cmd BO %u", boh); + ret = -ENOENT; + break; + } + + /* + * submit_one_cmd() blocks in wait_till_connected_hsa_not_full() until the + * ctx is connected and a slot is free, so a concurrent suspend/disconnect + * is waited out inline rather than returned here. The first sub-command + * (i == 0, nothing published yet) waits through a TDR reset and runs on + * the recreated ctx; a later sub-command returns -ECONNRESET if a reset + * landed while waiting for a slot, so the published prefix is not split + * across the reset. It also returns -ERESTARTSYS on a signal, or a + * validation error. Break on any; a published prefix is then reaped by + * the job worker's reset drain (see below). + */ + ret = submit_one_cmd(hwctx, abo, i + 1 == ccnt, i == 0, &job->seq); + amdxdna_gem_put_obj(abo); + if (ret) + break; + job->aie4_job_state = AIE4_JOB_STATE_SUBMITTING; + } + if (i == ccnt) + job->aie4_job_state = AIE4_JOB_STATE_SUBMITTED; + + /* + * As long as at least one sub-command was published, return success so the + * caller enqueues the job on the running list; the job worker then reaps the + * published prefix and reports the partial chain as failed (ABORT). Only when + * nothing was published (i == 0) is the error returned to the caller. + */ + if (i > 0) + return 0; + + return ret; +} + +/* + * Whole-job submission is serialized across submitters that share a ctx via the + * pending list: a job is appended on entry and only the head of the list is + * allowed to publish its command(s). Because the head stays on the list for the + * entire duration of submit_job_cmds() -- which may drop io_lock to wait for + * free queue slots -- no other submitter can interleave its commands into the + * middle of the head job's command chain. io_lock protects the lists; the + * job_list_wq waitqueue notifies parked submitters when the head changes. + */ +/* Publish the current pending-list head for the lockless submit wait condition. + * Caller holds io_lock. + */ +static void update_pending_head(struct amdxdna_hwctx_priv *priv) +{ + WRITE_ONCE(priv->pending_head, + list_first_entry_or_null(&priv->pending_job_list, + struct amdxdna_sched_job, aie4_job_list)); +} + +static void enqueue_pending_job(struct amdxdna_hwctx *hwctx, + struct amdxdna_sched_job *job) +{ + struct amdxdna_hwctx_priv *priv = hwctx->priv; + + mutex_lock(&priv->io_lock); + list_add_tail(&job->aie4_job_list, &priv->pending_job_list); + job->aie4_job_state = AIE4_JOB_STATE_PENDING; + update_pending_head(priv); + mutex_unlock(&priv->io_lock); + + /* Let the next pending submitter re-check whether it is now first. */ + wake_up_all(&priv->job_list_wq); +} + +static void cancel_pending_job(struct amdxdna_hwctx *hwctx, + struct amdxdna_sched_job *job) +{ + struct amdxdna_hwctx_priv *priv = hwctx->priv; + + mutex_lock(&priv->io_lock); + list_del(&job->aie4_job_list); + job->aie4_job_state = AIE4_JOB_STATE_INIT; + update_pending_head(priv); + mutex_unlock(&priv->io_lock); + /* Let the next pending submitter re-check whether it is now first. */ + wake_up_all(&priv->job_list_wq); +} + +int aie4_cmd_submit(struct amdxdna_hwctx *hwctx, struct amdxdna_sched_job *job, u64 *seq) +{ + struct amdxdna_hwctx_priv *priv = hwctx->priv; + struct amdxdna_dev *xdna = hwctx->client->xdna; + struct ww_acquire_ctx acquire_ctx; + struct amdxdna_gem_obj *abo; + u32 op; + int i; + int ret; + + XDNA_DBG(xdna, "ctx %s job %p received", hwctx->name, job); + + if (!job->cmd_bo) { + XDNA_ERR(xdna, "No command BO in job"); + return -EINVAL; + } + + op = amdxdna_cmd_get_op(job->cmd_bo); + if (op != ERT_START_DPU && op != ERT_CMD_CHAIN) { + XDNA_ERR(xdna, "Invalid cmd opcode %d", op); + return -EINVAL; + } + + INIT_LIST_HEAD(&job->aie4_job_list); + + /* + * Hold a reference on the submitter's address space until the job + * completes (job_done): on SVA/IOMMU platforms the device walks the + * submitter's page tables while the command runs. Balanced with the + * mmput_async() in job_done() and the mmput() on the failure paths below. + */ + if (!mmget_not_zero(job->mm)) { + XDNA_ERR(xdna, "Failed to get mm reference"); + return -ESRCH; + } + + /* + * Lock all job BOs and reserve fences. This attaches job->out_fence + * to each BO's reservation object, ensuring concurrent invalidation + * waits for the job to complete. + */ + ret = drm_gem_lock_reservations(job->bos, job->bo_cnt, &acquire_ctx); + if (ret) { + XDNA_WARN(xdna, "Failed to lock BOs, ret %d", ret); + goto put_mm; + } + + for (i = 0; i < job->bo_cnt; i++) { + ret = dma_resv_reserve_fences(job->bos[i]->resv, 1); + if (ret) { + XDNA_WARN(xdna, "Failed to reserve fences %d", ret); + drm_gem_unlock_reservations(job->bos, job->bo_cnt, &acquire_ctx); + goto put_mm; + } + } + + down_read(&xdna->notifier_lock); + for (i = 0; i < job->bo_cnt; i++) { + abo = to_xdna_obj(job->bos[i]); + if (abo->mem.map_invalid) { + up_read(&xdna->notifier_lock); + drm_gem_unlock_reservations(job->bos, job->bo_cnt, &acquire_ctx); + ret = -EINVAL; + goto put_mm; + } + } + + job->out_fence = dma_fence_get(job->fence); + for (i = 0; i < job->bo_cnt; i++) + dma_resv_add_fence(job->bos[i]->resv, job->out_fence, DMA_RESV_USAGE_WRITE); + + up_read(&xdna->notifier_lock); + drm_gem_unlock_reservations(job->bos, job->bo_cnt, &acquire_ctx); + + /* + * Wait until this job is at the head of the pending list before touching + * the queue (see enqueue_pending_job). Freezable so the freezer can + * suspend a parked submitter in place across S3/S4 rather than aborting + * the suspend; still interruptible so a signal (app exit/kill/^C) unwinds + * it and does not keep ctx teardown (synchronize_srcu) blocked. + */ + enqueue_pending_job(hwctx, job); + ret = wait_event_freezable(priv->job_list_wq, + READ_ONCE(priv->pending_head) == job); + if (ret) { + cancel_pending_job(hwctx, job); + goto signal_fence; + } + + mutex_lock(&priv->io_lock); + ret = submit_job_cmds(hwctx, job, op); + if (ret) { + mutex_unlock(&priv->io_lock); + cancel_pending_job(hwctx, job); + goto signal_fence; + } + + list_move_tail(&job->aie4_job_list, &priv->running_job_list); + update_pending_head(priv); + *seq = job->seq; + mutex_unlock(&priv->io_lock); + + /* Release the next pending submitter and kick the reaper. */ + wake_up_all(&priv->job_list_wq); + atomic64_inc(&hwctx->job_submit_cnt); + queue_work(priv->job_work_q, &priv->job_work); + return 0; + +signal_fence: + /* + * Map internal -ERESTARTSYS to -ECANCELED for the fence so downstream + * consumers (e.g. sync_file, dma-buf importers) do not observe internal + * signal restart codes, while preserving ret for the syscall return. + */ + dma_fence_set_error(job->fence, ret == -ERESTARTSYS ? -ECANCELED : ret); + dma_fence_signal(job->fence); + dma_fence_put(job->out_fence); + job->out_fence = NULL; +put_mm: + mmput(job->mm); + return ret; +} diff --git a/drivers/accel/amdxdna/aie4_pci.c b/drivers/accel/amdxdna/aie4_pci.c index f5fdc24689f8..f180983a692d 100644 --- a/drivers/accel/amdxdna/aie4_pci.c +++ b/drivers/accel/amdxdna/aie4_pci.c @@ -1109,6 +1109,7 @@ const struct amdxdna_dev_ops aie4_vf_ops = { .debugfs_init = aie4_debugfs_init, .hwctx_init = aie4_hwctx_init, .hwctx_fini = aie4_hwctx_fini, + .cmd_submit = aie4_cmd_submit, .cmd_wait = aie4_cmd_wait, .get_aie_info = aie4_get_info, .set_aie_state = aie4_set_state, @@ -1120,6 +1121,7 @@ const struct amdxdna_dev_ops aie4_classic_ops = { .debugfs_init = aie4_debugfs_init, .hwctx_init = aie4_hwctx_init, .hwctx_fini = aie4_hwctx_fini, + .cmd_submit = aie4_cmd_submit, .cmd_wait = aie4_cmd_wait, .get_aie_info = aie4_get_info, .set_aie_state = aie4_set_state, diff --git a/drivers/accel/amdxdna/aie4_pci.h b/drivers/accel/amdxdna/aie4_pci.h index f549d9e69d41..6b67c9da560e 100644 --- a/drivers/accel/amdxdna/aie4_pci.h +++ b/drivers/accel/amdxdna/aie4_pci.h @@ -50,6 +50,7 @@ struct amdxdna_hwctx_priv { struct cert_comp *cert_comp; u32 hw_ctx_id; + bool has_reset; /* Kernel-mode submission: driver fills the user HSA queue and rings * the doorbell. umq_pkts/umq_indirect_pkts alias the user umq_bo; @@ -165,8 +166,10 @@ enum aie4_hwctx_flags { int aie4_hwctx_init(struct amdxdna_hwctx *hwctx); void aie4_hwctx_fini(struct amdxdna_hwctx *hwctx); int aie4_cmd_wait(struct amdxdna_hwctx *hwctx, u64 seq, u32 timeout); +int aie4_cmd_submit(struct amdxdna_hwctx *hwctx, struct amdxdna_sched_job *job, u64 *seq); int aie4_hwctx_create(struct amdxdna_hwctx *hwctx); void aie4_hwctx_destroy(struct amdxdna_hwctx *hwctx, enum aie4_hwctx_flags); +int aie4_hwctx_wait_for_running(struct amdxdna_hwctx *hwctx); /* aie4_pci.c */ int aie4_restore_power_mode(struct amdxdna_dev_hdl *ndev); diff --git a/drivers/accel/amdxdna/amdxdna_ctx.c b/drivers/accel/amdxdna/amdxdna_ctx.c index 888e857ec558..2a635eab1e55 100644 --- a/drivers/accel/amdxdna/amdxdna_ctx.c +++ b/drivers/accel/amdxdna/amdxdna_ctx.c @@ -25,7 +25,7 @@ struct amdxdna_fence { struct dma_fence base; spinlock_t lock; /* for base */ - struct amdxdna_hwctx *hwctx; + struct device *dev; }; static const char *amdxdna_fence_get_driver_name(struct dma_fence *fence) @@ -39,7 +39,14 @@ static const char *amdxdna_fence_get_timeline_name(struct dma_fence *fence) xdna_fence = container_of(fence, struct amdxdna_fence, base); - return xdna_fence->hwctx->name; + /* + * Use device name rather than hwctx name: the fence is published into + * BO reservation objects via dma_resv_add_fence() and can outlive the + * hwctx (e.g. when a BO is exported as a dma-buf and imported by + * another process). The device outlives any individual context, so + * dev_name() is safe to call at any point during the fence's lifetime. + */ + return dev_name(xdna_fence->dev); } static const struct dma_fence_ops fence_ops = { @@ -55,9 +62,14 @@ static struct dma_fence *amdxdna_fence_create(struct amdxdna_hwctx *hwctx) if (!fence) return NULL; - fence->hwctx = hwctx; + fence->dev = hwctx->client->xdna->ddev.dev; spin_lock_init(&fence->lock); - dma_fence_init(&fence->base, &fence_ops, &fence->lock, hwctx->id, 0); + /* + * Each job fence needs a unique timeline context so dma_resv_add_fence() + * does not evict a prior in-flight job's fence from a shared BO's + * reservation object when concurrent submissions access the same BO. + */ + dma_fence_init(&fence->base, &fence_ops, &fence->lock, dma_fence_context_alloc(1), 0); return &fence->base; } -- 2.34.1
