The NTB rawdev framework was written around the Intel back-to-back topology and the built-in scratchpad handshake protocol. To allow other vendors to plug into the same framework, add vendor-neutral hooks and make the common code dispatch through them:
- Add NTB_TOPO_PRI/NTB_TOPO_SEC topology types for hardware that uses a primary/secondary topology instead of back-to-back. - Add optional ntb_dev_ops hooks: interrupt_handler (vendor-specific MSI-X handler), dev_handshake (vendor-specific link handshake) and read_peer_config (vendor-specific peer-config read at start). When a hook is NULL the common code keeps using the existing built-in path, so the Intel driver is unaffected. - Add a mem_align op and the rte_pmd_ntb_get_mem_align() API so an application can query the base-address alignment a memory window memzone needs, without embedding hardware-specific rules in the app. The Intel driver reports its memory-window size. - Add a pmd_private pointer to struct ntb_hw for vendor-specific state. - Guard the receive path against a malformed stream with no end-of-packet marker so it cannot overflow the descriptor ring. Signed-off-by: Raghavendra Ningoji <[email protected]> --- drivers/raw/ntb/ntb.c | 115 ++++++++++++++++++++++++--------- drivers/raw/ntb/ntb.h | 22 +++++++ drivers/raw/ntb/ntb_hw_intel.c | 11 ++++ drivers/raw/ntb/rte_pmd_ntb.h | 26 ++++++++ examples/ntb/ntb_fwd.c | 6 +- 5 files changed, 146 insertions(+), 34 deletions(-) diff --git a/drivers/raw/ntb/ntb.c b/drivers/raw/ntb/ntb.c index d54f2fb783..497c86b58c 100644 --- a/drivers/raw/ntb/ntb.c +++ b/drivers/raw/ntb/ntb.c @@ -18,6 +18,7 @@ #include <rte_memcpy.h> #include <rte_rawdev.h> #include <rte_rawdev_pmd.h> +#include <eal_export.h> #include "ntb_hw_intel.h" #include "rte_pmd_ntb.h" @@ -746,6 +747,11 @@ ntb_dequeue_bufs(struct rte_rawdev *dev, for (nb_rx = 0; nb_rx < count; nb_rx++) { i = 0; while (true) { + if (unlikely(nb_mbufs >= rxq->nb_rx_desc)) { + NTB_LOG(ERR, "Malformed rx stream (no EOP); " + "aborting to avoid desc overflow."); + goto end_of_rx; + } rx_item = rxq->rx_used_ring + rxq->last_used; rxm_t = sw_ring[rxq->last_used].mbuf; rxm_t->data_len = rx_item->len; @@ -855,6 +861,26 @@ ntb_dev_info_get(struct rte_rawdev *dev, rte_rawdev_obj_t dev_info, return 0; } +RTE_EXPORT_EXPERIMENTAL_SYMBOL(rte_pmd_ntb_get_mem_align, 26.11) +uint64_t +rte_pmd_ntb_get_mem_align(uint16_t dev_id, uint32_t mw_id, uint64_t mw_len) +{ + struct rte_rawdev *dev; + struct ntb_hw *hw; + + if (dev_id >= RTE_RAWDEV_MAX_DEVS) + return 0; + dev = rte_rawdev_pmd_get_dev(dev_id); + if (dev->dev_private == NULL) + return 0; + + hw = dev->dev_private; + if (mw_id >= hw->mw_cnt || hw->ntb_ops->mem_align == NULL) + return RTE_CACHE_LINE_SIZE; + + return (*hw->ntb_ops->mem_align)(dev, mw_id, mw_len); +} + static int ntb_dev_configure(const struct rte_rawdev *dev, rte_rawdev_obj_t config, size_t config_size) @@ -882,8 +908,13 @@ ntb_dev_configure(const struct rte_rawdev *dev, rte_rawdev_obj_t config, hw->ntb_xstats_off = rte_zmalloc("ntb_xstats_off", xstats_num * sizeof(uint64_t), 0); - /* Start handshake with the peer. */ - ret = ntb_handshake_work(dev); + /* Start handshake with the peer. Use the vendor-specific handshake + * if provided, otherwise the built-in scratchpad protocol. + */ + if (hw->ntb_ops->dev_handshake != NULL) + ret = (*hw->ntb_ops->dev_handshake)(dev); + else + ret = ntb_handshake_work(dev); if (ret < 0) { rte_free(hw->rx_queues); rte_free(hw->tx_queues); @@ -929,35 +960,44 @@ ntb_dev_start(struct rte_rawdev *dev) goto err_q_init; } - if (hw->ntb_ops->spad_read == NULL) { - ret = -ENOTSUP; - goto err_up; - } + /* Read/validate peer config. Use the vendor-specific reader if + * provided, otherwise the built-in scratchpad reads. + */ + if (hw->ntb_ops->read_peer_config != NULL) { + ret = (*hw->ntb_ops->read_peer_config)(dev); + if (ret < 0) + goto err_up; + } else { + if (hw->ntb_ops->spad_read == NULL) { + ret = -ENOTSUP; + goto err_up; + } - peer_val = (*hw->ntb_ops->spad_read)(dev, SPAD_Q_SZ, 0); - if (peer_val != hw->queue_size) { - NTB_LOG(ERR, "Inconsistent queue size! (local: %u peer: %u)", - hw->queue_size, peer_val); - ret = -EINVAL; - goto err_up; - } + peer_val = (*hw->ntb_ops->spad_read)(dev, SPAD_Q_SZ, 0); + if (peer_val != hw->queue_size) { + NTB_LOG(ERR, "Inconsistent queue size! (local: %u peer: %u)", + hw->queue_size, peer_val); + ret = -EINVAL; + goto err_up; + } - peer_val = (*hw->ntb_ops->spad_read)(dev, SPAD_NUM_QPS, 0); - if (peer_val != hw->queue_pairs) { - NTB_LOG(ERR, "Inconsistent number of queues! (local: %u peer:" - " %u)", hw->queue_pairs, peer_val); - ret = -EINVAL; - goto err_up; - } + peer_val = (*hw->ntb_ops->spad_read)(dev, SPAD_NUM_QPS, 0); + if (peer_val != hw->queue_pairs) { + NTB_LOG(ERR, "Inconsistent number of queues! (local: %u peer:" + " %u)", hw->queue_pairs, peer_val); + ret = -EINVAL; + goto err_up; + } - hw->peer_used_mws = (*hw->ntb_ops->spad_read)(dev, SPAD_USED_MWS, 0); + hw->peer_used_mws = (*hw->ntb_ops->spad_read)(dev, SPAD_USED_MWS, 0); - for (i = 0; i < hw->peer_used_mws; i++) { - peer_base_h = (*hw->ntb_ops->spad_read)(dev, - SPAD_MW0_BA_H + 2 * i, 0); - peer_base_l = (*hw->ntb_ops->spad_read)(dev, - SPAD_MW0_BA_L + 2 * i, 0); - hw->peer_mw_base[i] = (peer_base_h << 32) + peer_base_l; + for (i = 0; i < hw->peer_used_mws; i++) { + peer_base_h = (*hw->ntb_ops->spad_read)(dev, + SPAD_MW0_BA_H + 2 * i, 0); + peer_base_l = (*hw->ntb_ops->spad_read)(dev, + SPAD_MW0_BA_L + 2 * i, 0); + hw->peer_mw_base[i] = (peer_base_h << 32) + peer_base_l; + } } dev->started = 1; @@ -1057,8 +1097,13 @@ ntb_dev_close(struct rte_rawdev *dev) rte_intr_disable(intr_handle); /* Unregister callback func to eal lib */ - rte_intr_callback_unregister(intr_handle, - ntb_dev_intr_handler, dev); + if (hw->ntb_ops->interrupt_handler != NULL) + rte_intr_callback_unregister(intr_handle, + hw->ntb_ops->interrupt_handler, + dev); + else + rte_intr_callback_unregister(intr_handle, + ntb_dev_intr_handler, dev); return 0; } @@ -1409,9 +1454,15 @@ ntb_init_hw(struct rte_rawdev *dev, struct rte_pci_device *pci_dev) (*hw->ntb_ops->db_clear)(dev, hw->db_valid_mask); intr_handle = pci_dev->intr_handle; - /* Register callback func to eal lib */ - rte_intr_callback_register(intr_handle, - ntb_dev_intr_handler, dev); + /* Register callback func to eal lib. Use the vendor-specific handler + * if provided, otherwise fall back to the built-in handler. + */ + if (hw->ntb_ops->interrupt_handler != NULL) + rte_intr_callback_register(intr_handle, + hw->ntb_ops->interrupt_handler, dev); + else + rte_intr_callback_register(intr_handle, + ntb_dev_intr_handler, dev); ret = rte_intr_efd_enable(intr_handle, hw->db_cnt); if (ret) diff --git a/drivers/raw/ntb/ntb.h b/drivers/raw/ntb/ntb.h index 8c7a2230f9..57d09a2cd4 100644 --- a/drivers/raw/ntb/ntb.h +++ b/drivers/raw/ntb/ntb.h @@ -42,6 +42,9 @@ enum ntb_topo { NTB_TOPO_NONE = 0, NTB_TOPO_B2B_USD, NTB_TOPO_B2B_DSD, + /* Primary/secondary topology (e.g. AMD NTB). */ + NTB_TOPO_PRI, + NTB_TOPO_SEC, }; enum ntb_link { @@ -100,6 +103,10 @@ enum ntb_spad_idx { * for those db bits. * @peer_db_set: Set doorbell bit to generate peer interrupt for that bit. * @vector_bind: Bind vector source [intr] to msix vector [msix]. + * @interrupt_handler: Vendor-specific interrupt handler. If NULL, the + * built-in handler is used. + * @mem_align: Base-address alignment required for a memory window memzone + * of a given length. */ struct ntb_dev_ops { int (*ntb_dev_init)(const struct rte_rawdev *dev); @@ -119,6 +126,18 @@ struct ntb_dev_ops { int (*peer_db_set)(const struct rte_rawdev *dev, uint8_t db_bit); int (*vector_bind)(const struct rte_rawdev *dev, uint8_t intr, uint8_t msix); + void (*interrupt_handler)(void *param); + /* Optional vendor-specific handshake. If NULL, the built-in + * scratchpad handshake is used. Used by hardware (e.g. AMD) whose + * scratchpad layout differs from the built-in protocol. + */ + int (*dev_handshake)(const struct rte_rawdev *dev); + /* Optional vendor-specific peer-config read at device start. If NULL, + * the built-in scratchpad reads are used. + */ + int (*read_peer_config)(const struct rte_rawdev *dev); + uint64_t (*mem_align)(const struct rte_rawdev *dev, uint32_t mw_id, + uint64_t mw_len); }; struct ntb_desc { @@ -208,6 +227,9 @@ struct ntb_hw { const struct ntb_dev_ops *ntb_ops; + /* Vendor-specific hardware private data. */ + void *pmd_private; + struct rte_pci_device *pci_dev; char *hw_addr; diff --git a/drivers/raw/ntb/ntb_hw_intel.c b/drivers/raw/ntb/ntb_hw_intel.c index 956f411ea3..955b384614 100644 --- a/drivers/raw/ntb/ntb_hw_intel.c +++ b/drivers/raw/ntb/ntb_hw_intel.c @@ -613,6 +613,16 @@ intel_ntb_vector_bind(const struct rte_rawdev *dev, uint8_t intr, uint8_t msix) } /* operations for primary side of local ntb */ +static uint64_t +intel_ntb_get_mem_align(const struct rte_rawdev *dev, uint32_t mw_id, + uint64_t mw_len __rte_unused) +{ + struct ntb_hw *hw = dev->dev_private; + + /* The memzone base must be aligned to the memory window size. */ + return hw->mw_size[mw_id]; +} + const struct ntb_dev_ops intel_ntb_ops = { .ntb_dev_init = intel_ntb_dev_init, .get_peer_mw_addr = intel_ntb_get_peer_mw_addr, @@ -627,4 +637,5 @@ const struct ntb_dev_ops intel_ntb_ops = { .db_set_mask = intel_ntb_db_set_mask, .peer_db_set = intel_ntb_peer_db_set, .vector_bind = intel_ntb_vector_bind, + .mem_align = intel_ntb_get_mem_align, }; diff --git a/drivers/raw/ntb/rte_pmd_ntb.h b/drivers/raw/ntb/rte_pmd_ntb.h index 76da3be026..70c89dab11 100644 --- a/drivers/raw/ntb/rte_pmd_ntb.h +++ b/drivers/raw/ntb/rte_pmd_ntb.h @@ -7,6 +7,8 @@ #include <stdint.h> +#include <rte_compat.h> + /* App needs to set/get these attrs */ #define NTB_QUEUE_SZ_NAME "queue_size" #define NTB_QUEUE_NUM_NAME "queue_num" @@ -42,4 +44,28 @@ struct ntb_queue_conf { struct rte_mempool *rx_mp; }; +/** + * @warning + * @b EXPERIMENTAL: this API may change without prior notice. + * + * Get the base-address alignment a memory window memzone must be reserved + * with. Some NTB hardware constrains the address a memory window can be + * translated to (for example, hardware that forms the peer target as + * (base | offset) needs the base aligned to a power of two >= the window + * length). Applications should reserve the memzone for memory window + * @p mw_id, of length @p mw_len, with at least the returned alignment. + * + * @param dev_id + * The identifier of the raw device. + * @param mw_id + * The memory window index. + * @param mw_len + * The length, in bytes, of the memzone to be reserved. + * @return + * The required base-address alignment in bytes, or 0 on error. + */ +__rte_experimental +uint64_t +rte_pmd_ntb_get_mem_align(uint16_t dev_id, uint32_t mw_id, uint64_t mw_len); + #endif /* _RTE_PMD_NTB_H_ */ diff --git a/examples/ntb/ntb_fwd.c b/examples/ntb/ntb_fwd.c index 33f3c1ef17..a187022718 100644 --- a/examples/ntb/ntb_fwd.c +++ b/examples/ntb/ntb_fwd.c @@ -1146,8 +1146,6 @@ ntb_mbuf_pool_create(uint16_t mbuf_seg_size, uint32_t nb_mbuf, if (!left_sz) break; snprintf(mz_name, sizeof(mz_name), "ntb_mw_%d", mz_id); - align = ntb_info.mw_size_align ? ntb_info.mw_size[mz_id] : - RTE_CACHE_LINE_SIZE; /* Reserve ntb header space on memzone 0. */ max_mz_len = mz_id ? ntb_info.mw_size[mz_id] : ntb_info.mw_size[mz_id] - ntb_info.ntb_hdr_size; @@ -1155,6 +1153,10 @@ ntb_mbuf_pool_create(uint16_t mbuf_seg_size, uint32_t nb_mbuf, (max_mz_len / total_elt_sz * total_elt_sz); if (!mz_len) continue; + /* Let the driver report the base-address alignment its + * hardware needs for a memory window of this length. + */ + align = rte_pmd_ntb_get_mem_align(dev_id, mz_id, mz_len); mz = rte_memzone_reserve_aligned(mz_name, mz_len, socket_id, RTE_MEMZONE_IOVA_CONTIG, align); if (mz == NULL) { -- 2.34.1

