[PATCH v2 1/2] raw/ntb: generalize framework for multiple vendors
Raghavendra Ningoji
raghavendra.ningoji at amd.com
Mon Sep 28 12:49:28 CEST 2026
The NTB rawdev framework was written around the Intel back-to-back
topology and the built-in scratchpad handshake protocol. To allow
other vendors to plug into the same framework, add vendor-neutral
hooks and make the common code dispatch through them:
- Add NTB_TOPO_PRI/NTB_TOPO_SEC topology types for hardware that uses
a primary/secondary topology instead of back-to-back.
- Add optional ntb_dev_ops hooks: interrupt_handler (vendor-specific
MSI-X handler), dev_handshake (vendor-specific link handshake) and
read_peer_config (vendor-specific peer-config read at start). When a
hook is NULL the common code keeps using the existing built-in path,
so the Intel driver is unaffected.
- Add a mem_align op and the rte_pmd_ntb_get_mem_align() API so an
application can query the base-address alignment a memory window
memzone needs, without embedding hardware-specific rules in the app.
The Intel driver reports its memory-window size.
- Add a pmd_private pointer to struct ntb_hw for vendor-specific state.
- Guard the receive path against a malformed stream with no end-of-packet
marker so it cannot overflow the descriptor ring.
Signed-off-by: Raghavendra Ningoji <raghavendra.ningoji at amd.com>
---
drivers/raw/ntb/ntb.c | 115 ++++++++++++++++++++++++---------
drivers/raw/ntb/ntb.h | 22 +++++++
drivers/raw/ntb/ntb_hw_intel.c | 11 ++++
drivers/raw/ntb/rte_pmd_ntb.h | 26 ++++++++
examples/ntb/ntb_fwd.c | 6 +-
5 files changed, 146 insertions(+), 34 deletions(-)
diff --git a/drivers/raw/ntb/ntb.c b/drivers/raw/ntb/ntb.c
index d54f2fb783..497c86b58c 100644
--- a/drivers/raw/ntb/ntb.c
+++ b/drivers/raw/ntb/ntb.c
@@ -18,6 +18,7 @@
#include <rte_memcpy.h>
#include <rte_rawdev.h>
#include <rte_rawdev_pmd.h>
+#include <eal_export.h>
#include "ntb_hw_intel.h"
#include "rte_pmd_ntb.h"
@@ -746,6 +747,11 @@ ntb_dequeue_bufs(struct rte_rawdev *dev,
for (nb_rx = 0; nb_rx < count; nb_rx++) {
i = 0;
while (true) {
+ if (unlikely(nb_mbufs >= rxq->nb_rx_desc)) {
+ NTB_LOG(ERR, "Malformed rx stream (no EOP); "
+ "aborting to avoid desc overflow.");
+ goto end_of_rx;
+ }
rx_item = rxq->rx_used_ring + rxq->last_used;
rxm_t = sw_ring[rxq->last_used].mbuf;
rxm_t->data_len = rx_item->len;
@@ -855,6 +861,26 @@ ntb_dev_info_get(struct rte_rawdev *dev, rte_rawdev_obj_t dev_info,
return 0;
}
+RTE_EXPORT_EXPERIMENTAL_SYMBOL(rte_pmd_ntb_get_mem_align, 26.11)
+uint64_t
+rte_pmd_ntb_get_mem_align(uint16_t dev_id, uint32_t mw_id, uint64_t mw_len)
+{
+ struct rte_rawdev *dev;
+ struct ntb_hw *hw;
+
+ if (dev_id >= RTE_RAWDEV_MAX_DEVS)
+ return 0;
+ dev = rte_rawdev_pmd_get_dev(dev_id);
+ if (dev->dev_private == NULL)
+ return 0;
+
+ hw = dev->dev_private;
+ if (mw_id >= hw->mw_cnt || hw->ntb_ops->mem_align == NULL)
+ return RTE_CACHE_LINE_SIZE;
+
+ return (*hw->ntb_ops->mem_align)(dev, mw_id, mw_len);
+}
+
static int
ntb_dev_configure(const struct rte_rawdev *dev, rte_rawdev_obj_t config,
size_t config_size)
@@ -882,8 +908,13 @@ ntb_dev_configure(const struct rte_rawdev *dev, rte_rawdev_obj_t config,
hw->ntb_xstats_off = rte_zmalloc("ntb_xstats_off", xstats_num *
sizeof(uint64_t), 0);
- /* Start handshake with the peer. */
- ret = ntb_handshake_work(dev);
+ /* Start handshake with the peer. Use the vendor-specific handshake
+ * if provided, otherwise the built-in scratchpad protocol.
+ */
+ if (hw->ntb_ops->dev_handshake != NULL)
+ ret = (*hw->ntb_ops->dev_handshake)(dev);
+ else
+ ret = ntb_handshake_work(dev);
if (ret < 0) {
rte_free(hw->rx_queues);
rte_free(hw->tx_queues);
@@ -929,35 +960,44 @@ ntb_dev_start(struct rte_rawdev *dev)
goto err_q_init;
}
- if (hw->ntb_ops->spad_read == NULL) {
- ret = -ENOTSUP;
- goto err_up;
- }
+ /* Read/validate peer config. Use the vendor-specific reader if
+ * provided, otherwise the built-in scratchpad reads.
+ */
+ if (hw->ntb_ops->read_peer_config != NULL) {
+ ret = (*hw->ntb_ops->read_peer_config)(dev);
+ if (ret < 0)
+ goto err_up;
+ } else {
+ if (hw->ntb_ops->spad_read == NULL) {
+ ret = -ENOTSUP;
+ goto err_up;
+ }
- peer_val = (*hw->ntb_ops->spad_read)(dev, SPAD_Q_SZ, 0);
- if (peer_val != hw->queue_size) {
- NTB_LOG(ERR, "Inconsistent queue size! (local: %u peer: %u)",
- hw->queue_size, peer_val);
- ret = -EINVAL;
- goto err_up;
- }
+ peer_val = (*hw->ntb_ops->spad_read)(dev, SPAD_Q_SZ, 0);
+ if (peer_val != hw->queue_size) {
+ NTB_LOG(ERR, "Inconsistent queue size! (local: %u peer: %u)",
+ hw->queue_size, peer_val);
+ ret = -EINVAL;
+ goto err_up;
+ }
- peer_val = (*hw->ntb_ops->spad_read)(dev, SPAD_NUM_QPS, 0);
- if (peer_val != hw->queue_pairs) {
- NTB_LOG(ERR, "Inconsistent number of queues! (local: %u peer:"
- " %u)", hw->queue_pairs, peer_val);
- ret = -EINVAL;
- goto err_up;
- }
+ peer_val = (*hw->ntb_ops->spad_read)(dev, SPAD_NUM_QPS, 0);
+ if (peer_val != hw->queue_pairs) {
+ NTB_LOG(ERR, "Inconsistent number of queues! (local: %u peer:"
+ " %u)", hw->queue_pairs, peer_val);
+ ret = -EINVAL;
+ goto err_up;
+ }
- hw->peer_used_mws = (*hw->ntb_ops->spad_read)(dev, SPAD_USED_MWS, 0);
+ hw->peer_used_mws = (*hw->ntb_ops->spad_read)(dev, SPAD_USED_MWS, 0);
- for (i = 0; i < hw->peer_used_mws; i++) {
- peer_base_h = (*hw->ntb_ops->spad_read)(dev,
- SPAD_MW0_BA_H + 2 * i, 0);
- peer_base_l = (*hw->ntb_ops->spad_read)(dev,
- SPAD_MW0_BA_L + 2 * i, 0);
- hw->peer_mw_base[i] = (peer_base_h << 32) + peer_base_l;
+ for (i = 0; i < hw->peer_used_mws; i++) {
+ peer_base_h = (*hw->ntb_ops->spad_read)(dev,
+ SPAD_MW0_BA_H + 2 * i, 0);
+ peer_base_l = (*hw->ntb_ops->spad_read)(dev,
+ SPAD_MW0_BA_L + 2 * i, 0);
+ hw->peer_mw_base[i] = (peer_base_h << 32) + peer_base_l;
+ }
}
dev->started = 1;
@@ -1057,8 +1097,13 @@ ntb_dev_close(struct rte_rawdev *dev)
rte_intr_disable(intr_handle);
/* Unregister callback func to eal lib */
- rte_intr_callback_unregister(intr_handle,
- ntb_dev_intr_handler, dev);
+ if (hw->ntb_ops->interrupt_handler != NULL)
+ rte_intr_callback_unregister(intr_handle,
+ hw->ntb_ops->interrupt_handler,
+ dev);
+ else
+ rte_intr_callback_unregister(intr_handle,
+ ntb_dev_intr_handler, dev);
return 0;
}
@@ -1409,9 +1454,15 @@ ntb_init_hw(struct rte_rawdev *dev, struct rte_pci_device *pci_dev)
(*hw->ntb_ops->db_clear)(dev, hw->db_valid_mask);
intr_handle = pci_dev->intr_handle;
- /* Register callback func to eal lib */
- rte_intr_callback_register(intr_handle,
- ntb_dev_intr_handler, dev);
+ /* Register callback func to eal lib. Use the vendor-specific handler
+ * if provided, otherwise fall back to the built-in handler.
+ */
+ if (hw->ntb_ops->interrupt_handler != NULL)
+ rte_intr_callback_register(intr_handle,
+ hw->ntb_ops->interrupt_handler, dev);
+ else
+ rte_intr_callback_register(intr_handle,
+ ntb_dev_intr_handler, dev);
ret = rte_intr_efd_enable(intr_handle, hw->db_cnt);
if (ret)
diff --git a/drivers/raw/ntb/ntb.h b/drivers/raw/ntb/ntb.h
index 8c7a2230f9..57d09a2cd4 100644
--- a/drivers/raw/ntb/ntb.h
+++ b/drivers/raw/ntb/ntb.h
@@ -42,6 +42,9 @@ enum ntb_topo {
NTB_TOPO_NONE = 0,
NTB_TOPO_B2B_USD,
NTB_TOPO_B2B_DSD,
+ /* Primary/secondary topology (e.g. AMD NTB). */
+ NTB_TOPO_PRI,
+ NTB_TOPO_SEC,
};
enum ntb_link {
@@ -100,6 +103,10 @@ enum ntb_spad_idx {
* for those db bits.
* @peer_db_set: Set doorbell bit to generate peer interrupt for that bit.
* @vector_bind: Bind vector source [intr] to msix vector [msix].
+ * @interrupt_handler: Vendor-specific interrupt handler. If NULL, the
+ * built-in handler is used.
+ * @mem_align: Base-address alignment required for a memory window memzone
+ * of a given length.
*/
struct ntb_dev_ops {
int (*ntb_dev_init)(const struct rte_rawdev *dev);
@@ -119,6 +126,18 @@ struct ntb_dev_ops {
int (*peer_db_set)(const struct rte_rawdev *dev, uint8_t db_bit);
int (*vector_bind)(const struct rte_rawdev *dev, uint8_t intr,
uint8_t msix);
+ void (*interrupt_handler)(void *param);
+ /* Optional vendor-specific handshake. If NULL, the built-in
+ * scratchpad handshake is used. Used by hardware (e.g. AMD) whose
+ * scratchpad layout differs from the built-in protocol.
+ */
+ int (*dev_handshake)(const struct rte_rawdev *dev);
+ /* Optional vendor-specific peer-config read at device start. If NULL,
+ * the built-in scratchpad reads are used.
+ */
+ int (*read_peer_config)(const struct rte_rawdev *dev);
+ uint64_t (*mem_align)(const struct rte_rawdev *dev, uint32_t mw_id,
+ uint64_t mw_len);
};
struct ntb_desc {
@@ -208,6 +227,9 @@ struct ntb_hw {
const struct ntb_dev_ops *ntb_ops;
+ /* Vendor-specific hardware private data. */
+ void *pmd_private;
+
struct rte_pci_device *pci_dev;
char *hw_addr;
diff --git a/drivers/raw/ntb/ntb_hw_intel.c b/drivers/raw/ntb/ntb_hw_intel.c
index 956f411ea3..955b384614 100644
--- a/drivers/raw/ntb/ntb_hw_intel.c
+++ b/drivers/raw/ntb/ntb_hw_intel.c
@@ -613,6 +613,16 @@ intel_ntb_vector_bind(const struct rte_rawdev *dev, uint8_t intr, uint8_t msix)
}
/* operations for primary side of local ntb */
+static uint64_t
+intel_ntb_get_mem_align(const struct rte_rawdev *dev, uint32_t mw_id,
+ uint64_t mw_len __rte_unused)
+{
+ struct ntb_hw *hw = dev->dev_private;
+
+ /* The memzone base must be aligned to the memory window size. */
+ return hw->mw_size[mw_id];
+}
+
const struct ntb_dev_ops intel_ntb_ops = {
.ntb_dev_init = intel_ntb_dev_init,
.get_peer_mw_addr = intel_ntb_get_peer_mw_addr,
@@ -627,4 +637,5 @@ const struct ntb_dev_ops intel_ntb_ops = {
.db_set_mask = intel_ntb_db_set_mask,
.peer_db_set = intel_ntb_peer_db_set,
.vector_bind = intel_ntb_vector_bind,
+ .mem_align = intel_ntb_get_mem_align,
};
diff --git a/drivers/raw/ntb/rte_pmd_ntb.h b/drivers/raw/ntb/rte_pmd_ntb.h
index 76da3be026..70c89dab11 100644
--- a/drivers/raw/ntb/rte_pmd_ntb.h
+++ b/drivers/raw/ntb/rte_pmd_ntb.h
@@ -7,6 +7,8 @@
#include <stdint.h>
+#include <rte_compat.h>
+
/* App needs to set/get these attrs */
#define NTB_QUEUE_SZ_NAME "queue_size"
#define NTB_QUEUE_NUM_NAME "queue_num"
@@ -42,4 +44,28 @@ struct ntb_queue_conf {
struct rte_mempool *rx_mp;
};
+/**
+ * @warning
+ * @b EXPERIMENTAL: this API may change without prior notice.
+ *
+ * Get the base-address alignment a memory window memzone must be reserved
+ * with. Some NTB hardware constrains the address a memory window can be
+ * translated to (for example, hardware that forms the peer target as
+ * (base | offset) needs the base aligned to a power of two >= the window
+ * length). Applications should reserve the memzone for memory window
+ * @p mw_id, of length @p mw_len, with at least the returned alignment.
+ *
+ * @param dev_id
+ * The identifier of the raw device.
+ * @param mw_id
+ * The memory window index.
+ * @param mw_len
+ * The length, in bytes, of the memzone to be reserved.
+ * @return
+ * The required base-address alignment in bytes, or 0 on error.
+ */
+__rte_experimental
+uint64_t
+rte_pmd_ntb_get_mem_align(uint16_t dev_id, uint32_t mw_id, uint64_t mw_len);
+
#endif /* _RTE_PMD_NTB_H_ */
diff --git a/examples/ntb/ntb_fwd.c b/examples/ntb/ntb_fwd.c
index 33f3c1ef17..a187022718 100644
--- a/examples/ntb/ntb_fwd.c
+++ b/examples/ntb/ntb_fwd.c
@@ -1146,8 +1146,6 @@ ntb_mbuf_pool_create(uint16_t mbuf_seg_size, uint32_t nb_mbuf,
if (!left_sz)
break;
snprintf(mz_name, sizeof(mz_name), "ntb_mw_%d", mz_id);
- align = ntb_info.mw_size_align ? ntb_info.mw_size[mz_id] :
- RTE_CACHE_LINE_SIZE;
/* Reserve ntb header space on memzone 0. */
max_mz_len = mz_id ? ntb_info.mw_size[mz_id] :
ntb_info.mw_size[mz_id] - ntb_info.ntb_hdr_size;
@@ -1155,6 +1153,10 @@ ntb_mbuf_pool_create(uint16_t mbuf_seg_size, uint32_t nb_mbuf,
(max_mz_len / total_elt_sz * total_elt_sz);
if (!mz_len)
continue;
+ /* Let the driver report the base-address alignment its
+ * hardware needs for a memory window of this length.
+ */
+ align = rte_pmd_ntb_get_mem_align(dev_id, mz_id, mz_len);
mz = rte_memzone_reserve_aligned(mz_name, mz_len, socket_id,
RTE_MEMZONE_IOVA_CONTIG, align);
if (mz == NULL) {
--
2.34.1
More information about the dev
mailing list