[PATCH v2 1/2] raw/ntb: generalize framework for multiple vendors

Raghavendra Ningoji raghavendra.ningoji at amd.com
Mon Sep 28 12:49:28 CEST 2026


The NTB rawdev framework was written around the Intel back-to-back
topology and the built-in scratchpad handshake protocol. To allow
other vendors to plug into the same framework, add vendor-neutral
hooks and make the common code dispatch through them:

- Add NTB_TOPO_PRI/NTB_TOPO_SEC topology types for hardware that uses
  a primary/secondary topology instead of back-to-back.
- Add optional ntb_dev_ops hooks: interrupt_handler (vendor-specific
  MSI-X handler), dev_handshake (vendor-specific link handshake) and
  read_peer_config (vendor-specific peer-config read at start). When a
  hook is NULL the common code keeps using the existing built-in path,
  so the Intel driver is unaffected.
- Add a mem_align op and the rte_pmd_ntb_get_mem_align() API so an
  application can query the base-address alignment a memory window
  memzone needs, without embedding hardware-specific rules in the app.
  The Intel driver reports its memory-window size.
- Add a pmd_private pointer to struct ntb_hw for vendor-specific state.
- Guard the receive path against a malformed stream with no end-of-packet
  marker so it cannot overflow the descriptor ring.

Signed-off-by: Raghavendra Ningoji <raghavendra.ningoji at amd.com>
---
 drivers/raw/ntb/ntb.c          | 115 ++++++++++++++++++++++++---------
 drivers/raw/ntb/ntb.h          |  22 +++++++
 drivers/raw/ntb/ntb_hw_intel.c |  11 ++++
 drivers/raw/ntb/rte_pmd_ntb.h  |  26 ++++++++
 examples/ntb/ntb_fwd.c         |   6 +-
 5 files changed, 146 insertions(+), 34 deletions(-)

diff --git a/drivers/raw/ntb/ntb.c b/drivers/raw/ntb/ntb.c
index d54f2fb783..497c86b58c 100644
--- a/drivers/raw/ntb/ntb.c
+++ b/drivers/raw/ntb/ntb.c
@@ -18,6 +18,7 @@
 #include <rte_memcpy.h>
 #include <rte_rawdev.h>
 #include <rte_rawdev_pmd.h>
+#include <eal_export.h>
 
 #include "ntb_hw_intel.h"
 #include "rte_pmd_ntb.h"
@@ -746,6 +747,11 @@ ntb_dequeue_bufs(struct rte_rawdev *dev,
 	for (nb_rx = 0; nb_rx < count; nb_rx++) {
 		i = 0;
 		while (true) {
+			if (unlikely(nb_mbufs >= rxq->nb_rx_desc)) {
+				NTB_LOG(ERR, "Malformed rx stream (no EOP); "
+					"aborting to avoid desc overflow.");
+				goto end_of_rx;
+			}
 			rx_item = rxq->rx_used_ring + rxq->last_used;
 			rxm_t = sw_ring[rxq->last_used].mbuf;
 			rxm_t->data_len = rx_item->len;
@@ -855,6 +861,26 @@ ntb_dev_info_get(struct rte_rawdev *dev, rte_rawdev_obj_t dev_info,
 	return 0;
 }
 
+RTE_EXPORT_EXPERIMENTAL_SYMBOL(rte_pmd_ntb_get_mem_align, 26.11)
+uint64_t
+rte_pmd_ntb_get_mem_align(uint16_t dev_id, uint32_t mw_id, uint64_t mw_len)
+{
+	struct rte_rawdev *dev;
+	struct ntb_hw *hw;
+
+	if (dev_id >= RTE_RAWDEV_MAX_DEVS)
+		return 0;
+	dev = rte_rawdev_pmd_get_dev(dev_id);
+	if (dev->dev_private == NULL)
+		return 0;
+
+	hw = dev->dev_private;
+	if (mw_id >= hw->mw_cnt || hw->ntb_ops->mem_align == NULL)
+		return RTE_CACHE_LINE_SIZE;
+
+	return (*hw->ntb_ops->mem_align)(dev, mw_id, mw_len);
+}
+
 static int
 ntb_dev_configure(const struct rte_rawdev *dev, rte_rawdev_obj_t config,
 		size_t config_size)
@@ -882,8 +908,13 @@ ntb_dev_configure(const struct rte_rawdev *dev, rte_rawdev_obj_t config,
 	hw->ntb_xstats_off = rte_zmalloc("ntb_xstats_off", xstats_num *
 					 sizeof(uint64_t), 0);
 
-	/* Start handshake with the peer. */
-	ret = ntb_handshake_work(dev);
+	/* Start handshake with the peer. Use the vendor-specific handshake
+	 * if provided, otherwise the built-in scratchpad protocol.
+	 */
+	if (hw->ntb_ops->dev_handshake != NULL)
+		ret = (*hw->ntb_ops->dev_handshake)(dev);
+	else
+		ret = ntb_handshake_work(dev);
 	if (ret < 0) {
 		rte_free(hw->rx_queues);
 		rte_free(hw->tx_queues);
@@ -929,35 +960,44 @@ ntb_dev_start(struct rte_rawdev *dev)
 		goto err_q_init;
 	}
 
-	if (hw->ntb_ops->spad_read == NULL) {
-		ret = -ENOTSUP;
-		goto err_up;
-	}
+	/* Read/validate peer config. Use the vendor-specific reader if
+	 * provided, otherwise the built-in scratchpad reads.
+	 */
+	if (hw->ntb_ops->read_peer_config != NULL) {
+		ret = (*hw->ntb_ops->read_peer_config)(dev);
+		if (ret < 0)
+			goto err_up;
+	} else {
+		if (hw->ntb_ops->spad_read == NULL) {
+			ret = -ENOTSUP;
+			goto err_up;
+		}
 
-	peer_val = (*hw->ntb_ops->spad_read)(dev, SPAD_Q_SZ, 0);
-	if (peer_val != hw->queue_size) {
-		NTB_LOG(ERR, "Inconsistent queue size! (local: %u peer: %u)",
-			hw->queue_size, peer_val);
-		ret = -EINVAL;
-		goto err_up;
-	}
+		peer_val = (*hw->ntb_ops->spad_read)(dev, SPAD_Q_SZ, 0);
+		if (peer_val != hw->queue_size) {
+			NTB_LOG(ERR, "Inconsistent queue size! (local: %u peer: %u)",
+				hw->queue_size, peer_val);
+			ret = -EINVAL;
+			goto err_up;
+		}
 
-	peer_val = (*hw->ntb_ops->spad_read)(dev, SPAD_NUM_QPS, 0);
-	if (peer_val != hw->queue_pairs) {
-		NTB_LOG(ERR, "Inconsistent number of queues! (local: %u peer:"
-			" %u)", hw->queue_pairs, peer_val);
-		ret = -EINVAL;
-		goto err_up;
-	}
+		peer_val = (*hw->ntb_ops->spad_read)(dev, SPAD_NUM_QPS, 0);
+		if (peer_val != hw->queue_pairs) {
+			NTB_LOG(ERR, "Inconsistent number of queues! (local: %u peer:"
+				" %u)", hw->queue_pairs, peer_val);
+			ret = -EINVAL;
+			goto err_up;
+		}
 
-	hw->peer_used_mws = (*hw->ntb_ops->spad_read)(dev, SPAD_USED_MWS, 0);
+		hw->peer_used_mws = (*hw->ntb_ops->spad_read)(dev, SPAD_USED_MWS, 0);
 
-	for (i = 0; i < hw->peer_used_mws; i++) {
-		peer_base_h = (*hw->ntb_ops->spad_read)(dev,
-				SPAD_MW0_BA_H + 2 * i, 0);
-		peer_base_l = (*hw->ntb_ops->spad_read)(dev,
-				SPAD_MW0_BA_L + 2 * i, 0);
-		hw->peer_mw_base[i] = (peer_base_h << 32) + peer_base_l;
+		for (i = 0; i < hw->peer_used_mws; i++) {
+			peer_base_h = (*hw->ntb_ops->spad_read)(dev,
+					SPAD_MW0_BA_H + 2 * i, 0);
+			peer_base_l = (*hw->ntb_ops->spad_read)(dev,
+					SPAD_MW0_BA_L + 2 * i, 0);
+			hw->peer_mw_base[i] = (peer_base_h << 32) + peer_base_l;
+		}
 	}
 
 	dev->started = 1;
@@ -1057,8 +1097,13 @@ ntb_dev_close(struct rte_rawdev *dev)
 	rte_intr_disable(intr_handle);
 
 	/* Unregister callback func to eal lib */
-	rte_intr_callback_unregister(intr_handle,
-				     ntb_dev_intr_handler, dev);
+	if (hw->ntb_ops->interrupt_handler != NULL)
+		rte_intr_callback_unregister(intr_handle,
+					     hw->ntb_ops->interrupt_handler,
+					     dev);
+	else
+		rte_intr_callback_unregister(intr_handle,
+					     ntb_dev_intr_handler, dev);
 
 	return 0;
 }
@@ -1409,9 +1454,15 @@ ntb_init_hw(struct rte_rawdev *dev, struct rte_pci_device *pci_dev)
 	(*hw->ntb_ops->db_clear)(dev, hw->db_valid_mask);
 
 	intr_handle = pci_dev->intr_handle;
-	/* Register callback func to eal lib */
-	rte_intr_callback_register(intr_handle,
-				   ntb_dev_intr_handler, dev);
+	/* Register callback func to eal lib. Use the vendor-specific handler
+	 * if provided, otherwise fall back to the built-in handler.
+	 */
+	if (hw->ntb_ops->interrupt_handler != NULL)
+		rte_intr_callback_register(intr_handle,
+					   hw->ntb_ops->interrupt_handler, dev);
+	else
+		rte_intr_callback_register(intr_handle,
+					   ntb_dev_intr_handler, dev);
 
 	ret = rte_intr_efd_enable(intr_handle, hw->db_cnt);
 	if (ret)
diff --git a/drivers/raw/ntb/ntb.h b/drivers/raw/ntb/ntb.h
index 8c7a2230f9..57d09a2cd4 100644
--- a/drivers/raw/ntb/ntb.h
+++ b/drivers/raw/ntb/ntb.h
@@ -42,6 +42,9 @@ enum ntb_topo {
 	NTB_TOPO_NONE = 0,
 	NTB_TOPO_B2B_USD,
 	NTB_TOPO_B2B_DSD,
+	/* Primary/secondary topology (e.g. AMD NTB). */
+	NTB_TOPO_PRI,
+	NTB_TOPO_SEC,
 };
 
 enum ntb_link {
@@ -100,6 +103,10 @@ enum ntb_spad_idx {
  * for those db bits.
  * @peer_db_set: Set doorbell bit to generate peer interrupt for that bit.
  * @vector_bind: Bind vector source [intr] to msix vector [msix].
+ * @interrupt_handler: Vendor-specific interrupt handler. If NULL, the
+ * built-in handler is used.
+ * @mem_align: Base-address alignment required for a memory window memzone
+ * of a given length.
  */
 struct ntb_dev_ops {
 	int (*ntb_dev_init)(const struct rte_rawdev *dev);
@@ -119,6 +126,18 @@ struct ntb_dev_ops {
 	int (*peer_db_set)(const struct rte_rawdev *dev, uint8_t db_bit);
 	int (*vector_bind)(const struct rte_rawdev *dev, uint8_t intr,
 			   uint8_t msix);
+	void (*interrupt_handler)(void *param);
+	/* Optional vendor-specific handshake. If NULL, the built-in
+	 * scratchpad handshake is used. Used by hardware (e.g. AMD) whose
+	 * scratchpad layout differs from the built-in protocol.
+	 */
+	int (*dev_handshake)(const struct rte_rawdev *dev);
+	/* Optional vendor-specific peer-config read at device start. If NULL,
+	 * the built-in scratchpad reads are used.
+	 */
+	int (*read_peer_config)(const struct rte_rawdev *dev);
+	uint64_t (*mem_align)(const struct rte_rawdev *dev, uint32_t mw_id,
+			      uint64_t mw_len);
 };
 
 struct ntb_desc {
@@ -208,6 +227,9 @@ struct ntb_hw {
 
 	const struct ntb_dev_ops *ntb_ops;
 
+	/* Vendor-specific hardware private data. */
+	void *pmd_private;
+
 	struct rte_pci_device *pci_dev;
 	char *hw_addr;
 
diff --git a/drivers/raw/ntb/ntb_hw_intel.c b/drivers/raw/ntb/ntb_hw_intel.c
index 956f411ea3..955b384614 100644
--- a/drivers/raw/ntb/ntb_hw_intel.c
+++ b/drivers/raw/ntb/ntb_hw_intel.c
@@ -613,6 +613,16 @@ intel_ntb_vector_bind(const struct rte_rawdev *dev, uint8_t intr, uint8_t msix)
 }
 
 /* operations for primary side of local ntb */
+static uint64_t
+intel_ntb_get_mem_align(const struct rte_rawdev *dev, uint32_t mw_id,
+			uint64_t mw_len __rte_unused)
+{
+	struct ntb_hw *hw = dev->dev_private;
+
+	/* The memzone base must be aligned to the memory window size. */
+	return hw->mw_size[mw_id];
+}
+
 const struct ntb_dev_ops intel_ntb_ops = {
 	.ntb_dev_init       = intel_ntb_dev_init,
 	.get_peer_mw_addr   = intel_ntb_get_peer_mw_addr,
@@ -627,4 +637,5 @@ const struct ntb_dev_ops intel_ntb_ops = {
 	.db_set_mask        = intel_ntb_db_set_mask,
 	.peer_db_set        = intel_ntb_peer_db_set,
 	.vector_bind        = intel_ntb_vector_bind,
+	.mem_align          = intel_ntb_get_mem_align,
 };
diff --git a/drivers/raw/ntb/rte_pmd_ntb.h b/drivers/raw/ntb/rte_pmd_ntb.h
index 76da3be026..70c89dab11 100644
--- a/drivers/raw/ntb/rte_pmd_ntb.h
+++ b/drivers/raw/ntb/rte_pmd_ntb.h
@@ -7,6 +7,8 @@
 
 #include <stdint.h>
 
+#include <rte_compat.h>
+
 /* App needs to set/get these attrs */
 #define NTB_QUEUE_SZ_NAME           "queue_size"
 #define NTB_QUEUE_NUM_NAME          "queue_num"
@@ -42,4 +44,28 @@ struct ntb_queue_conf {
 	struct rte_mempool *rx_mp;
 };
 
+/**
+ * @warning
+ * @b EXPERIMENTAL: this API may change without prior notice.
+ *
+ * Get the base-address alignment a memory window memzone must be reserved
+ * with. Some NTB hardware constrains the address a memory window can be
+ * translated to (for example, hardware that forms the peer target as
+ * (base | offset) needs the base aligned to a power of two >= the window
+ * length). Applications should reserve the memzone for memory window
+ * @p mw_id, of length @p mw_len, with at least the returned alignment.
+ *
+ * @param dev_id
+ *   The identifier of the raw device.
+ * @param mw_id
+ *   The memory window index.
+ * @param mw_len
+ *   The length, in bytes, of the memzone to be reserved.
+ * @return
+ *   The required base-address alignment in bytes, or 0 on error.
+ */
+__rte_experimental
+uint64_t
+rte_pmd_ntb_get_mem_align(uint16_t dev_id, uint32_t mw_id, uint64_t mw_len);
+
 #endif /* _RTE_PMD_NTB_H_ */
diff --git a/examples/ntb/ntb_fwd.c b/examples/ntb/ntb_fwd.c
index 33f3c1ef17..a187022718 100644
--- a/examples/ntb/ntb_fwd.c
+++ b/examples/ntb/ntb_fwd.c
@@ -1146,8 +1146,6 @@ ntb_mbuf_pool_create(uint16_t mbuf_seg_size, uint32_t nb_mbuf,
 		if (!left_sz)
 			break;
 		snprintf(mz_name, sizeof(mz_name), "ntb_mw_%d", mz_id);
-		align = ntb_info.mw_size_align ? ntb_info.mw_size[mz_id] :
-			RTE_CACHE_LINE_SIZE;
 		/* Reserve ntb header space on memzone 0. */
 		max_mz_len = mz_id ? ntb_info.mw_size[mz_id] :
 			     ntb_info.mw_size[mz_id] - ntb_info.ntb_hdr_size;
@@ -1155,6 +1153,10 @@ ntb_mbuf_pool_create(uint16_t mbuf_seg_size, uint32_t nb_mbuf,
 			(max_mz_len / total_elt_sz * total_elt_sz);
 		if (!mz_len)
 			continue;
+		/* Let the driver report the base-address alignment its
+		 * hardware needs for a memory window of this length.
+		 */
+		align = rte_pmd_ntb_get_mem_align(dev_id, mz_id, mz_len);
 		mz = rte_memzone_reserve_aligned(mz_name, mz_len, socket_id,
 					RTE_MEMZONE_IOVA_CONTIG, align);
 		if (mz == NULL) {
-- 
2.34.1



More information about the dev mailing list