Page Menu
Home
FreeBSD
Search
Configure Global Search
Log In
Files
F175317578
D60416.diff
No One
Temporary
Actions
View File
Edit File
Delete File
View Transforms
Subscribe
Mute Notifications
Flag For Later
Award Token
Size
13 KB
Referenced Files
None
Subscribers
None
D60416.diff
View Options
diff --git a/share/man/man4/iflib.4 b/share/man/man4/iflib.4
--- a/share/man/man4/iflib.4
+++ b/share/man/man4/iflib.4
@@ -1,4 +1,4 @@
-.Dd September 16, 2026
+.Dd October 7, 2026
.Dt IFLIB 4
.Os
.Sh NAME
@@ -186,6 +186,23 @@
This setting applies only when
.Va simple_tx
is enabled.
+.It Va net.iflib.tx_db_batched
+When set, a receive task that passes a batch of frames to the network stack
+notifies the hardware of the frames the stack transmits from within that
+batch, for instance when forwarding, once per transmit queue at the end of
+the batch instead of once per frame.
+This saves a register write per frame, at the cost of the time it takes to
+pass the rest of the batch up.
+A transmit queue defers only while it is at most an eighth full; beyond that,
+.Nm
+already combines notifications on its own.
+It has no effect when
+.Va net.iflib.min_tx_latency
+is set, or on transmit queues serviced by a different task because
+.Va tx_abdicate
+is set.
+The default is zero; it is expected to become one once the deferral has
+seen wider use.
.It Va net.iflib.tx_watchdog_periods
Number of consecutive
.Va net.iflib.timer_default
diff --git a/sys/net/iflib.c b/sys/net/iflib.c
--- a/sys/net/iflib.c
+++ b/sys/net/iflib.c
@@ -428,6 +428,8 @@
ift_defer_mfree:1,
ift_spare_bits0:6;
uint16_t ift_npending;
+ uint16_t ift_db_flush;
+ uint32_t ift_db_held;
uint16_t ift_db_pending;
uint16_t ift_rs_pending;
uint32_t ift_last_reclaim;
@@ -680,6 +682,10 @@
SYSCTL_INT(_net_iflib, OID_AUTO, no_tx_batch, CTLFLAG_RW,
&iflib_no_tx_batch, 0,
"minimize transmit latency at the possible expense of throughput");
+static int iflib_tx_db_batched = 0;
+SYSCTL_INT(_net_iflib, OID_AUTO, tx_db_batched, CTLFLAG_RWTUN,
+ &iflib_tx_db_batched, 0,
+ "ring transmit doorbells once per receive batch");
static int iflib_timer_default = 1000;
SYSCTL_INT(_net_iflib, OID_AUTO, timer_default, CTLFLAG_RW,
&iflib_timer_default, 0, "number of ticks between iflib_timer calls");
@@ -855,6 +861,7 @@
static int iflib_legacy_setup(if_ctx_t ctx, driver_filter_t filter, void *filterarg, int *rid, const char *str);
static void iflib_txq_check_drain(iflib_txq_t txq, int budget);
static uint32_t iflib_txq_can_drain(struct ifmp_ring *);
+static bool iflib_txd_db_check(iflib_txq_t txq, int ring);
#ifdef ALTQ
static void iflib_altq_if_start(if_t ifp);
static int iflib_altq_if_transmit(if_t ifp, struct mbuf *m);
@@ -2995,7 +3002,7 @@
txq->ift_wdog_armed = 0;
txq->ift_in_use = txq->ift_gen = txq->ift_no_desc_avail = 0;
txq->ift_npending = txq->ift_db_pending = 0;
- txq->ift_rs_pending = 0;
+ txq->ift_db_flush = txq->ift_rs_pending = 0;
if (sctx->isc_flags & IFLIB_PRESERVE_TX_INDICES)
txq->ift_cidx = txq->ift_pidx;
else
@@ -3270,6 +3277,135 @@
GROUPTASK_ENQUEUE(&rxq->ifr_task);
}
+/*
+ * Transmit doorbells batched per receive pass.
+ *
+ * Telling the hardware about new transmit descriptors takes a register write
+ * that can cost hundreds of nanoseconds, and a ring that is at most an
+ * eighth full gets one for every packet (see txq_max_db_deferred()). With
+ * direct dispatch, the receive task forwards packets and sends replies from
+ * within if_input(), which it calls after its receive loop, from the LRO
+ * flush, and from tcp_lro_queue_mbuf() inside the loop for packets LRO
+ * cannot take (for instance while IP forwarding is enabled). For the whole
+ * receive pass, the transmit queues it sends on leave their doorbells alone,
+ * and at its end the task rings every queue it held back. A queue holds
+ * back only while it is at most an eighth full: above that,
+ * iflib_txd_db_check() batches on its own, and holding back more would only
+ * delay reclaim.
+ *
+ * The batch is described on the stack of the receive task, which stays on its
+ * CPU while the batch lasts. The transmit path looks for it on a per-CPU
+ * list by thread, so any other transmitter, including a thread that preempts
+ * the task, finds none and rings as before. Only the owner removes an entry,
+ * so the owner's entry stays valid after the list is left.
+ *
+ * A queue counts the batches holding its doorbell, so that the transmit task,
+ * which rings pending descriptors it finds on a completion interrupt, leaves
+ * them to the batches instead of taking over the mp_ring.
+ */
+#define IFLIB_TXDB_QUEUES 8 /* queues a batch keeps track of */
+
+struct iflib_txdb_batch {
+ LIST_ENTRY(iflib_txdb_batch) itb_link;
+ struct thread *itb_td;
+ int itb_nqueues;
+ iflib_txq_t itb_txq[IFLIB_TXDB_QUEUES];
+};
+
+LIST_HEAD(iflib_txdb_batches, iflib_txdb_batch);
+DPCPU_DEFINE_STATIC(struct iflib_txdb_batches, iflib_txdb_batches);
+
+static void
+iflib_txdb_batch_begin(struct iflib_txdb_batch *b)
+{
+
+ b->itb_td = curthread;
+ b->itb_nqueues = 0;
+ sched_pin();
+ critical_enter();
+ LIST_INSERT_HEAD(DPCPU_PTR(iflib_txdb_batches), b, itb_link);
+ critical_exit();
+}
+
+/* The receive batch the current thread is passing up, if any. */
+static struct iflib_txdb_batch *
+iflib_txdb_batch(void)
+{
+ struct iflib_txdb_batch *b;
+
+ /* A batch owner is pinned, so it never finds its CPU's list empty. */
+ if (LIST_EMPTY(DPCPU_PTR(iflib_txdb_batches)))
+ return (NULL);
+ critical_enter();
+ LIST_FOREACH(b, DPCPU_PTR(iflib_txdb_batches), itb_link) {
+ if (b->itb_td == curthread)
+ break;
+ }
+ critical_exit();
+ return (b);
+}
+
+/*
+ * Called by whoever may ring the doorbell of txq: the mp_ring consumer, or
+ * the holder of ift_mtx on the simple transmit path. Returns true if the
+ * doorbell may be left to the end of batch b, which may be NULL. The first
+ * time a queue holds back, the batch remembers it and counts itself on it.
+ */
+static bool
+iflib_txdb_hold(struct iflib_txdb_batch *b, iflib_txq_t txq)
+{
+ int i;
+
+ if (b == NULL || txq->ift_in_use > txq->ift_size / 8)
+ return (false);
+ for (i = 0; i < b->itb_nqueues; i++) {
+ if (b->itb_txq[i] == txq)
+ return (true);
+ }
+ if (i == nitems(b->itb_txq))
+ return (false);
+ b->itb_txq[b->itb_nqueues++] = txq;
+ atomic_add_32(&txq->ift_db_held, 1);
+ return (true);
+}
+
+static void
+iflib_txdb_batch_end(struct iflib_txdb_batch *b)
+{
+ iflib_txq_t txq;
+ int i;
+
+ critical_enter();
+ LIST_REMOVE(b, itb_link);
+ critical_exit();
+ sched_unpin();
+
+ for (i = 0; i < b->itb_nqueues; i++) {
+ txq = b->itb_txq[i];
+ atomic_subtract_32(&txq->ift_db_held, 1);
+ if (txq->ift_ctx->ifc_sysctl_simple_tx) {
+ mtx_lock(&txq->ift_mtx);
+ if (txq->ift_db_pending != 0 &&
+ (atomic_load_acq_int(&txq->ift_producers) &
+ IFLIB_TXQ_QUIESCING) == 0)
+ (void)iflib_txd_db_check(txq, true);
+ mtx_unlock(&txq->ift_mtx);
+ } else if (txq->ift_db_pending != 0) {
+ /*
+ * Only the mp_ring consumer may ring. Ask the next
+ * drain to ring whatever is pending, and enqueue the
+ * queue itself to become the consumer, as the transmit
+ * task does. A stale count costs at most an empty
+ * drain; if the ring is full, its consumer will see the
+ * request.
+ */
+ atomic_store_16(&txq->ift_db_flush, 1);
+ ifmp_ring_enqueue(txq->ift_br, (void **)&txq, 1,
+ TX_BATCH_SIZE, false);
+ }
+ }
+}
+
static uint8_t
iflib_rxeof(iflib_rxq_t rxq, qidx_t budget)
{
@@ -3282,6 +3418,8 @@
struct if_rxd_info ri;
int err, budget_left, rx_bytes, rx_pkts;
iflib_fl_t fl;
+ struct iflib_txdb_batch txdb;
+ bool txdb_batched;
#if defined(INET6) || defined(INET)
int lro_enabled;
#endif
@@ -3313,6 +3451,9 @@
#if defined(INET6) || defined(INET)
lro_enabled = (if_getcapenable(ifp) & IFCAP_LRO);
#endif
+ txdb_batched = iflib_tx_db_batched && !iflib_min_tx_latency;
+ if (txdb_batched)
+ iflib_txdb_batch_begin(&txdb);
/* pfil needs the vnet to be set */
CURVNET_SET_QUIET(if_getvnet(ifp));
@@ -3397,10 +3538,14 @@
#if defined(INET6) || defined(INET)
tcp_lro_flush_all(&rxq->ifr_lc);
#endif
+ if (txdb_batched)
+ iflib_txdb_batch_end(&txdb);
if (avail != 0 || iflib_rxd_avail(ctx, rxq, *cidxp, 1) != 0)
retval |= IFLIB_RXEOF_MORE;
return (retval);
err:
+ if (txdb_batched)
+ iflib_txdb_batch_end(&txdb);
STATE_LOCK(ctx);
ctx->ifc_flags |= IFC_DO_RESET;
iflib_admin_intr_deferred(ctx);
@@ -4261,6 +4406,7 @@
iflib_txq_t txq = r->cookie;
if_ctx_t ctx = txq->ift_ctx;
if_t ifp = ctx->ifc_ifp;
+ struct iflib_txdb_batch *txdb;
struct mbuf *m, **mp;
int avail, bytes_sent, consumed, count, err, i;
int mcast_sent, pkt_sent, reclaimed;
@@ -4271,7 +4417,12 @@
return (0);
}
reclaimed = iflib_completed_tx_reclaim(txq, NULL);
- rang = iflib_txd_db_check(txq, reclaimed && txq->ift_db_pending);
+ txdb = iflib_txdb_batch();
+ if (iflib_txdb_hold(txdb, txq))
+ rang = false;
+ else
+ rang = iflib_txd_db_check(txq,
+ reclaimed && txq->ift_db_pending);
avail = IDXDIFF(pidx, cidx, r->size);
if (__predict_false(ctx->ifc_flags & IFC_QFLUSH)) {
@@ -4338,12 +4489,20 @@
if (__predict_false(!iflib_is_running(ctx)))
break;
ETHER_BPF_MTAP(ifp, m);
- rang = iflib_txd_db_check(txq, false);
+ if (iflib_txdb_hold(txdb, txq))
+ rang = false;
+ else
+ rang = iflib_txd_db_check(txq, false);
}
/* deliberate use of bitwise or to avoid gratuitous short-circuit */
ring = rang ? false : (iflib_min_tx_latency | err | (!!txq->ift_reclaim_thresh));
- iflib_txd_db_check(txq, ring);
+ if (__predict_false(atomic_load_16(&txq->ift_db_flush) != 0)) {
+ /* A receive batch that sent on this queue has ended. */
+ atomic_store_16(&txq->ift_db_flush, 0);
+ iflib_txd_db_check(txq, true);
+ } else if (!iflib_txdb_hold(txdb, txq))
+ iflib_txd_db_check(txq, ring);
if_inc_counter(ifp, IFCOUNTER_OBYTES, bytes_sent);
if_inc_counter(ifp, IFCOUNTER_OPACKETS, pkt_sent);
if (mcast_sent)
@@ -4429,7 +4588,13 @@
if (if_altq_is_enabled(ifp))
iflib_altq_if_start(ifp);
#endif
- if (txq->ift_db_pending)
+ /*
+ * A doorbell that a receive batch holds is not a laggard: the batch
+ * rings it when it ends. Taking the consumer role here for it would
+ * leave this task, which rings per packet, draining the ring instead
+ * of the batch.
+ */
+ if (txq->ift_db_pending && atomic_load_32(&txq->ift_db_held) == 0)
ifmp_ring_enqueue(txq->ift_br, (void **)&txq, 1, TX_BATCH_SIZE, abdicate);
else if (!abdicate)
ifmp_ring_check_drainage(txq->ift_br, TX_BATCH_SIZE);
@@ -8165,8 +8330,8 @@
/* Consumes the mbuf in all cases */
static int
-iflib_simple_encap(iflib_txq_t txq, struct mbuf *m, int *bytes, int *pkts,
- int *mcasts)
+iflib_simple_encap(iflib_txq_t txq, struct iflib_txdb_batch *txdb,
+ struct mbuf *m, int *bytes, int *pkts, int *mcasts)
{
if_t ifp;
int error;
@@ -8186,14 +8351,15 @@
*mcasts += !!(m->m_flags & M_MCAST);
DBG_COUNTER_INC(tx_sent);
ETHER_BPF_MTAP(ifp, m);
- (void)iflib_txd_db_check(txq, false);
+ if (!iflib_txdb_hold(txdb, txq))
+ (void)iflib_txd_db_check(txq, false);
return (0);
}
/* Drain the deferral ring into the hardware. */
static void
-iflib_simple_drbr_drain(iflib_txq_t txq, u_int quota, int *bytes, int *pkts,
- int *mcasts)
+iflib_simple_drbr_drain(iflib_txq_t txq, struct iflib_txdb_batch *txdb,
+ u_int quota, int *bytes, int *pkts, int *mcasts)
{
if_ctx_t ctx;
struct mbuf *m;
@@ -8217,7 +8383,7 @@
m = drbr_dequeue(ifp, txq->ift_drbr);
if (m == NULL)
return;
- (void)iflib_simple_encap(txq, m, bytes, pkts, mcasts);
+ (void)iflib_simple_encap(txq, txdb, m, bytes, pkts, mcasts);
}
if (!drbr_empty(ifp, txq->ift_drbr)) {
txq->ift_drbr_stall++;
@@ -8252,7 +8418,7 @@
(void)iflib_completed_tx_reclaim(txq, NULL);
if (!drbr_empty(ifp, txq->ift_drbr))
- iflib_simple_drbr_drain(txq, iflib_simple_drain_quota,
+ iflib_simple_drbr_drain(txq, NULL, iflib_simple_drain_quota,
&bytes_sent, &pkt_sent, &mcast_sent);
if (txq->ift_db_pending != 0)
(void)iflib_txd_db_check(txq, true);
@@ -8270,8 +8436,8 @@
* never touched.
*/
static int
-iflib_simple_transmit_locked(iflib_txq_t txq, struct mbuf *m, int *bytes,
- int *pkts, int *mcasts)
+iflib_simple_transmit_locked(iflib_txq_t txq, struct iflib_txdb_batch *txdb,
+ struct mbuf *m, int *bytes, int *pkts, int *mcasts)
{
if_ctx_t ctx;
if_t ifp;
@@ -8284,7 +8450,7 @@
if (__predict_true(!drbr_needs_enqueue(ifp, txq->ift_drbr) &&
TXQ_AVAIL(txq) >= MAX_TX_DESC(ctx))) {
txq->ift_drbr_direct++;
- error = iflib_simple_encap(txq, m, bytes, pkts, mcasts);
+ error = iflib_simple_encap(txq, txdb, m, bytes, pkts, mcasts);
} else {
error = buf_ring_enqueue(txq->ift_drbr, m);
if (__predict_false(error != 0)) {
@@ -8301,8 +8467,8 @@
* are the only thread that can drain it.
*/
if (!drbr_empty(ifp, txq->ift_drbr))
- iflib_simple_drbr_drain(txq, iflib_simple_drain_quota_thread,
- bytes, pkts, mcasts);
+ iflib_simple_drbr_drain(txq, txdb,
+ iflib_simple_drain_quota_thread, bytes, pkts, mcasts);
return (error);
}
@@ -8315,6 +8481,7 @@
iflib_txq_t txq)
{
struct mbuf **m_defer;
+ struct iflib_txdb_batch *txdb;
enum iflib_txq_producer_status producer_status;
bool pinned;
int error, i, reclaimable;
@@ -8363,9 +8530,10 @@
goto net_down;
}
- error = iflib_simple_transmit_locked(txq, m, &bytes_sent, &pkt_sent,
- &mcast_sent);
- if (txq->ift_db_pending != 0)
+ txdb = iflib_txdb_batch();
+ error = iflib_simple_transmit_locked(txq, txdb, m, &bytes_sent,
+ &pkt_sent, &mcast_sent);
+ if (txq->ift_db_pending != 0 && !iflib_txdb_hold(txdb, txq))
(void)iflib_txd_db_check(txq, true);
m_defer = NULL;
reclaimable = iflib_txq_can_reclaim(txq);
@@ -8480,7 +8648,7 @@
mtx_lock(&txq->ift_mtx);
(void)iflib_completed_tx_reclaim(txq, NULL);
- iflib_simple_drbr_drain(txq, UINT_MAX, &bytes_sent, &pkt_sent,
+ iflib_simple_drbr_drain(txq, NULL, UINT_MAX, &bytes_sent, &pkt_sent,
&mcast_sent);
if (txq->ift_db_pending != 0)
(void)iflib_txd_db_check(txq, true);
File Metadata
Details
Attached
Mime Type
text/plain
Expires
Sat, Oct 10, 10:13 PM (1 h, 45 m)
Storage Engine
blob
Storage Format
Raw Data
Storage Handle
40519864
Default Alt Text
D60416.diff (13 KB)
Attached To
Mode
D60416: iflib: Ring transmit doorbells once per receive batch
Attached
Detach File
Event Timeline
Log In to Comment