Page MenuHomeFreeBSD

D60416.id188894.diff
No OneTemporary

D60416.id188894.diff

diff --git a/share/man/man4/iflib.4 b/share/man/man4/iflib.4
--- a/share/man/man4/iflib.4
+++ b/share/man/man4/iflib.4
@@ -1,4 +1,4 @@
-.Dd September 16, 2026
+.Dd October 6, 2026
.Dt IFLIB 4
.Os
.Sh NAME
@@ -186,6 +186,23 @@
This setting applies only when
.Va simple_tx
is enabled.
+.It Va net.iflib.tx_db_burst
+When set, a receive task that passes a burst of frames to the network stack
+notifies the hardware of the frames the stack transmits from within that
+burst, for instance when forwarding, once per transmit queue at the end of
+the burst instead of once per frame.
+This saves a register write per frame, at the cost of the time it takes to
+pass the rest of the burst up.
+A transmit queue defers only while it is at most an eighth full; beyond that,
+.Nm
+already combines notifications on its own.
+It has no effect when
+.Va net.iflib.min_tx_latency
+is set, or on transmit queues serviced by a different task because
+.Va tx_abdicate
+is set.
+The default is zero; it is expected to become one once the deferral has
+seen wider use.
.It Va net.iflib.tx_watchdog_periods
Number of consecutive
.Va net.iflib.timer_default
diff --git a/sys/net/iflib.c b/sys/net/iflib.c
--- a/sys/net/iflib.c
+++ b/sys/net/iflib.c
@@ -428,6 +428,8 @@
ift_defer_mfree:1,
ift_spare_bits0:6;
uint16_t ift_npending;
+ uint16_t ift_db_flush;
+ uint32_t ift_db_held;
uint16_t ift_db_pending;
uint16_t ift_rs_pending;
uint32_t ift_last_reclaim;
@@ -680,6 +682,10 @@
SYSCTL_INT(_net_iflib, OID_AUTO, no_tx_batch, CTLFLAG_RW,
&iflib_no_tx_batch, 0,
"minimize transmit latency at the possible expense of throughput");
+static int iflib_tx_db_burst = 0;
+SYSCTL_INT(_net_iflib, OID_AUTO, tx_db_burst, CTLFLAG_RWTUN,
+ &iflib_tx_db_burst, 0,
+ "ring transmit doorbells once per receive burst");
static int iflib_timer_default = 1000;
SYSCTL_INT(_net_iflib, OID_AUTO, timer_default, CTLFLAG_RW,
&iflib_timer_default, 0, "number of ticks between iflib_timer calls");
@@ -855,6 +861,7 @@
static int iflib_legacy_setup(if_ctx_t ctx, driver_filter_t filter, void *filterarg, int *rid, const char *str);
static void iflib_txq_check_drain(iflib_txq_t txq, int budget);
static uint32_t iflib_txq_can_drain(struct ifmp_ring *);
+static bool iflib_txd_db_check(iflib_txq_t txq, int ring);
#ifdef ALTQ
static void iflib_altq_if_start(if_t ifp);
static int iflib_altq_if_transmit(if_t ifp, struct mbuf *m);
@@ -2995,7 +3002,7 @@
txq->ift_wdog_armed = 0;
txq->ift_in_use = txq->ift_gen = txq->ift_no_desc_avail = 0;
txq->ift_npending = txq->ift_db_pending = 0;
- txq->ift_rs_pending = 0;
+ txq->ift_db_flush = txq->ift_rs_pending = 0;
if (sctx->isc_flags & IFLIB_PRESERVE_TX_INDICES)
txq->ift_cidx = txq->ift_pidx;
else
@@ -3270,6 +3277,132 @@
GROUPTASK_ENQUEUE(&rxq->ifr_task);
}
+/*
+ * Transmit doorbells per receive burst.
+ *
+ * Telling the hardware about new transmit descriptors takes a register write
+ * that can cost hundreds of nanoseconds, and a ring that is at most an
+ * eighth full gets one for every packet (see txq_max_db_deferred()). With
+ * direct dispatch, the receive task forwards packets and sends replies from
+ * within if_input(). While it does, the transmit queues it sends on leave
+ * their doorbells alone, and once the burst has been passed up the task rings
+ * every queue it held back. A queue holds back only while it is at most an
+ * eighth full: above that, iflib_txd_db_check() batches on its own, and
+ * holding back more would only delay reclaim.
+ *
+ * The burst is described on the stack of the receive task, which stays on its
+ * CPU while the burst lasts. The transmit path looks for it on a per-CPU
+ * list by thread, so any other transmitter, including a thread that preempts
+ * the task, finds none and rings as before. Only the owner removes an entry,
+ * so the owner's entry stays valid after the list is left.
+ *
+ * A queue counts the bursts holding its doorbell, so that the transmit task,
+ * which rings pending descriptors it finds on a completion interrupt, leaves
+ * them to the bursts instead of taking over the mp_ring.
+ */
+#define IFLIB_TXDB_QUEUES 8 /* queues a burst keeps track of */
+
+struct iflib_txdb_burst {
+ LIST_ENTRY(iflib_txdb_burst) itb_link;
+ struct thread *itb_td;
+ int itb_nqueues;
+ iflib_txq_t itb_txq[IFLIB_TXDB_QUEUES];
+};
+
+LIST_HEAD(iflib_txdb_bursts, iflib_txdb_burst);
+DPCPU_DEFINE_STATIC(struct iflib_txdb_bursts, iflib_txdb_bursts);
+
+static void
+iflib_txdb_burst_begin(struct iflib_txdb_burst *b)
+{
+
+ b->itb_td = curthread;
+ b->itb_nqueues = 0;
+ sched_pin();
+ critical_enter();
+ LIST_INSERT_HEAD(DPCPU_PTR(iflib_txdb_bursts), b, itb_link);
+ critical_exit();
+}
+
+/* The receive burst the current thread is passing up, if any. */
+static struct iflib_txdb_burst *
+iflib_txdb_burst(void)
+{
+ struct iflib_txdb_burst *b;
+
+ /* A burst owner is pinned, so it never finds its CPU's list empty. */
+ if (LIST_EMPTY(DPCPU_PTR(iflib_txdb_bursts)))
+ return (NULL);
+ critical_enter();
+ LIST_FOREACH(b, DPCPU_PTR(iflib_txdb_bursts), itb_link) {
+ if (b->itb_td == curthread)
+ break;
+ }
+ critical_exit();
+ return (b);
+}
+
+/*
+ * Called by whoever may ring the doorbell of txq: the mp_ring consumer, or
+ * the holder of ift_mtx on the simple transmit path. Returns true if the
+ * doorbell may be left to the end of burst b, which may be NULL. The first
+ * time a queue holds back, the burst remembers it and counts itself on it.
+ */
+static bool
+iflib_txdb_hold(struct iflib_txdb_burst *b, iflib_txq_t txq)
+{
+ int i;
+
+ if (b == NULL || txq->ift_in_use > txq->ift_size / 8)
+ return (false);
+ for (i = 0; i < b->itb_nqueues; i++) {
+ if (b->itb_txq[i] == txq)
+ return (true);
+ }
+ if (i == nitems(b->itb_txq))
+ return (false);
+ b->itb_txq[b->itb_nqueues++] = txq;
+ atomic_add_32(&txq->ift_db_held, 1);
+ return (true);
+}
+
+static void
+iflib_txdb_burst_end(struct iflib_txdb_burst *b)
+{
+ iflib_txq_t txq;
+ int i;
+
+ critical_enter();
+ LIST_REMOVE(b, itb_link);
+ critical_exit();
+ sched_unpin();
+
+ for (i = 0; i < b->itb_nqueues; i++) {
+ txq = b->itb_txq[i];
+ atomic_subtract_32(&txq->ift_db_held, 1);
+ if (txq->ift_ctx->ifc_sysctl_simple_tx) {
+ mtx_lock(&txq->ift_mtx);
+ if (txq->ift_db_pending != 0 &&
+ (atomic_load_acq_int(&txq->ift_producers) &
+ IFLIB_TXQ_QUIESCING) == 0)
+ (void)iflib_txd_db_check(txq, true);
+ mtx_unlock(&txq->ift_mtx);
+ } else if (txq->ift_db_pending != 0) {
+ /*
+ * Only the mp_ring consumer may ring. Ask the next
+ * drain to ring whatever is pending, and enqueue the
+ * queue itself to become the consumer, as the transmit
+ * task does. A stale count costs at most an empty
+ * drain; if the ring is full, its consumer will see the
+ * request.
+ */
+ atomic_store_16(&txq->ift_db_flush, 1);
+ ifmp_ring_enqueue(txq->ift_br, (void **)&txq, 1,
+ TX_BATCH_SIZE, false);
+ }
+ }
+}
+
static uint8_t
iflib_rxeof(iflib_rxq_t rxq, qidx_t budget)
{
@@ -3282,6 +3415,8 @@
struct if_rxd_info ri;
int err, budget_left, rx_bytes, rx_pkts;
iflib_fl_t fl;
+ struct iflib_txdb_burst txdb;
+ bool txdb_burst;
#if defined(INET6) || defined(INET)
int lro_enabled;
#endif
@@ -3383,6 +3518,9 @@
for (i = 0, fl = &rxq->ifr_fl[0]; i < sctx->isc_nfl; i++, fl++)
retval |= iflib_fl_refill_all(ctx, fl);
+ txdb_burst = rx_pkts != 0 && iflib_tx_db_burst && !iflib_min_tx_latency;
+ if (txdb_burst)
+ iflib_txdb_burst_begin(&txdb);
if (mh != NULL) {
if_input(ifp, mh);
DBG_COUNTER_INC(rx_if_input);
@@ -3397,6 +3535,8 @@
#if defined(INET6) || defined(INET)
tcp_lro_flush_all(&rxq->ifr_lc);
#endif
+ if (txdb_burst)
+ iflib_txdb_burst_end(&txdb);
if (avail != 0 || iflib_rxd_avail(ctx, rxq, *cidxp, 1) != 0)
retval |= IFLIB_RXEOF_MORE;
return (retval);
@@ -4261,6 +4401,7 @@
iflib_txq_t txq = r->cookie;
if_ctx_t ctx = txq->ift_ctx;
if_t ifp = ctx->ifc_ifp;
+ struct iflib_txdb_burst *txdb;
struct mbuf *m, **mp;
int avail, bytes_sent, consumed, count, err, i;
int mcast_sent, pkt_sent, reclaimed;
@@ -4271,7 +4412,12 @@
return (0);
}
reclaimed = iflib_completed_tx_reclaim(txq, NULL);
- rang = iflib_txd_db_check(txq, reclaimed && txq->ift_db_pending);
+ txdb = iflib_txdb_burst();
+ if (iflib_txdb_hold(txdb, txq))
+ rang = false;
+ else
+ rang = iflib_txd_db_check(txq,
+ reclaimed && txq->ift_db_pending);
avail = IDXDIFF(pidx, cidx, r->size);
if (__predict_false(ctx->ifc_flags & IFC_QFLUSH)) {
@@ -4338,12 +4484,20 @@
if (__predict_false(!iflib_is_running(ctx)))
break;
ETHER_BPF_MTAP(ifp, m);
- rang = iflib_txd_db_check(txq, false);
+ if (iflib_txdb_hold(txdb, txq))
+ rang = false;
+ else
+ rang = iflib_txd_db_check(txq, false);
}
/* deliberate use of bitwise or to avoid gratuitous short-circuit */
ring = rang ? false : (iflib_min_tx_latency | err | (!!txq->ift_reclaim_thresh));
- iflib_txd_db_check(txq, ring);
+ if (__predict_false(atomic_load_16(&txq->ift_db_flush) != 0)) {
+ /* A receive burst that sent on this queue has ended. */
+ atomic_store_16(&txq->ift_db_flush, 0);
+ iflib_txd_db_check(txq, true);
+ } else if (!iflib_txdb_hold(txdb, txq))
+ iflib_txd_db_check(txq, ring);
if_inc_counter(ifp, IFCOUNTER_OBYTES, bytes_sent);
if_inc_counter(ifp, IFCOUNTER_OPACKETS, pkt_sent);
if (mcast_sent)
@@ -4429,7 +4583,13 @@
if (if_altq_is_enabled(ifp))
iflib_altq_if_start(ifp);
#endif
- if (txq->ift_db_pending)
+ /*
+ * A doorbell that a receive burst holds is not a laggard: the burst
+ * rings it when it ends. Taking the consumer role here for it would
+ * leave this task, which rings per packet, draining the ring instead
+ * of the burst.
+ */
+ if (txq->ift_db_pending && atomic_load_32(&txq->ift_db_held) == 0)
ifmp_ring_enqueue(txq->ift_br, (void **)&txq, 1, TX_BATCH_SIZE, abdicate);
else if (!abdicate)
ifmp_ring_check_drainage(txq->ift_br, TX_BATCH_SIZE);
@@ -8165,8 +8325,8 @@
/* Consumes the mbuf in all cases */
static int
-iflib_simple_encap(iflib_txq_t txq, struct mbuf *m, int *bytes, int *pkts,
- int *mcasts)
+iflib_simple_encap(iflib_txq_t txq, struct iflib_txdb_burst *txdb,
+ struct mbuf *m, int *bytes, int *pkts, int *mcasts)
{
if_t ifp;
int error;
@@ -8186,14 +8346,15 @@
*mcasts += !!(m->m_flags & M_MCAST);
DBG_COUNTER_INC(tx_sent);
ETHER_BPF_MTAP(ifp, m);
- (void)iflib_txd_db_check(txq, false);
+ if (!iflib_txdb_hold(txdb, txq))
+ (void)iflib_txd_db_check(txq, false);
return (0);
}
/* Drain the deferral ring into the hardware. */
static void
-iflib_simple_drbr_drain(iflib_txq_t txq, u_int quota, int *bytes, int *pkts,
- int *mcasts)
+iflib_simple_drbr_drain(iflib_txq_t txq, struct iflib_txdb_burst *txdb,
+ u_int quota, int *bytes, int *pkts, int *mcasts)
{
if_ctx_t ctx;
struct mbuf *m;
@@ -8217,7 +8378,7 @@
m = drbr_dequeue(ifp, txq->ift_drbr);
if (m == NULL)
return;
- (void)iflib_simple_encap(txq, m, bytes, pkts, mcasts);
+ (void)iflib_simple_encap(txq, txdb, m, bytes, pkts, mcasts);
}
if (!drbr_empty(ifp, txq->ift_drbr)) {
txq->ift_drbr_stall++;
@@ -8252,7 +8413,7 @@
(void)iflib_completed_tx_reclaim(txq, NULL);
if (!drbr_empty(ifp, txq->ift_drbr))
- iflib_simple_drbr_drain(txq, iflib_simple_drain_quota,
+ iflib_simple_drbr_drain(txq, NULL, iflib_simple_drain_quota,
&bytes_sent, &pkt_sent, &mcast_sent);
if (txq->ift_db_pending != 0)
(void)iflib_txd_db_check(txq, true);
@@ -8270,8 +8431,8 @@
* never touched.
*/
static int
-iflib_simple_transmit_locked(iflib_txq_t txq, struct mbuf *m, int *bytes,
- int *pkts, int *mcasts)
+iflib_simple_transmit_locked(iflib_txq_t txq, struct iflib_txdb_burst *txdb,
+ struct mbuf *m, int *bytes, int *pkts, int *mcasts)
{
if_ctx_t ctx;
if_t ifp;
@@ -8284,7 +8445,7 @@
if (__predict_true(!drbr_needs_enqueue(ifp, txq->ift_drbr) &&
TXQ_AVAIL(txq) >= MAX_TX_DESC(ctx))) {
txq->ift_drbr_direct++;
- error = iflib_simple_encap(txq, m, bytes, pkts, mcasts);
+ error = iflib_simple_encap(txq, txdb, m, bytes, pkts, mcasts);
} else {
error = buf_ring_enqueue(txq->ift_drbr, m);
if (__predict_false(error != 0)) {
@@ -8301,8 +8462,8 @@
* are the only thread that can drain it.
*/
if (!drbr_empty(ifp, txq->ift_drbr))
- iflib_simple_drbr_drain(txq, iflib_simple_drain_quota_thread,
- bytes, pkts, mcasts);
+ iflib_simple_drbr_drain(txq, txdb,
+ iflib_simple_drain_quota_thread, bytes, pkts, mcasts);
return (error);
}
@@ -8315,6 +8476,7 @@
iflib_txq_t txq)
{
struct mbuf **m_defer;
+ struct iflib_txdb_burst *txdb;
enum iflib_txq_producer_status producer_status;
bool pinned;
int error, i, reclaimable;
@@ -8363,9 +8525,10 @@
goto net_down;
}
- error = iflib_simple_transmit_locked(txq, m, &bytes_sent, &pkt_sent,
- &mcast_sent);
- if (txq->ift_db_pending != 0)
+ txdb = iflib_txdb_burst();
+ error = iflib_simple_transmit_locked(txq, txdb, m, &bytes_sent,
+ &pkt_sent, &mcast_sent);
+ if (txq->ift_db_pending != 0 && !iflib_txdb_hold(txdb, txq))
(void)iflib_txd_db_check(txq, true);
m_defer = NULL;
reclaimable = iflib_txq_can_reclaim(txq);
@@ -8480,7 +8643,7 @@
mtx_lock(&txq->ift_mtx);
(void)iflib_completed_tx_reclaim(txq, NULL);
- iflib_simple_drbr_drain(txq, UINT_MAX, &bytes_sent, &pkt_sent,
+ iflib_simple_drbr_drain(txq, NULL, UINT_MAX, &bytes_sent, &pkt_sent,
&mcast_sent);
if (txq->ift_db_pending != 0)
(void)iflib_txd_db_check(txq, true);

File Metadata

Mime Type
text/plain
Expires
Sun, Oct 11, 7:26 AM (8 h, 41 m)
Storage Engine
blob
Storage Format
Raw Data
Storage Handle
40551361
Default Alt Text
D60416.id188894.diff (13 KB)

Event Timeline