Page Menu
Home
FreeBSD
Search
Configure Global Search
Log In
Files
F175363013
D60416.id188894.diff
No One
Temporary
Actions
View File
Edit File
Delete File
View Transforms
Subscribe
Mute Notifications
Flag For Later
Award Token
Size
13 KB
Referenced Files
None
Subscribers
None
D60416.id188894.diff
View Options
diff --git a/share/man/man4/iflib.4 b/share/man/man4/iflib.4
--- a/share/man/man4/iflib.4
+++ b/share/man/man4/iflib.4
@@ -1,4 +1,4 @@
-.Dd September 16, 2026
+.Dd October 6, 2026
.Dt IFLIB 4
.Os
.Sh NAME
@@ -186,6 +186,23 @@
This setting applies only when
.Va simple_tx
is enabled.
+.It Va net.iflib.tx_db_burst
+When set, a receive task that passes a burst of frames to the network stack
+notifies the hardware of the frames the stack transmits from within that
+burst, for instance when forwarding, once per transmit queue at the end of
+the burst instead of once per frame.
+This saves a register write per frame, at the cost of the time it takes to
+pass the rest of the burst up.
+A transmit queue defers only while it is at most an eighth full; beyond that,
+.Nm
+already combines notifications on its own.
+It has no effect when
+.Va net.iflib.min_tx_latency
+is set, or on transmit queues serviced by a different task because
+.Va tx_abdicate
+is set.
+The default is zero; it is expected to become one once the deferral has
+seen wider use.
.It Va net.iflib.tx_watchdog_periods
Number of consecutive
.Va net.iflib.timer_default
diff --git a/sys/net/iflib.c b/sys/net/iflib.c
--- a/sys/net/iflib.c
+++ b/sys/net/iflib.c
@@ -428,6 +428,8 @@
ift_defer_mfree:1,
ift_spare_bits0:6;
uint16_t ift_npending;
+ uint16_t ift_db_flush;
+ uint32_t ift_db_held;
uint16_t ift_db_pending;
uint16_t ift_rs_pending;
uint32_t ift_last_reclaim;
@@ -680,6 +682,10 @@
SYSCTL_INT(_net_iflib, OID_AUTO, no_tx_batch, CTLFLAG_RW,
&iflib_no_tx_batch, 0,
"minimize transmit latency at the possible expense of throughput");
+static int iflib_tx_db_burst = 0;
+SYSCTL_INT(_net_iflib, OID_AUTO, tx_db_burst, CTLFLAG_RWTUN,
+ &iflib_tx_db_burst, 0,
+ "ring transmit doorbells once per receive burst");
static int iflib_timer_default = 1000;
SYSCTL_INT(_net_iflib, OID_AUTO, timer_default, CTLFLAG_RW,
&iflib_timer_default, 0, "number of ticks between iflib_timer calls");
@@ -855,6 +861,7 @@
static int iflib_legacy_setup(if_ctx_t ctx, driver_filter_t filter, void *filterarg, int *rid, const char *str);
static void iflib_txq_check_drain(iflib_txq_t txq, int budget);
static uint32_t iflib_txq_can_drain(struct ifmp_ring *);
+static bool iflib_txd_db_check(iflib_txq_t txq, int ring);
#ifdef ALTQ
static void iflib_altq_if_start(if_t ifp);
static int iflib_altq_if_transmit(if_t ifp, struct mbuf *m);
@@ -2995,7 +3002,7 @@
txq->ift_wdog_armed = 0;
txq->ift_in_use = txq->ift_gen = txq->ift_no_desc_avail = 0;
txq->ift_npending = txq->ift_db_pending = 0;
- txq->ift_rs_pending = 0;
+ txq->ift_db_flush = txq->ift_rs_pending = 0;
if (sctx->isc_flags & IFLIB_PRESERVE_TX_INDICES)
txq->ift_cidx = txq->ift_pidx;
else
@@ -3270,6 +3277,132 @@
GROUPTASK_ENQUEUE(&rxq->ifr_task);
}
+/*
+ * Transmit doorbells per receive burst.
+ *
+ * Telling the hardware about new transmit descriptors takes a register write
+ * that can cost hundreds of nanoseconds, and a ring that is at most an
+ * eighth full gets one for every packet (see txq_max_db_deferred()). With
+ * direct dispatch, the receive task forwards packets and sends replies from
+ * within if_input(). While it does, the transmit queues it sends on leave
+ * their doorbells alone, and once the burst has been passed up the task rings
+ * every queue it held back. A queue holds back only while it is at most an
+ * eighth full: above that, iflib_txd_db_check() batches on its own, and
+ * holding back more would only delay reclaim.
+ *
+ * The burst is described on the stack of the receive task, which stays on its
+ * CPU while the burst lasts. The transmit path looks for it on a per-CPU
+ * list by thread, so any other transmitter, including a thread that preempts
+ * the task, finds none and rings as before. Only the owner removes an entry,
+ * so the owner's entry stays valid after the list is left.
+ *
+ * A queue counts the bursts holding its doorbell, so that the transmit task,
+ * which rings pending descriptors it finds on a completion interrupt, leaves
+ * them to the bursts instead of taking over the mp_ring.
+ */
+#define IFLIB_TXDB_QUEUES 8 /* queues a burst keeps track of */
+
+struct iflib_txdb_burst {
+ LIST_ENTRY(iflib_txdb_burst) itb_link;
+ struct thread *itb_td;
+ int itb_nqueues;
+ iflib_txq_t itb_txq[IFLIB_TXDB_QUEUES];
+};
+
+LIST_HEAD(iflib_txdb_bursts, iflib_txdb_burst);
+DPCPU_DEFINE_STATIC(struct iflib_txdb_bursts, iflib_txdb_bursts);
+
+static void
+iflib_txdb_burst_begin(struct iflib_txdb_burst *b)
+{
+
+ b->itb_td = curthread;
+ b->itb_nqueues = 0;
+ sched_pin();
+ critical_enter();
+ LIST_INSERT_HEAD(DPCPU_PTR(iflib_txdb_bursts), b, itb_link);
+ critical_exit();
+}
+
+/* The receive burst the current thread is passing up, if any. */
+static struct iflib_txdb_burst *
+iflib_txdb_burst(void)
+{
+ struct iflib_txdb_burst *b;
+
+ /* A burst owner is pinned, so it never finds its CPU's list empty. */
+ if (LIST_EMPTY(DPCPU_PTR(iflib_txdb_bursts)))
+ return (NULL);
+ critical_enter();
+ LIST_FOREACH(b, DPCPU_PTR(iflib_txdb_bursts), itb_link) {
+ if (b->itb_td == curthread)
+ break;
+ }
+ critical_exit();
+ return (b);
+}
+
+/*
+ * Called by whoever may ring the doorbell of txq: the mp_ring consumer, or
+ * the holder of ift_mtx on the simple transmit path. Returns true if the
+ * doorbell may be left to the end of burst b, which may be NULL. The first
+ * time a queue holds back, the burst remembers it and counts itself on it.
+ */
+static bool
+iflib_txdb_hold(struct iflib_txdb_burst *b, iflib_txq_t txq)
+{
+ int i;
+
+ if (b == NULL || txq->ift_in_use > txq->ift_size / 8)
+ return (false);
+ for (i = 0; i < b->itb_nqueues; i++) {
+ if (b->itb_txq[i] == txq)
+ return (true);
+ }
+ if (i == nitems(b->itb_txq))
+ return (false);
+ b->itb_txq[b->itb_nqueues++] = txq;
+ atomic_add_32(&txq->ift_db_held, 1);
+ return (true);
+}
+
+static void
+iflib_txdb_burst_end(struct iflib_txdb_burst *b)
+{
+ iflib_txq_t txq;
+ int i;
+
+ critical_enter();
+ LIST_REMOVE(b, itb_link);
+ critical_exit();
+ sched_unpin();
+
+ for (i = 0; i < b->itb_nqueues; i++) {
+ txq = b->itb_txq[i];
+ atomic_subtract_32(&txq->ift_db_held, 1);
+ if (txq->ift_ctx->ifc_sysctl_simple_tx) {
+ mtx_lock(&txq->ift_mtx);
+ if (txq->ift_db_pending != 0 &&
+ (atomic_load_acq_int(&txq->ift_producers) &
+ IFLIB_TXQ_QUIESCING) == 0)
+ (void)iflib_txd_db_check(txq, true);
+ mtx_unlock(&txq->ift_mtx);
+ } else if (txq->ift_db_pending != 0) {
+ /*
+ * Only the mp_ring consumer may ring. Ask the next
+ * drain to ring whatever is pending, and enqueue the
+ * queue itself to become the consumer, as the transmit
+ * task does. A stale count costs at most an empty
+ * drain; if the ring is full, its consumer will see the
+ * request.
+ */
+ atomic_store_16(&txq->ift_db_flush, 1);
+ ifmp_ring_enqueue(txq->ift_br, (void **)&txq, 1,
+ TX_BATCH_SIZE, false);
+ }
+ }
+}
+
static uint8_t
iflib_rxeof(iflib_rxq_t rxq, qidx_t budget)
{
@@ -3282,6 +3415,8 @@
struct if_rxd_info ri;
int err, budget_left, rx_bytes, rx_pkts;
iflib_fl_t fl;
+ struct iflib_txdb_burst txdb;
+ bool txdb_burst;
#if defined(INET6) || defined(INET)
int lro_enabled;
#endif
@@ -3383,6 +3518,9 @@
for (i = 0, fl = &rxq->ifr_fl[0]; i < sctx->isc_nfl; i++, fl++)
retval |= iflib_fl_refill_all(ctx, fl);
+ txdb_burst = rx_pkts != 0 && iflib_tx_db_burst && !iflib_min_tx_latency;
+ if (txdb_burst)
+ iflib_txdb_burst_begin(&txdb);
if (mh != NULL) {
if_input(ifp, mh);
DBG_COUNTER_INC(rx_if_input);
@@ -3397,6 +3535,8 @@
#if defined(INET6) || defined(INET)
tcp_lro_flush_all(&rxq->ifr_lc);
#endif
+ if (txdb_burst)
+ iflib_txdb_burst_end(&txdb);
if (avail != 0 || iflib_rxd_avail(ctx, rxq, *cidxp, 1) != 0)
retval |= IFLIB_RXEOF_MORE;
return (retval);
@@ -4261,6 +4401,7 @@
iflib_txq_t txq = r->cookie;
if_ctx_t ctx = txq->ift_ctx;
if_t ifp = ctx->ifc_ifp;
+ struct iflib_txdb_burst *txdb;
struct mbuf *m, **mp;
int avail, bytes_sent, consumed, count, err, i;
int mcast_sent, pkt_sent, reclaimed;
@@ -4271,7 +4412,12 @@
return (0);
}
reclaimed = iflib_completed_tx_reclaim(txq, NULL);
- rang = iflib_txd_db_check(txq, reclaimed && txq->ift_db_pending);
+ txdb = iflib_txdb_burst();
+ if (iflib_txdb_hold(txdb, txq))
+ rang = false;
+ else
+ rang = iflib_txd_db_check(txq,
+ reclaimed && txq->ift_db_pending);
avail = IDXDIFF(pidx, cidx, r->size);
if (__predict_false(ctx->ifc_flags & IFC_QFLUSH)) {
@@ -4338,12 +4484,20 @@
if (__predict_false(!iflib_is_running(ctx)))
break;
ETHER_BPF_MTAP(ifp, m);
- rang = iflib_txd_db_check(txq, false);
+ if (iflib_txdb_hold(txdb, txq))
+ rang = false;
+ else
+ rang = iflib_txd_db_check(txq, false);
}
/* deliberate use of bitwise or to avoid gratuitous short-circuit */
ring = rang ? false : (iflib_min_tx_latency | err | (!!txq->ift_reclaim_thresh));
- iflib_txd_db_check(txq, ring);
+ if (__predict_false(atomic_load_16(&txq->ift_db_flush) != 0)) {
+ /* A receive burst that sent on this queue has ended. */
+ atomic_store_16(&txq->ift_db_flush, 0);
+ iflib_txd_db_check(txq, true);
+ } else if (!iflib_txdb_hold(txdb, txq))
+ iflib_txd_db_check(txq, ring);
if_inc_counter(ifp, IFCOUNTER_OBYTES, bytes_sent);
if_inc_counter(ifp, IFCOUNTER_OPACKETS, pkt_sent);
if (mcast_sent)
@@ -4429,7 +4583,13 @@
if (if_altq_is_enabled(ifp))
iflib_altq_if_start(ifp);
#endif
- if (txq->ift_db_pending)
+ /*
+ * A doorbell that a receive burst holds is not a laggard: the burst
+ * rings it when it ends. Taking the consumer role here for it would
+ * leave this task, which rings per packet, draining the ring instead
+ * of the burst.
+ */
+ if (txq->ift_db_pending && atomic_load_32(&txq->ift_db_held) == 0)
ifmp_ring_enqueue(txq->ift_br, (void **)&txq, 1, TX_BATCH_SIZE, abdicate);
else if (!abdicate)
ifmp_ring_check_drainage(txq->ift_br, TX_BATCH_SIZE);
@@ -8165,8 +8325,8 @@
/* Consumes the mbuf in all cases */
static int
-iflib_simple_encap(iflib_txq_t txq, struct mbuf *m, int *bytes, int *pkts,
- int *mcasts)
+iflib_simple_encap(iflib_txq_t txq, struct iflib_txdb_burst *txdb,
+ struct mbuf *m, int *bytes, int *pkts, int *mcasts)
{
if_t ifp;
int error;
@@ -8186,14 +8346,15 @@
*mcasts += !!(m->m_flags & M_MCAST);
DBG_COUNTER_INC(tx_sent);
ETHER_BPF_MTAP(ifp, m);
- (void)iflib_txd_db_check(txq, false);
+ if (!iflib_txdb_hold(txdb, txq))
+ (void)iflib_txd_db_check(txq, false);
return (0);
}
/* Drain the deferral ring into the hardware. */
static void
-iflib_simple_drbr_drain(iflib_txq_t txq, u_int quota, int *bytes, int *pkts,
- int *mcasts)
+iflib_simple_drbr_drain(iflib_txq_t txq, struct iflib_txdb_burst *txdb,
+ u_int quota, int *bytes, int *pkts, int *mcasts)
{
if_ctx_t ctx;
struct mbuf *m;
@@ -8217,7 +8378,7 @@
m = drbr_dequeue(ifp, txq->ift_drbr);
if (m == NULL)
return;
- (void)iflib_simple_encap(txq, m, bytes, pkts, mcasts);
+ (void)iflib_simple_encap(txq, txdb, m, bytes, pkts, mcasts);
}
if (!drbr_empty(ifp, txq->ift_drbr)) {
txq->ift_drbr_stall++;
@@ -8252,7 +8413,7 @@
(void)iflib_completed_tx_reclaim(txq, NULL);
if (!drbr_empty(ifp, txq->ift_drbr))
- iflib_simple_drbr_drain(txq, iflib_simple_drain_quota,
+ iflib_simple_drbr_drain(txq, NULL, iflib_simple_drain_quota,
&bytes_sent, &pkt_sent, &mcast_sent);
if (txq->ift_db_pending != 0)
(void)iflib_txd_db_check(txq, true);
@@ -8270,8 +8431,8 @@
* never touched.
*/
static int
-iflib_simple_transmit_locked(iflib_txq_t txq, struct mbuf *m, int *bytes,
- int *pkts, int *mcasts)
+iflib_simple_transmit_locked(iflib_txq_t txq, struct iflib_txdb_burst *txdb,
+ struct mbuf *m, int *bytes, int *pkts, int *mcasts)
{
if_ctx_t ctx;
if_t ifp;
@@ -8284,7 +8445,7 @@
if (__predict_true(!drbr_needs_enqueue(ifp, txq->ift_drbr) &&
TXQ_AVAIL(txq) >= MAX_TX_DESC(ctx))) {
txq->ift_drbr_direct++;
- error = iflib_simple_encap(txq, m, bytes, pkts, mcasts);
+ error = iflib_simple_encap(txq, txdb, m, bytes, pkts, mcasts);
} else {
error = buf_ring_enqueue(txq->ift_drbr, m);
if (__predict_false(error != 0)) {
@@ -8301,8 +8462,8 @@
* are the only thread that can drain it.
*/
if (!drbr_empty(ifp, txq->ift_drbr))
- iflib_simple_drbr_drain(txq, iflib_simple_drain_quota_thread,
- bytes, pkts, mcasts);
+ iflib_simple_drbr_drain(txq, txdb,
+ iflib_simple_drain_quota_thread, bytes, pkts, mcasts);
return (error);
}
@@ -8315,6 +8476,7 @@
iflib_txq_t txq)
{
struct mbuf **m_defer;
+ struct iflib_txdb_burst *txdb;
enum iflib_txq_producer_status producer_status;
bool pinned;
int error, i, reclaimable;
@@ -8363,9 +8525,10 @@
goto net_down;
}
- error = iflib_simple_transmit_locked(txq, m, &bytes_sent, &pkt_sent,
- &mcast_sent);
- if (txq->ift_db_pending != 0)
+ txdb = iflib_txdb_burst();
+ error = iflib_simple_transmit_locked(txq, txdb, m, &bytes_sent,
+ &pkt_sent, &mcast_sent);
+ if (txq->ift_db_pending != 0 && !iflib_txdb_hold(txdb, txq))
(void)iflib_txd_db_check(txq, true);
m_defer = NULL;
reclaimable = iflib_txq_can_reclaim(txq);
@@ -8480,7 +8643,7 @@
mtx_lock(&txq->ift_mtx);
(void)iflib_completed_tx_reclaim(txq, NULL);
- iflib_simple_drbr_drain(txq, UINT_MAX, &bytes_sent, &pkt_sent,
+ iflib_simple_drbr_drain(txq, NULL, UINT_MAX, &bytes_sent, &pkt_sent,
&mcast_sent);
if (txq->ift_db_pending != 0)
(void)iflib_txd_db_check(txq, true);
File Metadata
Details
Attached
Mime Type
text/plain
Expires
Sun, Oct 11, 7:26 AM (8 h, 41 m)
Storage Engine
blob
Storage Format
Raw Data
Storage Handle
40551361
Default Alt Text
D60416.id188894.diff (13 KB)
Attached To
Mode
D60416: iflib: Ring transmit doorbells once per receive batch
Attached
Detach File
Event Timeline
Log In to Comment