From 0fee106139029fb270050e3747c34a8b1aa6510b Mon Sep 17 00:00:00 2001 From: snokvist Date: Tue, 29 Sep 2026 20:37:06 +0200 Subject: [PATCH 01/18] mt7612u bringup: put the retry-limit comment on txs_retry_limit_env txs_parse_long carried two comment blocks back to back; the first one documents txs_retry_limit_env (DEVOURER_TX_RETRY_LIMIT, 1/0/-1 return), which had no comment of its own. Move it there. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01VNC8xhn1rNCi5t6uLvE6M3 --- src/mt7612u/tools/bringup.cpp | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/src/mt7612u/tools/bringup.cpp b/src/mt7612u/tools/bringup.cpp index 346dadf9..9303380e 100644 --- a/src/mt7612u/tools/bringup.cpp +++ b/src/mt7612u/tools/bringup.cpp @@ -1933,11 +1933,6 @@ static int txs_drain(struct mt7612u_dev *d, struct txs_sum *o, return 0; } -/* DEVOURER_TX_RETRY_LIMIT, read with env_config's strictness: the whole - * string one number (base auto-detect), trailing whitespace by isspace() - * exactly as env_long_strict() takes it, clamped to the config's 0..63. - * Returns 1 and sets *out when present and valid, 0 when unset, -1 when - * present but not a number. */ /* The whole string one number (base auto-detect, leading and trailing * whitespace allowed, as strtol and isspace define them) - the rule * env_config's env_long_strict() applies. 0 and *out on success, -1 when no @@ -1987,6 +1982,11 @@ static void txs_print_escaped(const char *s) } } +/* DEVOURER_TX_RETRY_LIMIT, read with env_config's strictness: the whole + * string one number (base auto-detect), trailing whitespace by isspace() + * exactly as env_long_strict() takes it, clamped to the config's 0..63. + * Returns 1 and sets *out when present and valid, 0 when unset, -1 when + * present but not a number. */ static int txs_retry_limit_env(int *out) { const char *e = getenv("DEVOURER_TX_RETRY_LIMIT"); From 4d862ff035d90a9c47f48cbc6ed6bb015852187e Mon Sep 17 00:00:00 2001 From: snokvist Date: Tue, 29 Sep 2026 20:37:32 +0200 Subject: [PATCH 02/18] txdemo: retry StopBeacon when an armed beacon's disable is refused After a successful StartBeacon, the Jaguar2/3 StopBeacon returns false only when the disable itself was refused - the "nothing active" early return cannot fire, and _bcn_hw_touched makes a retry re-run the disable. TxBeaconGuard treated every false as done. Retry on false while armed; a clean false after a refused StartBeacon still ends the loop. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01VNC8xhn1rNCi5t6uLvE6M3 --- examples/tx/main.cpp | 18 ++++++++++-------- 1 file changed, 10 insertions(+), 8 deletions(-) diff --git a/examples/tx/main.cpp b/examples/tx/main.cpp index 33112639..9b30ebf3 100644 --- a/examples/tx/main.cpp +++ b/examples/tx/main.cpp @@ -1946,9 +1946,10 @@ int main(int argc, char **argv) { * explicitly, before Stop() powers the chip down; the destructor covers an * exception or an early return. It is declared after the DeviceSession, so * it runs before the device is destroyed. `attempted` is cleared only by a - * StopBeacon that returned (true, or a clean false = nothing active, per - * its contract); after three throws it stays set, so a later stop() - the - * destructor on an exception path - tries again. detach() drops the device + * StopBeacon that returned true, or a clean false when nothing was armed; + * after three failed attempts (throws, or a refused disable of an armed + * beacon) it stays set, so a later stop() - the destructor on an exception + * path - tries again. detach() drops the device * before the normal path destroys it. */ struct TxBeaconGuard { IRadio *dev; @@ -1960,17 +1961,18 @@ int main(int argc, char **argv) { return; for (int i = 0; i < 3; i++) { try { - /* true = stopped; a clean false = nothing active (StopBeacon - * contract), expected after a refused StartBeacon - either way - * there is nothing to retry. */ - dev->StopBeacon(); + /* After a successful arm, false means the disable was refused + * (Jaguar2/3), so retry. Otherwise a clean false is nothing + * active - expected after a refused StartBeacon. */ + if (!dev->StopBeacon() && armed) + continue; attempted = false; return; } catch (const std::exception &e) { log->warn("DEVOURER_TX_BEACON_TU: StopBeacon threw: {}", e.what()); } } - /* Three throws: `attempted` stays set so a later stop() tries again. */ + /* Three failures: `attempted` stays set so a later stop() tries again. */ log->error("DEVOURER_TX_BEACON_TU: StopBeacon failed 3 times - the " "beacon may keep airing until the adapter is re-enumerated " "or powered down (Jaguar2 has no teardown power-down)"); From 294a67196a186e15c110778aed2195314d336ebe Mon Sep 17 00:00:00 2001 From: snokvist Date: Sat, 3 Oct 2026 10:52:35 +0200 Subject: [PATCH 03/18] jaguar3 TX-DMA status: bit 13 "can latch" - in the doc and the header docs/jaguar3-tx-ring.md: the dated, attributed bench paragraph becomes plain measurement inside the 8812CU paragraph it qualifies. - On USB2 at DEVOURER_TX_GAP_US=0, bit 13 (BIT_PAYLOAD_OVF_8822C, 0x2000) latched while 8051/8051 frames completed. - It read 0 at the default gap. - One later USB2 gap-0 run did not reproduce the latch, so the bit can latch at max duty on USB2. - A second bench reproduced the defect and the fix clearing it. Git carries the provenance. src/IRtlRadio.h, the GetTxDmaStatus contract, said bit 13 "latches" under host-side max-duty backpressure. It now says "can latch", noting that a later such run did not latch it. docs/logging.md defers to that declaration. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01VNC8xhn1rNCi5t6uLvE6M3 --- docs/jaguar3-tx-ring.md | 21 ++++++++++----------- src/IRtlRadio.h | 9 +++++---- 2 files changed, 15 insertions(+), 15 deletions(-) diff --git a/docs/jaguar3-tx-ring.md b/docs/jaguar3-tx-ring.md index 14a90d40..2e7a60a5 100644 --- a/docs/jaguar3-tx-ring.md +++ b/docs/jaguar3-tx-ring.md @@ -343,7 +343,16 @@ Available with this PR: The verdict is the send failures, not `txdma_status`: on the faulting runs the periodic `tx.stats` still read `txdma_status` 0 up to its last sample (it is taken once per 500 frames, so a latch just before the stop would - not show). One adapter, one channel. On the 8822B (8812BU) this txdemo + not show). Nor is a nonzero `txdma_status` by itself the wedge: an 8812CU + on USB2 at `DEVOURER_TX_GAP_US=0` with 1400-byte QoS data read `0x2000` + (bit 13, `BIT_PAYLOAD_OVF_8822C`) from the first sample, on the fixed and + the control build alike, while TX completed 8051/8051; at the default 2 ms + gap it read 0, and one later USB2 run at gap 0 did not reproduce the + latch. So bit 13 can latch at max duty on USB2; bit 18 + (`BIT_TXPKTBUF_REQ_ERR`) is the bit measured with the wedge + (`IRtlRadio::GetTxDmaStatus`). One adapter, one channel; a second bench + reproduced the defect and the fix clearing it on an 8812CU, and gave the + same 8812BU LLT result as below. On the 8822B (8812BU) this txdemo form does NOT reproduce - unfixed and fixed alike ran clean (item 3); the Jaguar2 verification is the LLT check below. The 8822E form is unmeasured (the `ap_wpa2` stress, station-mode PR, is its record). Those runs used a radiotap-prefixed beacon; the demo now passes @@ -355,16 +364,6 @@ Available with this PR: submitted / 0 failed with the beacon armed and then stopped, the aggregated path 0 failed, and A-MPDU over QoS data 0 failed. - **The maintainer's bench (josephnef), 2026-09-27:** the reproducer - reproduces and the fix clears it on an 8812CU, and the 8812BU LLT check - gives the same result as above. And a counterpart for `txdma_status`: an - 8812CU on USB2 at `DEVOURER_TX_GAP_US=0` with 1400-byte QoS data read - `0x2000` (bit 13, `BIT_PAYLOAD_OVF_8822C`) from the first sample, on the - fixed and the control build alike, while TX completed 8051/8051; at the - default 2 ms gap it read 0. So a nonzero `txdma_status` is not by itself - the wedge - bit 18 (`BIT_TXPKTBUF_REQ_ERR`) is the bit measured with it - (`IRtlRadio::GetTxDmaStatus`). - **The canonical-frame form does not reproduce**: without `DEVOURER_TX_QOS_DATA`/`DEVOURER_TX_PAYLOAD_BYTES`/`DEVOURER_TX_WITH_RX`, the unfixed build ran 20051 submitted / 0 failed / `txdma_status` 0, the diff --git a/src/IRtlRadio.h b/src/IRtlRadio.h index 7e6498b6..90ba4ee9 100644 --- a/src/IRtlRadio.h +++ b/src/IRtlRadio.h @@ -167,10 +167,11 @@ class IRtlRadio : public IRadio { * armed it latched and the part transmitted nothing more for the life * of the process, while the receiver worked on (the 8812BU's wedge * read 0x10, then 0x15 - bits not decoded); - * - bit 13, BIT_PAYLOAD_OVF_8822C (0x00002000), latches under host-side - * max-duty backpressure while TX continues - an 8812CU on USB2 at - * DEVOURER_TX_GAP_US=0 read it from the first sample and still - * completed every frame. A poller must not treat it as the wedge. + * - bit 13, BIT_PAYLOAD_OVF_8822C (0x00002000), can latch under + * host-side max-duty backpressure while TX continues - an 8812CU on + * USB2 at DEVOURER_TX_GAP_US=0 read it from the first sample and still + * completed every frame (a later such run did not latch it). A poller + * must not treat it as the wedge. * Other bits are undecoded. Records: docs/jaguar3-tx-ring.md. * * NOT FOR THE SEND PATH. This is a register read over USB - see the From ddd84677681044fffffc57ac15e5e83771569740 Mon Sep 17 00:00:00 2001 From: snokvist Date: Tue, 29 Sep 2026 20:39:44 +0200 Subject: [PATCH 04/18] mt7612u bringup: stop the MAC when mt_mac_start fails mt_mac_start() writes MT_MAC_SYS_CTRL_ENABLE_TX before its WPDMA-idle poll and returns -1 without clearing it, so a gate that returned straight out of a failed start could leave TX enabled. Every such failure path now calls mt_mac_stop() before returning, after the gate's existing cleanup and in its normal teardown order: rx_teardown() first where a ring was draining EP 4 (gate_ap's bare rx_stop becomes rx_teardown), mt_mac_rx_disable() first in gate_rx as its normal exit does. The failed start never reaches mt_async_start() in gate_txs, and gate_caps' async drainer is torn down by rx_teardown() as on its normal exit. gate_tsfwrite already did this. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01VNC8xhn1rNCi5t6uLvE6M3 --- src/mt7612u/tools/bringup.cpp | 65 +++++++++++++++++++++++------------ 1 file changed, 43 insertions(+), 22 deletions(-) diff --git a/src/mt7612u/tools/bringup.cpp b/src/mt7612u/tools/bringup.cpp index 9303380e..eb13e533 100644 --- a/src/mt7612u/tools/bringup.cpp +++ b/src/mt7612u/tools/bringup.cpp @@ -620,7 +620,7 @@ static int gate_beacon(uint8_t chan, int secs) } /* TX-only: beaconing never reads EP 4. */ if (mt_mac_start(&dev, MT_RX_DRAIN_NONE)) { - printf("GATE A: FAIL - mac_start\n"); return 1; + printf("GATE A: FAIL - mac_start\n"); mt_mac_stop(&dev); return 1; } mt_beacon_init(&dev); @@ -792,7 +792,8 @@ static int gate_ap(uint8_t chan, int secs) } rx_up = 1; if (mt_mac_start(&dev, MT_RX_DRAIN_RING)) { - printf("GATE B: FAIL - mac_start\n"); mt7612u_rx_stop(&dev); return 1; + printf("GATE B: FAIL - mac_start\n"); + rx_teardown(); mt_mac_stop(&dev); return 1; } /* * AP receive filter. The managed default mt_mac_start() just wrote already @@ -957,7 +958,7 @@ static int gate_tx(uint8_t chan, int count, int phy, int mcs) } /* TX only: this gate never reads EP 4, so do not switch the receiver on. */ if (mt_mac_start(&dev, MT_RX_DRAIN_NONE)) { - printf("GATE E: FAIL - mac_start failed\n"); return 1; + printf("GATE E: FAIL - mac_start failed\n"); mt_mac_stop(&dev); return 1; } printf("MAC started: MT_MAC_SYS_CTRL=0x%08x (bit2 TX, bit3 RX)\n", mt_rr(&dev, MT_MAC_SYS_CTRL)); @@ -1014,7 +1015,8 @@ static int gate_rx(uint8_t chan, int want) printf("GATE F: FAIL - set_channel failed\n"); return 1; } if (mt_mac_start(&dev, MT_RX_DRAIN_SYNC)) { - printf("GATE F: FAIL - mac_start failed\n"); return 1; + printf("GATE F: FAIL - mac_start failed\n"); + mt_mac_rx_disable(&dev); mt_mac_stop(&dev); return 1; } /* Monitor: drop only CRC and PHY errors, accept everything else. The * initvals value 0x15f97 drops a great deal more than that. */ @@ -1140,7 +1142,7 @@ static int gate_g(uint8_t chan, int count) if (mt_eeprom_init(&dev)) return 1; if (mt_init_hardware(&dev, NULL)) return 1; if (mt_set_channel(&dev, chan, MT7612U_BW_20)) return 1; - if (mt_mac_start(&dev, MT_RX_DRAIN_NONE)) return 1; + if (mt_mac_start(&dev, MT_RX_DRAIN_NONE)) { mt_mac_stop(&dev); return 1; } memset(frame, 0, sizeof frame); frame[0] = 0x08; @@ -1251,7 +1253,7 @@ static int gate_mtu(uint8_t chan, int count) if (mt_eeprom_init(&dev)) return 1; if (mt_init_hardware(&dev, NULL)) return 1; if (mt_set_channel(&dev, chan, MT7612U_BW_20)) return 1; - if (mt_mac_start(&dev, MT_RX_DRAIN_NONE)) return 1; + if (mt_mac_start(&dev, MT_RX_DRAIN_NONE)) { mt_mac_stop(&dev); return 1; } memset(frame, 0, sizeof frame); frame[0] = 0x08; /* data, 3-address */ @@ -1339,7 +1341,7 @@ static int gate_soak(uint8_t chan, int secs, int framelen) if (mt_eeprom_init(&dev)) return 1; if (mt_init_hardware(&dev, NULL)) return 1; if (mt_set_channel(&dev, chan, MT7612U_BW_20)) return 1; - if (mt_mac_start(&dev, MT_RX_DRAIN_NONE)) return 1; + if (mt_mac_start(&dev, MT_RX_DRAIN_NONE)) { mt_mac_stop(&dev); return 1; } memset(frame, 0, sizeof frame); frame[0] = 0x08; @@ -1458,7 +1460,9 @@ static int gate_arx(uint8_t chan, int secs, int notick) if (mt7612u_rx_start(&dev, arx_cb, &ctx)) { printf("GATE arx: FAIL - rx_start failed\n"); return 1; } - if (mt_mac_start(&dev, MT_RX_DRAIN_RING)) { rx_teardown(); return 1; } + if (mt_mac_start(&dev, MT_RX_DRAIN_RING)) { + rx_teardown(); mt_mac_stop(&dev); return 1; + } mt7612u_set_monitor_rx(&dev, 0); t0 = now_ms(); /* notick is the negative control: without the 1 Hz PHY tick this gate @@ -1525,7 +1529,9 @@ static int gate_duplex(uint8_t chan, int secs) memcpy(frame + 24, "MT7612U-HAL ", 12); if (mt7612u_rx_start(&dev, arx_cb, &ctx)) return 1; - if (mt_mac_start(&dev, MT_RX_DRAIN_RING)) { rx_teardown(); return 1; } + if (mt_mac_start(&dev, MT_RX_DRAIN_RING)) { + rx_teardown(); mt_mac_stop(&dev); return 1; + } mt7612u_set_monitor_rx(&dev, 0); t0 = now_ms(); @@ -1676,7 +1682,7 @@ static int gate_ampdu(uint8_t chan, int count) if (mt_eeprom_init(&dev)) return 1; if (mt_init_hardware(&dev, NULL)) return 1; if (mt_set_channel(&dev, chan, MT7612U_BW_20)) return 1; - if (mt_mac_start(&dev, MT_RX_DRAIN_NONE)) return 1; + if (mt_mac_start(&dev, MT_RX_DRAIN_NONE)) { mt_mac_stop(&dev); return 1; } /* A real station-table entry: aggregation is a per-peer notion, and * wcid 0xff (what the injector normally uses) names no peer. */ @@ -2120,7 +2126,10 @@ static int gate_txs(uint8_t chan, int frames, const char *peer_str) ctr.acks.store(0); ctr.frames.store(0); - if (mt_mac_start(&dev, MT_RX_DRAIN_NONE)) return 1; + if (mt_mac_start(&dev, MT_RX_DRAIN_NONE)) { + mt_mac_stop(&dev); + return 1; + } if (mt_async_start(&dev, rx_on ? ucast_rx_cb : NULL, rx_on ? (void *)&ctr : NULL)) { mt_mac_stop(&dev); @@ -2365,7 +2374,9 @@ static int gate_caps(uint8_t chan) * with nothing reading, that is long enough to wedge the part below * the USB level, which no software reset recovers. */ if (mt_async_start(&dev, drain_cb, &drained)) return 1; - if (mt_mac_start(&dev, MT_RX_DRAIN_RING)) { rx_teardown(); return 1; } + if (mt_mac_start(&dev, MT_RX_DRAIN_RING)) { + rx_teardown(); mt_mac_stop(&dev); return 1; + } mt7612u_get_caps(&dev, &c); printf("caps: %s rev 0x%08x %dTx%dRx bw_mask 0x%02x (20%s%s)\n", @@ -2537,7 +2548,9 @@ static int gate_ack(uint8_t chan, int secs, int arm) /* Ring first, receiver second - see gate_caps. Arming the responder and * printing between the two would otherwise leave RX on and undrained. */ if (mt7612u_rx_start(&dev, ack_cb, &off)) return 1; - if (mt_mac_start(&dev, MT_RX_DRAIN_RING)) { rx_teardown(); return 1; } + if (mt_mac_start(&dev, MT_RX_DRAIN_RING)) { + rx_teardown(); mt_mac_stop(&dev); return 1; + } /* CRC and PHY errors only: DUP must stay clear so retries reach us. */ mt_wr(&dev, MT_RX_FILTR_CFG, MT_RX_FILTR_CFG_CRC_ERR | MT_RX_FILTR_CFG_PHY_ERR); @@ -2661,7 +2674,9 @@ static int gate_rxbytes(uint8_t chan, int secs) if (mt_init_hardware(&dev, NULL)) return 1; if (mt_set_channel(&dev, chan, MT7612U_BW_20)) return 1; if (mt7612u_rx_start(&dev, rxbytes_cb, NULL)) return 1; - if (mt_mac_start(&dev, MT_RX_DRAIN_RING)) { rx_teardown(); return 1; } + if (mt_mac_start(&dev, MT_RX_DRAIN_RING)) { + rx_teardown(); mt_mac_stop(&dev); return 1; + } mt7612u_set_monitor_rx(&dev, 0); mt7612u_link_stats_start(&dev); @@ -2741,7 +2756,11 @@ static int gate_linkstat(uint8_t chan, int secs, int with_rx) if (with_rx) { if (mt7612u_rx_start(&dev, drain_cb, &linkstat_drained)) return 1; } - if (mt_mac_start(&dev, with_rx)) { if (with_rx) rx_teardown(); return 1; } + if (mt_mac_start(&dev, with_rx)) { + if (with_rx) rx_teardown(); + mt_mac_stop(&dev); + return 1; + } if (with_rx) mt7612u_set_monitor_rx(&dev, 0); mt7612u_link_stats_start(&dev); @@ -2815,7 +2834,7 @@ static int gate_linktx(uint8_t chan, int count) if (mt_eeprom_init(&dev)) return 1; if (mt_init_hardware(&dev, NULL)) return 1; if (mt_set_channel(&dev, chan, MT7612U_BW_20)) return 1; - if (mt_mac_start(&dev, MT_RX_DRAIN_NONE)) return 1; + if (mt_mac_start(&dev, MT_RX_DRAIN_NONE)) { mt_mac_stop(&dev); return 1; } memset(f, 0, sizeof f); f[0] = 0x08; @@ -2904,7 +2923,9 @@ static int gate_linkrx(uint8_t chan, int secs) if (mt_init_hardware(&dev, NULL)) return 1; if (mt_set_channel(&dev, chan, MT7612U_BW_20)) return 1; if (mt7612u_rx_start(&dev, linkrx_cb, NULL)) return 1; - if (mt_mac_start(&dev, MT_RX_DRAIN_RING)) { rx_teardown(); return 1; } + if (mt_mac_start(&dev, MT_RX_DRAIN_RING)) { + rx_teardown(); mt_mac_stop(&dev); return 1; + } mt7612u_set_monitor_rx(&dev, 0); printf("RX on ch%u for %d s, filtering our own magic\n", chan, secs); @@ -2978,7 +2999,7 @@ static int gate_diversity(uint8_t chan, int count) if (mt_eeprom_init(&dev)) return 1; if (mt_init_hardware(&dev, NULL)) return 1; if (mt_set_channel(&dev, chan, MT7612U_BW_20)) return 1; - if (mt_mac_start(&dev, MT_RX_DRAIN_NONE)) return 1; + if (mt_mac_start(&dev, MT_RX_DRAIN_NONE)) { mt_mac_stop(&dev); return 1; } memset(f, 0, sizeof f); f[0] = 0x08; @@ -3082,7 +3103,7 @@ static int gate_coding(uint8_t chan, int count, int bw) if (mt_eeprom_init(&dev)) return 1; if (mt_init_hardware(&dev, NULL)) return 1; if (mt_set_channel(&dev, chan, (enum mt7612u_bw)bw)) return 1; - if (mt_mac_start(&dev, MT_RX_DRAIN_NONE)) return 1; + if (mt_mac_start(&dev, MT_RX_DRAIN_NONE)) { mt_mac_stop(&dev); return 1; } mt_chan_group(chan, (uint8_t)bw, &hw_chan, NULL, NULL); printf("ch%u (hw centre %u) at %d MHz, %d frames per arm\n\n", @@ -3189,7 +3210,7 @@ static int gate_sweep(uint8_t chan, int count, int bw) if (mt_eeprom_init(&dev)) return 1; if (mt_init_hardware(&dev, NULL)) return 1; if (mt_set_channel(&dev, chan, (enum mt7612u_bw)bw)) return 1; - if (mt_mac_start(&dev, MT_RX_DRAIN_NONE)) return 1; + if (mt_mac_start(&dev, MT_RX_DRAIN_NONE)) { mt_mac_stop(&dev); return 1; } /* Report the centre the hardware actually tuned, not the control * channel: at 80 MHz they differ by up to 6, and a witness listening on @@ -3318,7 +3339,7 @@ static int gate_vht(uint8_t chan, int count, int bw) if (mt_eeprom_init(&dev)) return 1; if (mt_init_hardware(&dev, NULL)) return 1; if (mt_set_channel(&dev, chan, (enum mt7612u_bw)bw)) return 1; - if (mt_mac_start(&dev, MT_RX_DRAIN_NONE)) return 1; + if (mt_mac_start(&dev, MT_RX_DRAIN_NONE)) { mt_mac_stop(&dev); return 1; } mt_chan_group(chan, (uint8_t)bw, &hw_chan, NULL, NULL); printf("chainmask 0x%04x -> %d spatial streams, txwi[17]=0x%02x\n", @@ -3400,7 +3421,7 @@ static int gate_rtap(uint8_t chan, int count) if (mt_eeprom_init(&dev)) return 1; if (mt_init_hardware(&dev, NULL)) return 1; if (mt_set_channel(&dev, chan, MT7612U_BW_20)) return 1; - if (mt_mac_start(&dev, MT_RX_DRAIN_NONE)) return 1; + if (mt_mac_start(&dev, MT_RX_DRAIN_NONE)) { mt_mac_stop(&dev); return 1; } memcpy(pkt, rtap, sizeof rtap); { From 2e7a3def3fab96a2fe51444980a281efa46bb103 Mon Sep 17 00:00:00 2001 From: snokvist Date: Sat, 3 Oct 2026 10:52:35 +0200 Subject: [PATCH 05/18] mt7612u bringup: txs - claim the arm's first entry back from a stale EXT read Refs #461: on a second MT7612U the ~6 fps txs arms read 199/200 (59/60) while the fast arms settle 200/200, which leaves the uplink harness INCONCLUSIVE. The slow rate is a symptom, not the cause: those are the arms that lost their first entry, after which every per-frame wait (entries >= n) times out. The status read is EXT, then the main word: two USB transfers. When the FIFO is empty at the EXT read and an entry is filed before the main read, the pop is paired with the stale EXT word of the previously popped entry. - Inside an arm that is the same pktid, and harmless. - On an arm's first entry it is the previous arm's pktid, so the entry was counted late (foreign on the session's first arm). That is the "one-step status lag" of docs/mt7612u-tx-retry.md. Its own table rules out the alternative, status posted only on the next TX: at limit 0, receiver ON, arm f lags with L1 after arm e settled 40/40 and owed nothing, and arm g after it shows no late entry. The late entry belongs to the lagging arm, not the one after it. A race on the poll timing also explains why it varies from pass to pass and from host to host. txs_drain now claims that one entry back when it can only be ours, which takes all of the following: - it is the arm's first popped entry, after the arm has submitted a frame; - it carries the pktid of the last arm that SENT a frame, and that arm settled with no entry owed; - or, on the session's first arm, it carries any pktid. An arm whose every submit failed popped nothing and leaves that reference untouched. The claimed entry counts toward entries, and toward success when its SUCCESS bit is set (the main word is fresh). It stays out of the retry columns (its retry count is the stale word's), and is reported as "stale-EXT entries claimed". After an UNSETTLED arm, a previous-pktid entry is still reported as late. The claim recovers at most one entry per arm. A larger deficit, like the uplink harness's multi-entry loss, is not this race. Hardware: in the form keyed to the previous arm, the author's unit settled 16/16 arms at limits 5 and 0 (issue #461). Keying it to the last arm that sent has not been run on hardware. No headless test: txs_drain reads the FIFO through mt_rr_chk on a live device inside the bring-up tool, which no selftest links. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01VNC8xhn1rNCi5t6uLvE6M3 --- docs/mt7612u-tx-retry.md | 47 ++++++++++++------- src/mt7612u/tools/bringup.cpp | 87 ++++++++++++++++++++++++++++------- 2 files changed, 102 insertions(+), 32 deletions(-) diff --git a/docs/mt7612u-tx-retry.md b/docs/mt7612u-tx-retry.md index 68735db3..a5a8b67d 100644 --- a/docs/mt7612u-tx-retry.md +++ b/docs/mt7612u-tx-retry.md @@ -158,19 +158,32 @@ What it shows: fps, the same per-frame cost as the No-Ack rows. At limit 5 and on the initvals they time out on every per-frame wait (T40), and three of the eight such rows are 39/40 with the lag below. -- **Characterised, unexplained: a one-step status lag.** In some arms EVERY - per-frame wait times out (T40), yet the entries do arrive - one step - behind: a frame's status becomes visible only after the next frame is - submitted. Such an arm ends 39/40 and its last entry lands in the next arm - as late (or, for the very first arm, as one foreign entry). Arm a lags in - every pass; which other arms lag varies from pass to pass (c, e, f and g - are each clean in some passes and lagging in others), so it tracks chip - state, not arm configuration. Counting 39/40 rows, it hit five of sixteen - at limit 0, eleven at limit 5 and seven on the initvals - one run each, - too few to call a trend. The two candidate explanations - (status posted only on the next TX; the EXT/FIFO pairing off by one) - produce identical signatures in this gate and are not distinguishable - here. The retry and success columns exclude every late and foreign entry. +- **A one-step status lag - a stale EXT read on the arm's first entry.** In + some arms EVERY per-frame wait times out (T40) and the arm ends 39/40, with + one late entry (or, for the very first arm, one foreign entry) in that + SAME arm's row. Counting 39/40 rows, it hit five of sixteen at limit 0, + eleven at limit 5 and seven on the initvals - one run each. Arm a lags in + every pass; which other arms lag varies from pass to pass. The table rules + out "status posted only on the next TX": at limit 0, receiver ON, arm f + lags with L1 although arm e before it settled 40/40 and owed nothing, and + arm g after it shows no late entry. What fits is the two-transfer read: + when the FIFO is empty at the EXT read and an entry is filed before the + main read, the popped entry is paired with the stale EXT word of the + previous entry. On an arm's first entry that is the previous arm's pktid, + so the entry was counted late, the arm stayed one short, and every + per-frame wait timed out. A race on the poll timing, which is why it + varies from pass to pass and with the host. The gate now claims that + entry back (`txs_drain`), under all of: it is the arm's first popped entry + (no own entry yet), the arm has submitted a frame, and its pktid is that + of the last arm that sent a frame, which must have settled with no entry + owed - or, on the session's first arm, any pktid. A claimed entry counts + in entries, and in success when its SUCCESS bit is set; it stays out of + the retry columns, and is reported as "stale-EXT entries claimed". These + tables were taken before that change. With the claim keyed to the + previous arm, the author's unit then settled 16/16 arms at limits 5 and 0 + (recorded on issue #461); keying it to the last arm that sent a frame, + so an arm with every submit failed is skipped, has not been run on + hardware. - **Arms e-h**: e, f and g read like c whenever they are clean; nothing distinguishes them. h (broadcast, WCID 1) lagged in five of six passes. @@ -244,9 +257,11 @@ why it is worth an issue of its own. tables; the delivery table above shows ACK-requesting retries working against a real AP. - UNSETTLED rows are read for the retry value, never for the counts. -- The one-step status lag is unexplained, and its two candidate causes are - indistinguishable in this gate. It moves an arm's last entry into the next - arm's late count, never into another arm's statistics. +- The one-step status lag is attributed to the stale-EXT race above from + this table's pattern, not from a bus trace. The claim recovers at most one + entry per arm - the first - and only on the conditions above; after an + UNSETTLED arm a previous-pktid entry is still reported as late. A deficit + of more than one entry is not this race. - `mt7612uprobe txs` reads two registers per status poll, so its `fps` is per-frame submit-to-status time, not comparable with any steady-state injection figure. diff --git a/src/mt7612u/tools/bringup.cpp b/src/mt7612u/tools/bringup.cpp index eb13e533..26559f00 100644 --- a/src/mt7612u/tools/bringup.cpp +++ b/src/mt7612u/tools/bringup.cpp @@ -1873,7 +1873,7 @@ static int parse_mac6(const char *s, uint8_t out[6]) * columns, making a No-Ack arm look as if it retried to exactly the * configured limit and displacing one of its own entries from entr/sent. * Entries carrying the previous arm's pktid are reported as late; any other - * pktid as foreign. + * pktid as foreign - with one exception, the stale EXT word (txs_drain). * * Exit: 0 reported, 1 device failure OR no status entry filed at all (the * measurement did not happen), 2 bad argument or retry limit refused, @@ -1883,6 +1883,8 @@ struct txs_sum { long entries, success, retry_total, retry_max; long late_prev; /* entries carrying the PREVIOUS arm's pktid */ long foreign; /* entries with any other pktid */ + long stale_ext; /* own entries popped with a stale EXT word: counted in + * entries/success, kept out of the retry columns */ }; /* mt76's skb pktid range starts at MT_PACKET_ID_FIRST (3) and the id must @@ -1893,14 +1895,35 @@ static unsigned txs_arm_pktid(int rx_on, unsigned arm) return 3u + 8u * (unsigned)rx_on + arm; } #define TXS_NO_PKTID 0x100u /* matches no 8-bit EXT_PKTID */ +#define TXS_ANY_PKTID 0x200u /* txs_drain's stale_id: unknown, so any */ /* Slack on top of frame_budget_ms for one frame's status wait: USB submit * latency plus the drain's two control reads. */ #define TXS_FRAME_MARGIN_MS 50.0 /* Returns 0 when the FIFO was drained (or is empty), -1 when a status read - * failed - the caller must not report the arm as measured then. */ + * failed - the caller must not report the arm as measured then. + * + * The EXT-then-main read is two USB transfers, not one atomic read. When the + * FIFO is EMPTY at the EXT read and an entry is filed before the main read, + * the main read pops that entry but the EXT word read just before it is + * stale - it still describes the last entry popped. Within an arm that is + * harmless (same pktid). On an arm's FIRST entry it is the previous arm's + * pktid (or, on the session's first arm, whatever EXT held), so the entry was + * counted late/foreign, the arm stayed one short for good, and every + * per-frame wait then timed out - the "one-step status lag" rows of + * docs/mt7612u-tx-retry.md (N-1/N, a timeout on every frame, ~5-6 fps, and + * the late entry in the SAME arm's row, after a fully settled arm). + * + * `stale_id` takes that one entry back. It is the pktid a stale EXT word + * would carry: the last arm that SENT a frame (an arm that sent none popped + * nothing, so EXT still describes the arm before it), or TXS_ANY_PKTID on the + * session's first. The caller passes it only once this arm has submitted a + * frame and that arm owed no entries, and TXS_NO_PKTID otherwise (no + * claim). The arm's first popped entry, carrying stale_id, can then only be + * ours. Its main word is fresh, so its SUCCESS bit counts; its retry count is + * the stale word's, so it is kept out of the retry columns. */ static int txs_drain(struct mt7612u_dev *d, struct txs_sum *o, - unsigned want, unsigned prev) + unsigned want, unsigned prev, unsigned stale_id) { int guard; @@ -1925,6 +1948,14 @@ static int txs_drain(struct mt7612u_dev *d, struct txs_sum *o, (unsigned)FIELD_GET(MT_TX_STAT_FIFO_EXT_PKTID, ext); if (id != want) { + if (stale_id != TXS_NO_PKTID && o->entries == 0 && + (stale_id == TXS_ANY_PKTID || id == stale_id)) { + o->entries++; + o->stale_ext++; + if (st & MT_TX_STAT_FIFO_SUCCESS) + o->success++; + continue; + } if (id == prev) o->late_prev++; else o->foreign++; continue; @@ -2015,6 +2046,10 @@ static int gate_txs(uint8_t chan, int frames, const char *peer_str) int rx_on, rl = 0, rl_given; long total_entries = 0, total_foreign = 0; unsigned prev_pktid = TXS_NO_PKTID; + /* txs_drain's stale_id: the last arm that sent a frame, and whether it + * settled with no entry owed. */ + unsigned stale_pktid = TXS_ANY_PKTID; + int stale_settled = 1; double frame_budget_ms; int io_fail = 0; /* status read / WCID setup failed: teardown, exit 1 */ double last_tick = 0.0; /* receiver passes: last mt7612u_phy_tick() */ @@ -2164,7 +2199,7 @@ static int gate_txs(uint8_t chan, int frames, const char *peer_str) for (a = 0; !io_fail && a < sizeof arms / sizeof arms[0]; a++) { struct mt7612u_tx_rate rate = { }; - struct txs_sum sum = { 0, 0, 0, 0, 0, 0 }; + struct txs_sum sum = { 0, 0, 0, 0, 0, 0, 0 }; const unsigned pktid = txs_arm_pktid(rx_on, a); const uint8_t *sa = arms[a].own_sa ? dev.macaddr : src; const uint8_t *a1 = arms[a].bcast_a1 ? bcast : peer; @@ -2188,7 +2223,8 @@ static int gate_txs(uint8_t chan, int frames, const char *peer_str) memcpy(frame + 26, "MT7612U-TXS", 11); /* The previous arm's tail: counted as late, never as ours. */ - if (txs_drain(&dev, &sum, pktid, prev_pktid)) io_fail = 1; + if (txs_drain(&dev, &sum, pktid, prev_pktid, TXS_NO_PKTID)) + io_fail = 1; /* Bounded twice, like gate_ampdu's wall clock: a submit * that keeps failing must end the arm, not spin it. The @@ -2207,7 +2243,9 @@ static int gate_txs(uint8_t chan, int frames, const char *peer_str) MT_TXOPT_TXS | MT_TXOPT_PKTID(pktid) | arms[a].opts) != 0) { submit_fail++; - if (txs_drain(&dev, &sum, pktid, prev_pktid)) + if (txs_drain(&dev, &sum, pktid, prev_pktid, + n > 0 && stale_settled ? + stale_pktid : TXS_NO_PKTID)) io_fail = 1; continue; } @@ -2233,7 +2271,9 @@ static int gate_txs(uint8_t chan, int frames, const char *peer_str) do { if (txs_drain(&dev, &sum, pktid, - prev_pktid)) { + prev_pktid, + stale_settled ? stale_pktid + : TXS_NO_PKTID)) { io_fail = 1; break; } @@ -2265,7 +2305,9 @@ static int gate_txs(uint8_t chan, int frames, const char *peer_str) && !g_stop && !io_fail) { txs_tick(rx_on, &last_tick); mt_usleep(2000); - if (txs_drain(&dev, &sum, pktid, prev_pktid)) + if (txs_drain(&dev, &sum, pktid, prev_pktid, + n > 0 && stale_settled ? + stale_pktid : TXS_NO_PKTID)) io_fail = 1; } settled = (sum.entries >= n); @@ -2282,22 +2324,35 @@ static int gate_txs(uint8_t chan, int frames, const char *peer_str) * entries, uncapped: more than `sent` would mean the MAC * filed duplicates, and is shown as such rather than * clipped. */ - printf(" %c %-28s %7.0f %4ld/%-4ld %8ld %9.1f %6ld%s\n", - arms[a].tag, arms[a].what, n * 1000.0 / wall, - sum.entries, n, sum.success, - sum.entries ? (double)sum.retry_total / sum.entries : 0.0, - sum.retry_max, settled ? "" : " UNSETTLED"); + { + /* The retry mean is over the entries whose EXT + * word is their own. */ + const long rtry_n = sum.entries - sum.stale_ext; + + printf(" %c %-28s %7.0f %4ld/%-4ld %8ld %9.1f %6ld%s\n", + arms[a].tag, arms[a].what, n * 1000.0 / wall, + sum.entries, n, sum.success, + rtry_n ? (double)sum.retry_total / rtry_n : 0.0, + sum.retry_max, settled ? "" : " UNSETTLED"); + } if (submit_fail || status_timeouts || sum.late_prev || - sum.foreign || n < frames) + sum.foreign || sum.stale_ext || n < frames) printf(" (pktid %u: %ld submit failures, %ld/%d " "frames sent, %ld per-frame status timeouts, " "%ld late entries from the previous arm, " - "%ld foreign)\n", + "%ld foreign, %ld stale-EXT entries claimed)\n", pktid, submit_fail, n, frames, status_timeouts, - sum.late_prev, sum.foreign); + sum.late_prev, sum.foreign, sum.stale_ext); total_entries += sum.entries; total_foreign += sum.foreign + sum.late_prev; prev_pktid = pktid; + /* More entries than frames would be MAC duplicates: then a + * stale-pktid entry is not provably ours either. An arm that + * sent nothing popped nothing and leaves both as they were. */ + if (n > 0) { + stale_pktid = pktid; + stale_settled = (sum.entries == n); + } if (io_fail) { printf(" (arm %c: a status-FIFO read FAILED - " "the row above is incomplete)\n", arms[a].tag); From 64ebce3f849b199899b604de303c24ec26efa717 Mon Sep 17 00:00:00 2001 From: snokvist Date: Fri, 2 Oct 2026 13:04:55 +0200 Subject: [PATCH 06/18] txdemo: join the optional IN drainers on every exit With DEVOURER_DRAIN_BULK_IN or DEVOURER_POLL_INTR_IN set, bulk_in_thread and intr_in_thread were joined only on the InitWrite-exception path. The normal end of main and the early returns after a PCIe-open or no-driver failure destroyed them still joinable, so the process ended in std::terminate. A small scope guard after the two threads now clears both run flags and joins them on every return. The normal path joins explicitly before session.close(), because both threads poll the handle it releases. The InitWrite catch block's open-coded join becomes the guard. A fork() child of the legacy DEVOURER_TX_WITH_RX path holds copies of the thread objects but not the threads, so it skips the join and keeps its pre-existing exit behaviour. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01VNC8xhn1rNCi5t6uLvE6M3 --- examples/tx/main.cpp | 37 ++++++++++++++++++++++++++++--------- 1 file changed, 28 insertions(+), 9 deletions(-) diff --git a/examples/tx/main.cpp b/examples/tx/main.cpp index 9b30ebf3..9c8d63a6 100644 --- a/examples/tx/main.cpp +++ b/examples/tx/main.cpp @@ -783,6 +783,28 @@ int main(int argc, char **argv) { }); logger->info("DEVOURER_POLL_INTR_IN — EP 0x85 interrupt-IN poller running"); } + /* Stops and joins the optional IN drainers on every exit from here: a + * still-joinable std::thread terminates the process when destroyed, and + * both threads poll `handle`, so the normal path joins them explicitly + * before session.close(). A fork() child has copies of the thread objects + * but not the threads, so it must not join them (in_fork_child). */ + struct DrainerJoin { + std::atomic &bulk_running, &intr_running; + std::thread &bulk, &intr; + bool in_fork_child = false; + void join() { + bulk_running = false; + intr_running = false; + if (bulk.joinable()) + bulk.join(); + if (intr.joinable()) + intr.join(); + } + ~DrainerJoin() { + if (!in_fork_child) + join(); + } + } drainers{bulk_in_running, intr_running, bulk_in_thread, intr_in_thread}; WiFiDriver wifi_driver{logger}; std::unique_ptr owned_device; @@ -1007,6 +1029,9 @@ int main(int argc, char **argv) { if (tx_with_rx && !rx_thread_mode) { pid_t fpid = fork(); if (fpid == 0) { +#if !defined(_MSC_VER) /* fork() is a real fork here, not the (0) stub */ + drainers.in_fork_child = true; +#endif rtlDevice->Init(packetProcessor, SelectedChannel{ .Channel = static_cast(channel), @@ -1031,16 +1056,9 @@ int main(int argc, char **argv) { } catch (const std::exception &e) { /* InitWrite returns void, so a refused bring-up (e.g. a channel/width/ * offset combination the chip rejects) surfaces as an exception. The - * device already tore itself down; exit cleanly instead of aborting — - * which means the optional IN-drainer threads above must be joined - * first, or their still-joinable std::thread destructors terminate. */ + * device already tore itself down; exit cleanly instead of aborting + * (the drainers guard joins the IN-drainer threads). */ logger->error("TX bring-up failed: {}", e.what()); - bulk_in_running = false; - intr_running = false; - if (bulk_in_thread.joinable()) - bulk_in_thread.join(); - if (intr_in_thread.joinable()) - intr_in_thread.join(); return 1; } @@ -2675,6 +2693,7 @@ int main(int argc, char **argv) { * does exactly the same on every other exit path. */ /* The beacon guard must not call into the device once it is gone. */ tx_beacon.detach(); + drainers.join(); /* they poll the handle session.close() releases */ session.close(); /* A truncated caller stream is a producer fault, and a harness that scored * the run as if it had ended cleanly would be scoring a short measurement. */ From 907a5b538c7bf3728a8ee29bb573c5081bb83167 Mon Sep 17 00:00:00 2001 From: snokvist Date: Fri, 2 Oct 2026 15:23:36 +0200 Subject: [PATCH 07/18] tests: realtek_station_onair - scale the submission floor to the aired span MIN_SUBMITTED defaulted to a quarter of SECS * 1e6 / GAP_US, but the SECS window also holds the transmitter's bring-up. On an 8812BU peer txdemo.init_write took 6.2 s and 8.7 s of the 10 s window, so the peer aired for only 1.3-3.7 s and submitted 283-327 frames against the 500 floor, with live=1 and max_gap under 35 ms. Every DOWN arm came back INCONCLUSIVE: a healthy transmitter scored as a stall. summarize now prints aired_ms, the span from the arm's first tx.report to its final tx.stats. Those are the first two timestamps in one timebase; txdemo.first_tx_submit counts ms from a different epoch. It also prints min_submitted, a quarter of the GAP_US rate over that span, and usable() holds each arm to its own floor. The span ends at the final tx.stats, not the last report, so a transmitter that slows or stops after MIN_REPORTS still owes the whole span. A set MIN_SUBMITTED stays a fixed floor. Checked on synthetic JSONL fixtures run through the script's own summarize and usable: - slow bring-up, 327 frames over 2.0 s: usable, floor 100 (refused under the old fixed 500); - rate stall, 110 frames over 10 s with sparse reports under MAX_GAP_MS: refused, floor 500; - hard stall: refused, live=0; - healthy full window: usable. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01VNC8xhn1rNCi5t6uLvE6M3 --- tests/realtek_station_onair.sh | 54 +++++++++++++++++++++------------- 1 file changed, 34 insertions(+), 20 deletions(-) diff --git a/tests/realtek_station_onair.sh b/tests/realtek_station_onair.sh index 1bb8208f..2af6da61 100755 --- a/tests/realtek_station_onair.sh +++ b/tests/realtek_station_onair.sh @@ -45,8 +45,8 @@ # # An arm is scored only when its transmitter kept airing through the window # (no silence over MAX_GAP_MS between its reports, or after the last one - -# see summarize) and carried MIN_REPORTS reports and MIN_SUBMITTED -# submissions. Reception in arm A is +# see summarize) and carried MIN_REPORTS reports and its submission floor +# (MIN_SUBMITTED, scaled to the span it aired). Reception in arm A is # judged against the peer's REPORTED frames, which aired; submitted frames # left unreported at window close are printed separately. # @@ -92,8 +92,13 @@ MIN_RX_PCT="${MIN_RX_PCT:-80}" # CCX reports, and between its last report and its final tx.stats. MAX_GAP_MS="${MAX_GAP_MS:-2000}" # Per-arm floor on frames the transmitter submitted. Default: a quarter of -# the nominal SECS / GAP_US rate (500 at the defaults; the slowest arm on -# record, an unacknowledged one at 12 retries, submitted ~900). +# the nominal GAP_US rate over the span the arm actually aired - its first +# report to its final tx.stats (see summarize) - not over SECS, which also +# holds the transmitter's bring-up: an 8812BU peer has spent 6-9 s of a 10 s +# window in InitWrite and been scored a stall at 283-327 healthy frames. +# The slowest arm on record, an unacknowledged one at 12 retries, submitted +# ~900 in 10 s against the 500 a full window asks. A set MIN_SUBMITTED is a +# fixed floor instead. MIN_SUBMITTED="${MIN_SUBMITTED:-}" EXPECT_UNARMED_SILENT="${EXPECT_UNARMED_SILENT:-1}" READY_TIMEOUT="${READY_TIMEOUT:-30}" @@ -106,14 +111,6 @@ case "$HALF" in both|down|up) ;; *) echo "HALF must be both, down or up"; exit 2 for v in SECS RETRY_LIMIT GAP_US MIN_REPORTS MIN_RX_PCT MAX_GAP_MS READY_TIMEOUT CLEAR_AFTER_MS CH; do case "${!v}" in ''|*[!0-9]*) echo "$v must be a non-negative integer"; exit 2 ;; esac done -# GAP_US=0 (max duty) has no nominal rate, so no default floor. -if [ -z "$MIN_SUBMITTED" ]; then - if [ "$GAP_US" -gt 0 ]; then - MIN_SUBMITTED=$(( SECS * 1000000 / GAP_US / 4 )) - else - MIN_SUBMITTED=0 - fi -fi case "$MIN_SUBMITTED" in *[!0-9]*) echo "MIN_SUBMITTED must be a non-negative integer"; exit 2 ;; esac # RETRY_LIMIT 0 would make every control indistinguishable from a one-shot # send: the retry count is the instrument. @@ -255,7 +252,8 @@ wait_for() { # $1 pid, $2 file, $3 regex, $4 timeout s # One summary line from a transmitter's JSONL (and, optionally, the DUT's # rx.seq stream for frames from TA): reports, submitted, unreported, -# ok_pct, retries_mean, max_gap_ms, tail_ms, live, rx_distinct. +# ok_pct, retries_mean, max_gap_ms, tail_ms, live, aired_ms, min_submitted, +# rx_distinct. # # LIVENESS. A report is a frame that aired, so the reports' own timestamps # (t, the host-monotonic tx.report timebase) show whether the transmitter @@ -264,10 +262,18 @@ wait_for() { # $1 pid, $2 file, $3 regex, $4 timeout s # tx.stats, which carries t in the same timebase. live=0 when either exceeds # MAX_GAP_MS or the final tx.stats has no t - an arm that aired a burst and # stalled, which MIN_REPORTS alone would accept. +# +# FLOOR. aired_ms runs from the first report to the final tx.stats: the +# first pair of timestamps in one timebase (txdemo.first_tx_submit carries +# ms from a different epoch). It ends at the final tx.stats, not the last +# report, so a transmitter that slows or stops after MIN_REPORTS still owes +# the whole span. min_submitted is a quarter of the GAP_US rate over it +# (0 at GAP_US=0, which has no nominal rate), or MIN_SUBMITTED when set. summarize() { # $1 tx jsonl, $2 tag, $3 dut jsonl or "" - python3 - "$1" "$2" "${3:-}" "$MAX_GAP_MS" <<'PYEOF' + python3 - "$1" "$2" "${3:-}" "$MAX_GAP_MS" "$GAP_US" "$MIN_SUBMITTED" <<'PYEOF' import json, sys tx, tag, rx, max_gap = sys.argv[1], sys.argv[2], sys.argv[3], int(sys.argv[4]) +gap_us, fixed_floor = int(sys.argv[5]), sys.argv[6] n = okc = retries = 0 submitted = 0 ts = [] @@ -298,6 +304,12 @@ out = (f"{tag} reports={n} submitted={submitted} " if n: out += f" ok_pct={100.0*okc/n:.1f} retries_mean={retries/n:.2f}" out += f" max_gap_ms={gap} tail_ms={'none' if tail is None else tail} live={live}" +aired = (final_t - ts[0]) if (final_t is not None and ts) else 0 +if fixed_floor: + floor = int(fixed_floor) +else: + floor = aired * 1000 // gap_us // 4 if gap_us > 0 else 0 +out += f" aired_ms={aired} min_submitted={floor}" if rx: seen = set() try: @@ -460,12 +472,12 @@ ok() { pass=$((pass+1)); printf ' PASS %s\n' "$*"; } bad() { fail=$((fail+1)); printf ' FAIL %s\n' "$*"; } inc() { inconclusive=$((inconclusive+1)); printf ' INCONCLUSIVE %s\n' "$*"; } field() { sed -n "s/.* $2=\\([0-9.]*\\).*/\\1/p" "$OUT/res_$1"; } -# A usable arm: not aborted, carrying at least MIN_REPORTS reports and -# MIN_SUBMITTED submissions, and with a transmitter that kept airing through +# A usable arm: not aborted, carrying at least MIN_REPORTS reports and its +# min_submitted submissions, and with a transmitter that kept airing through # the window (live=1, see summarize). Anything else leaves its verdicts # INCONCLUSIVE. usable() { - local r n sub + local r n sub floor r=$(cat "$OUT/res_$1" 2>/dev/null) case "$r" in *ABORTED*|*FAILCLEAR*|'') return 1 ;; @@ -473,7 +485,9 @@ usable() { case "$r" in *" live=1"*) ;; *) return 1 ;; esac n=$(field "$1" reports) sub=$(field "$1" submitted) - [ "${n:-0}" -ge "$MIN_REPORTS" ] && [ "${sub:-0}" -ge "$MIN_SUBMITTED" ] + floor=$(field "$1" min_submitted) + [ -n "$floor" ] || return 1 + [ "${n:-0}" -ge "$MIN_REPORTS" ] && [ "${sub:-0}" -ge "$floor" ] } show() { echo " $(cat "$OUT/res_$1" 2>/dev/null)"; } @@ -541,7 +555,7 @@ if [ "$HALF" != up ]; then bad "DOWN receive: the DUT delivered ${rxd:-0} distinct frames of the peer's ${rep:-0} reported (< ${MIN_RX_PCT}%; ${unr:-0} submitted unreported)" fi else - inc "DOWN: arm A, B or C aborted, carried under $MIN_REPORTS reports or $MIN_SUBMITTED submissions, or its transmitter stalled (live=0) - not a measurement" + inc "DOWN: arm A, B or C aborted, carried under $MIN_REPORTS reports or its min_submitted floor, or its transmitter stalled (live=0) - not a measurement" fi if usable A && usable D; then a=$(field A ok_pct); d=$(field D ok_pct) @@ -639,7 +653,7 @@ EOF bad "UP ack: F ${f}% is not clearly above the control G ${g}%" fi else - inc "UP: arm F or G aborted, carried under $MIN_REPORTS reports or $MIN_SUBMITTED submissions, or its transmitter stalled (live=0) - not a measurement" + inc "UP: arm F or G aborted, carried under $MIN_REPORTS reports or its min_submitted floor, or its transmitter stalled (live=0) - not a measurement" fi # H is information, not a verdict: whether TX ACK matching needs the arm # at all on this die. Never counted into pass/fail/inconclusive. From da8cd6d00b31b2f4fc734a8e05c999487cd41ea6 Mon Sep 17 00:00:00 2001 From: snokvist Date: Sat, 3 Oct 2026 10:52:39 +0200 Subject: [PATCH 08/18] tests: realtek_station_onair - run the AP guard in the preflight too The AP guard for the UP half ran only after the whole DOWN half. A run with the AP already off the bus spent 60 s on DOWN and then refused, and a hub, an adapter with no wireless netdev, one carrying a default route, or a phy without AP mode was refused just as late. The UP half's guard is now a function, ap_guard. It runs right after the lib is sourced, for the UP or both halves, and again at the start of UP, since the adapter can move during DOWN. The preflight only reads; nothing is written before the lock. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01VNC8xhn1rNCi5t6uLvE6M3 --- tests/realtek_station_onair.sh | 39 ++++++++++++++++++++++------------ 1 file changed, 25 insertions(+), 14 deletions(-) diff --git a/tests/realtek_station_onair.sh b/tests/realtek_station_onair.sh index 2af6da61..ef3bd00b 100755 --- a/tests/realtek_station_onair.sh +++ b/tests/realtek_station_onair.sh @@ -131,6 +131,29 @@ done # shellcheck source=tests/mt7612u_sta_lib.sh . "$ROOT/tests/mt7612u_sta_lib.sh" + +# The AP guard (tests/mt7612u_sta_identity.sh's): cleanup re-enumerates +# AP_SYSFS as root, so it must be a USB device that is not a hub, carrying a +# wireless netdev with no default route on a phy that supports AP mode. Sets +# AP_IF and PHY. Run in the preflight, before the DOWN half spends its +# minute, and again at the start of UP - the adapter can move in between. +ap_refuse() { echo "refusing AP_SYSFS=$AP_SYSFS: $* - cleanup would re-enumerate it."; exit 2; } +ap_guard() { + local fam + [ -n "$(cat "/sys/bus/usb/devices/$AP_SYSFS/idVendor" 2>/dev/null)" ] || + ap_refuse "not a USB device - if its driver just loaded it may have moved; re-read lsusb -t" + [ "$(cat "/sys/bus/usb/devices/$AP_SYSFS/bDeviceClass" 2>/dev/null)" != "09" ] || ap_refuse "a hub" + AP_IF=$(sta_first_netdev "$AP_SYSFS") + [ -n "$AP_IF" ] || ap_refuse "no network interface on it" + [ -e "/sys/class/net/$AP_IF/phy80211" ] || ap_refuse "$AP_IF is not wireless" + for fam in -4 -6; do + ip "$fam" route show default 2>/dev/null | grep -qw "dev $AP_IF" && + ap_refuse "$AP_IF carries a default route" + done + PHY=$(basename "$(readlink -f "/sys/class/net/$AP_IF/phy80211")") + iw phy "$PHY" info 2>/dev/null | grep -q '\* AP$' || ap_refuse "$AP_IF ($PHY) does not support AP mode" +} +[ "$HALF" = down ] || ap_guard sta_out_prepare || exit 2 sta_lock_take || exit 2 sta_pid_init dut peer hostapd @@ -591,20 +614,8 @@ fi # =============================== UP ========================================= if [ "$HALF" != down ]; then - # --- the AP: the guard tests/mt7612u_sta_identity.sh uses --- - ap_refuse() { echo "refusing AP_SYSFS=$AP_SYSFS: $* - cleanup would re-enumerate it."; exit 2; } - [ -n "$(cat "/sys/bus/usb/devices/$AP_SYSFS/idVendor" 2>/dev/null)" ] || - ap_refuse "not a USB device - if its driver just loaded it may have moved; re-read lsusb -t" - [ "$(cat "/sys/bus/usb/devices/$AP_SYSFS/bDeviceClass" 2>/dev/null)" != "09" ] || ap_refuse "a hub" - AP_IF=$(sta_first_netdev "$AP_SYSFS") - [ -n "$AP_IF" ] || ap_refuse "no network interface on it" - [ -e "/sys/class/net/$AP_IF/phy80211" ] || ap_refuse "$AP_IF is not wireless" - for fam in -4 -6; do - ip "$fam" route show default 2>/dev/null | grep -qw "dev $AP_IF" && - ap_refuse "$AP_IF carries a default route" - done - PHY=$(basename "$(readlink -f "/sys/class/net/$AP_IF/phy80211")") - iw phy "$PHY" info 2>/dev/null | grep -q '\* AP$' || ap_refuse "$AP_IF ($PHY) does not support AP mode" + # --- the AP: the full guard again (it can have moved during DOWN) --- + ap_guard AP_ID=$(sta_usb_id "$AP_SYSFS") nmcli device set "$AP_IF" managed no >/dev/null 2>&1 From 2cc53304df758f9b3f1a749594932e5aee524223 Mon Sep 17 00:00:00 2001 From: snokvist Date: Fri, 2 Oct 2026 15:25:26 +0200 Subject: [PATCH 09/18] jaguar1/2/3: refuse a station arm after Stop() Stop() cleared a held station arm but left _station_ready set, so a SetStationIdentity issued after Stop() was accepted: the same hole the gate closed for _brought_up. That arm went to a torn-down chip on Jaguar1/3, or to a still-powered chip on Jaguar2, which Stop() does not power down. All three Stop()s now clear _station_ready under the station lock, in the block that clears the arm (_port0_mu on Jaguar1, _reg_mu on Jaguar2/3). Nothing else reads the flag except the SetStationIdentity gate, and every bring-up already clears it on entry and commits it at the end, so a re-Init after Stop() arms again as before. StationArm.h's LIFETIME note says so. No selftest: _station_ready only becomes true at the end of a real Init/InitWrite, which no headless test can run. The existing station_arm selftest covers StationArm, not the device classes. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01VNC8xhn1rNCi5t6uLvE6M3 --- src/StationArm.h | 4 +++- src/jaguar1/RtlJaguarDevice.cpp | 3 ++- src/jaguar2/RtlJaguar2Device.cpp | 4 +++- src/jaguar3/RtlJaguar3Device.cpp | 4 +++- 4 files changed, 11 insertions(+), 4 deletions(-) diff --git a/src/StationArm.h b/src/StationArm.h index 49bf0fbc..9d26fc17 100644 --- a/src/StationArm.h +++ b/src/StationArm.h @@ -57,7 +57,9 @@ * (Jaguar2's does not; Jaguar1's is optional) and a port left on MACID = own * / Infra goes on acknowledging for a station whose process has gone. A * (re-)bring-up clears a held arm first (retire()); a clear that does not - * verify keeps the record for ClearStationIdentity to retry. */ + * verify keeps the record for ClearStationIdentity to retry. Stop() also + * clears _station_ready, so an arm after Stop() is refused until the next + * bring-up. */ #include #include diff --git a/src/jaguar1/RtlJaguarDevice.cpp b/src/jaguar1/RtlJaguarDevice.cpp index 399b4be0..b0e05df3 100644 --- a/src/jaguar1/RtlJaguarDevice.cpp +++ b/src/jaguar1/RtlJaguarDevice.cpp @@ -2487,9 +2487,10 @@ void RtlJaguarDevice::Stop() { /* A station arm ends with the session: restored before the optional * power-down (best effort; a failure is logged by the clear), so a chip * left powered (tuning.teardown_power_down=0) does not keep answering for - * the station. */ + * the station. A new arm is refused until the next bring-up. */ { std::lock_guard lock(_port0_mu); + _station_ready = false; if (_station.armed()) (void)_station.clear(_device, _logger, "Jaguar1"); } diff --git a/src/jaguar2/RtlJaguar2Device.cpp b/src/jaguar2/RtlJaguar2Device.cpp index ac80f318..6dd48472 100644 --- a/src/jaguar2/RtlJaguar2Device.cpp +++ b/src/jaguar2/RtlJaguar2Device.cpp @@ -2326,9 +2326,11 @@ void RtlJaguar2Device::Stop() { /* This Stop leaves the chip powered, so a station arm would outlive the * session: port 0 on MACID = own / Infra keeps acknowledging for the * station after the process has gone. Clear it here (best effort; a - * failure is logged by the clear). */ + * failure is logged by the clear). A new arm is refused until the next + * bring-up. */ { std::lock_guard lk(_reg_mu); + _station_ready = false; if (_station.armed()) (void)_station.clear(_device, _logger, "Jaguar2"); } diff --git a/src/jaguar3/RtlJaguar3Device.cpp b/src/jaguar3/RtlJaguar3Device.cpp index 1d86df2b..1f880965 100644 --- a/src/jaguar3/RtlJaguar3Device.cpp +++ b/src/jaguar3/RtlJaguar3Device.cpp @@ -858,9 +858,11 @@ void RtlJaguar3Device::Stop() { /* A station arm ends with the session, not with whatever the de-init * below leaves: restored first (best effort; a failure is logged by the * clear), so the port stops answering for the station even where the - * power-down does not complete. */ + * power-down does not complete. A new arm is refused until the next + * bring-up. */ { std::lock_guard lk(_reg_mu); + _station_ready = false; if (_station.armed()) (void)_station.clear(_device, _logger, "Jaguar3"); } From 53477e02f87314d725f7676002ba71db2e4b7556 Mon Sep 17 00:00:00 2001 From: snokvist Date: Sat, 3 Oct 2026 10:52:39 +0200 Subject: [PATCH 10/18] tests: mt7612u_ap_onair - stop on an interrupt; no hand-back under a live process - Interrupt. `trap cleanup EXIT INT TERM` ran cleanup on Ctrl-C and then carried on into the next cell against a re-enumerated adapter. The traps now follow the station harnesses: `cleanup; exit 130` on INT/TERM, cleanup on EXIT. A CLEANED flag, set only on the interrupt path, spares the EXIT pass a second re-enumeration. It is not a once-only guard, because CELLS=all calls cleanup between cells. - Hand-back. cleanup re-enumerated AP_SYSFS right after sending TERM, with the AP demo possibly still inside its de-init. reap now polls its children for up to 10 s with the station lib's zombie-aware sta_pid_alive (the lib is sourced for that alone). Anything still alive is KILLed, and cleanup then skips the re-enumeration, saying so. - PHASE 2. cell_stop waited up to 60 s for "PHASE 2" and then scored "beacon gone" whether or not it appeared. A bstop_onair that ended before StopBeacon - its teardown silencing the beacon - read as a StopBeacon PASS. Without PHASE 2 the cell now FAILs, naming that. - Ping loss. The open cell's failure message extracted the loss with `[0-9]+% packet loss`, which cut ping's "66.6667% packet loss" to "6667% packet loss". It now takes `[0-9.]+%`. The pass test is unchanged. - Dead loop counters (`local i`) dropped. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01VNC8xhn1rNCi5t6uLvE6M3 --- tests/mt7612u_ap_onair.sh | 57 ++++++++++++++++++++++++++++++--------- 1 file changed, 44 insertions(+), 13 deletions(-) diff --git a/tests/mt7612u_ap_onair.sh b/tests/mt7612u_ap_onair.sh index 2e9d5585..9a51ed8e 100755 --- a/tests/mt7612u_ap_onair.sh +++ b/tests/mt7612u_ap_onair.sh @@ -58,14 +58,36 @@ bad() { fail=$((fail+1)); printf ' FAIL %s\n' "$*"; } # wpa_supplicant` would drop every wireless client on the host, and a name kill # would reach a concurrent run of this same test. KIDS="" +# sta_pid_alive (zombie-aware) is all this takes from the station lib. +# shellcheck source=tests/mt7612u_sta_lib.sh +. "$ROOT/tests/mt7612u_sta_lib.sh" +# TERM every child, then WAIT for them to exit - up to 10 s: a demo's chip +# de-init runs after the signal, and re-enumerating the adapter under it is +# the hand-back this must not do. Anything still alive then is KILLed and +# reaped is 1, so cleanup leaves the adapter alone; 0 when all exited. reap() { - local pid + local pid live t=0 for pid in $KIDS; do kill "$pid" 2>/dev/null; done + while :; do + live="" + for pid in $KIDS; do sta_pid_alive "$pid" && live="$live $pid"; done + [ -z "$live" ] && break + if [ "$t" -ge 100 ]; then + for pid in $live; do kill -KILL "$pid" 2>/dev/null; done + echo "still running 10 s after TERM (KILLed):$live" + KIDS="" + return 1 + fi + sleep 0.1; t=$((t + 1)) + done + for pid in $KIDS; do wait "$pid" 2>/dev/null; done # reaps our own children KIDS="" + return 0 } cleanup() { - reap + local reaped=0 + reap || reaped=1 [ -n "${STA_IF:-}" ] && { ip addr flush dev "$STA_IF" 2>/dev/null iw dev "$STA_IF" disconnect 2>/dev/null; } # The MAC beacons autonomously, so a cell that died before its teardown can @@ -87,7 +109,9 @@ cleanup() { # Either way, confirmed against the VID:PID first: this runs as root and # writes to a path the caller supplied, and a stale AP_SYSFS would otherwise # yank whatever else is plugged there. - if [ "$(cat "/sys/bus/usb/devices/$AP_SYSFS/idVendor" 2>/dev/null)" = "0e8d" ] && + if [ "$reaped" != 0 ]; then + echo "a process outlived TERM - not re-enumerating AP_SYSFS=$AP_SYSFS" + elif [ "$(cat "/sys/bus/usb/devices/$AP_SYSFS/idVendor" 2>/dev/null)" = "0e8d" ] && [ "$(cat "/sys/bus/usb/devices/$AP_SYSFS/idProduct" 2>/dev/null)" = "7612" ]; then if [ -n "${AP_VBUS:-}" ]; then uhubctl -l "${AP_VBUS%%:*}" -p "${AP_VBUS##*:}" -a off >/dev/null 2>&1 @@ -102,7 +126,13 @@ cleanup() { fi fi } -trap cleanup EXIT INT TERM +# An interrupt must STOP the run: on the EXIT trap alone, INT/TERM would run +# cleanup and then carry on into the next cell against a re-enumerated +# adapter. CLEANED only spares the EXIT pass a second re-enumeration after +# that; the cleanups between cells (CELLS=all) still run every time. +CLEANED=no +trap '[ "$CLEANED" = yes ] || cleanup' EXIT +trap 'cleanup; CLEANED=yes; exit 130' INT TERM # --- the station ----------------------------------------------------------- STA_IF=$(ls "/sys/bus/usb/devices/$STA_SYSFS:1.0/net/" 2>/dev/null | head -1) @@ -126,8 +156,8 @@ say "AP $AP_SYSFS station $STA_SYSFS ($STA_IF) ch$CH ($FREQ MHz)" # live beacon as absent. Observed: a "beacon not scannable" FAIL in a run where # the station then associated, pinged, and got an auth at retry=0. seen() { # $1 = SSID, $2 = BSSID - local i n best=0 - for i in 1 2 3; do + local n best=0 + for _ in 1 2 3; do # Matched on BSSID *and* SSID: a neighbour running "devourerAP" would # otherwise pass an arm check, fail a stop check, or break the exact-count # comparison. awk keeps the pairing - grep -c on two patterns would count @@ -189,7 +219,7 @@ cell_open() { if ping -c 6 -W 1 -I "$STA_IF" "$APIP" 2>&1 | tee "$OUT/open.ping" | grep -q " 0% packet loss"; then ok "open: data plane ($(grep -oE 'rtt [^ ]+ = [0-9./]+' "$OUT/open.ping" | head -1))" else - bad "open: ping lost packets ($(grep -oE '[0-9]+% packet loss' "$OUT/open.ping" | head -1))" + bad "open: ping lost packets ($(grep -oE '[0-9.]+% packet loss' "$OUT/open.ping" | head -1))" fi # retry=0 on auth IS the hardware ACK: an un-ACKed frame comes back with FC # Retry set. This is the only evidence that the APC slot and port identity @@ -222,8 +252,7 @@ cell_wpa2() { ip addr flush dev "$STA_IF" 2>/dev/null wpa_supplicant -i "$STA_IF" -c "$wpa" -P "$OUT/wpa.pid" -B >/dev/null 2>&1 KIDS="$KIDS $(cat "$OUT/wpa.pid" 2>/dev/null)" - local i - for i in $(seq 1 20); do + for _ in $(seq 1 20); do grep -q "4-WAY HANDSHAKE COMPLETE" "$OUT/wpa2.log" && break sleep 1 done @@ -274,8 +303,7 @@ cell_stop() { printf '%s' "${n:-0}" } wait_arm() { # $1 = the count to exceed, $2 = seconds to wait - local i - for i in $(seq 1 "$2"); do [ "$(armed)" -gt "$1" ] && return 0; sleep 1; done + for _ in $(seq 1 "$2"); do [ "$(armed)" -gt "$1" ] && return 0; sleep 1; done return 1 } @@ -284,8 +312,11 @@ cell_stop() { [ "$(seen mtStopCheck 02:4d:54:53:54:50)" = 1 ] && ok "stop: armed - beacon on air" || bad "stop: armed but not scannable" local n_arms; n_arms=$(armed) - local i - for i in $(seq 1 60); do grep -q "PHASE 2" "$OUT/stop.log" && break; sleep 1; done + for _ in $(seq 1 60); do grep -q "PHASE 2" "$OUT/stop.log" && break; sleep 1; done + # No PHASE 2 means StopBeacon was never called: a beacon gone now was + # stopped by whatever ended the process, which says nothing about StopBeacon. + grep -q "PHASE 2" "$OUT/stop.log" || + { bad "stop: never reached PHASE 2 - StopBeacon not exercised"; kill $ap 2>/dev/null; return; } sleep 6 [ "$(seen mtStopCheck 02:4d:54:53:54:50)" = 0 ] && ok "stop: stopped - beacon gone" || bad "stop: STILL AIRING after StopBeacon" From 05bdaf72de72079790c257f99f7c3e2d7131f5d3 Mon Sep 17 00:00:00 2001 From: snokvist Date: Fri, 2 Oct 2026 22:19:15 +0200 Subject: [PATCH 11/18] tests: mt7612u_sta_autoack - no DUT hand-back while it is still running cleanup ignored sta_pid_kill's result for the DUT and re-enumerated it regardless, including while the bring-up process was still inside its de-init. The peer already had this rule; the hand-back is now gated the same way, as in realtek_station_onair.sh. That needed a CLEANED guard. sta_pid_kill forgets the PID on its first call, so the EXIT pass that follows an INT would have read "gone" and handed back the adapter the first pass had just refused to. The old comment called that second pass harmless. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01VNC8xhn1rNCi5t6uLvE6M3 --- tests/mt7612u_sta_autoack.sh | 14 ++++++++++---- 1 file changed, 10 insertions(+), 4 deletions(-) diff --git a/tests/mt7612u_sta_autoack.sh b/tests/mt7612u_sta_autoack.sh index f8091aa1..52f37ce3 100755 --- a/tests/mt7612u_sta_autoack.sh +++ b/tests/mt7612u_sta_autoack.sh @@ -68,17 +68,22 @@ ok() { pass=$((pass+1)); printf ' PASS %s\n' "$*"; } bad() { fail=$((fail+1)); printf ' FAIL %s\n' "$*"; } DUT_PID="" +CLEANED=no # shellcheck disable=SC2317 # reached through the traps below cleanup() { + [ "$CLEANED" = yes ] && return 0 + CLEANED=yes # arm() runs in a command substitution, so the PIDs it starts are recorded # in $OUT (tests/mt7612u_sta_lib.sh) for this trap to find. The peer first: # an orphan txdemo keeps its USB lock and fails the NEXT run's peer open # with "adapter already in use", which yields zero reports - and zero is a # control's passing value. INT, as timeout(1) forwards it to txdemo. sta_pid_kill peer INT; peer_gone=$? - sta_pid_kill dut + sta_pid_kill dut; dut_gone=$? DUT_PID="" - sta_dut_handback + # The same rule for the DUT: never re-enumerate it mid-de-init. + if [ "$dut_gone" = 0 ]; then sta_dut_handback + else echo "DUT still running - not re-enumerating DUT_SYSFS=$DUT_SYSFS"; fi # Only once the peer process has really exited: re-enumerating an adapter # still inside its de-init is what the hand-back must not do. if [ "$peer_gone" = 0 ]; then sta_peer_handback @@ -88,8 +93,9 @@ cleanup() { } trap cleanup EXIT # AND IT MUST STOP: with INT/TERM on the EXIT trap the shell runs cleanup -# and then CARRIES ON into the next arm. cleanup is idempotent, so the EXIT -# pass after it is harmless. +# and then CARRIES ON into the next arm. CLEANED makes the EXIT pass after it +# a no-op: sta_pid_kill forgets a PID on the first pass, so a second pass +# would hand back an adapter the first refused to. trap 'cleanup; exit 130' INT TERM sta_dut_take || exit 2 From 8a2ab144ee880d729014430308afec2f176be0e0 Mon Sep 17 00:00:00 2001 From: snokvist Date: Fri, 2 Oct 2026 22:19:44 +0200 Subject: [PATCH 12/18] tests: mt7612u_sta_uplink - bound the DUT gate; no hand-back while it runs The DUT's `mt7612uprobe txs` ran under a bare `wait`, so a wedged gate held the run and both adapters forever. It now runs under `timeout -s INT -k 10` with a bound derived from the gate's own worst case: 16 gate arms; per frame, the status wait (b + 50 ms) plus the settle (b), where b is gate_txs's frame_budget_ms for RETRY_LIMIT; and about 7 s of fixed cost per arm. On top of that, half again plus 2 min for bring-up: about 9 min at FRAMES=60 and 19 at FRAMES=200, against the ~9 min measured for a 200-frame arm. An overrun ABORTs the arm. cleanup re-enumerated the DUT whether or not sta_pid_kill saw its process exit. The hand-back is now gated on that, as for the peer, with a CLEANED guard so the EXIT pass after an INT cannot undo the refusal. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01VNC8xhn1rNCi5t6uLvE6M3 --- tests/mt7612u_sta_uplink.sh | 28 ++++++++++++++++++++++++---- 1 file changed, 24 insertions(+), 4 deletions(-) diff --git a/tests/mt7612u_sta_uplink.sh b/tests/mt7612u_sta_uplink.sh index 0eb064ea..76cec4c3 100755 --- a/tests/mt7612u_sta_uplink.sh +++ b/tests/mt7612u_sta_uplink.sh @@ -86,14 +86,19 @@ ok() { pass=$((pass+1)); printf ' PASS %s\n' "$*"; } bad() { fail=$((fail+1)); printf ' FAIL %s\n' "$*"; } RESP="" +CLEANED=no # shellcheck disable=SC2317 # reached through the traps below cleanup() { + [ "$CLEANED" = yes ] && return 0 + CLEANED=yes # arm() runs in a command substitution, so its PIDs are recorded in $OUT # (tests/mt7612u_sta_lib.sh) for this trap to find. - sta_pid_kill dut + sta_pid_kill dut; dut_gone=$? sta_pid_kill resp; peer_gone=$? RESP="" - sta_dut_handback + # Never re-enumerate the DUT while its process is still in de-init. + if [ "$dut_gone" = 0 ]; then sta_dut_handback + else echo "DUT still running - not re-enumerating DUT_SYSFS=$DUT_SYSFS"; fi # Only once the peer process has really exited: re-enumerating an adapter # still inside its de-init is what the hand-back must not do. if [ "$peer_gone" = 0 ]; then sta_peer_handback @@ -103,8 +108,9 @@ cleanup() { } trap cleanup EXIT # AND IT MUST STOP: with INT/TERM on the EXIT trap the shell runs cleanup -# and then CARRIES ON into the next arm. cleanup is idempotent, so the EXIT -# pass after it is harmless. +# and then CARRIES ON into the next arm. CLEANED makes the EXIT pass after it +# a no-op: sta_pid_kill forgets a PID on the first pass, so a second pass +# would hand back an adapter the first refused to. trap 'cleanup; exit 130' INT TERM sta_dut_take || exit 2 @@ -140,13 +146,27 @@ arm() { sta_pid_kill resp; RESP=""; return 1 fi + # BOUNDED, so a wedged gate cannot hold the run (and both adapters) forever. + # The gate's own worst case: 16 gate arms, each frame waiting at most its + # status bound (b + 50 ms) and the settle at most b per frame + 2 s, with b + # = 60 ms + 8 ms per retry past 15 (gate_txs's frame_budget_ms), plus ~7 s of + # fixed cost per arm. Half again on top, and 2 min for bring-up. INT lets the + # gate tear down (exit 3); KILL 10 s later if it does not. + b=$(( RETRY_LIMIT > 15 ? 60 + (RETRY_LIMIT - 15) * 8 : 60 )) + dut_bound=$(( 16 * (FRAMES * (2 * b + 50) / 1000 + 8) * 3 / 2 + 120 )) DEVOURER_TX_RETRY_LIMIT="$RETRY_LIMIT" \ + timeout -s INT -k 10 "$dut_bound" \ "$BUILD/mt7612uprobe" txs "$CH" "$FRAMES" "$TARGET" \ >"$OUT/dut_$tag.txt" 2>&1 & dut=$! sta_pid_record dut "$dut" wait "$dut" + dut_rc=$? rm -f "$OUT/.pid_dut" + if [ "$dut_rc" = 124 ] || [ "$dut_rc" = 137 ]; then + printf '%s ABORTED the DUT gate overran its %ss bound' "$tag" "$dut_bound" + sta_pid_kill resp; RESP=""; return 1 + fi # The limit must have LANDED, not merely been asked for: the gate prints # this line only after mt7612u_set_retry_limit() read it back. if ! grep -q "^retry limit set to $RETRY_LIMIT " "$OUT/dut_$tag.txt"; then From bf9c0dd6a2f1e01962f8b972599eb0068df28767 Mon Sep 17 00:00:00 2001 From: snokvist Date: Sat, 3 Oct 2026 10:52:43 +0200 Subject: [PATCH 13/18] tests: mt7612u_sta_identity - safe teardown, a bounded gate, a 0-3 exit - hostapd. AP_REENUM was set only after the hostapd launch and its 10 s wait loop, and cleanup killed hostapd only when the flag was set. A Ctrl-C in that window left hostapd running and the AP in AP mode, with the lock released. The flag is now set once the AP guard has passed, before the interface is touched, and hostapd is killed unconditionally, as in realtek_station_onair.sh. - DUT hand-back. It ignored sta_pid_kill's result and re-enumerated the DUT with the BSSID gate still running. It is now gated on the gate having exited. - Bounded gate. `wait "$sta_gate"` was unbounded. The gate now runs under `timeout -s INT -k 10` for six arms of SECS plus 183 s. An overrun is no measurement (2), and stays 2 even when the injector never started. - Monitor vif. It was probed only after the contract and probe-response gates. It is now also probed (create, then delete) right after the AP guard. - Exit status. Everything collapsed into 0/1. It is now 0 pass, 1 a gate failed, 2 INCONCLUSIVE, 3 interrupted, combined as: 1 if any gate failed, else 3, else 2, else 0. - Rig refusals exit 2: a hostapd that never brought the AP up, and no monitor vif. realtek_station_onair scores the same hostapd failure as 2. - The INT/TERM trap exits 3, as documented. - Nothing in the tree reads this exit code. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01VNC8xhn1rNCi5t6uLvE6M3 --- tests/mt7612u_sta_identity.sh | 67 ++++++++++++++++++++++++++++------- 1 file changed, 54 insertions(+), 13 deletions(-) diff --git a/tests/mt7612u_sta_identity.sh b/tests/mt7612u_sta_identity.sh index 9ea8b3ed..7d3ebe98 100755 --- a/tests/mt7612u_sta_identity.sh +++ b/tests/mt7612u_sta_identity.sh @@ -41,6 +41,10 @@ # first binds (seen: bus 9 -> 10, 3-2.3.3 -> 4-2.3.3). Read AP_SYSFS from # `lsusb -t` after the driver has loaded; a stale one is refused. # +# Exit status: 0 every gate passed; 1 a gate failed; 2 INCONCLUSIVE (a gate +# could not measure, or the rig was refused); 3 interrupted (a gate, or the +# run itself by INT/TERM). +# # Env: AP_SYSFS, DUT_SYSFS, CH, BSSID, SECS, OUT, FW_DIR. set -u @@ -79,9 +83,10 @@ AP_IF="" # The accepted AP's idVendor:idProduct:serial, recorded once the guard has # passed; cleanup re-enumerates AP_SYSFS only while it still names this device. AP_ID="" -# Set only once hostapd is up on a verified AP-capable interface: before -# that, the trap has no business re-enumerating anything (a wrong or default -# AP_SYSFS naming a hub would power-cycle every device under it). +# Set only once AP_SYSFS has passed the AP guard below, and before anything +# touches the interface: before that, the trap has no business +# re-enumerating anything (a wrong or default AP_SYSFS naming a hub would +# power-cycle every device under it). AP_REENUM=no CLEANED=no # shellcheck disable=SC2317 # reached through the traps below @@ -90,11 +95,15 @@ cleanup() { CLEANED=yes sta_fw_unlink sta_pid_kill inject - sta_pid_kill gate - sta_dut_handback - [ "$AP_REENUM" = yes ] || { sta_lock_release; return 0; } + local gate_gone=0 + sta_pid_kill gate || gate_gone=1 + # Never re-enumerate the DUT while its gate is still in de-init. + if [ "$gate_gone" = 0 ]; then sta_dut_handback + else echo "DUT gate still running - not re-enumerating DUT_SYSFS=$DUT_SYSFS"; fi # hostapd -B daemonizes; its PID is the one it wrote to -P for this run. + # Unconditional: nothing is recorded unless hostapd started. sta_pid_kill hostapd + [ "$AP_REENUM" = yes ] || { sta_lock_release; return 0; } sleep 1 iw dev staid_mon del 2>/dev/null # RE-ENUMERATE the AP adapter, do not just bounce the link. @@ -125,7 +134,7 @@ trap cleanup EXIT # AND IT MUST STOP: with INT/TERM on the EXIT trap the shell runs cleanup # and then CARRIES ON into the next arm. cleanup is idempotent, so the EXIT # pass after it is harmless. -trap 'cleanup; exit 130' INT TERM +trap 'cleanup; exit 3' INT TERM # --- the AP ---------------------------------------------------------------- # THE AP GUARD. Cleanup re-enumerates AP_SYSFS as root, so it is accepted only @@ -162,6 +171,17 @@ PHY=$(basename "$(readlink -f "/sys/class/net/$AP_IF/phy80211")") iw phy "$PHY" info 2>/dev/null | grep -q '\* AP$' || ap_refuse "$AP_IF ($PHY) does not support AP mode" AP_ID=$(sta_usb_id "$AP_SYSFS") +# The BSSID gate needs a monitor vif on the AP's phy. Probed here, before +# the gates spend their minute, and again (unchanged) where it is used. +iw dev staid_mon del 2>/dev/null +if ! iw phy "$PHY" interface add staid_mon type monitor 2>/dev/null; then + echo "no monitor vif on $PHY: the BSSID gate would measure broadcast" + echo "reception only - refusing this AP." + exit 2 +fi +iw dev staid_mon del 2>/dev/null +# From here on the trap restores the AP: everything below changes it. +AP_REENUM=yes echo "AP $AP_IF ($PHY) bssid $BSSID ch$CH" echo "DUT $DUT_SYSFS (MT7612U)" @@ -196,11 +216,10 @@ for _ in 1 2 3 4 5 6 7 8 9 10; do fi sleep 1 done -AP_REENUM=yes # hostapd may have run and left its bssid behind if [ "$ap_up" != yes ]; then echo "hostapd did not bring $AP_IF up in AP mode:" tail -12 "$OUT/hostapd.log" 2>/dev/null || echo "(no hostapd log written)" - exit 1 + exit 2 # the rig, not the DUT fi # --- free the DUT ---------------------------------------------------------- @@ -253,7 +272,7 @@ if ! { iw phy "$PHY" interface add staid_mon type monitor 2>/dev/null && ip link set staid_mon up 2>/dev/null; }; then echo "no monitor vif on $PHY: the BSSID gate would measure broadcast" echo "reception only, which is not the question - refusing to run it." - exit 1 + exit 2 # the rig, not the DUT fi # ORDER MATTERS. The gate's bring-up runs the MT7612U's calibrations, whose # MCU replies arrive late under a strong transmitter nearby (mcu.cpp); the @@ -261,7 +280,11 @@ fi # the injector only once the gate prints "bring-up done", and the gate then # pauses 3 s before arm A so the stimulus covers every arm. : > "$OUT/bssid.txt" -"$BUILD/mt7612uprobe" sta "$CH" "$SECS" "$BSSID" > "$OUT/bssid.txt" 2>&1 & +# BOUNDED: six arms of SECS, a 3 s pause, and 3 min for bring-up and slack. +# INT lets the gate restore the registers (exit 3); KILL 10 s later if not. +gate_bound=$(( SECS * 6 + 183 )) +timeout -s INT -k 10 "$gate_bound" \ + "$BUILD/mt7612uprobe" sta "$CH" "$SECS" "$BSSID" > "$OUT/bssid.txt" 2>&1 & sta_gate=$! sta_pid_record gate "$sta_gate" waited=0 @@ -284,6 +307,11 @@ fi wait "$sta_gate" r_bss=$? rm -f "$OUT/.pid_gate" +gate_overran=no +if [ "$r_bss" = 124 ] || [ "$r_bss" = 137 ]; then + echo "the BSSID gate overran its ${gate_bound}s bound - no measurement" + r_bss=2; gate_overran=yes +fi inj_secs=$(( $(date +%s) - inj_t0 )) cat "$OUT/bssid.txt" # Stop the injector (it prints its count on SIGTERM) and require that it @@ -304,7 +332,9 @@ if [ "${injected:-0}" -gt 0 ] 2>/dev/null; then else echo "the unicast injector injected NOTHING (see $OUT/inject.log) - the BSSID" echo "table measured broadcast reception only." - [ "$r_bss" = 3 ] || r_bss=1 # an interrupted gate stays "no verdict" + # An interrupted or overrun gate stays "no verdict": a bring-up that + # wedged before the injector started is not a failed measurement. + [ "$r_bss" = 3 ] || [ "$gate_overran" = yes ] || r_bss=1 fi echo @@ -321,4 +351,15 @@ case "$r_bss" in 3) echo "the BSSID gate was INTERRUPTED - no verdict" ;; *) echo "the BSSID gate did not pass - see $OUT/bssid.txt" ;; esac -exit $(( r_bss != 0 || r_ack != 0 || ${staid:-0} != 0 )) +# 1 if any gate failed, else 3 if any was interrupted, else 2 if any could +# not measure, else 0. +rc=0 +for r in "$r_bss" "$r_ack" "${staid:-0}"; do + case "$r" in + 0) ;; + 3) [ "$rc" = 1 ] || rc=3 ;; + 2) [ "$rc" = 0 ] && rc=2 ;; + *) rc=1 ;; + esac +done +exit "$rc" From d47302753cc6c155e90d9662135802947692f8da Mon Sep 17 00:00:00 2001 From: snokvist Date: Sat, 3 Oct 2026 10:52:43 +0200 Subject: [PATCH 14/18] tests: mt7612u_sta_onair - make the AP netdev ready before each hostapd cell_end returned as soon as the previous hostapd exited, and ap_up launched the next one at once. On one rig, one restart in four started 30 ms after AP-DISABLED and died with "Could not read interface flags: No such device" / "nl80211 driver initialization failed", which the cell scored INCONCLUSIVE. Before each launch, ap_up now spends up to 10 s, inside the netns, on these steps: - it waits for the netdev to be present; - once present, if it reports a type other than managed, it takes the interface down and sets type managed, retrying until the type reads managed; - it brings the interface up. Forcing the type matters because a driver may leave the vif in AP type after hostapd exits, and hostapd then fails with "Match already configured". realtek_station_onair.sh and mt7612u_sta_identity.sh force the type before every hostapd for the same reason. Nothing here is driver-specific, so it holds for an mt76x2u AP as well as rtw88. If the window runs out, the cell is refused as before (INCONCLUSIVE). The reason is written into the hostapd log that message points at, and the interface is left up. No hostapd retry: the wait removes the race. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01VNC8xhn1rNCi5t6uLvE6M3 --- tests/mt7612u_sta_onair.sh | 28 +++++++++++++++++++++++++++- 1 file changed, 27 insertions(+), 1 deletion(-) diff --git a/tests/mt7612u_sta_onair.sh b/tests/mt7612u_sta_onair.sh index ebb31fa9..f4bc009d 100755 --- a/tests/mt7612u_sta_onair.sh +++ b/tests/mt7612u_sta_onair.sh @@ -238,9 +238,35 @@ ap_up() { # $1 open | wpa2, $2 cell printf 'wpa_group_rekey=%s\nwpa_ptk_rekey=%s\n' "$REKEY_S" "$PTK_REKEY_S" fi } > "$OUT/hostapd_$2.conf" + # The previous cell's hostapd exiting is not its interface being back: a + # launch 30 ms after AP-DISABLED found the netdev gone ("Could not read + # interface flags: No such device" / "nl80211 driver initialization + # failed"). Wait, bounded, until the netdev is present, and FORCE it to a + # station once it is: a driver may leave the vif in AP type after hostapd + # exits, and hostapd then fails with "Match already configured" rather + # than anything that names the problem (tests/mt7612u_sta_identity.sh). + local t=0 info + while :; do + info=$(ip netns exec "$NS" iw dev "$AP_IF" info 2>/dev/null) + case "$info" in + *'type managed'*) break ;; + '') ;; # not back yet + *) ip netns exec "$NS" ip link set "$AP_IF" down 2>/dev/null + ip netns exec "$NS" iw dev "$AP_IF" set type managed 2>/dev/null ;; + esac + if [ "$t" -ge 100 ]; then + echo "rig: $AP_IF not back as a managed netdev in netns $NS within 10 s" \ + "- hostapd not started" | tee "$OUT/hostapd_$2.log" + # The loop may just have taken it down; leave it up (best effort). + ip netns exec "$NS" ip link set "$AP_IF" up 2>/dev/null + return 1 + fi + sleep 0.1; t=$((t + 1)) + done + ip netns exec "$NS" ip link set "$AP_IF" up 2>/dev/null ip netns exec "$NS" hostapd -t "$OUT/hostapd_$2.conf" > "$OUT/hostapd_$2.log" 2>&1 & sta_pid_record hostapd $! - local t=0 + t=0 until ip netns exec "$NS" iw dev "$AP_IF" info 2>/dev/null | grep -q 'type AP'; do [ "$t" -ge 15 ] && return 1 sleep 1; t=$((t + 1)) From 6b31d5f6dcc0fed815eef72451c5851c50662b45 Mon Sep 17 00:00:00 2001 From: snokvist Date: Sat, 3 Oct 2026 10:52:43 +0200 Subject: [PATCH 15/18] txdemo: join the RX and USB event threads on every exit usb_thread and rx_thread were joined only at the end of main. The early returns after them - the three `return 1`s on the TX path - destroyed them still joinable, so the process ended in std::terminate. It is the same class as the IN drainers fixed earlier on this branch. A scope guard, IoThreadsJoin, declared right after rx_thread, runs the normal teardown's sequence on every exit: 1. StopRxLoop. 2. Join the RX thread. 3. Set g_devourer_should_stop. 4. Join the event pump. The normal path calls it explicitly where the joins were, before Stop(), so the order against Stop(), the drainers' join and session.close() is unchanged. On an early return it runs before the session releases the device. Both threads start after the DEVOURER_TX_WITH_RX fork, whose child returns first, so no fork-child skip is needed. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01VNC8xhn1rNCi5t6uLvE6M3 --- examples/tx/main.cpp | 30 +++++++++++++++++++++++++----- 1 file changed, 25 insertions(+), 5 deletions(-) diff --git a/examples/tx/main.cpp b/examples/tx/main.cpp index 9c8d63a6..d380cbcc 100644 --- a/examples/tx/main.cpp +++ b/examples/tx/main.cpp @@ -1090,6 +1090,30 @@ int main(int argc, char **argv) { }); logger->info("DEVOURER_TX_WITH_RX=thread: RX loop started alongside TX"); } + /* Stops and joins the RX and USB event threads on every exit from here, + * the early returns below included - a joinable std::thread destructor + * terminates the process. The normal teardown's order: StopRxLoop and the + * RX join, then the event pump (which polls g_devourer_should_stop). It is + * declared after the session, so on an early return it runs while the + * device and libusb are still alive. Both threads start after the + * DEVOURER_TX_WITH_RX fork, so a fork child never reaches here. */ + struct IoThreadsJoin { + IRadio *dev; + std::thread &rx, &usb; + bool done = false; + void join() { + if (done) + return; + done = true; + dev->StopRxLoop(); + if (rx.joinable()) + rx.join(); + g_devourer_should_stop = true; + if (usb.joinable()) + usb.join(); + } + ~IoThreadsJoin() { join(); } + } io_threads{rtlDevice, rx_thread, usb_thread}; /* DEVOURER_STA_IDENTITY: arm before the first frame, once the RX loop is * shown running - its first received frame (3 s cap: a silent channel still @@ -2677,11 +2701,7 @@ int main(int argc, char **argv) { sta_stop = true; if (sta_clear_thread.joinable()) sta_clear_thread.join(); - rtlDevice->StopRxLoop(); - if (rx_thread.joinable()) - rx_thread.join(); - if (usb_thread.joinable()) - usb_thread.join(); + io_threads.join(); /* Clean chip de-init before releasing the interface: card-disable PWR_SEQ on * the HalMAC families, TX quiesce on Jaguar1 — so the adapter re-enumerates From fcdec63db3a1d7d2f92c7d34f927e298060d12fa Mon Sep 17 00:00:00 2001 From: snokvist Date: Sat, 3 Oct 2026 11:55:00 +0200 Subject: [PATCH 16/18] tests: realtek_station_onair - date the aired span from the first submit The floor's aired_ms started at the arm's FIRST tx.report. A transmitter that started, stalled, and then burst 50 reports in its last second met MIN_REPORTS and a ~50-frame floor, where the old fixed 500 would have refused it (qodo, PR #466). txdemo's tx.frame now carries t, the same host-monotonic timebase as tx.report and the final tx.stats. The first tx.frame follows the first submit, so a harness can date it. docs/logging.md lists the field. Existing consumers match on "n" and "rc" or parse JSON, so the extra trailing field changes nothing for them. summarize uses that first-submit time in two places: - aired_ms now runs from the first submit to the final tx.stats. - Liveness gains lead_ms, the silence from the first submit to the first report. live=0 when it exceeds MAX_GAP_MS, or when no tx.frame carries t. The bring-up stays outside the window either way, because the first submit follows InitWrite. Synthetic JSONL through the script's own summarize/usable: | Fixture | Result | |---|---| | slow bring-up, 327 frames in 2.0 s | usable, floor 101 | | start, stall, 60 reports in the last second | refused: live=0 on lead_ms 8800, and 60 < floor 490 | | rate stall | refused, floor 500 | | hard stall | refused, live=0 | | healthy full window | usable | | no tx.frame t | refused, live=0 | Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01VNC8xhn1rNCi5t6uLvE6M3 --- docs/logging.md | 2 +- examples/tx/main.cpp | 3 +- tests/realtek_station_onair.sh | 57 ++++++++++++++++++++-------------- 3 files changed, 37 insertions(+), 25 deletions(-) diff --git a/docs/logging.md b/docs/logging.md index cfc43f83..82ffcdbf 100644 --- a/docs/logging.md +++ b/docs/logging.md @@ -110,7 +110,7 @@ Emitters: L = library, RX/TX/... = demo. Optional fields in [brackets]; ### TX plane | ev | emitter | fields | |---|---|---| -| `tx.frame` | TX | n, rc — precoder demo variant: n, ok | +| `tx.frame` | TX | n, rc, t (the `tx.report` timebase: the first one dates the first submit) — precoder demo variant: n, ok | | `tx.stats` | TX | submitted, failed, was_timeout, last_rc; the `final:1` event also carries t (the `tx.report` timebase, so a harness can tell how long before the end a transmitter last reported); periodic events (not the `final:1` one) also carry `txdma_status` (the raw `REG_TXDMA_STATUS` latch; which bits mean a stopped transmitter: `IRtlRadio::GetTxDmaStatus`) where `IRtlRadio::HasTxDmaStatus()` (which backends: its declaration), or `txdma_read_failed:1` when that sample's register read failed | | `tx.agg` | L (`DEVOURER_TX_USB_AGG`, send_packets) | frames, bytes, shim, ok — one per multi-frame bulk-OUT URB. The sync-TX generations (Jaguar2/Jaguar3/RTL8733B) also emit `sent` — bytes actually transferred, OR the negative libusb rc on a transport error (deliberately raw: this event is the only machine-readable carrier of the aggregated-path error code) — and set `ok` only on a FULL write, so `ok=false` splits as `sent < 0` transport error vs `0 <= sent < bytes` short write. Jaguar1 TX is async: its `ok` means URB accepted by the transport and there is no `sent` field (bytes resolve at completion reaping) | | `tx.report` | L (`DEVOURER_TX_REPORT`, CCX decode) | t, state (0=delivered, 1=retry-drop), ok, retries, final_rate, queue_time_raw, bmc, macid, fmt ("8812"\|"halmac"); halmac adds tag (SW_DEFINE echo), rts_retries, missed (fw-stuffed constant on Jaguar3 — tag gaps are the drop signal; `tests/txrpt_coverage_attrib.py`) — t is the achieved-report-rate timebase (the CCX emission ceiling is reports/s) | diff --git a/examples/tx/main.cpp b/examples/tx/main.cpp index d380cbcc..43ebfd29 100644 --- a/examples/tx/main.cpp +++ b/examples/tx/main.cpp @@ -2525,7 +2525,8 @@ int main(int argc, char **argv) { ++frames_in_dwell >= hop_dwell) frames_in_dwell = 0; if (tx_count <= 10 || tx_count % 500 == 0) { - devourer::Ev(*g_ev, "tx.frame").f("n", tx_count).f("rc", rc); + /* t: the tx.report timebase, so a harness can date the first submit. */ + devourer::Ev(*g_ev, "tx.frame").f("n", tx_count).f("rc", rc).t(); /* TX submission health — the driver-drop / congestion feed (xtx). A * climbing failed with was_timeout=1 is a full TX FIFO (recoverable * back-pressure); a hard rc is a broken path. */ diff --git a/tests/realtek_station_onair.sh b/tests/realtek_station_onair.sh index ef3bd00b..3da2613c 100755 --- a/tests/realtek_station_onair.sh +++ b/tests/realtek_station_onair.sh @@ -44,8 +44,8 @@ # station (the AP then deauths the "class 3" sender - expected, harmless). # # An arm is scored only when its transmitter kept airing through the window -# (no silence over MAX_GAP_MS between its reports, or after the last one - -# see summarize) and carried MIN_REPORTS reports and its submission floor +# (no silence over MAX_GAP_MS before its first report, between its +# reports, or after the last one - see summarize) and carried MIN_REPORTS reports and its submission floor # (MIN_SUBMITTED, scaled to the span it aired). Reception in arm A is # judged against the peer's REPORTED frames, which aired; submitted frames # left unreported at window close are printed separately. @@ -88,12 +88,13 @@ GAP_US="${GAP_US:-5000}" HALF="${HALF:-both}" MIN_REPORTS="${MIN_REPORTS:-50}" MIN_RX_PCT="${MIN_RX_PCT:-80}" -# Transmitter liveness: the longest silence allowed between two of an arm's -# CCX reports, and between its last report and its final tx.stats. +# Transmitter liveness: the longest silence allowed from an arm's first +# submit to its first CCX report, between two reports, and from its last +# report to its final tx.stats. MAX_GAP_MS="${MAX_GAP_MS:-2000}" # Per-arm floor on frames the transmitter submitted. Default: a quarter of # the nominal GAP_US rate over the span the arm actually aired - its first -# report to its final tx.stats (see summarize) - not over SECS, which also +# submit to its final tx.stats (see summarize) - not over SECS, which also # holds the transmitter's bring-up: an 8812BU peer has spent 6-9 s of a 10 s # window in InitWrite and been scored a stall at 283-327 healthy frames. # The slowest arm on record, an unacknowledged one at 12 retries, submitted @@ -275,23 +276,26 @@ wait_for() { # $1 pid, $2 file, $3 regex, $4 timeout s # One summary line from a transmitter's JSONL (and, optionally, the DUT's # rx.seq stream for frames from TA): reports, submitted, unreported, -# ok_pct, retries_mean, max_gap_ms, tail_ms, live, aired_ms, min_submitted, -# rx_distinct. +# ok_pct, retries_mean, lead_ms, max_gap_ms, tail_ms, live, aired_ms, +# min_submitted, rx_distinct. +# +# One clock: tx.report, the first tx.frame (txdemo's first submit) and the +# final tx.stats all carry t in the host-monotonic tx.report timebase. # # LIVENESS. A report is a frame that aired, so the reports' own timestamps -# (t, the host-monotonic tx.report timebase) show whether the transmitter -# kept airing through the window: max_gap_ms is the longest silence between -# two reports, tail_ms the silence from the last report to the final -# tx.stats, which carries t in the same timebase. live=0 when either exceeds -# MAX_GAP_MS or the final tx.stats has no t - an arm that aired a burst and -# stalled, which MIN_REPORTS alone would accept. +# show whether the transmitter kept airing through the window: lead_ms is +# the silence from the first submit to the first report, max_gap_ms the +# longest silence between two reports, tail_ms the silence from the last +# report to the final tx.stats. live=0 when any of them exceeds MAX_GAP_MS, +# or a timestamp it needs is missing - an arm that aired a burst and +# stalled, or that started, stalled and burst at the end, which MIN_REPORTS +# alone would accept. # -# FLOOR. aired_ms runs from the first report to the final tx.stats: the -# first pair of timestamps in one timebase (txdemo.first_tx_submit carries -# ms from a different epoch). It ends at the final tx.stats, not the last -# report, so a transmitter that slows or stops after MIN_REPORTS still owes -# the whole span. min_submitted is a quarter of the GAP_US rate over it -# (0 at GAP_US=0, which has no nominal rate), or MIN_SUBMITTED when set. +# FLOOR. aired_ms runs from the first submit to the final tx.stats - not +# over SECS, which also holds the transmitter's bring-up, and not from the +# first report, which a late burst would move to the end. min_submitted +# is a quarter of the GAP_US rate over it (0 at GAP_US=0, which has no +# nominal rate), or MIN_SUBMITTED when set. summarize() { # $1 tx jsonl, $2 tag, $3 dut jsonl or "" python3 - "$1" "$2" "${3:-}" "$MAX_GAP_MS" "$GAP_US" "$MIN_SUBMITTED" <<'PYEOF' import json, sys @@ -301,6 +305,7 @@ n = okc = retries = 0 submitted = 0 ts = [] final_t = None +first_submit_t = None for line in open(tx, errors='replace'): if not line.startswith('{'): continue @@ -314,20 +319,26 @@ for line in open(tx, errors='replace'): retries += int(e.get('retries', 0) or 0) if 't' in e: ts.append(int(e['t'])) + elif e.get('ev') == 'tx.frame': + if first_submit_t is None and 't' in e: + first_submit_t = int(e['t']) elif e.get('ev') == 'tx.stats': submitted = int(e.get('submitted', submitted) or submitted) if e.get('final') and 't' in e: final_t = int(e['t']) gap = max((b - a for a, b in zip(ts, ts[1:])), default=0) tail = (final_t - ts[-1]) if (final_t is not None and ts) else None -live = int(bool(ts) and tail is not None and gap <= max_gap - and tail <= max_gap) +lead = (ts[0] - first_submit_t) if (first_submit_t is not None and ts) else None +live = int(bool(ts) and tail is not None and lead is not None + and lead <= max_gap and gap <= max_gap and tail <= max_gap) out = (f"{tag} reports={n} submitted={submitted} " f"unreported={max(submitted - n, 0)}") if n: out += f" ok_pct={100.0*okc/n:.1f} retries_mean={retries/n:.2f}" -out += f" max_gap_ms={gap} tail_ms={'none' if tail is None else tail} live={live}" -aired = (final_t - ts[0]) if (final_t is not None and ts) else 0 +out += (f" lead_ms={'none' if lead is None else lead} max_gap_ms={gap}" + f" tail_ms={'none' if tail is None else tail} live={live}") +aired = (final_t - first_submit_t) \ + if (final_t is not None and first_submit_t is not None) else 0 if fixed_floor: floor = int(fixed_floor) else: From 92dfa1d0e5ae82e110089c296f52333ffd42610f Mon Sep 17 00:00:00 2001 From: snokvist Date: Sat, 3 Oct 2026 11:55:47 +0200 Subject: [PATCH 17/18] tests: mt7612u_ap_onair - stop the run when the AP cannot be reset When reap() had to KILL a child, cleanup rightly skipped the authorized toggle. Under CELLS=all, though, that cleanup also runs between cells, and the run carried on into the next cell on an adapter that was never reset and whose autonomous beacon could still be airing (qodo, PR #466). cleanup now returns 1 when it could not reset the AP. In the CELLS=all sequence, the cells after such a cleanup are not run; they are listed as NOT RUN, and the run exits 2 (INCONCLUSIVE) unless a check already failed (1). It never scores those cells. A single-cell run is unchanged. The exit status is otherwise as before: 0, or 1 on a failed check. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01VNC8xhn1rNCi5t6uLvE6M3 --- tests/mt7612u_ap_onair.sh | 23 ++++++++++++++++++++--- 1 file changed, 20 insertions(+), 3 deletions(-) diff --git a/tests/mt7612u_ap_onair.sh b/tests/mt7612u_ap_onair.sh index 9a51ed8e..2289c157 100755 --- a/tests/mt7612u_ap_onair.sh +++ b/tests/mt7612u_ap_onair.sh @@ -64,7 +64,7 @@ KIDS="" # TERM every child, then WAIT for them to exit - up to 10 s: a demo's chip # de-init runs after the signal, and re-enumerating the adapter under it is # the hand-back this must not do. Anything still alive then is KILLed and -# reaped is 1, so cleanup leaves the adapter alone; 0 when all exited. +# reap returns 1, so cleanup leaves the adapter alone; 0 when all exited. reap() { local pid live t=0 for pid in $KIDS; do kill "$pid" 2>/dev/null; done @@ -85,6 +85,9 @@ reap() { return 0 } +# Returns 1 when it could not reset the AP (a process outlived TERM): the +# adapter may still be airing an autonomous beacon, so no later cell may be +# scored against it. cleanup() { local reaped=0 reap || reaped=1 @@ -125,6 +128,7 @@ cleanup() { sleep 3 fi fi + return "$reaped" } # An interrupt must STOP the run: on the EXIT trap alone, INT/TERM would run # cleanup and then carry on into the next cell against a re-enumerated @@ -336,10 +340,23 @@ case "$CELLS" in open) cell_open ;; wpa2) cell_wpa2 ;; stop) cell_stop ;; - all) cell_open; cleanup; cell_wpa2; cleanup; cell_stop ;; + all) # A between-cell cleanup that could not reset the AP ends the run: + # the cells after it are recorded as not run, never scored. + cell_open + if ! cleanup; then not_run="wpa2 stop" + else + cell_wpa2 + if ! cleanup; then not_run="stop"; else cell_stop; fi + fi ;; *) echo "usage: $0 [open|wpa2|stop|all]"; exit 2 ;; esac say "" +if [ -n "${not_run:-}" ]; then + say " NOT RUN $not_run - the AP could not be reset between cells" +fi say "=== $pass passed, $fail failed (logs: $OUT) ===" -exit $(( fail > 0 )) +# 1 a check failed; else 2 when cells were left unrun (INCONCLUSIVE); else 0. +[ "$fail" -gt 0 ] && exit 1 +[ -n "${not_run:-}" ] && exit 2 +exit 0 From 18f0a200f7ad77e1346b3622fc9909437c0a53d3 Mon Sep 17 00:00:00 2001 From: snokvist Date: Sat, 3 Oct 2026 11:56:26 +0200 Subject: [PATCH 18/18] txdemo: the TX_WITH_RX fork child leaves through std::_Exit The legacy DEVOURER_TX_WITH_RX fork child returned from main holding copies of the parent's objects, including the IN drainers' std::threads. Those are joinable copies of threads that do not exist in the child, so ~std::thread terminated it, and the DrainerJoin guard's in_fork_child skip could not prevent that (qodo, PR #466). The child now follows the standard post-fork rule: - It runs Init inside a try, so an exception does not unwind it either. - It flushes stdio. - It leaves through std::_Exit(1), so no destructor runs in the child. That is the child's only exit path. The in_fork_child flag is gone, since the child never reaches the guard's destructor. On MSVC fork() is the (0) stub, which makes that branch the only process, so it keeps its normal return and teardown. Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_01VNC8xhn1rNCi5t6uLvE6M3 --- examples/tx/main.cpp | 35 ++++++++++++++++++++++++++--------- 1 file changed, 26 insertions(+), 9 deletions(-) diff --git a/examples/tx/main.cpp b/examples/tx/main.cpp index 43ebfd29..636d9f43 100644 --- a/examples/tx/main.cpp +++ b/examples/tx/main.cpp @@ -2,6 +2,7 @@ #include #include #include +#include #include #include #include @@ -786,12 +787,11 @@ int main(int argc, char **argv) { /* Stops and joins the optional IN drainers on every exit from here: a * still-joinable std::thread terminates the process when destroyed, and * both threads poll `handle`, so the normal path joins them explicitly - * before session.close(). A fork() child has copies of the thread objects - * but not the threads, so it must not join them (in_fork_child). */ + * before session.close(). (The DEVOURER_TX_WITH_RX fork child never runs + * this: it leaves through std::_Exit - see there.) */ struct DrainerJoin { std::atomic &bulk_running, &intr_running; std::thread &bulk, &intr; - bool in_fork_child = false; void join() { bulk_running = false; intr_running = false; @@ -800,10 +800,7 @@ int main(int argc, char **argv) { if (intr.joinable()) intr.join(); } - ~DrainerJoin() { - if (!in_fork_child) - join(); - } + ~DrainerJoin() { join(); } } drainers{bulk_in_running, intr_running, bulk_in_thread, intr_in_thread}; WiFiDriver wifi_driver{logger}; @@ -1030,8 +1027,27 @@ int main(int argc, char **argv) { pid_t fpid = fork(); if (fpid == 0) { #if !defined(_MSC_VER) /* fork() is a real fork here, not the (0) stub */ - drainers.in_fork_child = true; -#endif + /* The post-fork rule: the child holds copies of the parent's objects - + * the IN-drainer std::threads among them, joinable copies of threads + * that do not exist here - so it must run no destructor. It flushes + * stdio and leaves through std::_Exit, never through a return; an + * exception from Init must not unwind it either. */ + try { + rtlDevice->Init(packetProcessor, + SelectedChannel{ + .Channel = static_cast(channel), + .ChannelOffset = 0, + .ChannelWidth = CHANNEL_WIDTH_20, + }); + } catch (const std::exception &e) { + logger->error("RX child: {}", e.what()); + } catch (...) { + logger->error("RX child: unknown exception"); + } + std::fflush(nullptr); + std::_Exit(1); +#else + /* The stub: this IS the only process, so it tears down normally. */ rtlDevice->Init(packetProcessor, SelectedChannel{ .Channel = static_cast(channel), @@ -1039,6 +1055,7 @@ int main(int argc, char **argv) { .ChannelWidth = CHANNEL_WIDTH_20, }); return 1; +#endif } }