@alteriom/painlessmesh 1.9.20 → 1.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -73,29 +73,40 @@ struct QueueStats {
73
73
  typedef std::function<void(QueueState state, uint32_t messageCount)> queueStateChangedCallback_t;
74
74
 
75
75
  /**
76
- * Message Queue for offline/Internet-unavailable mode
77
- *
76
+ * Manual priority buffer for offline / Internet-unavailable mode
77
+ *
78
+ * This is a *storage* container only. It does NOT transmit anything on
79
+ * its own: painlessMesh will not enqueue, flush, or deliver messages
80
+ * automatically, and there is no connectivity-restored trigger inside
81
+ * the library. The application owns the send loop — enqueue when your
82
+ * upstream (MQTT/HTTP/etc.) is down, drain via `getMessages()` /
83
+ * `mesh.flushMessageQueue()` and `remove()` / `mesh.removeQueuedMessage()`
84
+ * when your own connectivity check tells you it is safe to resend.
85
+ *
78
86
  * Provides priority-based message queueing with support for:
79
- * - Priority levels (CRITICAL messages never dropped)
87
+ * - Priority levels (CRITICAL messages evicted last, but may still be dropped
88
+ * if the queue is full and all entries are CRITICAL)
80
89
  * - Queue size limits with intelligent eviction
81
90
  * - Statistics tracking
82
91
  * - State change callbacks
83
- *
92
+ *
84
93
  * Example usage:
85
94
  * \code
86
95
  * MessageQueue queue(1000); // Max 1000 messages
87
- *
88
- * // Queue a critical message
96
+ *
97
+ * // Queue a critical message while upstream is offline
89
98
  * uint32_t msgId = queue.enqueue(PRIORITY_CRITICAL, "alarm_data", "mqtt://...");
90
- *
99
+ *
91
100
  * // Get queue size
92
101
  * uint32_t size = queue.size();
93
- *
94
- * // Get all messages (for sending)
95
- * std::vector<QueuedMessage> messages = queue.getMessages();
96
- *
97
- * // Remove sent message
98
- * queue.remove(msgId);
102
+ *
103
+ * // When your app detects the upstream is back, drain manually:
104
+ * auto messages = queue.getMessages();
105
+ * for (auto& msg : messages) {
106
+ * if (sendToCloud(msg.payload, msg.destination)) {
107
+ * queue.remove(msg.id);
108
+ * }
109
+ * }
99
110
  * \endcode
100
111
  */
101
112
  class MessageQueue {
@@ -173,12 +173,34 @@ class BridgeCoordinationPackage : public plugin::BroadcastPackage {
173
173
  template <typename T>
174
174
  class PackageHandler : public layout::Layout<T> {
175
175
  public:
176
- void stop() {
177
- for (auto&& task : taskList) {
178
- task->disable();
179
- task->setCallback(NULL);
176
+ // NOTE: scheduler is optional (defaults to nullptr) for backward
177
+ // compatibility with existing call sites; without it we can't detect
178
+ // the "current" task and behaviour falls back to the original one.
179
+ // Passing the scheduler (mesh.hpp already has it as mScheduler) avoids
180
+ // the use-after-free described below.
181
+ void stop(Scheduler* scheduler = nullptr) {
182
+ // If stop() is called from within the callback of one of the tasks in
183
+ // taskList (e.g. promoteToBridge()), that task is
184
+ // "scheduler->getCurrentTask()" at this exact moment. Disabling it,
185
+ // clearing its callback, or letting its shared_ptr refcount drop to
186
+ // zero here would destroy its closure - which is still on the stack,
187
+ // in the middle of its own execution - causing a use-after-free crash
188
+ // (same bug family as upstream issue #373, but triggered by the
189
+ // shared_ptr refcount instead of a raw delete in onDisable).
190
+ // We simply leave it in the list: addTask() will recognise it as
191
+ // "disabled with a single reference" and reuse it on the next call,
192
+ // exactly like it already does for disabled anonymous tasks (see the
193
+ // comment on addTask() above).
194
+ Task* current = scheduler ? scheduler->getCurrentTask() : nullptr;
195
+ for (auto it = taskList.begin(); it != taskList.end();) {
196
+ if (current != nullptr && it->get() == current) {
197
+ ++it;
198
+ continue;
199
+ }
200
+ (*it)->disable();
201
+ (*it)->setCallback(NULL);
202
+ it = taskList.erase(it);
180
203
  }
181
- taskList.clear();
182
204
  callbackList.clear();
183
205
  }
184
206
 
@@ -30,6 +30,84 @@ static const uint32_t TCP_EXHAUSTION_RECONNECT_DELAY_MS = 10000; // 10 seconds b
30
30
  // This prevents repeatedly trying to connect to nodes with unresponsive TCP servers
31
31
  static const uint32_t TCP_FAILURE_BLOCK_DURATION_MS = 60000;
32
32
 
33
+ // Safe operating bounds applied by clampTcpRetryConfig().
34
+ // Each retry allocates a new AsyncClient and schedules a task, so an unbounded
35
+ // maxRetries is a heap and recursion-depth hazard on ESP8266. A zero retry
36
+ // delay would schedule retry tasks with no spacing, i.e. a hot loop allocating
37
+ // an AsyncClient per scheduler tick.
38
+ static const uint8_t TCP_RETRY_MAX_RETRIES_LIMIT = 10;
39
+ static const uint32_t TCP_RETRY_MIN_DELAY_MS = 50;
40
+ static const uint32_t TCP_RETRY_MAX_DELAY_MS = 60000;
41
+
42
+ /**
43
+ * User-tunable TCP connection retry parameters
44
+ *
45
+ * One instance is stored per mesh (painlessmesh::Mesh), configured through
46
+ * Mesh::setTcpRetryConfig(). The member defaults are spelled as the legacy
47
+ * constants above rather than as literals, so the "defaults match the previous
48
+ * hardcoded behaviour exactly" guarantee is enforced by the compiler and cannot
49
+ * silently drift.
50
+ *
51
+ * NOTE: this deliberately lives on the mesh instance rather than at namespace
52
+ * scope. A mutable namespace-scope instance in this header would give every
53
+ * translation unit its own private copy, so a sketch that configured the mesh
54
+ * in one TU and connected from another would silently fall back to defaults
55
+ * with no diagnostic.
56
+ */
57
+ struct TcpRetryConfig {
58
+ /// Max TCP connect retry attempts before falling back to a WiFi reconnect
59
+ uint8_t maxRetries = TCP_CONNECT_MAX_RETRIES;
60
+ /// Base delay between retry attempts (ms), scaled by exponential backoff
61
+ uint32_t retryDelayMs = TCP_CONNECT_RETRY_DELAY_MS;
62
+ /// Delay after IP acquisition before the TCP connect is attempted (ms)
63
+ uint32_t stabilizationDelayMs = TCP_CONNECT_STABILIZATION_DELAY_MS;
64
+ /// Delay before the WiFi reconnect that follows retry exhaustion (ms)
65
+ uint32_t exhaustionReconnectDelayMs = TCP_EXHAUSTION_RECONNECT_DELAY_MS;
66
+ /// Duration a node is blocklisted after retry exhaustion (ms, 0 = never)
67
+ uint32_t failureBlockDurationMs = TCP_FAILURE_BLOCK_DURATION_MS;
68
+ };
69
+
70
+ /**
71
+ * Delay before retry attempt number `retryCount`
72
+ *
73
+ * Exponential backoff with the multiplier capped at 8, giving the default
74
+ * sequence 1s, 2s, 4s, 8s, 8s. Capping prevents excessive delays and keeps the
75
+ * product clear of uint32_t overflow.
76
+ *
77
+ * @param cfg Active retry configuration
78
+ * @param retryCount Zero-based retry attempt number
79
+ * @return Delay in milliseconds
80
+ */
81
+ inline uint32_t retryBackoffDelay(const TcpRetryConfig &cfg,
82
+ uint8_t retryCount) {
83
+ uint8_t backoffMultiplier = (retryCount < 3) ? (1U << retryCount) : 8;
84
+ return cfg.retryDelayMs * backoffMultiplier;
85
+ }
86
+
87
+ /**
88
+ * Coerce a retry configuration into safe operating bounds
89
+ *
90
+ * Only the two values that can render a node unusable are clamped:
91
+ * maxRetries (heap/recursion pressure) and retryDelayMs (hot-loop floor and
92
+ * overflow ceiling). maxRetries == 0 is explicitly allowed - it means "fall
93
+ * back to a WiFi reconnect on the first TCP error", which is the whole point
94
+ * of the low-latency profile. stabilizationDelayMs, exhaustionReconnectDelayMs
95
+ * and failureBlockDurationMs are left alone because 0 is meaningful for each
96
+ * (skip stabilization / reconnect immediately / never blocklist).
97
+ *
98
+ * @param cfg Configuration to clamp (taken by value, returned clamped)
99
+ * @return The clamped configuration
100
+ */
101
+ inline TcpRetryConfig clampTcpRetryConfig(TcpRetryConfig cfg) {
102
+ if (cfg.maxRetries > TCP_RETRY_MAX_RETRIES_LIMIT)
103
+ cfg.maxRetries = TCP_RETRY_MAX_RETRIES_LIMIT;
104
+ if (cfg.retryDelayMs < TCP_RETRY_MIN_DELAY_MS)
105
+ cfg.retryDelayMs = TCP_RETRY_MIN_DELAY_MS;
106
+ if (cfg.retryDelayMs > TCP_RETRY_MAX_DELAY_MS)
107
+ cfg.retryDelayMs = TCP_RETRY_MAX_DELAY_MS;
108
+ return cfg;
109
+ }
110
+
33
111
  inline uint32_t encodeNodeId(const uint8_t *hwaddr) {
34
112
  using namespace painlessmesh::logger;
35
113
  Log(GENERAL, "encodeNodeId():\n");
@@ -93,8 +171,9 @@ void initServer(AsyncServer &server, M &mesh) {
93
171
  *
94
172
  * This function attempts to connect to the mesh network via TCP.
95
173
  * If the connection fails (error -14 ERR_CONN or other errors), it will
96
- * retry up to TCP_CONNECT_MAX_RETRIES times before triggering a full
97
- * WiFi reconnection cycle.
174
+ * retry up to the configured maxRetries times before triggering a full
175
+ * WiFi reconnection cycle. See Mesh::setTcpRetryConfig() to tune the retry
176
+ * envelope; the defaults reproduce the historic hardcoded behaviour.
98
177
  *
99
178
  * The retry mechanism helps handle timing issues where:
100
179
  * - The TCP server may not be immediately ready after AP initialization
@@ -117,16 +196,49 @@ template <class T, class M>
117
196
  void connect(AsyncClient &client, IPAddress ip, uint16_t port, M &mesh,
118
197
  uint8_t retryCount = 0) {
119
198
  using namespace logger;
120
-
199
+
200
+ // Snapshot the retry configuration once, before any lambda is built. The
201
+ // onError lambda outlives this stack frame, so it captures this snapshot by
202
+ // value rather than reading mesh's member later: a user callback could
203
+ // otherwise mutate the config between the connect attempt and the error,
204
+ // tearing the backoff schedule mid-sequence. Each retry re-enters connect()
205
+ // and re-snapshots, so a config change still takes effect on the next
206
+ // attempt.
207
+ const TcpRetryConfig cfg = mesh.getTcpRetryConfig();
208
+
121
209
  Log(CONNECTION, "tcp::connect(): Attempting connection to port %d (attempt %d/%d)\n",
122
- port, retryCount + 1, TCP_CONNECT_MAX_RETRIES + 1);
210
+ port, retryCount + 1, cfg.maxRetries + 1);
211
+
212
+ // Guard shared between onError and onConnect: under normal conditions a
213
+ // TCP connection attempt leads to only ONE of the two outcomes. But if
214
+ // WiFi drops right as the TCP handshake completes, AsyncTCP can queue
215
+ // BOTH events (connection succeeded at the TCP level + abort due to the
216
+ // WiFi link being lost) for the same AsyncClient, and both callbacks end
217
+ // up firing for the SAME object. Without this guard, the client would be
218
+ // "handed over" twice: once to a BufferedConnection (via onConnect, which
219
+ // becomes its owner and will delete it via its own destructor) and once
220
+ // to the retry path (via onError, which schedules it again for deletion)
221
+ // - two scheduleAsyncClientDeletion() calls on the same pointer, and
222
+ // therefore a double deferred deletion on the same object once both
223
+ // tasks fire.
224
+ auto claimed = std::make_shared<bool>(false);
123
225
 
124
226
  // Store retry count and connection parameters for the error handler
125
227
  // We need to capture these by value since they're used in the lambda
126
- client.onError([&mesh, ip, port, retryCount](void *, AsyncClient *client, int8_t err) {
228
+ client.onError([&mesh, ip, port, retryCount, claimed, cfg](void *, AsyncClient *client, int8_t err) {
229
+ if (*claimed) {
230
+ // onConnect has already claimed this client (the connection actually
231
+ // succeeded, and it's already been wrapped in a BufferedConnection
232
+ // that owns it): don't touch the same object again.
233
+ Log(CONNECTION,
234
+ "tcp_err(): onError fired after onConnect had already claimed "
235
+ "the client - ignored to avoid double handling\n");
236
+ return;
237
+ }
238
+ *claimed = true;
127
239
  if (mesh.semaphoreTake()) {
128
- Log(CONNECTION, "tcp_err(): error trying to connect %d (attempt %d/%d)\n",
129
- err, retryCount + 1, TCP_CONNECT_MAX_RETRIES + 1);
240
+ Log(CONNECTION, "tcp_err(): error trying to connect %d (attempt %d/%d)\n",
241
+ err, retryCount + 1, cfg.maxRetries + 1);
130
242
 
131
243
  // Check if we have retries left - retry logic only works on real hardware
132
244
  // In test environment (PAINLESSMESH_BOOST), fall through to dropped connection
@@ -134,27 +246,21 @@ void connect(AsyncClient &client, IPAddress ip, uint16_t port, M &mesh,
134
246
  (void)ip;
135
247
  (void)port;
136
248
  #if !defined(PAINLESSMESH_BOOST) && (defined(ESP32) || defined(ESP8266))
137
- if (retryCount < TCP_CONNECT_MAX_RETRIES) {
138
- // Calculate delay with exponential backoff: base_delay * 2^retryCount
139
- // This gives increasing time between retries as failures accumulate:
140
- // - retryCount=0: 1000ms * 1 = 1s
141
- // - retryCount=1: 1000ms * 2 = 2s
142
- // - retryCount=2: 1000ms * 4 = 4s
143
- // - retryCount=3: 1000ms * 8 = 8s (capped at 8)
144
- // - retryCount=4: 1000ms * 8 = 8s (capped at 8)
145
- // Cap multiplier at 8 to prevent excessive delays
146
- uint8_t backoffMultiplier = (retryCount < 3) ? (1U << retryCount) : 8;
147
- uint32_t retryDelay = TCP_CONNECT_RETRY_DELAY_MS * backoffMultiplier;
148
-
149
- Log(CONNECTION, "tcp_err(): Scheduling retry in %u ms (backoff x%d)\n",
150
- retryDelay, backoffMultiplier);
249
+ if (retryCount < cfg.maxRetries) {
250
+ // Delay grows with exponential backoff, multiplier capped at 8.
251
+ // With the default 1000ms base that is: 1s, 2s, 4s, 8s, 8s.
252
+ uint32_t retryDelay = retryBackoffDelay(cfg, retryCount);
253
+
254
+ Log(CONNECTION, "tcp_err(): Scheduling retry in %u ms\n", retryDelay);
151
255
 
152
256
  // Schedule a retry after a delay using the mesh's task scheduler
153
257
  // Note: &mesh is captured by reference because:
154
258
  // 1. Mesh is a singleton that lives for the program's lifetime
155
259
  // 2. The task scheduler belongs to the mesh, so mesh is always valid when task runs
156
260
  // 3. Copying the mesh object is not possible/allowed
157
- // Recursion depth is strictly bounded by TCP_CONNECT_MAX_RETRIES (default: 5)
261
+ // Recursion depth is strictly bounded by cfg.maxRetries (default 5),
262
+ // which clampTcpRetryConfig() never lets exceed
263
+ // TCP_RETRY_MAX_RETRIES_LIMIT (10).
158
264
  mesh.addTask([&mesh, ip, port, retryCount]() {
159
265
  Log(CONNECTION, "tcp_err(): Retrying TCP connection...\n");
160
266
 
@@ -178,7 +284,7 @@ void connect(AsyncClient &client, IPAddress ip, uint16_t port, M &mesh,
178
284
  // Adding a significant delay before reconnection prevents rapid reconnection loops
179
285
  // when the TCP server is persistently unavailable or overloaded
180
286
  Log(CONNECTION, "tcp_err(): All %d retries exhausted for IP %s\n",
181
- TCP_CONNECT_MAX_RETRIES + 1, ip.toString().c_str());
287
+ cfg.maxRetries + 1, ip.toString().c_str());
182
288
 
183
289
  // Block this node temporarily to prevent immediate reconnection to the same unresponsive node
184
290
  // This helps when the bridge's TCP server is down but WiFi AP is still running
@@ -186,15 +292,19 @@ void connect(AsyncClient &client, IPAddress ip, uint16_t port, M &mesh,
186
292
  #if !defined(PAINLESSMESH_BOOST)
187
293
  // Try to decode nodeId from IP and block it
188
294
  // Only works for mesh IPs (format: 10.x.x.1)
295
+ // A failureBlockDurationMs of 0 is an explicit "never blocklist" opt-out.
296
+ // Passing it through would set blockUntil = millis() + 0, which
297
+ // isNodeBlocked() reads as already expired - so the entry would do
298
+ // nothing except linger in the blocklist until cleanup sweeps it.
189
299
  uint32_t failedNodeId = decodeNodeIdFromIP(ip);
190
- if (failedNodeId != 0) {
300
+ if (failedNodeId != 0 && cfg.failureBlockDurationMs > 0) {
191
301
  // Note: This requires M to be wifi::Mesh which has blockNodeAfterTCPFailure
192
- mesh.blockNodeAfterTCPFailure(ip, TCP_FAILURE_BLOCK_DURATION_MS);
302
+ mesh.blockNodeAfterTCPFailure(ip, cfg.failureBlockDurationMs);
193
303
  }
194
304
  #endif
195
-
305
+
196
306
  Log(CONNECTION, "tcp_err(): Scheduling WiFi reconnection in %u ms\n",
197
- TCP_EXHAUSTION_RECONNECT_DELAY_MS);
307
+ cfg.exhaustionReconnectDelayMs);
198
308
 
199
309
  // Defer deletion of the failed AsyncClient to prevent heap corruption
200
310
  // Use the centralized deletion scheduler to ensure proper spacing between deletions
@@ -208,13 +318,32 @@ void connect(AsyncClient &client, IPAddress ip, uint16_t port, M &mesh,
208
318
  mesh.addTask([&mesh]() {
209
319
  Log(CONNECTION, "tcp_err(): Executing delayed WiFi reconnection after retry exhaustion\n");
210
320
  mesh.droppedConnectionCallbacks.execute(0, true);
211
- }, TCP_EXHAUSTION_RECONNECT_DELAY_MS);
321
+ }, cfg.exhaustionReconnectDelayMs);
212
322
  mesh.semaphoreGive();
213
323
  }
214
324
  });
215
325
 
216
326
  client.onConnect(
217
- [&mesh](void *, AsyncClient *client) {
327
+ [&mesh, claimed](void *, AsyncClient *client) {
328
+ if (*claimed) {
329
+ // onError has already fired for this client (e.g. an abort
330
+ // caused by losing WiFi right while the TCP handshake was
331
+ // completing): the client is already scheduled for deletion by
332
+ // the retry path. Wrapping it in a new BufferedConnection now
333
+ // would give it two owners at once. This connection did however
334
+ // genuinely succeed at the TCP level (that's why onConnect
335
+ // fired): close it explicitly here, so that when the deferred
336
+ // deletion already scheduled by onError fires, it finds a
337
+ // properly closed client instead of one still "connected" -
338
+ // otherwise we'd fall right back into the bug we're fixing.
339
+ Log(CONNECTION,
340
+ "tcp::connect(): onConnect fired after onError had already "
341
+ "claimed the client - closing the connection and discarding "
342
+ "it\n");
343
+ client->close();
344
+ return;
345
+ }
346
+ *claimed = true;
218
347
  if (mesh.semaphoreTake()) {
219
348
  Log(CONNECTION, "New STA connection incoming\n");
220
349
  auto conn = std::make_shared<T>(client, &mesh, true);