@alteriom/painlessmesh 2.0.3 → 2.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -23,6 +23,14 @@
23
23
 
24
24
  extern painlessmesh::logger::LogClass Log;
25
25
 
26
+ /**
27
+ * Defined when Mesh::tcpListening() exists, so a node can be asked whether
28
+ * its TCP listener is in LISTEN rather than having that inferred from a
29
+ * peer eventually connecting. Code that must build against older releases
30
+ * too can test for it with #ifdef.
31
+ */
32
+ #define PAINLESSMESH_HAS_TCP_LISTENING 1
33
+
26
34
  namespace painlessmesh {
27
35
  namespace wifi {
28
36
  class Mesh : public painlessmesh::Mesh<Connection> {
@@ -883,7 +891,7 @@ class Mesh : public painlessmesh::Mesh<Connection> {
883
891
  // then had a bridge's listener. AsyncTCP keeps its pcb private, so
884
892
  // SO_REUSEADDR cannot be set from here; not re-binding is the fix.
885
893
  if (_tcpListener != nullptr) {
886
- if (_tcpListener->status() == 1) {
894
+ if (_tcpListener->status() == TCP_STATE_LISTEN) {
887
895
  Log(CONNECTION,
888
896
  "tcpServerInit(): listener on port %d already listening, kept\n",
889
897
  _meshPort);
@@ -905,6 +913,28 @@ class Mesh : public painlessmesh::Mesh<Connection> {
905
913
  return;
906
914
  }
907
915
 
916
+ /**
917
+ * Whether this node's TCP listener exists and is in LISTEN.
918
+ *
919
+ * The state peers depend on and nothing else reports: a node whose
920
+ * listener was never created (#466 crashed before it could be) or was
921
+ * re-created not listening (#435's promoted bridge served nothing for two
922
+ * minutes) is reachable only through connections it made outbound, and
923
+ * nothing can join through it. Cheap enough to put in a health report.
924
+ */
925
+ bool tcpListening() {
926
+ return _tcpListener != nullptr && _tcpListener->status() == TCP_STATE_LISTEN;
927
+ }
928
+
929
+ /**
930
+ * lwIP's `LISTEN`, which AsyncServer::status() returns as the pcb's state
931
+ * on both cores (ESPAsyncTCP and ESP32Async/AsyncTCP alike). Named rather
932
+ * than written as 1 because status() also answers 0 -- lwIP's CLOSED --
933
+ * when the server holds no pcb at all, so the literal read as if it
934
+ * conflated the two. It does not: a listener is listening, or it is not.
935
+ */
936
+ static constexpr uint8_t TCP_STATE_LISTEN = 1;
937
+
908
938
  /**
909
939
  * Establish TCP connection to mesh network
910
940
  *
@@ -1521,12 +1551,15 @@ class Mesh : public painlessmesh::Mesh<Connection> {
1521
1551
 
1522
1552
  protected:
1523
1553
  friend class ::StationScan;
1554
+ // init() sets all of these; the defaults are init()'s own, so a node
1555
+ // that is asked about its AP before init() answers something true
1556
+ // rather than whatever the stack held.
1524
1557
  TSTRING _meshSSID;
1525
1558
  TSTRING _meshPassword;
1526
- uint8_t _meshChannel;
1527
- uint8_t _meshHidden;
1528
- uint8_t _meshMaxConn;
1529
- uint16_t _meshPort;
1559
+ uint8_t _meshChannel = 1;
1560
+ uint8_t _meshHidden = 0;
1561
+ uint8_t _meshMaxConn = MAX_CONN;
1562
+ uint16_t _meshPort = 5555;
1530
1563
 
1531
1564
  IPAddress _apIp;
1532
1565
  StationScan stationScan;
@@ -2467,7 +2500,7 @@ class Mesh : public painlessmesh::Mesh<Connection> {
2467
2500
  // stop/re-init, was not listening, and nothing looked. This task runs
2468
2501
  // every thirty seconds on a bridge: if the listener is not in LISTEN
2469
2502
  // (1 on both cores) it is re-created, and the log says so.
2470
- if (_tcpListener != nullptr && _tcpListener->status() != 1) {
2503
+ if (_tcpListener != nullptr && _tcpListener->status() != TCP_STATE_LISTEN) {
2471
2504
  Log(ERROR,
2472
2505
  "sendBridgeStatus(): TCP listener on port %d is in state %u, not "
2473
2506
  "LISTEN; re-creating it\n",
@@ -2727,41 +2760,44 @@ class Mesh : public painlessmesh::Mesh<Connection> {
2727
2760
 
2728
2761
  #if defined(ESP32) || defined(ESP8266)
2729
2762
  /**
2730
- * Read the start of an HTTP response body, bounded in bytes and in time
2731
- *
2732
- * The gateway used to discard the body, so a service that answers a refusal
2733
- * with a 2xx (CallMeBot, issue #450) was reported as delivered, and a
2734
- * failure reached the origin node as a bare status. When the length is
2735
- * known and small the whole body is taken; otherwise up to maxBytes are
2736
- * read from the stream, waiting at most GATEWAY_RESPONSE_HEAD_TIMEOUT_MS,
2737
- * which keeps a large or chunked page from eating the heap or the
2738
- * cooperative scheduler.
2763
+ * The start and the end of the response body (gateway::ResponseExcerpt),
2764
+ * read from the stream for at most GATEWAY_RESPONSE_HEAD_TIMEOUT_MS and
2765
+ * GATEWAY_RESPONSE_SCAN_BYTES. A body that fits whole is read whole. A
2766
+ * chunked body is read through gateway::ChunkedBodyDecoder: the raw stream
2767
+ * carries the framing, which getString() would remove but only by keeping
2768
+ * the whole body.
2739
2769
  */
2740
- static TSTRING readResponseHead(HTTPClient& http, size_t maxBytes) {
2770
+ static TSTRING readResponseExcerpt(HTTPClient& http) {
2741
2771
  const int size = http.getSize();
2742
- if (size >= 0 && static_cast<size_t>(size) <= maxBytes) {
2772
+ if (size >= 0 &&
2773
+ static_cast<size_t>(size) <=
2774
+ gateway::GATEWAY_RESPONSE_HEAD_BYTES + gateway::GATEWAY_RESPONSE_TAIL_BYTES) {
2743
2775
  return http.getString();
2744
2776
  }
2745
- TSTRING head;
2777
+ gateway::ResponseExcerpt excerpt;
2746
2778
  WiFiClient* stream = http.getStreamPtr();
2747
- if (stream == nullptr) return head;
2779
+ if (stream == nullptr) return excerpt.text();
2780
+ const bool chunked =
2781
+ size < 0 && gateway::transferEncodingIsChunked(http.header("Transfer-Encoding"));
2782
+ gateway::ChunkedBodyDecoder decoder(excerpt);
2748
2783
  const uint32_t deadline =
2749
2784
  millis() + gateway::GATEWAY_RESPONSE_HEAD_TIMEOUT_MS;
2750
- while (head.length() < maxBytes &&
2751
- static_cast<int32_t>(deadline - millis()) > 0) {
2785
+ bool more = true;
2786
+ while (more && static_cast<int32_t>(deadline - millis()) > 0) {
2752
2787
  int available = stream->available();
2753
2788
  if (available <= 0) {
2754
2789
  if (!stream->connected()) break;
2755
2790
  delay(1);
2756
2791
  continue;
2757
2792
  }
2758
- while (available-- > 0 && head.length() < maxBytes) {
2793
+ while (more && available-- > 0) {
2759
2794
  const int c = stream->read();
2760
2795
  if (c < 0) break;
2761
- head += static_cast<char>(c);
2796
+ more = chunked ? decoder.add(static_cast<char>(c))
2797
+ : excerpt.add(static_cast<char>(c));
2762
2798
  }
2763
2799
  }
2764
- return head;
2800
+ return excerpt.text();
2765
2801
  }
2766
2802
  #endif
2767
2803
 
@@ -2836,8 +2872,13 @@ class Mesh : public painlessmesh::Mesh<Connection> {
2836
2872
  // that genuinely stopped answering NODE_SYNC still gets reaped even
2837
2873
  // under continuous gateway traffic from other peers.
2838
2874
  const auto blockingStartedMs = millis();
2875
+ // `retryable` is the gateway's word to the origin node on whether
2876
+ // the identical request may be sent again: 0 whenever it may have
2877
+ // reached the server, so a retry cannot deliver it twice.
2839
2878
  auto finish = [this, &pkg, ingress, blockingStartedMs](
2840
- bool ok, uint16_t code, const TSTRING& err) {
2879
+ bool ok, uint16_t code, const TSTRING& err,
2880
+ int8_t retryable, const TSTRING& response = TSTRING(),
2881
+ uint32_t retryAfterMs = 0) {
2841
2882
  const auto stalledMs = millis() - blockingStartedMs;
2842
2883
  auto refreshed = gateway::refreshPeerWatchdogs(*this, stalledMs);
2843
2884
  if (refreshed > 0) {
@@ -2847,27 +2888,28 @@ class Mesh : public painlessmesh::Mesh<Connection> {
2847
2888
  static_cast<unsigned>(refreshed),
2848
2889
  static_cast<unsigned long>(stalledMs));
2849
2890
  }
2850
- this->sendGatewayAck(pkg, ok, code, err, ingress);
2891
+ this->sendGatewayAck(pkg, ok, code, err, ingress, response, retryable,
2892
+ retryAfterMs);
2851
2893
  };
2852
2894
 
2853
2895
  // Check Internet connectivity
2854
2896
  // First check WiFi status for quick fail
2855
2897
  if (WiFi.status() != WL_CONNECTED) {
2856
- finish(false, 0, "Gateway WiFi not connected");
2898
+ finish(false, 0, "Gateway WiFi not connected", 0);
2857
2899
  return true; // Consume package - we handled it (with error)
2858
2900
  }
2859
2901
 
2860
2902
  // Then check actual internet access (DNS resolution)
2861
2903
  // This detects when WiFi is connected but router has no internet
2862
2904
  if (!hasActualInternetAccess()) {
2863
- finish(false, 0, "Router has no internet access - check WAN connection");
2905
+ finish(false, 0, "Router has no internet access - check WAN connection", 0);
2864
2906
  return true; // Consume package - we handled it (with error)
2865
2907
  }
2866
2908
 
2867
2909
  // Finally, check for captive portal interference
2868
2910
  // This detects when DNS works but HTTP requests are intercepted
2869
2911
  if (!detectCaptivePortal()) {
2870
- finish(false, 0, "Captive portal detected - requires web authentication. Check router/WiFi settings");
2912
+ finish(false, 0, "Captive portal detected - requires web authentication. Check router/WiFi settings", 0);
2871
2913
  return true; // Consume package - we handled it (with error)
2872
2914
  }
2873
2915
 
@@ -2899,7 +2941,7 @@ class Mesh : public painlessmesh::Mesh<Connection> {
2899
2941
  (gateway::GATEWAY_DNS_NEGATIVE_TTL_MS - failedAgoMs) /
2900
2942
  1000));
2901
2943
  Log(ERROR, "%s\n", dnsBuf);
2902
- finish(false, 0, TSTRING(dnsBuf));
2944
+ finish(false, 0, TSTRING(dnsBuf), 0);
2903
2945
  return true;
2904
2946
  }
2905
2947
  IPAddress resolved;
@@ -2909,7 +2951,7 @@ class Mesh : public painlessmesh::Mesh<Connection> {
2909
2951
  snprintf(dnsBuf, sizeof(dnsBuf), "DNS lookup failed for %s",
2910
2952
  host.c_str());
2911
2953
  Log(ERROR, "%s\n", dnsBuf);
2912
- finish(false, 0, TSTRING(dnsBuf));
2954
+ finish(false, 0, TSTRING(dnsBuf), 0);
2913
2955
  return true;
2914
2956
  }
2915
2957
  }
@@ -2959,6 +3001,17 @@ class Mesh : public painlessmesh::Mesh<Connection> {
2959
3001
  #endif
2960
3002
  }
2961
3003
 
3004
+ // The same id on every attempt at this call. A service that
3005
+ // honours Idempotency-Key treats a retry as the request it already
3006
+ // has; anything recording requests can count the copies.
3007
+ const TSTRING requestId =
3008
+ gateway::requestIdFor(pkg.originNode, pkg.messageId, pkg.requestNonce);
3009
+ http.addHeader("X-Request-Id", requestId.c_str());
3010
+ http.addHeader("Idempotency-Key", requestId.c_str());
3011
+ // Not const: both cores declare collectHeaders(const char* keys[], ...).
3012
+ const char* collected[] = {"Retry-After", "Transfer-Encoding"};
3013
+ http.collectHeaders(collected, 2);
3014
+
2962
3015
  // Make request (GET if no payload, POST if payload)
2963
3016
  if (pkg.payload.length() > 0) {
2964
3017
  http.addHeader("Content-Type", pkg.contentType.c_str());
@@ -2967,11 +3020,11 @@ class Mesh : public painlessmesh::Mesh<Connection> {
2967
3020
  httpCode = http.GET();
2968
3021
  }
2969
3022
 
2970
- // A 2xx is the origin's acceptance, except 203 (a proxy transformed
2971
- // the response) and except when the body says the service refused
2972
- // the request: CallMeBot answers "Too many requests" under 201 and
2973
- // 203 (issue #450). The body is therefore read -- bounded -- before
2974
- // classifying, and it is what the origin node is told on failure.
3023
+ // HTTP semantics only: 200/201/202/204 succeed, anything else fails,
3024
+ // and a retry is allowed only when it cannot deliver the request
3025
+ // twice (see gateway::classifyHttpResult). Whether a service's reply
3026
+ // means what the application wanted is the application's decision,
3027
+ // so the start of the body goes back on success as well as failure.
2975
3028
  //
2976
3029
  // 3xx redirects are not automatically followed.
2977
3030
  //
@@ -2981,72 +3034,68 @@ class Mesh : public painlessmesh::Mesh<Connection> {
2981
3034
  TSTRING responseHead;
2982
3035
  if (httpCode > 0) {
2983
3036
  responseHead =
2984
- readResponseHead(http, gateway::GATEWAY_RESPONSE_HEAD_BYTES);
3037
+ readResponseExcerpt(http);
2985
3038
  }
2986
3039
  const auto outcome =
2987
3040
  gateway::classifyHttpResult(httpCode, responseHead);
2988
3041
  success = outcome.success;
3042
+ const TSTRING responseSummary =
3043
+ httpCode > 0 ? gateway::summarizeResponseBody(responseHead) : TSTRING();
3044
+ uint32_t retryAfterMs = 0;
3045
+ if (outcome.retryable && httpCode > 0) {
3046
+ retryAfterMs = gateway::parseRetryAfterMs(http.header("Retry-After"));
3047
+ }
2989
3048
 
2990
3049
  if (!outcome.transportError) {
2991
- char errorBuf[192];
3050
+ char errorBuf[320];
2992
3051
  const char* reason = outcome.reason.c_str();
2993
3052
  const char* sep = outcome.reason.length() > 0 ? ": " : "";
2994
3053
  if (success) {
2995
3054
  Log(COMMUNICATION, "HTTP request completed: code=%d\n", httpCode);
2996
- } else if (outcome.refusedByBody) {
2997
- // Success-class status, refusing body. Not retryable: the
2998
- // service answered, and the origin node gets its words.
2999
- snprintf(errorBuf, sizeof(errorBuf),
3000
- "HTTP %d: service refused the request%s%s", httpCode,
3001
- sep, reason);
3002
- error = TSTRING(errorBuf);
3003
- Log(ERROR, "HTTP request refused by the service: %s\n", errorBuf);
3004
3055
  } else if (outcome.unverifiedStatus) {
3005
- // A 2xx outside 200/201/202/204. CallMeBot answers 208 to a
3006
- // message that never arrives (issue #452), so this is a
3007
- // failure, and the body -- the only place the service says
3008
- // what it did -- goes to the origin node and to the log at
3009
- // ERROR level, where a sketch running the default levels
3010
- // sees it. 203 stays retryable; the rest are final.
3056
+ // A 2xx outside 200/201/202/204: the server answered, but not
3057
+ // with a status this library can call a delivery. Never
3058
+ // retried. Logged at ERROR, where a sketch on the default
3059
+ // levels sees what the service said.
3011
3060
  snprintf(errorBuf, sizeof(errorBuf),
3012
3061
  "HTTP %d: not a delivery the gateway can confirm%s%s",
3013
3062
  httpCode, sep, reason);
3014
3063
  error = TSTRING(errorBuf);
3015
3064
  Log(ERROR, "HTTP response unverified: %s\n", errorBuf);
3016
- } else if (httpCode >= 500 && httpCode < 600) {
3017
- // 5xx server errors are retryable, log at COMMUNICATION level
3065
+ } else if (outcome.retryable) {
3066
+ // 429 or 503: the server says it did not take the request.
3018
3067
  snprintf(errorBuf, sizeof(errorBuf), "HTTP %d%s%s", httpCode, sep,
3019
3068
  reason);
3020
3069
  error = TSTRING(errorBuf);
3021
- Log(COMMUNICATION, "HTTP server error: code=%d (will retry)\n", httpCode);
3022
- } else if (httpCode == 429) {
3023
- // HTTP 429 rate limit is retryable, log at COMMUNICATION level
3024
- snprintf(errorBuf, sizeof(errorBuf), "HTTP %d%s%s", httpCode, sep,
3025
- reason);
3026
- error = TSTRING(errorBuf);
3027
- Log(COMMUNICATION, "HTTP rate limit: code=%d (will retry)\n", httpCode);
3070
+ Log(COMMUNICATION, "HTTP %d: the server asked to retry (after %lu ms)\n",
3071
+ httpCode, static_cast<unsigned long>(retryAfterMs));
3028
3072
  } else {
3029
- // 1xx, 3xx, 4xx (except 429) - non-retryable, log at ERROR level
3030
3073
  snprintf(errorBuf, sizeof(errorBuf), "HTTP %d%s%s", httpCode, sep,
3031
3074
  reason);
3032
3075
  error = TSTRING(errorBuf);
3033
3076
  Log(ERROR, "HTTP request failed: %s\n", errorBuf);
3034
3077
  }
3035
3078
  } else {
3036
- // Network errors (httpCode <= 0) are retryable but indicate serious issues
3037
- // Keep at ERROR level as they may indicate gateway connectivity problems
3079
+ // Below HTTP. When the request may already be at the server -- a
3080
+ // read timeout, a connection lost after sending -- it is not
3081
+ // retried, and the error says why, so an application that
3082
+ // resends does so knowing it may duplicate.
3038
3083
  error = http.errorToString(httpCode);
3084
+ if (!outcome.retryable) {
3085
+ error += " (the request may have reached the server; not retried)";
3086
+ }
3039
3087
  Log(ERROR, "HTTP request failed: %s\n", error.c_str());
3040
3088
  }
3041
3089
 
3042
3090
  http.end();
3043
3091
 
3044
- // Send acknowledgment back. Transport errors carry status 0, the
3045
- // value handleGatewayAck() treats as a retryable network error.
3046
- finish(success, outcome.ackStatus, error);
3092
+ // Send acknowledgment back, with the verdict on retrying and the
3093
+ // start of whatever the server said.
3094
+ finish(success, outcome.ackStatus, error, outcome.retryable ? 1 : 0,
3095
+ responseSummary, retryAfterMs);
3047
3096
  #else
3048
3097
  // Non-ESP platform - send error
3049
- finish(false, 0, "HTTP client not available on this platform");
3098
+ finish(false, 0, "HTTP client not available on this platform", 0);
3050
3099
  #endif
3051
3100
 
3052
3101
  return true; // Consume package - we have processed it and sent
@@ -3220,16 +3269,25 @@ class Mesh : public painlessmesh::Mesh<Connection> {
3220
3269
  }
3221
3270
 
3222
3271
  #ifdef ESP32
3223
- WiFiEventId_t eventScanDoneHandler;
3224
- WiFiEventId_t eventSTAStartHandler;
3225
- WiFiEventId_t eventSTADisconnectedHandler;
3226
- WiFiEventId_t eventSTAGotIPHandler;
3272
+ // Event ids the core hands back from onEvent(); stop() removes all four.
3273
+ // 0 is never issued (the core's ids start at 1 and 0 means "no
3274
+ // callback"), so removeEvent(0) is a no-op on a node stopped before
3275
+ // init() rather than a walk of the core's callback list for garbage.
3276
+ WiFiEventId_t eventScanDoneHandler = 0;
3277
+ WiFiEventId_t eventSTAStartHandler = 0;
3278
+ WiFiEventId_t eventSTADisconnectedHandler = 0;
3279
+ WiFiEventId_t eventSTAGotIPHandler = 0;
3227
3280
  #elif defined(ESP8266)
3228
3281
  WiFiEventHandler eventSTAConnectedHandler;
3229
3282
  WiFiEventHandler eventSTADisconnectedHandler;
3230
3283
  WiFiEventHandler eventSTAGotIPHandler;
3231
3284
  #endif // ESP8266
3232
- AsyncServer* _tcpListener;
3285
+ // Null until tcpServerInit() creates it. Every reader checks for null
3286
+ // first, and one of them (tcpServerInit() itself, since 2.1.0) runs before
3287
+ // anything has assigned it -- on an uninitialised pointer that check
3288
+ // passed on garbage and the delete that followed crashed the node before
3289
+ // it served a single connection (#466).
3290
+ AsyncServer* _tcpListener = nullptr;
3233
3291
  std::shared_ptr<Task> bridgeStatusTask;
3234
3292
  // millis() of the last status broadcast, periodic or brought forward by a
3235
3293
  // topology change; the latter is held to one every five seconds.
@@ -3242,11 +3300,11 @@ class Mesh : public painlessmesh::Mesh<Connection> {
3242
3300
  enum ElectionState { ELECTION_IDLE, ELECTION_SCANNING, ELECTION_COLLECTING };
3243
3301
 
3244
3302
  struct BridgeCandidate {
3245
- uint32_t nodeId;
3246
- int8_t routerRSSI;
3247
- uint8_t routerChannel;
3248
- uint32_t uptime;
3249
- uint32_t freeMemory;
3303
+ uint32_t nodeId = 0;
3304
+ int8_t routerRSSI = 0;
3305
+ uint8_t routerChannel = 0;
3306
+ uint32_t uptime = 0;
3307
+ uint32_t freeMemory = 0;
3250
3308
  };
3251
3309
 
3252
3310
  bool bridgeFailoverEnabled = true;
@@ -3301,11 +3359,19 @@ class Mesh : public painlessmesh::Mesh<Connection> {
3301
3359
  size_t lastSelectedBridgeIndex = 0; // For round-robin selection
3302
3360
 
3303
3361
  // Bridge coordination monitoring callbacks and state
3362
+ // Constructed by name, not as an aggregate: with default member
3363
+ // initializers this is not an aggregate under gnu++11, which the ESP32
3364
+ // Arduino 2.x core still builds with, and `= {priority, role, load,
3365
+ // millis()}` below needs a constructor to land on there.
3304
3366
  struct BridgeCoordinationState {
3305
- uint8_t priority;
3367
+ uint8_t priority = 0;
3306
3368
  TSTRING role;
3307
- uint8_t load;
3308
- uint32_t lastSeen;
3369
+ uint8_t load = 0;
3370
+ uint32_t lastSeen = 0;
3371
+ BridgeCoordinationState() {}
3372
+ BridgeCoordinationState(uint8_t priority_, const TSTRING& role_,
3373
+ uint8_t load_, uint32_t lastSeen_)
3374
+ : priority(priority_), role(role_), load(load_), lastSeen(lastSeen_) {}
3309
3375
  };
3310
3376
  std::map<uint32_t, BridgeCoordinationState> lastBridgeCoordinationState;
3311
3377
  std::function<void(const plugin::BridgeCoordinationPackage&, uint32_t)> bridgeCoordinationCallback;
@@ -258,7 +258,7 @@ class AsyncServer {
258
258
 
259
259
  protected:
260
260
  boost::asio::io_context& _io_service;
261
- uint16_t _port;
261
+ uint16_t _port = 0;
262
262
  tcp::acceptor mAcceptor;
263
263
  AcConnectHandler _connect_cb = 0;
264
264
  void* _connect_cb_arg = 0;
@@ -628,6 +628,9 @@ void ICACHE_FLASH_ATTR StationScan::connectToAP() {
628
628
  #endif
629
629
  bool isRooted = layout::isRooted(mesh->asNodeTree());
630
630
  if (isRooted) everRooted = true;
631
+ // The immediate-rescan allowance is per outage, and this is the one place
632
+ // every scan passes through: a node with a connection has not spent it.
633
+ if (painlessmesh::layout::liveSubs(*mesh) > 0) rejoinScanned = false;
631
634
  if (aps.empty()) {
632
635
  // No unknown nodes found
633
636
  consecutiveEmptyScans++;
@@ -732,11 +735,42 @@ void ICACHE_FLASH_ATTR StationScan::connectToAP() {
732
735
  task.delay(interval);
733
736
  if (orphanScanBackoff < 2) orphanScanBackoff++;
734
737
  } else {
735
- // else scan fast (SCAN_INTERVAL)
736
- Log(CONNECTION,
737
- "connectToAP(): No unknown nodes found scan rate set to "
738
- "fast\n");
738
+ // Not connected, and this scan turned up nothing to connect to.
739
+ //
740
+ // "Fast" here is 0.5 * SCAN_INTERVAL -- fifteen seconds. For a node
741
+ // that still has a live connection (an AP-side child) that is soon
742
+ // enough: it is in the mesh and reachable while it looks for more.
743
+ // For a node with none it is the whole cost of the outage, and it is
744
+ // paid while the node's neighbours still have it in their routing
745
+ // tables: they keep sending to it until their own NODE_TIMEOUT, and
746
+ // those messages are dropped.
747
+ //
748
+ // Measured on the Alteriom HIL rig (issue #459): an ESP32-C3 closed
749
+ // its only uplink on a momentary nodeSync contradiction, logged
750
+ // "scan rate set to fast" 12 ms later, and did not scan again for
751
+ // exactly 15 000 ms. It was out of the mesh for 16.2 s and answered
752
+ // an empty node list throughout; a unicast sent to it in that window
753
+ // was accepted by the sender and lost. The scan that finally ran
754
+ // found five mesh APs at -30 to -58 dBm and was associated 1.2 s
755
+ // later. There was nothing to wait for.
756
+ //
757
+ // So a node with no connections left scans at once. Once per outage:
758
+ // `rejoinScanned` is cleared as soon as it has a connection again,
759
+ // so a node that is simply alone falls back to the interval instead
760
+ // of scanning back to back.
739
761
  task.setInterval(0.5 * SCAN_INTERVAL);
762
+ if (painlessmesh::layout::liveSubs(*mesh) == 0 && !rejoinScanned) {
763
+ rejoinScanned = true;
764
+ Log(CONNECTION,
765
+ "connectToAP(): No connections left; scanning again now rather "
766
+ "than in %d s\n",
767
+ (int)(0.5 * SCAN_INTERVAL / TASK_SECOND));
768
+ task.forceNextIteration();
769
+ } else {
770
+ Log(CONNECTION,
771
+ "connectToAP(): No unknown nodes found scan rate set to "
772
+ "fast\n");
773
+ }
740
774
  }
741
775
  mesh->stability += min(1000 - mesh->stability, (size_t)25);
742
776
  } else {
@@ -62,7 +62,14 @@ class StationScan {
62
62
  // The station got an address: its link is up. Cleared once the drop
63
63
  // callbacks have judged a disconnect, so they can tell a link that was
64
64
  // up and went away from an attempt that never got that far.
65
- void stationUp() { stationLinkUp = true; }
65
+ void stationUp() {
66
+ stationLinkUp = true;
67
+ // A link again, so the immediate-rescan allowance is restored now
68
+ // rather than at the next scan. Without this a node that dropped
69
+ // twice inside one scan interval paid the full fifteen seconds for
70
+ // the second drop (#459).
71
+ rejoinScanned = false;
72
+ }
66
73
  void stationDown() { stationLinkUp = false; }
67
74
  bool stationLinkUp = false;
68
75
  // The station link was closed by this node's own channel move a moment
@@ -118,10 +125,11 @@ class StationScan {
118
125
  protected:
119
126
  TSTRING ssid;
120
127
  TSTRING password;
121
- painlessMesh *mesh;
122
- uint16_t port;
123
- uint8_t channel;
124
- bool hidden;
128
+ // init() sets these; until then the scan belongs to no mesh and no port.
129
+ painlessMesh *mesh = nullptr;
130
+ uint16_t port = 0;
131
+ uint8_t channel = 0;
132
+ bool hidden = false;
125
133
  std::list<WiFi_AP_Record_t> aps;
126
134
 
127
135
  void requestIP(WiFi_AP_Record_t &ap);
@@ -147,6 +155,11 @@ class StationScan {
147
155
  // channel re-detection runs, and consuming those results here found them
148
156
  // already deleted, reported "wifi scan failed", and rescanned at once.
149
157
  bool scanRequested = false;
158
+ // Whether this node has already scanned at once on losing its last
159
+ // connection. Cleared the moment it has one again, so the immediate
160
+ // rescan is one per outage and not a scan loop: a node that is out of
161
+ // the mesh and finds nothing still falls back to the interval below.
162
+ bool rejoinScanned = false;
150
163
  // Set when the next station scan must cover every channel: the empty
151
164
  // scans have piled up and the node is looking for the channel the mesh
152
165
  // moved to. The re-detection is this task's ordinary asynchronous scan
@@ -124,7 +124,7 @@ inline void ReceiveBuffer<std::string>::stringAppend(std::string &buffer,
124
124
  template <class T>
125
125
  struct PrioritizedMessage {
126
126
  T message;
127
- uint8_t priority; // 0=CRITICAL, 1=HIGH, 2=NORMAL, 3=LOW
127
+ uint8_t priority = 2; // 0=CRITICAL, 1=HIGH, 2=NORMAL, 3=LOW
128
128
 
129
129
  PrioritizedMessage(const T& msg, uint8_t prio = 2) : message(msg), priority(prio) {}
130
130
  };
@@ -206,12 +206,17 @@ class SentBuffer {
206
206
  size_t requestLength(size_t buffer_length) {
207
207
  // Use highest priority message available
208
208
  auto* msg = getNextMessage();
209
- if (!msg)
210
- return 0;
211
- else
212
- // String.toCharArray automatically turns the last character into
213
- // a \0, we need the extra space to deal with that annoyance
214
- return (std::min)(buffer_length - 1, msg->length() + 1);
209
+ if (!msg) return 0;
210
+ // A caller with no room is owed nothing. The subtraction below leaves
211
+ // space for the terminator toCharArray() always writes; on an unsigned
212
+ // zero it wraps to SIZE_MAX instead, min() then picks the message, and
213
+ // the answer comes back *larger* than was asked for -- which the
214
+ // documented contract above forbids and read() then acts on, writing
215
+ // length + 1 bytes into a buffer the caller said had none.
216
+ if (buffer_length == 0) return 0;
217
+ // String.toCharArray automatically turns the last character into
218
+ // a \0, we need the extra space to deal with that annoyance
219
+ return (std::min)(buffer_length - 1, msg->length() + 1);
215
220
  }
216
221
 
217
222
  /**
@@ -313,18 +318,18 @@ class SentBuffer {
313
318
  * Get statistics about queued and sent messages
314
319
  */
315
320
  struct SendStats {
316
- uint32_t totalQueued;
317
- uint32_t criticalQueued;
318
- uint32_t highQueued;
319
- uint32_t normalQueued;
320
- uint32_t lowQueued;
321
- uint32_t criticalSent;
322
- uint32_t highSent;
323
- uint32_t normalSent;
324
- uint32_t lowSent;
321
+ uint32_t totalQueued = 0;
322
+ uint32_t criticalQueued = 0;
323
+ uint32_t highQueued = 0;
324
+ uint32_t normalQueued = 0;
325
+ uint32_t lowQueued = 0;
326
+ uint32_t criticalSent = 0;
327
+ uint32_t highSent = 0;
328
+ uint32_t normalSent = 0;
329
+ uint32_t lowSent = 0;
325
330
  // Messages lost to the PAINLESSMESH_MAX_SENT_BUFFER_MESSAGES cap:
326
331
  // evicted from the buffer or rejected on push (issue #388)
327
- uint32_t dropped;
332
+ uint32_t dropped = 0;
328
333
  };
329
334
 
330
335
  SendStats getStats() const {