@camstack/addon-post-analysis 1.2.124 → 1.2.125

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -49927,44 +49927,65 @@ var TrackStore = class {
49927
49927
  async listDeviceIdsSince(since) {
49928
49928
  return this.collectDeviceIds(Math.max(0, Math.floor(since)));
49929
49929
  }
49930
+ /**
49931
+ * The distinct `deviceId`s with a track at or after `fromLastSeen`.
49932
+ *
49933
+ * ## Why this is a histogram and not a paged read
49934
+ *
49935
+ * It used to drain the table in 501-row pages on a `lastSeen` range with no
49936
+ * `deviceId`, ordered by `lastSeen`. The indexes on this table are
49937
+ * `(deviceId, lastSeen)` and `(deviceId, firstSeen)` — neither can serve a
49938
+ * `lastSeen` range from the left — so `EXPLAIN QUERY PLAN` on the live
49939
+ * database answered:
49940
+ *
49941
+ * ```
49942
+ * SCAN pipeline-analytics:tracks
49943
+ * USE TEMP B-TREE FOR ORDER BY
49944
+ * ```
49945
+ *
49946
+ * A full scan of a 155 MB table, plus a sorter holding every matching row —
49947
+ * and the store has no column projection, so each row carries its ~8.5 KB
49948
+ * `positions`/`snapshots` JSON through the sort. PER PAGE. Draining 17 940
49949
+ * rows took 36 such pages: measured on a copy of the live database, 36 full
49950
+ * scans, and on the hub itself single statements of 1 196–1 615 ms each,
49951
+ * every one of them blocking the one synchronous thread the whole cluster
49952
+ * reads configuration through. `listDeviceIds` has ~10 callers.
49953
+ *
49954
+ * The answer being asked for is at most a couple of dozen integers. A
49955
+ * histogram on `deviceId` with `bucketSize: 1` is `GROUP BY deviceId`, and
49956
+ * the planner serves it from an index that already exists:
49957
+ *
49958
+ * ```
49959
+ * SCAN pipeline-analytics:tracks USING COVERING INDEX idx_tracks_device_lastSeen
49960
+ * USE TEMP B-TREE FOR GROUP BY
49961
+ * ```
49962
+ *
49963
+ * 384 KB of index instead of 155 MB of table, ONE statement instead of 36:
49964
+ * **6.0 ms against 678 ms** on the same copy, warm cache both times. No new
49965
+ * index, so no write amplification and no extra bytes — the cheap question
49966
+ * came first, it just was not being asked (D56).
49967
+ *
49968
+ * `bucket` IS the deviceId: the column is INTEGER and the bucket expression
49969
+ * is `CAST((deviceId - 0) / 1 AS INTEGER)`. Best-effort, as before — a
49970
+ * failure yields [] and the caller treats it as "no devices", which is the
49971
+ * same answer it got from a failed page.
49972
+ */
49930
49973
  async collectDeviceIds(fromLastSeen) {
49931
- const seenDevices = /* @__PURE__ */ new Set();
49932
- const seenIds = /* @__PURE__ */ new Set();
49933
- const PAGE = 500;
49934
- let cursor = fromLastSeen;
49935
49974
  try {
49936
- for (;;) {
49937
- const rows = await this.store.query.query({
49938
- collection: TRACKS_COLLECTION,
49939
- filter: {
49940
- whereBetween: { lastSeen: [cursor, Number.MAX_SAFE_INTEGER] },
49941
- orderBy: {
49942
- field: "lastSeen",
49943
- direction: "asc"
49944
- },
49945
- limit: PAGE
49946
- }
49947
- });
49948
- if (rows.length === 0) break;
49949
- let newInPage = 0;
49950
- let maxLastSeen = cursor;
49951
- for (const r of rows) {
49952
- if (typeof r.id === "string" && !seenIds.has(r.id)) {
49953
- seenIds.add(r.id);
49954
- newInPage++;
49955
- }
49956
- const ls = Number(r.data["lastSeen"]);
49957
- if (Number.isFinite(ls) && ls > maxLastSeen) maxLastSeen = ls;
49958
- const deviceId = Number(r.data["deviceId"]);
49959
- if (Number.isFinite(deviceId)) seenDevices.add(deviceId);
49960
- }
49961
- if (rows.length < PAGE || newInPage === 0) break;
49962
- cursor = maxLastSeen;
49963
- }
49975
+ const buckets = await this.store.histogram.query({
49976
+ collection: TRACKS_COLLECTION,
49977
+ field: "deviceId",
49978
+ bucketSize: 1,
49979
+ origin: 0,
49980
+ ...fromLastSeen > 0 ? { filter: { whereBetween: { lastSeen: [fromLastSeen, Number.MAX_SAFE_INTEGER] } } } : {}
49981
+ });
49982
+ const ids = [];
49983
+ for (const b of buckets) if (b.count > 0 && Number.isFinite(b.bucket)) ids.push(b.bucket);
49984
+ return ids;
49964
49985
  } catch (err) {
49965
49986
  this.logger.warn("TrackStore.listDeviceIds failed", { meta: { error: String(err) } });
49987
+ return [];
49966
49988
  }
49967
- return [...seenDevices];
49968
49989
  }
49969
49990
  /** Historical query — hits the persisted collection. With `zone` set,
49970
49991
  * candidates are SQL-prefiltered on the envelope columns (overlap test via
@@ -49864,44 +49864,65 @@ var TrackStore = class {
49864
49864
  async listDeviceIdsSince(since) {
49865
49865
  return this.collectDeviceIds(Math.max(0, Math.floor(since)));
49866
49866
  }
49867
+ /**
49868
+ * The distinct `deviceId`s with a track at or after `fromLastSeen`.
49869
+ *
49870
+ * ## Why this is a histogram and not a paged read
49871
+ *
49872
+ * It used to drain the table in 501-row pages on a `lastSeen` range with no
49873
+ * `deviceId`, ordered by `lastSeen`. The indexes on this table are
49874
+ * `(deviceId, lastSeen)` and `(deviceId, firstSeen)` — neither can serve a
49875
+ * `lastSeen` range from the left — so `EXPLAIN QUERY PLAN` on the live
49876
+ * database answered:
49877
+ *
49878
+ * ```
49879
+ * SCAN pipeline-analytics:tracks
49880
+ * USE TEMP B-TREE FOR ORDER BY
49881
+ * ```
49882
+ *
49883
+ * A full scan of a 155 MB table, plus a sorter holding every matching row —
49884
+ * and the store has no column projection, so each row carries its ~8.5 KB
49885
+ * `positions`/`snapshots` JSON through the sort. PER PAGE. Draining 17 940
49886
+ * rows took 36 such pages: measured on a copy of the live database, 36 full
49887
+ * scans, and on the hub itself single statements of 1 196–1 615 ms each,
49888
+ * every one of them blocking the one synchronous thread the whole cluster
49889
+ * reads configuration through. `listDeviceIds` has ~10 callers.
49890
+ *
49891
+ * The answer being asked for is at most a couple of dozen integers. A
49892
+ * histogram on `deviceId` with `bucketSize: 1` is `GROUP BY deviceId`, and
49893
+ * the planner serves it from an index that already exists:
49894
+ *
49895
+ * ```
49896
+ * SCAN pipeline-analytics:tracks USING COVERING INDEX idx_tracks_device_lastSeen
49897
+ * USE TEMP B-TREE FOR GROUP BY
49898
+ * ```
49899
+ *
49900
+ * 384 KB of index instead of 155 MB of table, ONE statement instead of 36:
49901
+ * **6.0 ms against 678 ms** on the same copy, warm cache both times. No new
49902
+ * index, so no write amplification and no extra bytes — the cheap question
49903
+ * came first, it just was not being asked (D56).
49904
+ *
49905
+ * `bucket` IS the deviceId: the column is INTEGER and the bucket expression
49906
+ * is `CAST((deviceId - 0) / 1 AS INTEGER)`. Best-effort, as before — a
49907
+ * failure yields [] and the caller treats it as "no devices", which is the
49908
+ * same answer it got from a failed page.
49909
+ */
49867
49910
  async collectDeviceIds(fromLastSeen) {
49868
- const seenDevices = /* @__PURE__ */ new Set();
49869
- const seenIds = /* @__PURE__ */ new Set();
49870
- const PAGE = 500;
49871
- let cursor = fromLastSeen;
49872
49911
  try {
49873
- for (;;) {
49874
- const rows = await this.store.query.query({
49875
- collection: TRACKS_COLLECTION,
49876
- filter: {
49877
- whereBetween: { lastSeen: [cursor, Number.MAX_SAFE_INTEGER] },
49878
- orderBy: {
49879
- field: "lastSeen",
49880
- direction: "asc"
49881
- },
49882
- limit: PAGE
49883
- }
49884
- });
49885
- if (rows.length === 0) break;
49886
- let newInPage = 0;
49887
- let maxLastSeen = cursor;
49888
- for (const r of rows) {
49889
- if (typeof r.id === "string" && !seenIds.has(r.id)) {
49890
- seenIds.add(r.id);
49891
- newInPage++;
49892
- }
49893
- const ls = Number(r.data["lastSeen"]);
49894
- if (Number.isFinite(ls) && ls > maxLastSeen) maxLastSeen = ls;
49895
- const deviceId = Number(r.data["deviceId"]);
49896
- if (Number.isFinite(deviceId)) seenDevices.add(deviceId);
49897
- }
49898
- if (rows.length < PAGE || newInPage === 0) break;
49899
- cursor = maxLastSeen;
49900
- }
49912
+ const buckets = await this.store.histogram.query({
49913
+ collection: TRACKS_COLLECTION,
49914
+ field: "deviceId",
49915
+ bucketSize: 1,
49916
+ origin: 0,
49917
+ ...fromLastSeen > 0 ? { filter: { whereBetween: { lastSeen: [fromLastSeen, Number.MAX_SAFE_INTEGER] } } } : {}
49918
+ });
49919
+ const ids = [];
49920
+ for (const b of buckets) if (b.count > 0 && Number.isFinite(b.bucket)) ids.push(b.bucket);
49921
+ return ids;
49901
49922
  } catch (err) {
49902
49923
  this.logger.warn("TrackStore.listDeviceIds failed", { meta: { error: String(err) } });
49924
+ return [];
49903
49925
  }
49904
- return [...seenDevices];
49905
49926
  }
49906
49927
  /** Historical query — hits the persisted collection. With `zone` set,
49907
49928
  * candidates are SQL-prefiltered on the envelope columns (overlap test via
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@camstack/addon-post-analysis",
3
- "version": "1.2.124",
3
+ "version": "1.2.125",
4
4
  "description": "Post-Analysis bundle — enrichment, embedding-encoder, pipeline-analytics. Multi-entry npm package shipping addons that consume pipeline output.",
5
5
  "keywords": [
6
6
  "camstack",