@uptimizr/agent-core 1.1.0 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. package/AGENTS.md +238 -25
  2. package/README.md +57 -21
  3. package/dist/client.d.ts +22 -5
  4. package/dist/client.d.ts.map +1 -1
  5. package/dist/client.js +39 -10
  6. package/dist/client.js.map +1 -1
  7. package/dist/context.d.ts +93 -0
  8. package/dist/context.d.ts.map +1 -0
  9. package/dist/context.js +137 -0
  10. package/dist/context.js.map +1 -0
  11. package/dist/index.d.ts +10 -3
  12. package/dist/index.d.ts.map +1 -1
  13. package/dist/index.js +20 -2
  14. package/dist/index.js.map +1 -1
  15. package/dist/nonRegistryTools.d.ts +34 -0
  16. package/dist/nonRegistryTools.d.ts.map +1 -0
  17. package/dist/nonRegistryTools.js +70 -0
  18. package/dist/nonRegistryTools.js.map +1 -0
  19. package/dist/prompt.d.ts +33 -0
  20. package/dist/prompt.d.ts.map +1 -0
  21. package/dist/prompt.js +49 -0
  22. package/dist/prompt.js.map +1 -0
  23. package/dist/provider.d.ts +18 -0
  24. package/dist/provider.d.ts.map +1 -1
  25. package/dist/providers/anthropic.d.ts +18 -1
  26. package/dist/providers/anthropic.d.ts.map +1 -1
  27. package/dist/providers/anthropic.js +35 -3
  28. package/dist/providers/anthropic.js.map +1 -1
  29. package/dist/providers/openai.d.ts +14 -1
  30. package/dist/providers/openai.d.ts.map +1 -1
  31. package/dist/providers/openai.js +23 -1
  32. package/dist/providers/openai.js.map +1 -1
  33. package/dist/queryTool.d.ts +36 -0
  34. package/dist/queryTool.d.ts.map +1 -0
  35. package/dist/queryTool.js +123 -0
  36. package/dist/queryTool.js.map +1 -0
  37. package/dist/registryTools.d.ts +23 -1
  38. package/dist/registryTools.d.ts.map +1 -1
  39. package/dist/registryTools.js +185 -31
  40. package/dist/registryTools.js.map +1 -1
  41. package/dist/skills.d.ts +105 -0
  42. package/dist/skills.d.ts.map +1 -0
  43. package/dist/skills.generated.d.ts +44 -0
  44. package/dist/skills.generated.d.ts.map +1 -0
  45. package/dist/skills.generated.js +511 -0
  46. package/dist/skills.generated.js.map +1 -0
  47. package/dist/skills.js +144 -0
  48. package/dist/skills.js.map +1 -0
  49. package/dist/tools.d.ts +55 -8
  50. package/dist/tools.d.ts.map +1 -1
  51. package/dist/tools.js +41 -1
  52. package/dist/tools.js.map +1 -1
  53. package/dist/writeTools.d.ts +68 -0
  54. package/dist/writeTools.d.ts.map +1 -0
  55. package/dist/writeTools.js +256 -0
  56. package/dist/writeTools.js.map +1 -0
  57. package/llms.txt +164 -15
  58. package/package.json +7 -5
  59. package/skills/attention-hotspots/SKILL.md +88 -0
  60. package/skills/conversion-investigation/SKILL.md +97 -0
  61. package/skills/performance-regression-triage/SKILL.md +106 -0
  62. package/skills/weekly-scene-health/SKILL.md +105 -0
  63. package/skills/xr-comfort-audit/SKILL.md +95 -0
@@ -0,0 +1 @@
1
+ {"version":3,"file":"writeTools.js","sourceRoot":"","sources":["../src/writeTools.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,CAAC,EAAE,MAAM,KAAK,CAAC;AACxB,OAAO,EAAE,MAAM,EAAE,MAAM,kBAAkB,CAAC;AA4C1C;;;;GAIG;AACH,MAAM,OAAO,sBAAuB,SAAQ,KAAK;IAC/C,YAAY,MAAc;QACxB,KAAK,CACH,mDAAmD,MAAM,4DAA4D,CACtH,CAAC;QACF,IAAI,CAAC,IAAI,GAAG,wBAAwB,CAAC;IACvC,CAAC;CACF;AAED,SAAS,WAAW,CAAC,MAAuB;IAC1C,IAAI,CAAC,MAAM,CAAC,IAAI;QAAE,MAAM,IAAI,sBAAsB,CAAC,MAAM,CAAC,CAAC;IAC3D,OAAO,MAAM,CAAC,IAAI,CAAC,IAAI,CAAC,MAAM,CAAC,CAAC;AAClC,CAAC;AAED,SAAS,UAAU,CAAC,MAAuB;IACzC,IAAI,CAAC,MAAM,CAAC,GAAG;QAAE,MAAM,IAAI,sBAAsB,CAAC,KAAK,CAAC,CAAC;IACzD,OAAO,MAAM,CAAC,GAAG,CAAC,IAAI,CAAC,MAAM,CAAC,CAAC;AACjC,CAAC;AAED,qFAAqF;AACrF,SAAS,OAAO,CAAC,IAA6B;IAC5C,OAAO,MAAM,CAAC,WAAW,CAAC,MAAM,CAAC,OAAO,CAAC,IAAI,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,KAAK,CAAC,EAAE,EAAE,CAAC,KAAK,KAAK,SAAS,CAAC,CAAC,CAAC;AAC7F,CAAC;AAED,MAAM,UAAU,GAAG,CAAC;KACjB,IAAI,CAAC,CAAC,SAAS,EAAE,OAAO,EAAE,MAAM,EAAE,QAAQ,EAAE,QAAQ,EAAE,QAAQ,CAAC,CAAC;KAChE,QAAQ,CACP,qGAAqG,CACtG,CAAC;AAEJ,MAAM,CAAC,MAAM,YAAY,GAAc;IACrC,IAAI,EAAE,UAAU;IAChB,KAAK,EAAE,oBAAoB;IAC3B,WAAW,EACT,qUAAqU;IACvU,WAAW,EAAE;QACX,UAAU;QACV,QAAQ,EAAE,CAAC;aACR,MAAM,EAAE;aACR,GAAG,CAAC,CAAC,CAAC;aACN,GAAG,CAAC,MAAM,CAAC,2BAA2B,CAAC;aACvC,QAAQ,EAAE;aACV,QAAQ,CAAC,oEAAoE,CAAC;QACjF,KAAK,EAAE,CAAC;aACL,MAAM,EAAE;aACR,GAAG,EAAE;aACL,WAAW,EAAE;aACb,QAAQ,EAAE;aACV,QAAQ,CAAC,8DAA8D,CAAC;QAC3E,KAAK,EAAE,CAAC;aACL,MAAM,EAAE;aACR,GAAG,EAAE;aACL,WAAW,EAAE;aACb,QAAQ,EAAE;aACV,QAAQ,CAAC,4DAA4D,CAAC;QACzE,IAAI,EAAE,CAAC;aACJ,MAAM,EAAE;aACR,GAAG,CAAC,CAAC,CAAC;aACN,GAAG,CAAC,MAAM,CAAC,uBAAuB,CAAC;aACnC,QAAQ,CAAC,kCAAkC,CAAC;KAChD;IACD,OAAO,EAAE,IAAI;IACb,OAAO,EAAE,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,EAAE,CAC9B,WAAW,CAAC,MAAM,CAAC,CACjB,qBAAqB,EACrB,OAAO,CAAC;QACN,UAAU,EAAE,IAAI,CAAC,UAAU;QAC3B,QAAQ,EAAE,IAAI,CAAC,QAAQ;QACvB,KAAK,EAAE,IAAI,CAAC,KAAK;QACjB,KAAK,EAAE,IAAI,CAAC,KAAK;QACjB,IAAI,EAAE,IAAI,CAAC,IAAI;KAChB,CAAC,CACH;CACJ,CAAC;AAEF,MAAM,CAAC,MAAM,cAAc,GAAc;IACvC,IAAI,EAAE,aAAa;IACnB,KAAK,EAAE,eAAe;IACtB,WAAW,EACT,gPAAgP;IAClP,WAAW,EAAE;QACX,IAAI,EAAE,CAAC;aACJ,MAAM,EAAE;aACR,GAAG,CAAC,CAAC,CAAC;aACN,GAAG,CAAC,MAAM,CAAC,qBAAqB,CAAC;aACjC,QAAQ,CAAC,4DAA4D,CAAC;QACzE,OAAO,EAAE,CAAC;aACP,MAAM,EAAE;aACR,GAAG,CAAC,CAAC,CAAC;aACN,GAAG,CAAC,MAAM,CAAC,wBAAwB,CAAC;aACpC,QAAQ,CAAC,gCAAgC,CAAC;KAC9C;IACD,OAAO,EAAE,IAAI;IACb,OAAO,EAAE,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,EAAE,CAC9B,UAAU,CAAC,MAAM,CAAC,CAAC,oBAAoB,kBAAkB,CAAC,MAAM,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC,EAAE,EAAE;QAC9E,OAAO,EAAE,IAAI,CAAC,OAAO;KACtB,CAAC;CACL,CAAC;AAEF,MAAM,CAAC,MAAM,gBAAgB,GAAc;IACzC,IAAI,EAAE,eAAe;IACrB,KAAK,EAAE,kBAAkB;IACzB,WAAW,EACT,6PAA6P;IAC/P,WAAW,EAAE;QACX,KAAK,EAAE,CAAC;aACL,MAAM,EAAE;aACR,GAAG,CAAC,CAAC,CAAC;aACN,GAAG,CAAC,MAAM,CAAC,2BAA2B,CAAC;aACvC,QAAQ,CAAC,gEAAgE,CAAC;QAC7E,KAAK,EAAE,CAAC;aACL,MAAM,CAAC,CAAC,CAAC,MAAM,EAAE,EAAE,CAAC,CAAC,OAAO,EAAE,CAAC;aAC/B,QAAQ,CACP,uHAAuH,CACxH;QACH,UAAU,EAAE,CAAC;aACV,MAAM,EAAE;aACR,GAAG,CAAC,MAAM,CAAC,gCAAgC,CAAC;aAC5C,QAAQ,EAAE;aACV,QAAQ,CAAC,mDAAmD,CAAC;KACjE;IACD,OAAO,EAAE,IAAI;IACb,OAAO,EAAE,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,EAAE,CAC9B,WAAW,CAAC,MAAM,CAAC,CACjB,kBAAkB,EAClB,OAAO,CAAC,EAAE,KAAK,EAAE,IAAI,CAAC,KAAK,EAAE,KAAK,EAAE,IAAI,CAAC,KAAK,EAAE,UAAU,EAAE,IAAI,CAAC,UAAU,EAAE,CAAC,CAC/E;CACJ,CAAC;AAEF,MAAM,CAAC,MAAM,mBAAmB,GAAc;IAC5C,IAAI,EAAE,kBAAkB;IACxB,KAAK,EAAE,gCAAgC;IACvC,WAAW,EACT,sOAAsO;IACxO,WAAW,EAAE;QACX,UAAU,EAAE,UAAU,CAAC,QAAQ,EAAE;QACjC,QAAQ,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC,MAAM,CAAC,2BAA2B,CAAC,CAAC,QAAQ,EAAE;QAC9E,KAAK,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,GAAG,EAAE,CAAC,WAAW,EAAE,CAAC,QAAQ,EAAE;QAChD,KAAK,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,GAAG,EAAE,CAAC,WAAW,EAAE,CAAC,QAAQ,EAAE;QAChD,KAAK,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,GAAG,EAAE,CAAC,QAAQ,EAAE,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC,QAAQ,EAAE;KACvD;IACD,OAAO,EAAE,KAAK;IACd,OAAO,EAAE,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,EAAE,CAC9B,MAAM,CAAC,GAAG,CAAC,qBAAqB,EAAE;QAChC,UAAU,EAAE,IAAI,CAAC,UAAgC;QACjD,QAAQ,EAAE,IAAI,CAAC,QAA8B;QAC7C,KAAK,EAAE,IAAI,CAAC,KAA2B;QACvC,KAAK,EAAE,IAAI,CAAC,KAA2B;QACvC,KAAK,EAAE,IAAI,CAAC,KAA2B;KACxC,CAAC;CACL,CAAC;AAEF,MAAM,CAAC,MAAM,gBAAgB,GAAc;IACzC,IAAI,EAAE,eAAe;IACrB,KAAK,EAAE,6BAA6B;IACpC,WAAW,EACT,wJAAwJ;IAC1J,WAAW,EAAE,EAAE,KAAK,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,GAAG,EAAE,CAAC,QAAQ,EAAE,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC,QAAQ,EAAE,EAAE;IACvE,OAAO,EAAE,KAAK;IACd,OAAO,EAAE,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,EAAE,CAC9B,MAAM,CAAC,GAAG,CAAC,kBAAkB,EAAE,EAAE,KAAK,EAAE,IAAI,CAAC,KAA2B,EAAE,CAAC;CAC9E,CAAC;AAEF,MAAM,CAAC,MAAM,gBAAgB,GAAc;IACzC,IAAI,EAAE,eAAe;IACrB,KAAK,EAAE,mCAAmC;IAC1C,WAAW,EACT,qIAAqI;IACvI,WAAW,EAAE,EAAE,KAAK,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,GAAG,EAAE,CAAC,QAAQ,EAAE,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC,QAAQ,EAAE,EAAE;IACvE,OAAO,EAAE,KAAK;IACd,OAAO,EAAE,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,EAAE,CAC9B,MAAM,CAAC,GAAG,CAAC,kBAAkB,EAAE,EAAE,KAAK,EAAE,IAAI,CAAC,KAA2B,EAAE,CAAC;CAC9E,CAAC;AAEF,8EAA8E;AAC9E,EAAE;AACF,0EAA0E;AAC1E,yEAAyE;AACzE,2EAA2E;AAC3E,8EAA8E;AAC9E,4EAA4E;AAC5E,4EAA4E;AAE5E,MAAM,CAAC,MAAM,YAAY,GAAc;IACrC,IAAI,EAAE,WAAW;IACjB,KAAK,EAAE,2CAA2C;IAClD,WAAW,EACT,0FAA0F;QAC1F,4FAA4F;QAC5F,8FAA8F;QAC9F,uFAAuF;QACvF,0FAA0F;QAC1F,6FAA6F;QAC7F,0FAA0F;QAC1F,2FAA2F;QAC3F,0FAA0F;IAC5F,WAAW,EAAE;QACX,KAAK,EAAE,CAAC;aACL,MAAM,EAAE;aACR,GAAG,CAAC,CAAC,CAAC;aACN,GAAG,CAAC,MAAM,CAAC,uBAAuB,CAAC;aACnC,QAAQ,CAAC,4EAA4E,CAAC;QACzF,KAAK,EAAE,CAAC;aACL,MAAM,CAAC,CAAC,CAAC,MAAM,EAAE,EAAE,CAAC,CAAC,OAAO,EAAE,CAAC;aAC/B,QAAQ,CACP,qJAAqJ,CACtJ;QACH,KAAK,EAAE,CAAC;aACL,IAAI,CAAC,CAAC,MAAM,EAAE,OAAO,EAAE,KAAK,EAAE,MAAM,EAAE,MAAM,EAAE,WAAW,EAAE,SAAS,CAAC,CAAC;aACtE,QAAQ,CAAC,uDAAuD,CAAC;QACpE,QAAQ,EAAE,CAAC;aACR,MAAM,CAAC;YACN,CAAC,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,GAAG,CAAC,MAAM,CAAC,4BAA4B,CAAC,CAAC,QAAQ,EAAE;YACjE,CAAC,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,GAAG,CAAC,MAAM,CAAC,4BAA4B,CAAC,CAAC,QAAQ,EAAE;YACjE,MAAM,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,GAAG,CAAC,MAAM,CAAC,4BAA4B,CAAC,CAAC,QAAQ,EAAE;SACvE,CAAC;aACD,QAAQ,EAAE;aACV,QAAQ,CACP,uGAAuG,CACxG;QACH,IAAI,EAAE,CAAC;aACJ,KAAK,CAAC,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,CAAC,CAAC;aACnC,QAAQ,EAAE;aACV,QAAQ,CAAC,kDAAkD,CAAC;QAC/D,IAAI,EAAE,CAAC;aACJ,MAAM,EAAE;aACR,GAAG,CAAC,MAAM,CAAC,sBAAsB,CAAC;aAClC,QAAQ,EAAE;aACV,QAAQ,CAAC,qEAAqE,CAAC;KACnF;IACD,OAAO,EAAE,IAAI;IACb,OAAO,EAAE,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,EAAE,CAC9B,WAAW,CAAC,MAAM,CAAC,CACjB,gBAAgB,EAChB,OAAO,CAAC;QACN,CAAC,EAAE,CAAC;QACJ,KAAK,EAAE,IAAI,CAAC,KAAK;QACjB,KAAK,EAAE,IAAI,CAAC,KAAK;QACjB,KAAK,EAAE,IAAI,CAAC,KAAK;QACjB,QAAQ,EAAE,IAAI,CAAC,QAAQ;QACvB,IAAI,EAAE,IAAI,CAAC,IAAI;QACf,IAAI,EAAE,IAAI,CAAC,IAAI;KAChB,CAAC,CACH;CACJ,CAAC;AAEF,MAAM,CAAC,MAAM,cAAc,GAAc;IACvC,IAAI,EAAE,aAAa;IACnB,KAAK,EAAE,kCAAkC;IACzC,WAAW,EACT,8FAA8F;QAC9F,qFAAqF;QACrF,8FAA8F;QAC9F,iFAAiF;IACnF,WAAW,EAAE,EAAE,KAAK,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,GAAG,EAAE,CAAC,QAAQ,EAAE,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC,QAAQ,EAAE,EAAE;IACvE,OAAO,EAAE,KAAK;IACd,OAAO,EAAE,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,EAAE,CAC9B,MAAM,CAAC,GAAG,CAAC,gBAAgB,EAAE,EAAE,KAAK,EAAE,IAAI,CAAC,KAA2B,EAAE,CAAC;CAC5E,CAAC;AAEF,MAAM,CAAC,MAAM,cAAc,GAAc;IACvC,IAAI,EAAE,aAAa;IACnB,KAAK,EAAE,uBAAuB;IAC9B,WAAW,EACT,uFAAuF;QACvF,6FAA6F;QAC7F,0EAA0E;IAC5E,WAAW,EAAE;QACX,EAAE,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC,QAAQ,CAAC,mDAAmD,CAAC;KAC7F;IACD,OAAO,EAAE,IAAI;IACb,OAAO,EAAE,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,EAAE;QAC9B,IAAI,CAAC,MAAM,CAAC,MAAM;YAAE,MAAM,IAAI,sBAAsB,CAAC,QAAQ,CAAC,CAAC;QAC/D,OAAO,MAAM,CAAC,MAAM,CAAC,kBAAkB,kBAAkB,CAAC,MAAM,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC,EAAE,CAAC,CAAC;IAChF,CAAC;CACF,CAAC;AAEF;;;;;GAKG;AACH,MAAM,CAAC,MAAM,UAAU,GAAyB;IAC9C,YAAY;IACZ,cAAc;IACd,gBAAgB;IAChB,YAAY;IACZ,cAAc;IACd,mBAAmB;IACnB,gBAAgB;IAChB,gBAAgB;IAChB,cAAc;CACf,CAAC;AAEF,mDAAmD;AACnD,MAAM,CAAC,MAAM,kBAAkB,GAAyB,UAAU,CAAC,MAAM,CAAC,CAAC,IAAI,EAAE,EAAE,CAAC,IAAI,CAAC,OAAO,CAAC,CAAC"}
package/llms.txt CHANGED
@@ -1,8 +1,9 @@
1
1
  # @uptimizr/agent-core
2
2
 
3
3
  > The framework-agnostic, browser-safe core for Uptimizr analytics agents: the single read-only
4
- > tool catalog over the collector query API, a headless LLM provider-adapter interface, and the
5
- > tool-calling loop. Defined once, consumed by MCP, the dashboard, and the demo (ADR 0050).
4
+ > analytics tool catalog over the collector query API, the `annotate`-gated project-metadata tools,
5
+ > a headless LLM provider-adapter interface, and the tool-calling loop. Defined once, consumed by
6
+ > MCP, the dashboard, and the demo (ADR 0050). Events are read-only (ADR 0051 §9).
6
7
 
7
8
  ## Docs
8
9
 
@@ -11,7 +12,7 @@
11
12
  - [Integration & API reference](https://github.com/RaananW/Uptimizr/blob/main/docs/integration.md): the underlying query endpoints.
12
13
  - [Architecture Decision Records](https://github.com/RaananW/Uptimizr/tree/main/docs/adr): in-browser assistant (0050), privacy model (0003), thin backends (0005), consumer-facing agents (0017).
13
14
 
14
- ## Tools (read-only)
15
+ ## Analytics tools (read-only)
15
16
 
16
17
  <!-- generated:registry-tool-names:start — generated by `pnpm gen:docs`; edit the metric registry, not this list -->
17
18
 
@@ -21,27 +22,175 @@
21
22
  `position_heatmap`, `session_trajectory`, `aggregate_paths`, `scene_coverage`, `camera_distance`,
22
23
  `click_rays`, `flow_links`, `top_meshes`, `mesh_sources`, `mesh_trend`, `mesh_dwell`,
23
24
  `mesh_blind_spots`, `mesh_interaction_kinds`, `mesh_reachability`, `dead_clicks`, `rage_clicks`,
24
- `hover_dwell`, `interaction_sources`, `top_input_actions`, `camera_gestures`, `navigation_stats`,
25
- `backtrack_ratio`, `perf_summary`, `render_scale_truth`, `perf_distribution`, `fps_histogram`,
26
- `frame_time_percentiles`, `jank_rate`, `perf_churn`, `perf_by_device`, `perf_by_scene`,
27
- `perf_heatmap`, `compile_stalls`, `resource_summary`, `resource_percentiles`, `stability_counts`,
28
- `graphics_diagnostics`, `error_heatmap`, `rendering_technology`, `capability_changes`,
29
- `xr_rotation`, `xr_sources`, `xr_abandonment`, `xr_locomotion`, `xr_tracking_quality`,
30
- `boundary_heatmap`, `boundary_heatmap_stats`, `xr_boundary_contacts`,
31
- `ar_placement_time_to_place`, `ar_placement_attempts`, `ar_placement_surfaces`, `funnel`,
32
- `scene_retention`, `load_bounce_funnel`, `variant_leaderboard`
25
+ `hover_dwell`, `interaction_sources`, `top_input_actions`, `custom_event_vocabulary`,
26
+ `camera_gestures`, `navigation_stats`, `backtrack_ratio`, `perf_summary`, `render_scale_truth`,
27
+ `perf_distribution`, `fps_histogram`, `frame_time_percentiles`, `jank_rate`, `perf_churn`,
28
+ `perf_by_device`, `perf_by_scene`, `perf_heatmap`, `compile_stalls`, `resource_summary`,
29
+ `resource_percentiles`, `stability_counts`, `graphics_diagnostics`, `error_heatmap`,
30
+ `rendering_technology`, `capability_changes`, `xr_rotation`, `xr_sources`, `xr_abandonment`,
31
+ `xr_locomotion`, `xr_tracking_quality`, `boundary_heatmap`, `boundary_heatmap_stats`,
32
+ `xr_boundary_contacts`, `ar_placement_time_to_place`, `ar_placement_attempts`,
33
+ `ar_placement_surfaces`, `funnel`, `scene_retention`, `load_bounce_funnel`, `variant_leaderboard`,
34
+ `insight_baseline`, `insight_movers`, `insight_anomalies`, `insight_significance`,
35
+ `insight_scene_health`
36
+
37
+ Only on a key holding `query:raw`, and only when the collector runs with
38
+ `ENABLE_RAW_SESSION_RETENTION` (ADR 0003):
39
+
40
+ `session_narrative`
33
41
 
34
42
  <!-- generated:registry-tool-names:end -->
35
43
 
44
+ ## Insight primitives (ADR 0051 §4)
45
+
46
+ Five of the tools are readings *about* the other metrics, and they are the right first call on an
47
+ open-ended question — "how are things?" does not mean "call thirty tools".
48
+
49
+ - `insight_scene_health` — one 0-100 score per scene over six weighted factors (perf stability,
50
+ jank, errors, dead clicks, coverage, XR abandonment), least healthy first. The right *unscoped*
51
+ starting point: it answers which scene to open. Every factor carries the `metric` id behind it,
52
+ its `raw` value and the project `baseline` it was compared with, so the next call and the
53
+ sentence to write are both already in the row. **50 is the project norm, not a pass mark** — the
54
+ score compares a scene with the rest of the project's recent past — and a factor with
55
+ `score: null` was not counted (its `note` says why).
56
+ - `insight_movers` — what changed against a reference window (the previous equal one by default),
57
+ for every comparable metric in scope, ranked by a robust z-score: the change divided by how much
58
+ that metric normally swings, so a noisy metric must move much further than a steady one to reach
59
+ the top. Bounded: at most 24 metrics are scanned per request.
60
+ - `insight_baseline` — what is normal for one metric in one scene: `median` and `mad`, the
61
+ `p10`..`p90` band, and the drift `slope`. Call it on whatever moved, to say whether the new level
62
+ is actually outside normal rather than merely different.
63
+ - `insight_anomalies` — *when* one metric went wrong. One row per bucket that does not belong in
64
+ the series: `kind: "spike" | "drop"` for a bucket more than `sensitivity` (default 3) standard
65
+ deviations from the trailing window, and `kind: "shift"` at the bucket where the level moved and
66
+ stayed moved. Where the metric declares a dimension it can be split by, `contributor` names the
67
+ mesh, source, input action, event type or scene holding the largest `share` of the excess.
68
+ Bounded: one scan for the series plus at most three attribution scans per request.
69
+ - `insight_significance` — whether a difference between two windows could be chance: `effect`,
70
+ `ci95`, `p`, and the `test` that produced them (a two-proportion z with Wilson intervals for a
71
+ declared rate, Welch's t over the per-bucket values for a level, an exact Poisson rate test for a
72
+ bare count — picked from the measure, not from you). **Read `ci95` before `p`**: an interval that
73
+ straddles 0 means you cannot tell yet, and `powerNote` says what this much data could have
74
+ detected at all. Welch's `n` is the number of **buckets**, not events. It compares two *windows*,
75
+ not two segments: a variant-versus-variant question is a `400`, not a wrong answer.
76
+
77
+ Read `insight_anomalies`' `z` as standard deviations and `insight_movers`' as the same ratio
78
+ unscaled — they differ by a constant factor (1.4826) and must not be compared directly. A sustained
79
+ level change is reported both as a `shift` and, for the days right after it, as `drop`/`spike`
80
+ rows: two true statements about one event.
81
+
82
+ Two fields decide whether a mover is worth reporting, and both are easy to misread:
83
+
84
+ - `direction` is the registry's opinion of what a **rise** in that metric means — `up` (good),
85
+ `down` (bad) or `neutral`. It is *not* the direction of the move: read it with the sign of
86
+ `delta`, so a rise in a `down` metric (errors, dead clicks, jank) is a regression.
87
+ - `aboveMinSample: false` means the delta is real arithmetic but the denominator is below the
88
+ metric's declared minimum. Such rows are returned rather than dropped — "we cannot tell" and
89
+ "nothing changed" are different answers — and must never be reported as findings.
90
+
91
+ Statistics are computed in pure TypeScript over portable day/hour buckets, so the same data yields
92
+ the same answer on every storage engine. Only `comparable` metrics with a faithful per-bucket form
93
+ participate; asking for one that has none returns `400` naming every id that does.
94
+
95
+ ## Result formats (`format`)
96
+
97
+ Every aggregate tool declares a `format` argument — the envelope its rows arrive in, not a filter
98
+ (ADR 0051 §2). `buildRequest` applies `DEFAULT_TOOL_FORMAT` (`table`) when the caller names none
99
+ and sends it explicitly, so the collector's own default stays `full`. A tool's output schema
100
+ describes all three envelopes, so whichever one comes back validates.
101
+
102
+ - `summary` — **prefer this whenever a model reads the result.** A bounded digest: `ranked` top
103
+ rows, a `series` trend, merged spatial `clusters`, or a single `record`, with shares, a sample
104
+ size, the metric's `caveats` and a templated `reading` sentence. Capped at the metric's
105
+ `limits.maxSummaryRows`, so a 500-bin heatmap costs the same as a 5-bin one — which is what keeps
106
+ one heatmap from filling a small local model's whole context.
107
+ - `table` — **the tools' default.** `{ meta, rows }`: every row plus the metric, range, applied
108
+ filters, sample size, row count, a truncation flag and the registry limits. Self-describing but
109
+ not bounded — ask for `summary` when the result could be large.
110
+ - `full` — the bare rows, with no envelope at all.
111
+
112
+ `reading` and `caveats` are templated by pure code, never a model, so identical rows always produce
113
+ identical words. Shares appear only where the measure can honestly be summed (an FPS or ratio metric
114
+ reports `total: null` and no shares). Cluster coordinates are grid indices — multiply by the
115
+ effective `cellSize` for world space. `session_meta` and `scene_representation` are stored records,
116
+ not aggregations, and declare no `format`.
117
+
118
+ ## Packaged methodology skills (ADR 0051 §7)
119
+
120
+ Shipped in this tarball as Agent Skills files under `skills/<name>/SKILL.md` — frontmatter (tools,
121
+ capabilities, arguments) plus a numbered method. `AGENT_SKILLS` is those files compiled to data;
122
+ `getAgentSkill(name)` resolves either spelling and `render(args)` substitutes the scene and range.
123
+ The same files back `@uptimizr/mcp`'s prompt templates and `uptimizr agent report --skill`.
124
+
125
+ <!-- generated:registry-skill-names:start — generated by `pnpm gen:docs`; edit the SKILL.md files, not this list -->
126
+
127
+ - `attention_hotspots` (scene (required), range) — Find where visitors look and click in a scene: view-direction concentration, gaze→mesh flow, the objects that draw the most interaction, and the ones nobody ever notices. USE FOR: deciding where to put a call to action, finding ignored or invisible content, explaining why an object gets no clicks, laying out a scene around what people actually look at.
128
+ Method: `skills/attention-hotspots/SKILL.md`. Tools: `camera_heatmap`, `flow_links`, `click_rays`, `top_meshes`, `mesh_dwell`, `mesh_blind_spots`, `query`.
129
+ - `conversion_investigation` (scene, range) — Find out where a funnel loses people and whether the loss is real: step-by-step drop-off, the bounce that happens before the funnel even starts, scene-to-scene retention, variant performance, and the interaction failures (dead clicks, rage clicks, unreachable meshes) that explain a stalled step. USE FOR: a funnel that converts worse than expected, an A/B variant comparison, "where do people drop off", diagnosing a step nobody completes.
130
+ Method: `skills/conversion-investigation/SKILL.md`. Tools: `funnel`, `load_bounce_funnel`, `scene_retention`, `variant_leaderboard`, `dead_clicks`, `rage_clicks`, `mesh_reachability`, `flow_links`, `insight_significance`, `insight_movers`, `query`.
131
+ - `performance_regression_triage` (scene, range) — Triage a frame-rate or stability regression: confirm it moved, date it, locate it (which scene, device class, place in the scene), and name the mechanism — jank, shader compile stalls, memory pressure, a render-scale change or a rendering-technology shift. USE FOR: "the app got slower", a FPS drop after a release, stutter reports, deciding whether a regression is real or noise.
132
+ Method: `skills/performance-regression-triage/SKILL.md`. Tools: `insight_movers`, `insight_anomalies`, `insight_significance`, `insight_baseline`, `perf_summary`, `perf_distribution`, `frame_time_percentiles`, `jank_rate`, `perf_by_device`, `perf_by_scene`, `perf_heatmap`, `compile_stalls`, `resource_percentiles`, `render_scale_truth`, `rendering_technology`, `query`.
133
+ - `weekly_scene_health` (scene, range) — A weekly health check for a scene (or the whole project): a weighted health score with every factor traced back to the metric behind it, what changed against last week, traffic, event mix, performance, and the most-interacted meshes. USE FOR: the recurring "how is the scene doing?" review, a scheduled weekly or monthly report, a first look at a project you do not know yet, deciding which scene to investigate next.
134
+ Method: `skills/weekly-scene-health/SKILL.md`. Tools: `insight_scene_health`, `insight_movers`, `insight_baseline`, `insight_significance`, `insight_anomalies`, `event_counts`, `timeseries`, `perf_summary`, `top_meshes`, `list_sessions`, `query`.
135
+ - `xr_comfort_audit` (scene, range) — Audit VR/AR comfort for a scene (or the whole project): rapid head rotation, locomotion style, tracking quality, guardian/boundary contacts, input-source mix, and the short sessions that mean someone took the headset off. USE FOR: motion-sickness complaints, immersive sessions that end early, choosing a locomotion scheme, checking whether a play space is big enough.
136
+ Method: `skills/xr-comfort-audit/SKILL.md`. Tools: `xr_rotation`, `xr_locomotion`, `xr_abandonment`, `xr_sources`, `xr_tracking_quality`, `xr_boundary_contacts`, `boundary_heatmap_stats`, `insight_scene_health`, `insight_movers`, `query`.
137
+
138
+ <!-- generated:registry-skill-names:end -->
139
+
36
140
  ## Key exports
37
141
 
38
142
  - `readTools` — the read-only tool catalog (one entry per query endpoint), generated from the
39
- `@uptimizr/metrics` metric registry: 69 tools, each with an input schema, an output schema and the
40
- metric's interpretation notes and caveats in its description (ADR 0051).
143
+ `@uptimizr/metrics` metric registry plus `query` and `list_subscriptions`: 77 tools, each with an
144
+ input schema, an output schema covering all three `format` envelopes, and the metric's
145
+ interpretation notes and caveats in its description (ADR 0051).
146
+ - `writeTools` / `mutatingWriteTools` — the project-metadata tools `annotate`, `define_term`,
147
+ `save_analysis` (ADR 0051 §5), `pin_panel` and `unpin_panel` (ADR 0051 §7), plus the reads
148
+ `list_annotations`, `list_glossary`, `list_analyses`, `list_panels`. Exported individually too
149
+ (`annotateTool`, `defineTermTool`, `saveAnalysisTool`, `pinPanelTool`, `unpinPanelTool`,
150
+ `listAnnotationsTool`, `listGlossaryTool`, `listAnalysesTool`, `listPanelsTool`). A
151
+ separate export from `readTools` on purpose, so an integration's read-only stance stays
152
+ inspectable. They need an `annotate` key, write metadata only, and cannot touch an event.
153
+ `pin_panel` stores a panel **spec** — a metric id, a chart name, some column names — that the
154
+ dashboard renders with panels it already ships, so nothing is loaded and nothing is evaluated;
155
+ send the `query` document with `range: "inherit"` and a chart the metric's grain supports, or the
156
+ collector refuses it and names the charts that would have worked. `unpin_panel` removes a panel
157
+ for everyone on the project and is the one tool here that uses `CollectorClient.delete`.
41
158
  - `coreReadTools` / `selectReadTools(kind)` / `filterReadTools(names)` — narrowed views of the
42
159
  catalog for models that cannot carry every tool schema.
43
160
  - `registryToTools(metrics?)` — generate the catalog from metric-registry definitions.
44
- - `createCollectorClient(config)` — the thin GET-only collector client.
161
+ - `createCollectorClient(config)` — the thin collector client: `get` for every analytics read, plus
162
+ `post`/`put`/`delete` used only by the metadata write tools.
45
163
  - `runAgent(options)` — the headless tool-calling loop (LLM ↔ tools ↔ collector).
46
164
  - `toToolSchemas(tools?)` — convert catalog tools to JSON-Schema tool descriptors for an LLM.
165
+ - `renderContextForPrompt(context, nowMs?)` — render the collector's project context document
166
+ (`GET /api/v1/context`, ADR 0051 §5) as a compact system-prompt block: the project's real scene
167
+ ids, named region ids, custom-event names with their prop keys and coarse types, top meshes, input
168
+ actions, data freshness, retention flags and the metrics that are empty because their capture
169
+ channel is off. Pure, ~1.5 k characters at most, tolerant of a partial document, and `""` when
170
+ there is nothing to say — so an older collector degrades instead of failing. Read the context
171
+ first and inject this before the first question.
172
+ - `AGENT_SKILLS` / `getAgentSkill(name)` / `renderAgentSkill(name, args)` — the curated
173
+ investigations packaged as `skills/<name>/SKILL.md` in this tarball (`attention_hotspots`,
174
+ `conversion_investigation`, `performance_regression_triage`, `weekly_scene_health`,
175
+ `xr_comfort_audit`): title, description, the registry tools each method relies on, its arguments,
176
+ and a pure `render(args)` producing the user turn. `@uptimizr/mcp` registers them as prompt
177
+ templates and `uptimizr agent report --skill` runs them headlessly, from this one definition.
178
+ `getAgentSkill` takes either spelling, and the pre-#316 name `xr_comfort_review` still resolves.
179
+ - `ANALYTICS_AGENT_GUIDELINES` / `renderCurrentTimeLine(nowMs)` — the system-prompt fragments every
180
+ analytics agent shares (what it may claim about the data; the clock it resolves "this week"
181
+ against). Clients add only their own role sentence and output format.
47
182
  - `LlmProvider` — the pluggable, user-controlled LLM backend interface (no bundled model or key).
183
+ - `ProviderResponse.usage` — optional `{ inputTokens?, outputTokens? }` per turn, populated by the
184
+ hosted adapters from Anthropic's `usage` and OpenAI's `prompt_tokens`/`completion_tokens`
185
+ (buffered and streamed). Absent means **not reported**, never zero.
186
+
187
+ ## Capability-gated tools (ADR 0051 §7)
188
+
189
+ - `readTools` is the `query` surface; `rawTools` holds the tools whose registry endpoint declares `capability: "query:raw"` (today `session_narrative`, the compacted account of one session).
190
+ - Keep them apart: the capability belongs to the key the agent was handed. Register `rawTools` only after `GET /api/v1/whoami` confirms `query:raw` — the collector also requires `ENABLE_RAW_SESSION_RETENTION`, so the capability is necessary but not sufficient (ADR 0003).
191
+
192
+ ## Non-metric reads
193
+
194
+ - `list_subscriptions` — the project's conditional subscriptions (ADR 0051 §6). Not a registry
195
+ metric: configuration, not a measurement. No arguments, no time range. Creating/deleting one
196
+ needs `annotate` and is not a tool.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@uptimizr/agent-core",
3
- "version": "1.1.0",
3
+ "version": "1.2.0",
4
4
  "description": "Framework-agnostic, browser-safe core for Uptimizr analytics agents — the single read-only tool catalog over the collector query API, a headless LLM provider-adapter interface, and the tool-calling loop.",
5
5
  "keywords": [
6
6
  "uptimizr",
@@ -51,6 +51,7 @@
51
51
  },
52
52
  "files": [
53
53
  "dist",
54
+ "skills",
54
55
  "README.md",
55
56
  "LICENSE",
56
57
  "AGENTS.md",
@@ -58,8 +59,9 @@
58
59
  ],
59
60
  "sideEffects": false,
60
61
  "dependencies": {
61
- "zod": "^4.5.4",
62
- "@uptimizr/metrics": "0.1.0"
62
+ "zod": "^4.6.5",
63
+ "@uptimizr/metrics": "0.2.0",
64
+ "@uptimizr/schema": "1.2.0"
63
65
  },
64
66
  "peerDependencies": {
65
67
  "@mlc-ai/web-llm": "^0.2.0"
@@ -70,9 +72,9 @@
70
72
  }
71
73
  },
72
74
  "devDependencies": {
73
- "@mlc-ai/web-llm": "^0.2.84",
75
+ "@mlc-ai/web-llm": "^0.2.85",
74
76
  "esbuild": "^0.28.2",
75
- "vitest": "^4.1.11"
77
+ "vitest": "^5.0.1"
76
78
  },
77
79
  "engines": {
78
80
  "node": ">=22"
@@ -0,0 +1,88 @@
1
+ ---
2
+ name: attention_hotspots
3
+ title: Attention hot-spots for a scene
4
+ description: >-
5
+ Find where visitors look and click in a scene: view-direction concentration, gaze→mesh flow, the
6
+ objects that draw the most interaction, and the ones nobody ever notices. USE FOR: deciding where
7
+ to put a call to action, finding ignored or invisible content, explaining why an object gets no
8
+ clicks, laying out a scene around what people actually look at. Trigger phrases: what do people
9
+ look at, attention hotspots, where do visitors click, which meshes get ignored, blind spots,
10
+ gaze heatmap, is anyone seeing this object.
11
+ tools:
12
+ - camera_heatmap
13
+ - flow_links
14
+ - click_rays
15
+ - top_meshes
16
+ - mesh_dwell
17
+ - mesh_blind_spots
18
+ - query
19
+ capabilities:
20
+ - query
21
+ args:
22
+ - name: scene
23
+ required: true
24
+ description: The scene id to analyse (see the uptimizr://scenes resource).
25
+ - name: range
26
+ required: false
27
+ default: the last 7 days
28
+ description: The window to analyse, in words — e.g. "the last 7 days", "since launch".
29
+ ---
30
+
31
+ Where does attention concentrate in scene "{{scene}}" over {{range}}?
32
+
33
+ Work through the method below with the read-only tools it names — all of them scoped with
34
+ `scene="{{scene}}"` — and synthesise one answer.
35
+
36
+ 1. **Orient before you ask anything.** Read the `uptimizr://context` resource first: it gives the
37
+ real scene ids, the scene's **named regions** and the custom-event names this project emits, and
38
+ it tells you which metrics are empty because their capture channel is off. Name regions the way
39
+ the project names them — "the checkout counter", not "the cluster at x≈3".
40
+
41
+ 2. **Where do they look?** `camera_heatmap` (`scene="{{scene}}"`) gives the view-direction
42
+ distribution — what people point the camera at, whether or not they ever click it. Ask for
43
+ `format: "summary"`: the digest merges neighbouring cells into a handful of clusters with a
44
+ share each and a plain-language `reading`, which is what you want here; the raw grid is
45
+ thousands of cells you cannot describe.
46
+
47
+ 3. **Does looking turn into touching?** `flow_links` (`scene="{{scene}}"`) links where the gaze was
48
+ to the mesh that was then clicked. A strong link is a working call to action; a heavy look with
49
+ no outgoing link is content that draws the eye and then disappoints.
50
+
51
+ 4. **Where do the clicks land?** `click_rays` (`scene="{{scene}}"`) gives view-gated clicks per
52
+ voxel and mesh — clicks attributed to what the visitor could actually see, not to whatever the
53
+ ray happened to pass through.
54
+
55
+ 5. **Rank the objects.** `top_meshes` (`scene="{{scene}}"`) for the most-interacted meshes, and
56
+ `mesh_dwell` (`scene="{{scene}}"`) for how long attention rests on each one. Dwell without
57
+ interaction is hesitation, and it usually means the object looks clickable and is not, or is
58
+ clickable and does not look it.
59
+
60
+ 6. **Name the cold half.** `mesh_blind_spots` (`scene="{{scene}}"`) lists the meshes that are
61
+ present and essentially never noticed. A hot-spot report that only names hot spots tells you
62
+ nothing about the content you paid to build.
63
+
64
+ 7. **Narrow it with the DSL.** For anything the canned tools do not expose, use the single `query`
65
+ tool: pick the `metric`, bound it with `range`, filter it, and set `format: "summary"` for a
66
+ bounded digest — each summary row carries a `drillQuery` you can send straight back instead of
67
+ rebuilding the filter. Set `compare: { range: <previous window> }` to see whether a hot spot is
68
+ new, and `explain: true` when a result looks wrong or empty: the plan names the capture channel,
69
+ the sample size and the row cap behind it.
70
+
71
+ ## What to report
72
+
73
+ - The two or three real hot-spots, named with the project's own region and mesh names, each with
74
+ its share of attention.
75
+ - The cold areas and the meshes nobody notices.
76
+ - Where gaze fails to convert into interaction, and what that implies for layout and
77
+ call-to-action placement.
78
+
79
+ Carry the caveats: heatmaps are gated on the view/pointer capture channels and are sampled
80
+ (ADR 0012), so a share is a share _of the sampled events_; say so, and say when a result was
81
+ truncated (`meta.truncated`) or sits below the metric's own minimum sample. Do not turn a voxel
82
+ cluster into a claim about one object unless `click_rays` or `flow_links` attributes it to that
83
+ mesh.
84
+
85
+ End with 2–3 concrete layout or content recommendations. If an `annotate` tool is available, leave
86
+ a note on the region you want revisited — a region-scoped annotation is what makes the next report
87
+ open where this one ended. If a `pin_panel` tool is available, pin the heatmap panel you reasoned
88
+ from.
@@ -0,0 +1,97 @@
1
+ ---
2
+ name: conversion_investigation
3
+ title: Conversion investigation
4
+ description: >-
5
+ Find out where a funnel loses people and whether the loss is real: step-by-step drop-off, the
6
+ bounce that happens before the funnel even starts, scene-to-scene retention, variant performance,
7
+ and the interaction failures (dead clicks, rage clicks, unreachable meshes) that explain a stalled
8
+ step. USE FOR: a funnel that converts worse than expected, an A/B variant comparison, "where do
9
+ people drop off", diagnosing a step nobody completes. Trigger phrases: conversion, funnel,
10
+ drop-off, why are people leaving, which variant wins, bounce rate, retention, people get stuck.
11
+ tools:
12
+ - funnel
13
+ - load_bounce_funnel
14
+ - scene_retention
15
+ - variant_leaderboard
16
+ - dead_clicks
17
+ - rage_clicks
18
+ - mesh_reachability
19
+ - flow_links
20
+ - insight_significance
21
+ - insight_movers
22
+ - query
23
+ capabilities:
24
+ - query
25
+ args:
26
+ - name: scene
27
+ required: false
28
+ description: Optional scene id to scope the investigation to (see the uptimizr://scenes resource).
29
+ - name: range
30
+ required: false
31
+ default: the last 7 days
32
+ description: The window to investigate, in words — e.g. "the last 7 days", "since the release".
33
+ ---
34
+
35
+ Investigate conversion for {{scope}} over {{range}}: where do people drop off, and is the drop real?
36
+
37
+ Work through the method below with the read-only tools it names, then answer.
38
+
39
+ 1. **Orient before you ask anything.** Read the `uptimizr://context` resource first. A funnel is
40
+ built out of **this project's own event types and custom-event names** — invent one and every
41
+ step reads zero. The context document lists the vocabulary the application actually emits, the
42
+ real scene ids, and which metrics are empty because their capture channel is off.
43
+
44
+ 2. **Check the step before the first step.** `load_bounce_funnel`{{#scene}} (`scene="{{scene}}"`){{/scene}}
45
+ measures load → first interaction → stay. If people leave before the funnel starts, nothing
46
+ inside it will explain the number, and a "conversion problem" is really a load or a first-impression
47
+ problem.
48
+
49
+ 3. **Run the funnel itself.** `funnel`{{#scene}} (`scene="{{scene}}"`){{/scene}} with the `steps`
50
+ built from the context document's vocabulary. Read it as the _transition_ rates, not the totals:
51
+ the step with the worst step-to-step rate is the one to investigate, even when a later step has
52
+ fewer people in absolute terms.
53
+
54
+ 4. **Follow them out of the scene.** `scene_retention`{{#scene}} (`scene="{{scene}}"`){{/scene}}
55
+ shows where a visitor goes next. A step that "loses" people to the next scene is not a loss at
56
+ all; one that loses them to nothing is.
57
+
58
+ 5. **Explain the stalled step.** At the worst step, look for interaction failure rather than
59
+ intent: `dead_clicks` (clicks that hit nothing actionable), `rage_clicks` (repeated clicking in
60
+ one spot — frustration you can locate), `mesh_reachability` (the target is too far away or
61
+ behind something to be clicked at all) and `flow_links` (people look at the target and never
62
+ click it). One of these usually _is_ the drop-off.
63
+
64
+ 6. **Compare variants honestly.** `variant_leaderboard` ranks variants by conversion. A leaderboard
65
+ is not a verdict: check each variant's sample size before repeating its rate, and say plainly
66
+ when two variants are too close or too small to separate. `insight_significance` can test a
67
+ metric with a portable bucket series across two **windows**; it does not test one segment against
68
+ another, and if you ask it to it will say so — report that limitation rather than inventing a
69
+ p-value.
70
+
71
+ 7. **See whether this is new.** `insight_movers`{{#scene}} (`scene="{{scene}}"`){{/scene}} ranks
72
+ what changed against the previous equal window, so you can tell "this funnel has always been bad"
73
+ from "this funnel broke last Tuesday". Ignore any row with `aboveMinSample: false`.
74
+
75
+ 8. **Use the DSL for the cuts the canned tools do not expose.** The single `query` tool takes a
76
+ `metric`, a `range`, that metric's filters, and `compare: { range: <previous window> }` to return
77
+ `{ current, previous, delta, deltaPct }` already joined — do not subtract two runs by hand. Use
78
+ `dimensions` to regroup a portable count (by device class, source or scene) and
79
+ `format: "summary"` for a bounded digest with a `reading` and a `drillQuery` per row. When a step
80
+ reads zero, re-send it with `explain: true` before reporting it: the plan will tell you whether
81
+ the number is real or whether the channel behind it is switched off.
82
+
83
+ ## What to report
84
+
85
+ - The step that actually loses people, with its entry and exit counts and its transition rate.
86
+ - The mechanism, named: dead clicks on a specific mesh, an unreachable target, a bounce before the
87
+ first interaction, or a genuine loss of interest.
88
+ - What each variant did, with sample sizes, and whether the difference can be told apart from noise.
89
+
90
+ Carry the caveats into the text: funnel steps are only as good as the event vocabulary they were
91
+ built from, a rate over a handful of sessions is not a rate, `meta.truncated` means you are looking
92
+ at a cut-off list, and a disabled capture channel produces a zero that means "not measured".
93
+
94
+ End with 2–3 concrete recommendations tied to the step and the mesh they apply to. If an `annotate`
95
+ tool is available, leave a note on the failing step so the next investigation starts there, and use
96
+ `save_analysis` to store the funnel definition you settled on — the next run should not have to
97
+ guess the steps again. If a `pin_panel` tool is available, pin the funnel panel.
@@ -0,0 +1,106 @@
1
+ ---
2
+ name: performance_regression_triage
3
+ title: Performance regression triage
4
+ description: >-
5
+ Triage a frame-rate or stability regression: confirm it moved, date it, locate it (which scene,
6
+ device class, place in the scene), and name the mechanism — jank, shader compile stalls, memory
7
+ pressure, a render-scale change or a rendering-technology shift. USE FOR: "the app got slower",
8
+ a FPS drop after a release, stutter reports, deciding whether a regression is real or noise.
9
+ Trigger phrases: performance regression, FPS dropped, why is it slow, stutter, jank, frame drops,
10
+ did the last release slow things down, triage performance.
11
+ tools:
12
+ - insight_movers
13
+ - insight_anomalies
14
+ - insight_significance
15
+ - insight_baseline
16
+ - perf_summary
17
+ - perf_distribution
18
+ - frame_time_percentiles
19
+ - jank_rate
20
+ - perf_by_device
21
+ - perf_by_scene
22
+ - perf_heatmap
23
+ - compile_stalls
24
+ - resource_percentiles
25
+ - render_scale_truth
26
+ - rendering_technology
27
+ - query
28
+ capabilities:
29
+ - query
30
+ args:
31
+ - name: scene
32
+ required: false
33
+ description: Optional scene id to scope the triage to (see the uptimizr://scenes resource).
34
+ - name: range
35
+ required: false
36
+ default: the last 14 days
37
+ description: The window to triage, in words — e.g. "the last 14 days", "since the release".
38
+ ---
39
+
40
+ Triage the performance regression in {{scope}} over {{range}}: is it real, when did it start, who
41
+ does it hit, and what is causing it?
42
+
43
+ Work through the method below with the read-only tools it names, then answer.
44
+
45
+ 1. **Orient before you ask anything.** Read the `uptimizr://context` resource first: the real scene
46
+ ids, the data freshness, and which metrics are empty because their capture channel is off. A
47
+ performance channel that was never enabled looks exactly like a scene with no problem.
48
+
49
+ 2. **Confirm something moved.** `insight_movers`{{#scene}} (`scene="{{scene}}"`){{/scene}} ranks
50
+ every comparable metric against the previous equal window by how unusual the change is. Read
51
+ `direction` with the sign of `delta` — a _rise_ in jank or errors is a regression — and drop any
52
+ row with `aboveMinSample: false`.
53
+
54
+ 3. **Ask whether the new level is outside normal.** `insight_baseline` on `perf_summary` gives the
55
+ project's own median and spread; compare the new reading with `median` give or take a few `mad`,
56
+ or with the p10..p90 band. Frame rates are noisy, and "down 6 FPS" is routine in some projects
57
+ and an incident in others.
58
+
59
+ 4. **Prove it rather than asserting it.** `insight_significance` on the metric that moved reports
60
+ the effect, a 95 % interval and a p-value across the two windows. An interval straddling 0 means
61
+ you cannot tell yet; `powerNote` says what this much data could have detected at all. Say "not
62
+ yet distinguishable from noise" when that is the truth — it is a finding.
63
+
64
+ 5. **Put a date on it.** `insight_anomalies` (`metric=perf_summary`, `window=28`{{#scene}},
65
+ `scene="{{scene}}"`{{/scene}}) separates a one-day `spike`/`drop` from a `shift` — a level that
66
+ changed and stayed changed, which is what a release looks like. Quote the `bucketStart` and the
67
+ `contributor`. For the day-by-day shape around that date, ask the `query` tool for
68
+ `metric: "perf_daily"` — it is a registry metric with no canned tool of its own.
69
+
70
+ 6. **Locate it.** `perf_by_scene` (which scene), `perf_by_device` (which device class — a
71
+ regression that only hits low-end hardware is a different bug from one that hits everyone), and
72
+ `perf_heatmap`{{#scene}} (`scene="{{scene}}"`){{/scene}} for _where in the scene_ the frames are
73
+ being lost. Ask for `format: "summary"` on the heatmap: merged clusters with shares, not a grid.
74
+
75
+ 7. **Name the mechanism.** `frame_time_percentiles` and `jank_rate` separate "uniformly slower"
76
+ from "occasionally catastrophic" — the second is what users report and the average hides.
77
+ `perf_distribution` shows whether the whole population shifted or a tail got worse.
78
+ `compile_stalls` finds shader/pipeline compilation blocking the first seconds.
79
+ `resource_percentiles` finds GPU/memory pressure. `render_scale_truth` catches a resolution
80
+ change quietly doing the work the frame rate is getting credit for, and `rendering_technology`
81
+ catches a shift in the engine/renderer mix between the two windows — a "regression" that is
82
+ really a change in who is measuring.
83
+
84
+ 8. **Cut it any way you need with the DSL.** The single `query` tool takes a `metric`, a `range`,
85
+ that metric's filters and `compare: { range: <the window before the shift> }`, returning
86
+ `{ current, previous, delta, deltaPct }` already joined — never subtract two runs yourself. Use
87
+ `dimensions` to regroup a portable count, `format: "summary"` for a bounded digest with a
88
+ `reading` and a per-row `drillQuery`, and `explain: true` whenever a number looks impossible:
89
+ the plan names the sample size, the row cap and every capture channel that could make it lie.
90
+
91
+ ## What to report
92
+
93
+ - Whether the regression is real, with the effect, the interval and the honest verdict when it is
94
+ not yet distinguishable from noise.
95
+ - The date it started and whether it is a spike or a sustained shift.
96
+ - Who it hits: scene, device class, and where in the scene.
97
+ - The mechanism, named, with the metric that shows it.
98
+
99
+ Carry the caveats: percentiles below the metric's minimum sample, a session count too small to
100
+ generalise, sampled capture (ADR 0012), truncated results, and any device class whose share of
101
+ traffic changed between the windows — a mix shift moves the average without anything getting slower.
102
+
103
+ End with 2–3 concrete recommendations naming the scene, the device class or the asset they apply
104
+ to. If an `annotate` tool is available, leave a dated note on the shift so the next report can see
105
+ what happened; `save_analysis` keeps the triage for the post-mortem. If a `pin_panel` tool is
106
+ available, pin the panel that shows the regression.