@uptimizr/agent-core 1.1.1 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +221 -34
- package/README.md +57 -21
- package/dist/client.d.ts +22 -5
- package/dist/client.d.ts.map +1 -1
- package/dist/client.js +39 -10
- package/dist/client.js.map +1 -1
- package/dist/context.d.ts +93 -0
- package/dist/context.d.ts.map +1 -0
- package/dist/context.js +137 -0
- package/dist/context.js.map +1 -0
- package/dist/index.d.ts +10 -3
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +20 -2
- package/dist/index.js.map +1 -1
- package/dist/nonRegistryTools.d.ts +34 -0
- package/dist/nonRegistryTools.d.ts.map +1 -0
- package/dist/nonRegistryTools.js +70 -0
- package/dist/nonRegistryTools.js.map +1 -0
- package/dist/prompt.d.ts +33 -0
- package/dist/prompt.d.ts.map +1 -0
- package/dist/prompt.js +49 -0
- package/dist/prompt.js.map +1 -0
- package/dist/provider.d.ts +18 -0
- package/dist/provider.d.ts.map +1 -1
- package/dist/providers/anthropic.d.ts +18 -1
- package/dist/providers/anthropic.d.ts.map +1 -1
- package/dist/providers/anthropic.js +35 -3
- package/dist/providers/anthropic.js.map +1 -1
- package/dist/providers/openai.d.ts +14 -1
- package/dist/providers/openai.d.ts.map +1 -1
- package/dist/providers/openai.js +23 -1
- package/dist/providers/openai.js.map +1 -1
- package/dist/queryTool.d.ts +36 -0
- package/dist/queryTool.d.ts.map +1 -0
- package/dist/queryTool.js +123 -0
- package/dist/queryTool.js.map +1 -0
- package/dist/registryTools.d.ts +23 -1
- package/dist/registryTools.d.ts.map +1 -1
- package/dist/registryTools.js +185 -31
- package/dist/registryTools.js.map +1 -1
- package/dist/skills.d.ts +105 -0
- package/dist/skills.d.ts.map +1 -0
- package/dist/skills.generated.d.ts +44 -0
- package/dist/skills.generated.d.ts.map +1 -0
- package/dist/skills.generated.js +511 -0
- package/dist/skills.generated.js.map +1 -0
- package/dist/skills.js +144 -0
- package/dist/skills.js.map +1 -0
- package/dist/tools.d.ts +55 -8
- package/dist/tools.d.ts.map +1 -1
- package/dist/tools.js +41 -1
- package/dist/tools.js.map +1 -1
- package/dist/writeTools.d.ts +68 -0
- package/dist/writeTools.d.ts.map +1 -0
- package/dist/writeTools.js +256 -0
- package/dist/writeTools.js.map +1 -0
- package/llms.txt +148 -20
- package/package.json +7 -5
- package/skills/attention-hotspots/SKILL.md +88 -0
- package/skills/conversion-investigation/SKILL.md +97 -0
- package/skills/performance-regression-triage/SKILL.md +106 -0
- package/skills/weekly-scene-health/SKILL.md +105 -0
- package/skills/xr-comfort-audit/SKILL.md +95 -0
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"writeTools.js","sourceRoot":"","sources":["../src/writeTools.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,CAAC,EAAE,MAAM,KAAK,CAAC;AACxB,OAAO,EAAE,MAAM,EAAE,MAAM,kBAAkB,CAAC;AA4C1C;;;;GAIG;AACH,MAAM,OAAO,sBAAuB,SAAQ,KAAK;IAC/C,YAAY,MAAc;QACxB,KAAK,CACH,mDAAmD,MAAM,4DAA4D,CACtH,CAAC;QACF,IAAI,CAAC,IAAI,GAAG,wBAAwB,CAAC;IACvC,CAAC;CACF;AAED,SAAS,WAAW,CAAC,MAAuB;IAC1C,IAAI,CAAC,MAAM,CAAC,IAAI;QAAE,MAAM,IAAI,sBAAsB,CAAC,MAAM,CAAC,CAAC;IAC3D,OAAO,MAAM,CAAC,IAAI,CAAC,IAAI,CAAC,MAAM,CAAC,CAAC;AAClC,CAAC;AAED,SAAS,UAAU,CAAC,MAAuB;IACzC,IAAI,CAAC,MAAM,CAAC,GAAG;QAAE,MAAM,IAAI,sBAAsB,CAAC,KAAK,CAAC,CAAC;IACzD,OAAO,MAAM,CAAC,GAAG,CAAC,IAAI,CAAC,MAAM,CAAC,CAAC;AACjC,CAAC;AAED,qFAAqF;AACrF,SAAS,OAAO,CAAC,IAA6B;IAC5C,OAAO,MAAM,CAAC,WAAW,CAAC,MAAM,CAAC,OAAO,CAAC,IAAI,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,EAAE,KAAK,CAAC,EAAE,EAAE,CAAC,KAAK,KAAK,SAAS,CAAC,CAAC,CAAC;AAC7F,CAAC;AAED,MAAM,UAAU,GAAG,CAAC;KACjB,IAAI,CAAC,CAAC,SAAS,EAAE,OAAO,EAAE,MAAM,EAAE,QAAQ,EAAE,QAAQ,EAAE,QAAQ,CAAC,CAAC;KAChE,QAAQ,CACP,qGAAqG,CACtG,CAAC;AAEJ,MAAM,CAAC,MAAM,YAAY,GAAc;IACrC,IAAI,EAAE,UAAU;IAChB,KAAK,EAAE,oBAAoB;IAC3B,WAAW,EACT,qUAAqU;IACvU,WAAW,EAAE;QACX,UAAU;QACV,QAAQ,EAAE,CAAC;aACR,MAAM,EAAE;aACR,GAAG,CAAC,CAAC,CAAC;aACN,GAAG,CAAC,MAAM,CAAC,2BAA2B,CAAC;aACvC,QAAQ,EAAE;aACV,QAAQ,CAAC,oEAAoE,CAAC;QACjF,KAAK,EAAE,CAAC;aACL,MAAM,EAAE;aACR,GAAG,EAAE;aACL,WAAW,EAAE;aACb,QAAQ,EAAE;aACV,QAAQ,CAAC,8DAA8D,CAAC;QAC3E,KAAK,EAAE,CAAC;aACL,MAAM,EAAE;aACR,GAAG,EAAE;aACL,WAAW,EAAE;aACb,QAAQ,EAAE;aACV,QAAQ,CAAC,4DAA4D,CAAC;QACzE,IAAI,EAAE,CAAC;aACJ,MAAM,EAAE;aACR,GAAG,CAAC,CAAC,CAAC;aACN,GAAG,CAAC,MAAM,CAAC,uBAAuB,CAAC;aACnC,QAAQ,CAAC,kCAAkC,CAAC;KAChD;IACD,OAAO,EAAE,IAAI;IACb,OAAO,EAAE,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,EAAE,CAC9B,WAAW,CAAC,MAAM,CAAC,CACjB,qBAAqB,EACrB,OAAO,CAAC;QACN,UAAU,EAAE,IAAI,CAAC,UAAU;QAC3B,QAAQ,EAAE,IAAI,CAAC,QAAQ;QACvB,KAAK,EAAE,IAAI,CAAC,KAAK;QACjB,KAAK,EAAE,IAAI,CAAC,KAAK;QACjB,IAAI,EAAE,IAAI,CAAC,IAAI;KAChB,CAAC,CACH;CACJ,CAAC;AAEF,MAAM,CAAC,MAAM,cAAc,GAAc;IACvC,IAAI,EAAE,aAAa;IACnB,KAAK,EAAE,eAAe;IACtB,WAAW,EACT,gPAAgP;IAClP,WAAW,EAAE;QACX,IAAI,EAAE,CAAC;aACJ,MAAM,EAAE;aACR,GAAG,CAAC,CAAC,CAAC;aACN,GAAG,CAAC,MAAM,CAAC,qBAAqB,CAAC;aACjC,QAAQ,CAAC,4DAA4D,CAAC;QACzE,OAAO,EAAE,CAAC;aACP,MAAM,EAAE;aACR,GAAG,CAAC,CAAC,CAAC;aACN,GAAG,CAAC,MAAM,CAAC,wBAAwB,CAAC;aACpC,QAAQ,CAAC,gCAAgC,CAAC;KAC9C;IACD,OAAO,EAAE,IAAI;IACb,OAAO,EAAE,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,EAAE,CAC9B,UAAU,CAAC,MAAM,CAAC,CAAC,oBAAoB,kBAAkB,CAAC,MAAM,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC,EAAE,EAAE;QAC9E,OAAO,EAAE,IAAI,CAAC,OAAO;KACtB,CAAC;CACL,CAAC;AAEF,MAAM,CAAC,MAAM,gBAAgB,GAAc;IACzC,IAAI,EAAE,eAAe;IACrB,KAAK,EAAE,kBAAkB;IACzB,WAAW,EACT,6PAA6P;IAC/P,WAAW,EAAE;QACX,KAAK,EAAE,CAAC;aACL,MAAM,EAAE;aACR,GAAG,CAAC,CAAC,CAAC;aACN,GAAG,CAAC,MAAM,CAAC,2BAA2B,CAAC;aACvC,QAAQ,CAAC,gEAAgE,CAAC;QAC7E,KAAK,EAAE,CAAC;aACL,MAAM,CAAC,CAAC,CAAC,MAAM,EAAE,EAAE,CAAC,CAAC,OAAO,EAAE,CAAC;aAC/B,QAAQ,CACP,uHAAuH,CACxH;QACH,UAAU,EAAE,CAAC;aACV,MAAM,EAAE;aACR,GAAG,CAAC,MAAM,CAAC,gCAAgC,CAAC;aAC5C,QAAQ,EAAE;aACV,QAAQ,CAAC,mDAAmD,CAAC;KACjE;IACD,OAAO,EAAE,IAAI;IACb,OAAO,EAAE,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,EAAE,CAC9B,WAAW,CAAC,MAAM,CAAC,CACjB,kBAAkB,EAClB,OAAO,CAAC,EAAE,KAAK,EAAE,IAAI,CAAC,KAAK,EAAE,KAAK,EAAE,IAAI,CAAC,KAAK,EAAE,UAAU,EAAE,IAAI,CAAC,UAAU,EAAE,CAAC,CAC/E;CACJ,CAAC;AAEF,MAAM,CAAC,MAAM,mBAAmB,GAAc;IAC5C,IAAI,EAAE,kBAAkB;IACxB,KAAK,EAAE,gCAAgC;IACvC,WAAW,EACT,sOAAsO;IACxO,WAAW,EAAE;QACX,UAAU,EAAE,UAAU,CAAC,QAAQ,EAAE;QACjC,QAAQ,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC,MAAM,CAAC,2BAA2B,CAAC,CAAC,QAAQ,EAAE;QAC9E,KAAK,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,GAAG,EAAE,CAAC,WAAW,EAAE,CAAC,QAAQ,EAAE;QAChD,KAAK,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,GAAG,EAAE,CAAC,WAAW,EAAE,CAAC,QAAQ,EAAE;QAChD,KAAK,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,GAAG,EAAE,CAAC,QAAQ,EAAE,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC,QAAQ,EAAE;KACvD;IACD,OAAO,EAAE,KAAK;IACd,OAAO,EAAE,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,EAAE,CAC9B,MAAM,CAAC,GAAG,CAAC,qBAAqB,EAAE;QAChC,UAAU,EAAE,IAAI,CAAC,UAAgC;QACjD,QAAQ,EAAE,IAAI,CAAC,QAA8B;QAC7C,KAAK,EAAE,IAAI,CAAC,KAA2B;QACvC,KAAK,EAAE,IAAI,CAAC,KAA2B;QACvC,KAAK,EAAE,IAAI,CAAC,KAA2B;KACxC,CAAC;CACL,CAAC;AAEF,MAAM,CAAC,MAAM,gBAAgB,GAAc;IACzC,IAAI,EAAE,eAAe;IACrB,KAAK,EAAE,6BAA6B;IACpC,WAAW,EACT,wJAAwJ;IAC1J,WAAW,EAAE,EAAE,KAAK,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,GAAG,EAAE,CAAC,QAAQ,EAAE,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC,QAAQ,EAAE,EAAE;IACvE,OAAO,EAAE,KAAK;IACd,OAAO,EAAE,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,EAAE,CAC9B,MAAM,CAAC,GAAG,CAAC,kBAAkB,EAAE,EAAE,KAAK,EAAE,IAAI,CAAC,KAA2B,EAAE,CAAC;CAC9E,CAAC;AAEF,MAAM,CAAC,MAAM,gBAAgB,GAAc;IACzC,IAAI,EAAE,eAAe;IACrB,KAAK,EAAE,mCAAmC;IAC1C,WAAW,EACT,qIAAqI;IACvI,WAAW,EAAE,EAAE,KAAK,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,GAAG,EAAE,CAAC,QAAQ,EAAE,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC,QAAQ,EAAE,EAAE;IACvE,OAAO,EAAE,KAAK;IACd,OAAO,EAAE,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,EAAE,CAC9B,MAAM,CAAC,GAAG,CAAC,kBAAkB,EAAE,EAAE,KAAK,EAAE,IAAI,CAAC,KAA2B,EAAE,CAAC;CAC9E,CAAC;AAEF,8EAA8E;AAC9E,EAAE;AACF,0EAA0E;AAC1E,yEAAyE;AACzE,2EAA2E;AAC3E,8EAA8E;AAC9E,4EAA4E;AAC5E,4EAA4E;AAE5E,MAAM,CAAC,MAAM,YAAY,GAAc;IACrC,IAAI,EAAE,WAAW;IACjB,KAAK,EAAE,2CAA2C;IAClD,WAAW,EACT,0FAA0F;QAC1F,4FAA4F;QAC5F,8FAA8F;QAC9F,uFAAuF;QACvF,0FAA0F;QAC1F,6FAA6F;QAC7F,0FAA0F;QAC1F,2FAA2F;QAC3F,0FAA0F;IAC5F,WAAW,EAAE;QACX,KAAK,EAAE,CAAC;aACL,MAAM,EAAE;aACR,GAAG,CAAC,CAAC,CAAC;aACN,GAAG,CAAC,MAAM,CAAC,uBAAuB,CAAC;aACnC,QAAQ,CAAC,4EAA4E,CAAC;QACzF,KAAK,EAAE,CAAC;aACL,MAAM,CAAC,CAAC,CAAC,MAAM,EAAE,EAAE,CAAC,CAAC,OAAO,EAAE,CAAC;aAC/B,QAAQ,CACP,qJAAqJ,CACtJ;QACH,KAAK,EAAE,CAAC;aACL,IAAI,CAAC,CAAC,MAAM,EAAE,OAAO,EAAE,KAAK,EAAE,MAAM,EAAE,MAAM,EAAE,WAAW,EAAE,SAAS,CAAC,CAAC;aACtE,QAAQ,CAAC,uDAAuD,CAAC;QACpE,QAAQ,EAAE,CAAC;aACR,MAAM,CAAC;YACN,CAAC,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,GAAG,CAAC,MAAM,CAAC,4BAA4B,CAAC,CAAC,QAAQ,EAAE;YACjE,CAAC,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,GAAG,CAAC,MAAM,CAAC,4BAA4B,CAAC,CAAC,QAAQ,EAAE;YACjE,MAAM,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,GAAG,CAAC,MAAM,CAAC,4BAA4B,CAAC,CAAC,QAAQ,EAAE;SACvE,CAAC;aACD,QAAQ,EAAE;aACV,QAAQ,CACP,uGAAuG,CACxG;QACH,IAAI,EAAE,CAAC;aACJ,KAAK,CAAC,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,OAAO,CAAC,CAAC,CAAC,CAAC,CAAC;aACnC,QAAQ,EAAE;aACV,QAAQ,CAAC,kDAAkD,CAAC;QAC/D,IAAI,EAAE,CAAC;aACJ,MAAM,EAAE;aACR,GAAG,CAAC,MAAM,CAAC,sBAAsB,CAAC;aAClC,QAAQ,EAAE;aACV,QAAQ,CAAC,qEAAqE,CAAC;KACnF;IACD,OAAO,EAAE,IAAI;IACb,OAAO,EAAE,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,EAAE,CAC9B,WAAW,CAAC,MAAM,CAAC,CACjB,gBAAgB,EAChB,OAAO,CAAC;QACN,CAAC,EAAE,CAAC;QACJ,KAAK,EAAE,IAAI,CAAC,KAAK;QACjB,KAAK,EAAE,IAAI,CAAC,KAAK;QACjB,KAAK,EAAE,IAAI,CAAC,KAAK;QACjB,QAAQ,EAAE,IAAI,CAAC,QAAQ;QACvB,IAAI,EAAE,IAAI,CAAC,IAAI;QACf,IAAI,EAAE,IAAI,CAAC,IAAI;KAChB,CAAC,CACH;CACJ,CAAC;AAEF,MAAM,CAAC,MAAM,cAAc,GAAc;IACvC,IAAI,EAAE,aAAa;IACnB,KAAK,EAAE,kCAAkC;IACzC,WAAW,EACT,8FAA8F;QAC9F,qFAAqF;QACrF,8FAA8F;QAC9F,iFAAiF;IACnF,WAAW,EAAE,EAAE,KAAK,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,GAAG,EAAE,CAAC,QAAQ,EAAE,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC,QAAQ,EAAE,EAAE;IACvE,OAAO,EAAE,KAAK;IACd,OAAO,EAAE,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,EAAE,CAC9B,MAAM,CAAC,GAAG,CAAC,gBAAgB,EAAE,EAAE,KAAK,EAAE,IAAI,CAAC,KAA2B,EAAE,CAAC;CAC5E,CAAC;AAEF,MAAM,CAAC,MAAM,cAAc,GAAc;IACvC,IAAI,EAAE,aAAa;IACnB,KAAK,EAAE,uBAAuB;IAC9B,WAAW,EACT,uFAAuF;QACvF,6FAA6F;QAC7F,0EAA0E;IAC5E,WAAW,EAAE;QACX,EAAE,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC,GAAG,CAAC,CAAC,QAAQ,CAAC,mDAAmD,CAAC;KAC7F;IACD,OAAO,EAAE,IAAI;IACb,OAAO,EAAE,KAAK,EAAE,MAAM,EAAE,IAAI,EAAE,EAAE;QAC9B,IAAI,CAAC,MAAM,CAAC,MAAM;YAAE,MAAM,IAAI,sBAAsB,CAAC,QAAQ,CAAC,CAAC;QAC/D,OAAO,MAAM,CAAC,MAAM,CAAC,kBAAkB,kBAAkB,CAAC,MAAM,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC,EAAE,CAAC,CAAC;IAChF,CAAC;CACF,CAAC;AAEF;;;;;GAKG;AACH,MAAM,CAAC,MAAM,UAAU,GAAyB;IAC9C,YAAY;IACZ,cAAc;IACd,gBAAgB;IAChB,YAAY;IACZ,cAAc;IACd,mBAAmB;IACnB,gBAAgB;IAChB,gBAAgB;IAChB,cAAc;CACf,CAAC;AAEF,mDAAmD;AACnD,MAAM,CAAC,MAAM,kBAAkB,GAAyB,UAAU,CAAC,MAAM,CAAC,CAAC,IAAI,EAAE,EAAE,CAAC,IAAI,CAAC,OAAO,CAAC,CAAC"}
|
package/llms.txt
CHANGED
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
# @uptimizr/agent-core
|
|
2
2
|
|
|
3
3
|
> The framework-agnostic, browser-safe core for Uptimizr analytics agents: the single read-only
|
|
4
|
-
> tool catalog over the collector query API,
|
|
5
|
-
> tool-calling loop. Defined once, consumed by
|
|
4
|
+
> analytics tool catalog over the collector query API, the `annotate`-gated project-metadata tools,
|
|
5
|
+
> a headless LLM provider-adapter interface, and the tool-calling loop. Defined once, consumed by
|
|
6
|
+
> MCP, the dashboard, and the demo (ADR 0050). Events are read-only (ADR 0051 §9).
|
|
6
7
|
|
|
7
8
|
## Docs
|
|
8
9
|
|
|
@@ -11,7 +12,7 @@
|
|
|
11
12
|
- [Integration & API reference](https://github.com/RaananW/Uptimizr/blob/main/docs/integration.md): the underlying query endpoints.
|
|
12
13
|
- [Architecture Decision Records](https://github.com/RaananW/Uptimizr/tree/main/docs/adr): in-browser assistant (0050), privacy model (0003), thin backends (0005), consumer-facing agents (0017).
|
|
13
14
|
|
|
14
|
-
##
|
|
15
|
+
## Analytics tools (read-only)
|
|
15
16
|
|
|
16
17
|
<!-- generated:registry-tool-names:start — generated by `pnpm gen:docs`; edit the metric registry, not this list -->
|
|
17
18
|
|
|
@@ -21,32 +22,92 @@
|
|
|
21
22
|
`position_heatmap`, `session_trajectory`, `aggregate_paths`, `scene_coverage`, `camera_distance`,
|
|
22
23
|
`click_rays`, `flow_links`, `top_meshes`, `mesh_sources`, `mesh_trend`, `mesh_dwell`,
|
|
23
24
|
`mesh_blind_spots`, `mesh_interaction_kinds`, `mesh_reachability`, `dead_clicks`, `rage_clicks`,
|
|
24
|
-
`hover_dwell`, `interaction_sources`, `top_input_actions`, `
|
|
25
|
-
`
|
|
26
|
-
`
|
|
27
|
-
`
|
|
28
|
-
`
|
|
29
|
-
`
|
|
30
|
-
`
|
|
31
|
-
`
|
|
32
|
-
`scene_retention`, `load_bounce_funnel`, `variant_leaderboard
|
|
25
|
+
`hover_dwell`, `interaction_sources`, `top_input_actions`, `custom_event_vocabulary`,
|
|
26
|
+
`camera_gestures`, `navigation_stats`, `backtrack_ratio`, `perf_summary`, `render_scale_truth`,
|
|
27
|
+
`perf_distribution`, `fps_histogram`, `frame_time_percentiles`, `jank_rate`, `perf_churn`,
|
|
28
|
+
`perf_by_device`, `perf_by_scene`, `perf_heatmap`, `compile_stalls`, `resource_summary`,
|
|
29
|
+
`resource_percentiles`, `stability_counts`, `graphics_diagnostics`, `error_heatmap`,
|
|
30
|
+
`rendering_technology`, `capability_changes`, `xr_rotation`, `xr_sources`, `xr_abandonment`,
|
|
31
|
+
`xr_locomotion`, `xr_tracking_quality`, `boundary_heatmap`, `boundary_heatmap_stats`,
|
|
32
|
+
`xr_boundary_contacts`, `ar_placement_time_to_place`, `ar_placement_attempts`,
|
|
33
|
+
`ar_placement_surfaces`, `funnel`, `scene_retention`, `load_bounce_funnel`, `variant_leaderboard`,
|
|
34
|
+
`insight_baseline`, `insight_movers`, `insight_anomalies`, `insight_significance`,
|
|
35
|
+
`insight_scene_health`
|
|
36
|
+
|
|
37
|
+
Only on a key holding `query:raw`, and only when the collector runs with
|
|
38
|
+
`ENABLE_RAW_SESSION_RETENTION` (ADR 0003):
|
|
39
|
+
|
|
40
|
+
`session_narrative`
|
|
33
41
|
|
|
34
42
|
<!-- generated:registry-tool-names:end -->
|
|
35
43
|
|
|
44
|
+
## Insight primitives (ADR 0051 §4)
|
|
45
|
+
|
|
46
|
+
Five of the tools are readings *about* the other metrics, and they are the right first call on an
|
|
47
|
+
open-ended question — "how are things?" does not mean "call thirty tools".
|
|
48
|
+
|
|
49
|
+
- `insight_scene_health` — one 0-100 score per scene over six weighted factors (perf stability,
|
|
50
|
+
jank, errors, dead clicks, coverage, XR abandonment), least healthy first. The right *unscoped*
|
|
51
|
+
starting point: it answers which scene to open. Every factor carries the `metric` id behind it,
|
|
52
|
+
its `raw` value and the project `baseline` it was compared with, so the next call and the
|
|
53
|
+
sentence to write are both already in the row. **50 is the project norm, not a pass mark** — the
|
|
54
|
+
score compares a scene with the rest of the project's recent past — and a factor with
|
|
55
|
+
`score: null` was not counted (its `note` says why).
|
|
56
|
+
- `insight_movers` — what changed against a reference window (the previous equal one by default),
|
|
57
|
+
for every comparable metric in scope, ranked by a robust z-score: the change divided by how much
|
|
58
|
+
that metric normally swings, so a noisy metric must move much further than a steady one to reach
|
|
59
|
+
the top. Bounded: at most 24 metrics are scanned per request.
|
|
60
|
+
- `insight_baseline` — what is normal for one metric in one scene: `median` and `mad`, the
|
|
61
|
+
`p10`..`p90` band, and the drift `slope`. Call it on whatever moved, to say whether the new level
|
|
62
|
+
is actually outside normal rather than merely different.
|
|
63
|
+
- `insight_anomalies` — *when* one metric went wrong. One row per bucket that does not belong in
|
|
64
|
+
the series: `kind: "spike" | "drop"` for a bucket more than `sensitivity` (default 3) standard
|
|
65
|
+
deviations from the trailing window, and `kind: "shift"` at the bucket where the level moved and
|
|
66
|
+
stayed moved. Where the metric declares a dimension it can be split by, `contributor` names the
|
|
67
|
+
mesh, source, input action, event type or scene holding the largest `share` of the excess.
|
|
68
|
+
Bounded: one scan for the series plus at most three attribution scans per request.
|
|
69
|
+
- `insight_significance` — whether a difference between two windows could be chance: `effect`,
|
|
70
|
+
`ci95`, `p`, and the `test` that produced them (a two-proportion z with Wilson intervals for a
|
|
71
|
+
declared rate, Welch's t over the per-bucket values for a level, an exact Poisson rate test for a
|
|
72
|
+
bare count — picked from the measure, not from you). **Read `ci95` before `p`**: an interval that
|
|
73
|
+
straddles 0 means you cannot tell yet, and `powerNote` says what this much data could have
|
|
74
|
+
detected at all. Welch's `n` is the number of **buckets**, not events. It compares two *windows*,
|
|
75
|
+
not two segments: a variant-versus-variant question is a `400`, not a wrong answer.
|
|
76
|
+
|
|
77
|
+
Read `insight_anomalies`' `z` as standard deviations and `insight_movers`' as the same ratio
|
|
78
|
+
unscaled — they differ by a constant factor (1.4826) and must not be compared directly. A sustained
|
|
79
|
+
level change is reported both as a `shift` and, for the days right after it, as `drop`/`spike`
|
|
80
|
+
rows: two true statements about one event.
|
|
81
|
+
|
|
82
|
+
Two fields decide whether a mover is worth reporting, and both are easy to misread:
|
|
83
|
+
|
|
84
|
+
- `direction` is the registry's opinion of what a **rise** in that metric means — `up` (good),
|
|
85
|
+
`down` (bad) or `neutral`. It is *not* the direction of the move: read it with the sign of
|
|
86
|
+
`delta`, so a rise in a `down` metric (errors, dead clicks, jank) is a regression.
|
|
87
|
+
- `aboveMinSample: false` means the delta is real arithmetic but the denominator is below the
|
|
88
|
+
metric's declared minimum. Such rows are returned rather than dropped — "we cannot tell" and
|
|
89
|
+
"nothing changed" are different answers — and must never be reported as findings.
|
|
90
|
+
|
|
91
|
+
Statistics are computed in pure TypeScript over portable day/hour buckets, so the same data yields
|
|
92
|
+
the same answer on every storage engine. Only `comparable` metrics with a faithful per-bucket form
|
|
93
|
+
participate; asking for one that has none returns `400` naming every id that does.
|
|
94
|
+
|
|
36
95
|
## Result formats (`format`)
|
|
37
96
|
|
|
38
97
|
Every aggregate tool declares a `format` argument — the envelope its rows arrive in, not a filter
|
|
39
|
-
(ADR 0051 §2).
|
|
98
|
+
(ADR 0051 §2). `buildRequest` applies `DEFAULT_TOOL_FORMAT` (`table`) when the caller names none
|
|
99
|
+
and sends it explicitly, so the collector's own default stays `full`. A tool's output schema
|
|
100
|
+
describes all three envelopes, so whichever one comes back validates.
|
|
40
101
|
|
|
41
102
|
- `summary` — **prefer this whenever a model reads the result.** A bounded digest: `ranked` top
|
|
42
103
|
rows, a `series` trend, merged spatial `clusters`, or a single `record`, with shares, a sample
|
|
43
104
|
size, the metric's `caveats` and a templated `reading` sentence. Capped at the metric's
|
|
44
105
|
`limits.maxSummaryRows`, so a 500-bin heatmap costs the same as a 5-bin one — which is what keeps
|
|
45
106
|
one heatmap from filling a small local model's whole context.
|
|
46
|
-
- `table` — `{ meta, rows }`: every row plus the metric, range, applied
|
|
47
|
-
count, a truncation flag and the registry limits.
|
|
48
|
-
|
|
49
|
-
|
|
107
|
+
- `table` — **the tools' default.** `{ meta, rows }`: every row plus the metric, range, applied
|
|
108
|
+
filters, sample size, row count, a truncation flag and the registry limits. Self-describing but
|
|
109
|
+
not bounded — ask for `summary` when the result could be large.
|
|
110
|
+
- `full` — the bare rows, with no envelope at all.
|
|
50
111
|
|
|
51
112
|
`reading` and `caveats` are templated by pure code, never a model, so identical rows always produce
|
|
52
113
|
identical words. Shares appear only where the measure can honestly be summed (an FPS or ratio metric
|
|
@@ -54,15 +115,82 @@ reports `total: null` and no shares). Cluster coordinates are grid indices — m
|
|
|
54
115
|
effective `cellSize` for world space. `session_meta` and `scene_representation` are stored records,
|
|
55
116
|
not aggregations, and declare no `format`.
|
|
56
117
|
|
|
118
|
+
## Packaged methodology skills (ADR 0051 §7)
|
|
119
|
+
|
|
120
|
+
Shipped in this tarball as Agent Skills files under `skills/<name>/SKILL.md` — frontmatter (tools,
|
|
121
|
+
capabilities, arguments) plus a numbered method. `AGENT_SKILLS` is those files compiled to data;
|
|
122
|
+
`getAgentSkill(name)` resolves either spelling and `render(args)` substitutes the scene and range.
|
|
123
|
+
The same files back `@uptimizr/mcp`'s prompt templates and `uptimizr agent report --skill`.
|
|
124
|
+
|
|
125
|
+
<!-- generated:registry-skill-names:start — generated by `pnpm gen:docs`; edit the SKILL.md files, not this list -->
|
|
126
|
+
|
|
127
|
+
- `attention_hotspots` (scene (required), range) — Find where visitors look and click in a scene: view-direction concentration, gaze→mesh flow, the objects that draw the most interaction, and the ones nobody ever notices. USE FOR: deciding where to put a call to action, finding ignored or invisible content, explaining why an object gets no clicks, laying out a scene around what people actually look at.
|
|
128
|
+
Method: `skills/attention-hotspots/SKILL.md`. Tools: `camera_heatmap`, `flow_links`, `click_rays`, `top_meshes`, `mesh_dwell`, `mesh_blind_spots`, `query`.
|
|
129
|
+
- `conversion_investigation` (scene, range) — Find out where a funnel loses people and whether the loss is real: step-by-step drop-off, the bounce that happens before the funnel even starts, scene-to-scene retention, variant performance, and the interaction failures (dead clicks, rage clicks, unreachable meshes) that explain a stalled step. USE FOR: a funnel that converts worse than expected, an A/B variant comparison, "where do people drop off", diagnosing a step nobody completes.
|
|
130
|
+
Method: `skills/conversion-investigation/SKILL.md`. Tools: `funnel`, `load_bounce_funnel`, `scene_retention`, `variant_leaderboard`, `dead_clicks`, `rage_clicks`, `mesh_reachability`, `flow_links`, `insight_significance`, `insight_movers`, `query`.
|
|
131
|
+
- `performance_regression_triage` (scene, range) — Triage a frame-rate or stability regression: confirm it moved, date it, locate it (which scene, device class, place in the scene), and name the mechanism — jank, shader compile stalls, memory pressure, a render-scale change or a rendering-technology shift. USE FOR: "the app got slower", a FPS drop after a release, stutter reports, deciding whether a regression is real or noise.
|
|
132
|
+
Method: `skills/performance-regression-triage/SKILL.md`. Tools: `insight_movers`, `insight_anomalies`, `insight_significance`, `insight_baseline`, `perf_summary`, `perf_distribution`, `frame_time_percentiles`, `jank_rate`, `perf_by_device`, `perf_by_scene`, `perf_heatmap`, `compile_stalls`, `resource_percentiles`, `render_scale_truth`, `rendering_technology`, `query`.
|
|
133
|
+
- `weekly_scene_health` (scene, range) — A weekly health check for a scene (or the whole project): a weighted health score with every factor traced back to the metric behind it, what changed against last week, traffic, event mix, performance, and the most-interacted meshes. USE FOR: the recurring "how is the scene doing?" review, a scheduled weekly or monthly report, a first look at a project you do not know yet, deciding which scene to investigate next.
|
|
134
|
+
Method: `skills/weekly-scene-health/SKILL.md`. Tools: `insight_scene_health`, `insight_movers`, `insight_baseline`, `insight_significance`, `insight_anomalies`, `event_counts`, `timeseries`, `perf_summary`, `top_meshes`, `list_sessions`, `query`.
|
|
135
|
+
- `xr_comfort_audit` (scene, range) — Audit VR/AR comfort for a scene (or the whole project): rapid head rotation, locomotion style, tracking quality, guardian/boundary contacts, input-source mix, and the short sessions that mean someone took the headset off. USE FOR: motion-sickness complaints, immersive sessions that end early, choosing a locomotion scheme, checking whether a play space is big enough.
|
|
136
|
+
Method: `skills/xr-comfort-audit/SKILL.md`. Tools: `xr_rotation`, `xr_locomotion`, `xr_abandonment`, `xr_sources`, `xr_tracking_quality`, `xr_boundary_contacts`, `boundary_heatmap_stats`, `insight_scene_health`, `insight_movers`, `query`.
|
|
137
|
+
|
|
138
|
+
<!-- generated:registry-skill-names:end -->
|
|
139
|
+
|
|
57
140
|
## Key exports
|
|
58
141
|
|
|
59
142
|
- `readTools` — the read-only tool catalog (one entry per query endpoint), generated from the
|
|
60
|
-
`@uptimizr/metrics` metric registry
|
|
61
|
-
|
|
143
|
+
`@uptimizr/metrics` metric registry plus `query` and `list_subscriptions`: 77 tools, each with an
|
|
144
|
+
input schema, an output schema covering all three `format` envelopes, and the metric's
|
|
145
|
+
interpretation notes and caveats in its description (ADR 0051).
|
|
146
|
+
- `writeTools` / `mutatingWriteTools` — the project-metadata tools `annotate`, `define_term`,
|
|
147
|
+
`save_analysis` (ADR 0051 §5), `pin_panel` and `unpin_panel` (ADR 0051 §7), plus the reads
|
|
148
|
+
`list_annotations`, `list_glossary`, `list_analyses`, `list_panels`. Exported individually too
|
|
149
|
+
(`annotateTool`, `defineTermTool`, `saveAnalysisTool`, `pinPanelTool`, `unpinPanelTool`,
|
|
150
|
+
`listAnnotationsTool`, `listGlossaryTool`, `listAnalysesTool`, `listPanelsTool`). A
|
|
151
|
+
separate export from `readTools` on purpose, so an integration's read-only stance stays
|
|
152
|
+
inspectable. They need an `annotate` key, write metadata only, and cannot touch an event.
|
|
153
|
+
`pin_panel` stores a panel **spec** — a metric id, a chart name, some column names — that the
|
|
154
|
+
dashboard renders with panels it already ships, so nothing is loaded and nothing is evaluated;
|
|
155
|
+
send the `query` document with `range: "inherit"` and a chart the metric's grain supports, or the
|
|
156
|
+
collector refuses it and names the charts that would have worked. `unpin_panel` removes a panel
|
|
157
|
+
for everyone on the project and is the one tool here that uses `CollectorClient.delete`.
|
|
62
158
|
- `coreReadTools` / `selectReadTools(kind)` / `filterReadTools(names)` — narrowed views of the
|
|
63
159
|
catalog for models that cannot carry every tool schema.
|
|
64
160
|
- `registryToTools(metrics?)` — generate the catalog from metric-registry definitions.
|
|
65
|
-
- `createCollectorClient(config)` — the thin
|
|
161
|
+
- `createCollectorClient(config)` — the thin collector client: `get` for every analytics read, plus
|
|
162
|
+
`post`/`put`/`delete` used only by the metadata write tools.
|
|
66
163
|
- `runAgent(options)` — the headless tool-calling loop (LLM ↔ tools ↔ collector).
|
|
67
164
|
- `toToolSchemas(tools?)` — convert catalog tools to JSON-Schema tool descriptors for an LLM.
|
|
165
|
+
- `renderContextForPrompt(context, nowMs?)` — render the collector's project context document
|
|
166
|
+
(`GET /api/v1/context`, ADR 0051 §5) as a compact system-prompt block: the project's real scene
|
|
167
|
+
ids, named region ids, custom-event names with their prop keys and coarse types, top meshes, input
|
|
168
|
+
actions, data freshness, retention flags and the metrics that are empty because their capture
|
|
169
|
+
channel is off. Pure, ~1.5 k characters at most, tolerant of a partial document, and `""` when
|
|
170
|
+
there is nothing to say — so an older collector degrades instead of failing. Read the context
|
|
171
|
+
first and inject this before the first question.
|
|
172
|
+
- `AGENT_SKILLS` / `getAgentSkill(name)` / `renderAgentSkill(name, args)` — the curated
|
|
173
|
+
investigations packaged as `skills/<name>/SKILL.md` in this tarball (`attention_hotspots`,
|
|
174
|
+
`conversion_investigation`, `performance_regression_triage`, `weekly_scene_health`,
|
|
175
|
+
`xr_comfort_audit`): title, description, the registry tools each method relies on, its arguments,
|
|
176
|
+
and a pure `render(args)` producing the user turn. `@uptimizr/mcp` registers them as prompt
|
|
177
|
+
templates and `uptimizr agent report --skill` runs them headlessly, from this one definition.
|
|
178
|
+
`getAgentSkill` takes either spelling, and the pre-#316 name `xr_comfort_review` still resolves.
|
|
179
|
+
- `ANALYTICS_AGENT_GUIDELINES` / `renderCurrentTimeLine(nowMs)` — the system-prompt fragments every
|
|
180
|
+
analytics agent shares (what it may claim about the data; the clock it resolves "this week"
|
|
181
|
+
against). Clients add only their own role sentence and output format.
|
|
68
182
|
- `LlmProvider` — the pluggable, user-controlled LLM backend interface (no bundled model or key).
|
|
183
|
+
- `ProviderResponse.usage` — optional `{ inputTokens?, outputTokens? }` per turn, populated by the
|
|
184
|
+
hosted adapters from Anthropic's `usage` and OpenAI's `prompt_tokens`/`completion_tokens`
|
|
185
|
+
(buffered and streamed). Absent means **not reported**, never zero.
|
|
186
|
+
|
|
187
|
+
## Capability-gated tools (ADR 0051 §7)
|
|
188
|
+
|
|
189
|
+
- `readTools` is the `query` surface; `rawTools` holds the tools whose registry endpoint declares `capability: "query:raw"` (today `session_narrative`, the compacted account of one session).
|
|
190
|
+
- Keep them apart: the capability belongs to the key the agent was handed. Register `rawTools` only after `GET /api/v1/whoami` confirms `query:raw` — the collector also requires `ENABLE_RAW_SESSION_RETENTION`, so the capability is necessary but not sufficient (ADR 0003).
|
|
191
|
+
|
|
192
|
+
## Non-metric reads
|
|
193
|
+
|
|
194
|
+
- `list_subscriptions` — the project's conditional subscriptions (ADR 0051 §6). Not a registry
|
|
195
|
+
metric: configuration, not a measurement. No arguments, no time range. Creating/deleting one
|
|
196
|
+
needs `annotate` and is not a tool.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@uptimizr/agent-core",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.2.0",
|
|
4
4
|
"description": "Framework-agnostic, browser-safe core for Uptimizr analytics agents — the single read-only tool catalog over the collector query API, a headless LLM provider-adapter interface, and the tool-calling loop.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"uptimizr",
|
|
@@ -51,6 +51,7 @@
|
|
|
51
51
|
},
|
|
52
52
|
"files": [
|
|
53
53
|
"dist",
|
|
54
|
+
"skills",
|
|
54
55
|
"README.md",
|
|
55
56
|
"LICENSE",
|
|
56
57
|
"AGENTS.md",
|
|
@@ -58,8 +59,9 @@
|
|
|
58
59
|
],
|
|
59
60
|
"sideEffects": false,
|
|
60
61
|
"dependencies": {
|
|
61
|
-
"zod": "^4.5
|
|
62
|
-
"@uptimizr/metrics": "0.
|
|
62
|
+
"zod": "^4.6.5",
|
|
63
|
+
"@uptimizr/metrics": "0.2.0",
|
|
64
|
+
"@uptimizr/schema": "1.2.0"
|
|
63
65
|
},
|
|
64
66
|
"peerDependencies": {
|
|
65
67
|
"@mlc-ai/web-llm": "^0.2.0"
|
|
@@ -70,9 +72,9 @@
|
|
|
70
72
|
}
|
|
71
73
|
},
|
|
72
74
|
"devDependencies": {
|
|
73
|
-
"@mlc-ai/web-llm": "^0.2.
|
|
75
|
+
"@mlc-ai/web-llm": "^0.2.85",
|
|
74
76
|
"esbuild": "^0.28.2",
|
|
75
|
-
"vitest": "^
|
|
77
|
+
"vitest": "^5.0.1"
|
|
76
78
|
},
|
|
77
79
|
"engines": {
|
|
78
80
|
"node": ">=22"
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: attention_hotspots
|
|
3
|
+
title: Attention hot-spots for a scene
|
|
4
|
+
description: >-
|
|
5
|
+
Find where visitors look and click in a scene: view-direction concentration, gaze→mesh flow, the
|
|
6
|
+
objects that draw the most interaction, and the ones nobody ever notices. USE FOR: deciding where
|
|
7
|
+
to put a call to action, finding ignored or invisible content, explaining why an object gets no
|
|
8
|
+
clicks, laying out a scene around what people actually look at. Trigger phrases: what do people
|
|
9
|
+
look at, attention hotspots, where do visitors click, which meshes get ignored, blind spots,
|
|
10
|
+
gaze heatmap, is anyone seeing this object.
|
|
11
|
+
tools:
|
|
12
|
+
- camera_heatmap
|
|
13
|
+
- flow_links
|
|
14
|
+
- click_rays
|
|
15
|
+
- top_meshes
|
|
16
|
+
- mesh_dwell
|
|
17
|
+
- mesh_blind_spots
|
|
18
|
+
- query
|
|
19
|
+
capabilities:
|
|
20
|
+
- query
|
|
21
|
+
args:
|
|
22
|
+
- name: scene
|
|
23
|
+
required: true
|
|
24
|
+
description: The scene id to analyse (see the uptimizr://scenes resource).
|
|
25
|
+
- name: range
|
|
26
|
+
required: false
|
|
27
|
+
default: the last 7 days
|
|
28
|
+
description: The window to analyse, in words — e.g. "the last 7 days", "since launch".
|
|
29
|
+
---
|
|
30
|
+
|
|
31
|
+
Where does attention concentrate in scene "{{scene}}" over {{range}}?
|
|
32
|
+
|
|
33
|
+
Work through the method below with the read-only tools it names — all of them scoped with
|
|
34
|
+
`scene="{{scene}}"` — and synthesise one answer.
|
|
35
|
+
|
|
36
|
+
1. **Orient before you ask anything.** Read the `uptimizr://context` resource first: it gives the
|
|
37
|
+
real scene ids, the scene's **named regions** and the custom-event names this project emits, and
|
|
38
|
+
it tells you which metrics are empty because their capture channel is off. Name regions the way
|
|
39
|
+
the project names them — "the checkout counter", not "the cluster at x≈3".
|
|
40
|
+
|
|
41
|
+
2. **Where do they look?** `camera_heatmap` (`scene="{{scene}}"`) gives the view-direction
|
|
42
|
+
distribution — what people point the camera at, whether or not they ever click it. Ask for
|
|
43
|
+
`format: "summary"`: the digest merges neighbouring cells into a handful of clusters with a
|
|
44
|
+
share each and a plain-language `reading`, which is what you want here; the raw grid is
|
|
45
|
+
thousands of cells you cannot describe.
|
|
46
|
+
|
|
47
|
+
3. **Does looking turn into touching?** `flow_links` (`scene="{{scene}}"`) links where the gaze was
|
|
48
|
+
to the mesh that was then clicked. A strong link is a working call to action; a heavy look with
|
|
49
|
+
no outgoing link is content that draws the eye and then disappoints.
|
|
50
|
+
|
|
51
|
+
4. **Where do the clicks land?** `click_rays` (`scene="{{scene}}"`) gives view-gated clicks per
|
|
52
|
+
voxel and mesh — clicks attributed to what the visitor could actually see, not to whatever the
|
|
53
|
+
ray happened to pass through.
|
|
54
|
+
|
|
55
|
+
5. **Rank the objects.** `top_meshes` (`scene="{{scene}}"`) for the most-interacted meshes, and
|
|
56
|
+
`mesh_dwell` (`scene="{{scene}}"`) for how long attention rests on each one. Dwell without
|
|
57
|
+
interaction is hesitation, and it usually means the object looks clickable and is not, or is
|
|
58
|
+
clickable and does not look it.
|
|
59
|
+
|
|
60
|
+
6. **Name the cold half.** `mesh_blind_spots` (`scene="{{scene}}"`) lists the meshes that are
|
|
61
|
+
present and essentially never noticed. A hot-spot report that only names hot spots tells you
|
|
62
|
+
nothing about the content you paid to build.
|
|
63
|
+
|
|
64
|
+
7. **Narrow it with the DSL.** For anything the canned tools do not expose, use the single `query`
|
|
65
|
+
tool: pick the `metric`, bound it with `range`, filter it, and set `format: "summary"` for a
|
|
66
|
+
bounded digest — each summary row carries a `drillQuery` you can send straight back instead of
|
|
67
|
+
rebuilding the filter. Set `compare: { range: <previous window> }` to see whether a hot spot is
|
|
68
|
+
new, and `explain: true` when a result looks wrong or empty: the plan names the capture channel,
|
|
69
|
+
the sample size and the row cap behind it.
|
|
70
|
+
|
|
71
|
+
## What to report
|
|
72
|
+
|
|
73
|
+
- The two or three real hot-spots, named with the project's own region and mesh names, each with
|
|
74
|
+
its share of attention.
|
|
75
|
+
- The cold areas and the meshes nobody notices.
|
|
76
|
+
- Where gaze fails to convert into interaction, and what that implies for layout and
|
|
77
|
+
call-to-action placement.
|
|
78
|
+
|
|
79
|
+
Carry the caveats: heatmaps are gated on the view/pointer capture channels and are sampled
|
|
80
|
+
(ADR 0012), so a share is a share _of the sampled events_; say so, and say when a result was
|
|
81
|
+
truncated (`meta.truncated`) or sits below the metric's own minimum sample. Do not turn a voxel
|
|
82
|
+
cluster into a claim about one object unless `click_rays` or `flow_links` attributes it to that
|
|
83
|
+
mesh.
|
|
84
|
+
|
|
85
|
+
End with 2–3 concrete layout or content recommendations. If an `annotate` tool is available, leave
|
|
86
|
+
a note on the region you want revisited — a region-scoped annotation is what makes the next report
|
|
87
|
+
open where this one ended. If a `pin_panel` tool is available, pin the heatmap panel you reasoned
|
|
88
|
+
from.
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: conversion_investigation
|
|
3
|
+
title: Conversion investigation
|
|
4
|
+
description: >-
|
|
5
|
+
Find out where a funnel loses people and whether the loss is real: step-by-step drop-off, the
|
|
6
|
+
bounce that happens before the funnel even starts, scene-to-scene retention, variant performance,
|
|
7
|
+
and the interaction failures (dead clicks, rage clicks, unreachable meshes) that explain a stalled
|
|
8
|
+
step. USE FOR: a funnel that converts worse than expected, an A/B variant comparison, "where do
|
|
9
|
+
people drop off", diagnosing a step nobody completes. Trigger phrases: conversion, funnel,
|
|
10
|
+
drop-off, why are people leaving, which variant wins, bounce rate, retention, people get stuck.
|
|
11
|
+
tools:
|
|
12
|
+
- funnel
|
|
13
|
+
- load_bounce_funnel
|
|
14
|
+
- scene_retention
|
|
15
|
+
- variant_leaderboard
|
|
16
|
+
- dead_clicks
|
|
17
|
+
- rage_clicks
|
|
18
|
+
- mesh_reachability
|
|
19
|
+
- flow_links
|
|
20
|
+
- insight_significance
|
|
21
|
+
- insight_movers
|
|
22
|
+
- query
|
|
23
|
+
capabilities:
|
|
24
|
+
- query
|
|
25
|
+
args:
|
|
26
|
+
- name: scene
|
|
27
|
+
required: false
|
|
28
|
+
description: Optional scene id to scope the investigation to (see the uptimizr://scenes resource).
|
|
29
|
+
- name: range
|
|
30
|
+
required: false
|
|
31
|
+
default: the last 7 days
|
|
32
|
+
description: The window to investigate, in words — e.g. "the last 7 days", "since the release".
|
|
33
|
+
---
|
|
34
|
+
|
|
35
|
+
Investigate conversion for {{scope}} over {{range}}: where do people drop off, and is the drop real?
|
|
36
|
+
|
|
37
|
+
Work through the method below with the read-only tools it names, then answer.
|
|
38
|
+
|
|
39
|
+
1. **Orient before you ask anything.** Read the `uptimizr://context` resource first. A funnel is
|
|
40
|
+
built out of **this project's own event types and custom-event names** — invent one and every
|
|
41
|
+
step reads zero. The context document lists the vocabulary the application actually emits, the
|
|
42
|
+
real scene ids, and which metrics are empty because their capture channel is off.
|
|
43
|
+
|
|
44
|
+
2. **Check the step before the first step.** `load_bounce_funnel`{{#scene}} (`scene="{{scene}}"`){{/scene}}
|
|
45
|
+
measures load → first interaction → stay. If people leave before the funnel starts, nothing
|
|
46
|
+
inside it will explain the number, and a "conversion problem" is really a load or a first-impression
|
|
47
|
+
problem.
|
|
48
|
+
|
|
49
|
+
3. **Run the funnel itself.** `funnel`{{#scene}} (`scene="{{scene}}"`){{/scene}} with the `steps`
|
|
50
|
+
built from the context document's vocabulary. Read it as the _transition_ rates, not the totals:
|
|
51
|
+
the step with the worst step-to-step rate is the one to investigate, even when a later step has
|
|
52
|
+
fewer people in absolute terms.
|
|
53
|
+
|
|
54
|
+
4. **Follow them out of the scene.** `scene_retention`{{#scene}} (`scene="{{scene}}"`){{/scene}}
|
|
55
|
+
shows where a visitor goes next. A step that "loses" people to the next scene is not a loss at
|
|
56
|
+
all; one that loses them to nothing is.
|
|
57
|
+
|
|
58
|
+
5. **Explain the stalled step.** At the worst step, look for interaction failure rather than
|
|
59
|
+
intent: `dead_clicks` (clicks that hit nothing actionable), `rage_clicks` (repeated clicking in
|
|
60
|
+
one spot — frustration you can locate), `mesh_reachability` (the target is too far away or
|
|
61
|
+
behind something to be clicked at all) and `flow_links` (people look at the target and never
|
|
62
|
+
click it). One of these usually _is_ the drop-off.
|
|
63
|
+
|
|
64
|
+
6. **Compare variants honestly.** `variant_leaderboard` ranks variants by conversion. A leaderboard
|
|
65
|
+
is not a verdict: check each variant's sample size before repeating its rate, and say plainly
|
|
66
|
+
when two variants are too close or too small to separate. `insight_significance` can test a
|
|
67
|
+
metric with a portable bucket series across two **windows**; it does not test one segment against
|
|
68
|
+
another, and if you ask it to it will say so — report that limitation rather than inventing a
|
|
69
|
+
p-value.
|
|
70
|
+
|
|
71
|
+
7. **See whether this is new.** `insight_movers`{{#scene}} (`scene="{{scene}}"`){{/scene}} ranks
|
|
72
|
+
what changed against the previous equal window, so you can tell "this funnel has always been bad"
|
|
73
|
+
from "this funnel broke last Tuesday". Ignore any row with `aboveMinSample: false`.
|
|
74
|
+
|
|
75
|
+
8. **Use the DSL for the cuts the canned tools do not expose.** The single `query` tool takes a
|
|
76
|
+
`metric`, a `range`, that metric's filters, and `compare: { range: <previous window> }` to return
|
|
77
|
+
`{ current, previous, delta, deltaPct }` already joined — do not subtract two runs by hand. Use
|
|
78
|
+
`dimensions` to regroup a portable count (by device class, source or scene) and
|
|
79
|
+
`format: "summary"` for a bounded digest with a `reading` and a `drillQuery` per row. When a step
|
|
80
|
+
reads zero, re-send it with `explain: true` before reporting it: the plan will tell you whether
|
|
81
|
+
the number is real or whether the channel behind it is switched off.
|
|
82
|
+
|
|
83
|
+
## What to report
|
|
84
|
+
|
|
85
|
+
- The step that actually loses people, with its entry and exit counts and its transition rate.
|
|
86
|
+
- The mechanism, named: dead clicks on a specific mesh, an unreachable target, a bounce before the
|
|
87
|
+
first interaction, or a genuine loss of interest.
|
|
88
|
+
- What each variant did, with sample sizes, and whether the difference can be told apart from noise.
|
|
89
|
+
|
|
90
|
+
Carry the caveats into the text: funnel steps are only as good as the event vocabulary they were
|
|
91
|
+
built from, a rate over a handful of sessions is not a rate, `meta.truncated` means you are looking
|
|
92
|
+
at a cut-off list, and a disabled capture channel produces a zero that means "not measured".
|
|
93
|
+
|
|
94
|
+
End with 2–3 concrete recommendations tied to the step and the mesh they apply to. If an `annotate`
|
|
95
|
+
tool is available, leave a note on the failing step so the next investigation starts there, and use
|
|
96
|
+
`save_analysis` to store the funnel definition you settled on — the next run should not have to
|
|
97
|
+
guess the steps again. If a `pin_panel` tool is available, pin the funnel panel.
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: performance_regression_triage
|
|
3
|
+
title: Performance regression triage
|
|
4
|
+
description: >-
|
|
5
|
+
Triage a frame-rate or stability regression: confirm it moved, date it, locate it (which scene,
|
|
6
|
+
device class, place in the scene), and name the mechanism — jank, shader compile stalls, memory
|
|
7
|
+
pressure, a render-scale change or a rendering-technology shift. USE FOR: "the app got slower",
|
|
8
|
+
a FPS drop after a release, stutter reports, deciding whether a regression is real or noise.
|
|
9
|
+
Trigger phrases: performance regression, FPS dropped, why is it slow, stutter, jank, frame drops,
|
|
10
|
+
did the last release slow things down, triage performance.
|
|
11
|
+
tools:
|
|
12
|
+
- insight_movers
|
|
13
|
+
- insight_anomalies
|
|
14
|
+
- insight_significance
|
|
15
|
+
- insight_baseline
|
|
16
|
+
- perf_summary
|
|
17
|
+
- perf_distribution
|
|
18
|
+
- frame_time_percentiles
|
|
19
|
+
- jank_rate
|
|
20
|
+
- perf_by_device
|
|
21
|
+
- perf_by_scene
|
|
22
|
+
- perf_heatmap
|
|
23
|
+
- compile_stalls
|
|
24
|
+
- resource_percentiles
|
|
25
|
+
- render_scale_truth
|
|
26
|
+
- rendering_technology
|
|
27
|
+
- query
|
|
28
|
+
capabilities:
|
|
29
|
+
- query
|
|
30
|
+
args:
|
|
31
|
+
- name: scene
|
|
32
|
+
required: false
|
|
33
|
+
description: Optional scene id to scope the triage to (see the uptimizr://scenes resource).
|
|
34
|
+
- name: range
|
|
35
|
+
required: false
|
|
36
|
+
default: the last 14 days
|
|
37
|
+
description: The window to triage, in words — e.g. "the last 14 days", "since the release".
|
|
38
|
+
---
|
|
39
|
+
|
|
40
|
+
Triage the performance regression in {{scope}} over {{range}}: is it real, when did it start, who
|
|
41
|
+
does it hit, and what is causing it?
|
|
42
|
+
|
|
43
|
+
Work through the method below with the read-only tools it names, then answer.
|
|
44
|
+
|
|
45
|
+
1. **Orient before you ask anything.** Read the `uptimizr://context` resource first: the real scene
|
|
46
|
+
ids, the data freshness, and which metrics are empty because their capture channel is off. A
|
|
47
|
+
performance channel that was never enabled looks exactly like a scene with no problem.
|
|
48
|
+
|
|
49
|
+
2. **Confirm something moved.** `insight_movers`{{#scene}} (`scene="{{scene}}"`){{/scene}} ranks
|
|
50
|
+
every comparable metric against the previous equal window by how unusual the change is. Read
|
|
51
|
+
`direction` with the sign of `delta` — a _rise_ in jank or errors is a regression — and drop any
|
|
52
|
+
row with `aboveMinSample: false`.
|
|
53
|
+
|
|
54
|
+
3. **Ask whether the new level is outside normal.** `insight_baseline` on `perf_summary` gives the
|
|
55
|
+
project's own median and spread; compare the new reading with `median` give or take a few `mad`,
|
|
56
|
+
or with the p10..p90 band. Frame rates are noisy, and "down 6 FPS" is routine in some projects
|
|
57
|
+
and an incident in others.
|
|
58
|
+
|
|
59
|
+
4. **Prove it rather than asserting it.** `insight_significance` on the metric that moved reports
|
|
60
|
+
the effect, a 95 % interval and a p-value across the two windows. An interval straddling 0 means
|
|
61
|
+
you cannot tell yet; `powerNote` says what this much data could have detected at all. Say "not
|
|
62
|
+
yet distinguishable from noise" when that is the truth — it is a finding.
|
|
63
|
+
|
|
64
|
+
5. **Put a date on it.** `insight_anomalies` (`metric=perf_summary`, `window=28`{{#scene}},
|
|
65
|
+
`scene="{{scene}}"`{{/scene}}) separates a one-day `spike`/`drop` from a `shift` — a level that
|
|
66
|
+
changed and stayed changed, which is what a release looks like. Quote the `bucketStart` and the
|
|
67
|
+
`contributor`. For the day-by-day shape around that date, ask the `query` tool for
|
|
68
|
+
`metric: "perf_daily"` — it is a registry metric with no canned tool of its own.
|
|
69
|
+
|
|
70
|
+
6. **Locate it.** `perf_by_scene` (which scene), `perf_by_device` (which device class — a
|
|
71
|
+
regression that only hits low-end hardware is a different bug from one that hits everyone), and
|
|
72
|
+
`perf_heatmap`{{#scene}} (`scene="{{scene}}"`){{/scene}} for _where in the scene_ the frames are
|
|
73
|
+
being lost. Ask for `format: "summary"` on the heatmap: merged clusters with shares, not a grid.
|
|
74
|
+
|
|
75
|
+
7. **Name the mechanism.** `frame_time_percentiles` and `jank_rate` separate "uniformly slower"
|
|
76
|
+
from "occasionally catastrophic" — the second is what users report and the average hides.
|
|
77
|
+
`perf_distribution` shows whether the whole population shifted or a tail got worse.
|
|
78
|
+
`compile_stalls` finds shader/pipeline compilation blocking the first seconds.
|
|
79
|
+
`resource_percentiles` finds GPU/memory pressure. `render_scale_truth` catches a resolution
|
|
80
|
+
change quietly doing the work the frame rate is getting credit for, and `rendering_technology`
|
|
81
|
+
catches a shift in the engine/renderer mix between the two windows — a "regression" that is
|
|
82
|
+
really a change in who is measuring.
|
|
83
|
+
|
|
84
|
+
8. **Cut it any way you need with the DSL.** The single `query` tool takes a `metric`, a `range`,
|
|
85
|
+
that metric's filters and `compare: { range: <the window before the shift> }`, returning
|
|
86
|
+
`{ current, previous, delta, deltaPct }` already joined — never subtract two runs yourself. Use
|
|
87
|
+
`dimensions` to regroup a portable count, `format: "summary"` for a bounded digest with a
|
|
88
|
+
`reading` and a per-row `drillQuery`, and `explain: true` whenever a number looks impossible:
|
|
89
|
+
the plan names the sample size, the row cap and every capture channel that could make it lie.
|
|
90
|
+
|
|
91
|
+
## What to report
|
|
92
|
+
|
|
93
|
+
- Whether the regression is real, with the effect, the interval and the honest verdict when it is
|
|
94
|
+
not yet distinguishable from noise.
|
|
95
|
+
- The date it started and whether it is a spike or a sustained shift.
|
|
96
|
+
- Who it hits: scene, device class, and where in the scene.
|
|
97
|
+
- The mechanism, named, with the metric that shows it.
|
|
98
|
+
|
|
99
|
+
Carry the caveats: percentiles below the metric's minimum sample, a session count too small to
|
|
100
|
+
generalise, sampled capture (ADR 0012), truncated results, and any device class whose share of
|
|
101
|
+
traffic changed between the windows — a mix shift moves the average without anything getting slower.
|
|
102
|
+
|
|
103
|
+
End with 2–3 concrete recommendations naming the scene, the device class or the asset they apply
|
|
104
|
+
to. If an `annotate` tool is available, leave a dated note on the shift so the next report can see
|
|
105
|
+
what happened; `save_analysis` keeps the triage for the post-mortem. If a `pin_panel` tool is
|
|
106
|
+
available, pin the panel that shows the regression.
|