@datalayer/agent-runtimes 1.3.27 → 1.3.29

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/lib/chat/ChatFloating.d.ts +12 -2
  2. package/lib/chat/ChatFloating.js +196 -25
  3. package/lib/chat/assistant/AssistantStage.d.ts +70 -0
  4. package/lib/chat/assistant/AssistantStage.js +192 -0
  5. package/lib/chat/assistant/SpeechBalloon.d.ts +26 -0
  6. package/lib/chat/assistant/SpeechBalloon.js +46 -0
  7. package/lib/chat/assistant/SpriteCharacter.d.ts +26 -0
  8. package/lib/chat/assistant/SpriteCharacter.js +73 -0
  9. package/lib/chat/assistant/characters.d.ts +30 -0
  10. package/lib/chat/assistant/characters.js +41 -0
  11. package/lib/chat/assistant/formats/acs.d.ts +88 -0
  12. package/lib/chat/assistant/formats/acs.js +493 -0
  13. package/lib/chat/assistant/formats/agentCompression.d.ts +5 -0
  14. package/lib/chat/assistant/formats/agentCompression.js +92 -0
  15. package/lib/chat/assistant/formats/blobs.d.ts +6 -0
  16. package/lib/chat/assistant/formats/blobs.js +52 -0
  17. package/lib/chat/assistant/formats/clippy.d.ts +21 -0
  18. package/lib/chat/assistant/formats/clippy.js +167 -0
  19. package/lib/chat/assistant/formats/index.d.ts +13 -0
  20. package/lib/chat/assistant/formats/index.js +15 -0
  21. package/lib/chat/assistant/formats/objectLiteral.d.ts +23 -0
  22. package/lib/chat/assistant/formats/objectLiteral.js +233 -0
  23. package/lib/chat/assistant/formats/stateAnimations.d.ts +16 -0
  24. package/lib/chat/assistant/formats/stateAnimations.js +115 -0
  25. package/lib/chat/assistant/formats/types.d.ts +56 -0
  26. package/lib/chat/assistant/formats/types.js +11 -0
  27. package/lib/chat/assistant/state.d.ts +62 -0
  28. package/lib/chat/assistant/state.js +116 -0
  29. package/lib/chat/header/ChatViewModeToggle.js +6 -3
  30. package/lib/chat/presence/Presence.d.ts +9 -0
  31. package/lib/chat/presence/Presence.js +57 -0
  32. package/lib/chat/presence/presenceStatus.d.ts +32 -0
  33. package/lib/chat/presence/presenceStatus.js +46 -0
  34. package/lib/chat/viewModes.d.ts +6 -0
  35. package/lib/chat/viewModes.js +10 -1
  36. package/lib/examples/ChatAssistantExample.d.ts +8 -0
  37. package/lib/examples/ChatAssistantExample.js +59 -0
  38. package/lib/examples/example-selector.js +1 -0
  39. package/lib/examples/main.js +3 -0
  40. package/lib/loop/apps/AppRenderer.d.ts +8 -1
  41. package/lib/loop/apps/AppRenderer.js +2 -1
  42. package/lib/loop/apps/appspec.d.ts +5 -1
  43. package/lib/loop/apps/appspec.js +21 -1
  44. package/lib/loop/apps/checks.js +3 -1
  45. package/lib/loop/core/index.d.ts +15 -0
  46. package/lib/loop/core/index.js +1 -0
  47. package/lib/loop/plugins/assistant-characters/index.d.ts +30 -0
  48. package/lib/loop/plugins/assistant-characters/index.js +41 -0
  49. package/lib/loop/plugins/chat/ChatView.js +21 -1
  50. package/lib/loop/plugins/chat/index.d.ts +8 -0
  51. package/lib/loop/plugins/page-layout/index.js +5 -1
  52. package/lib/specs/actions.js +6 -0
  53. package/lib/specs/apps.d.ts +3 -0
  54. package/lib/specs/apps.js +793 -65
  55. package/lib/specs/appspecSchema.js +20 -2
  56. package/lib/types/agentspecs.d.ts +10 -1
  57. package/lib/types/chat.d.ts +1 -1
  58. package/package.json +1 -1
package/lib/specs/apps.js CHANGED
@@ -2,6 +2,135 @@
2
2
  * Copyright (c) 2025-2026 Datalayer, Inc.
3
3
  * Distributed under the terms of the Modified BSD License.
4
4
  */
5
+ export const DATA_QUALITY_APP_0_0_1 = {
6
+ schema: 'loop.app/v1',
7
+ id: 'data-quality',
8
+ version: '0.0.1',
9
+ name: 'Data Quality Investigation',
10
+ kind: 'decision',
11
+ description: 'Which anomalies in this dataset should we fix first? For a data team, before a dataset is used for a decision.',
12
+ owner: 'Datalayer <info@datalayer.io>',
13
+ agent: 'jupyter-data-analyst:0.0.1',
14
+ team: '',
15
+ instructions: '',
16
+ model: '',
17
+ skills: [],
18
+ tools: [],
19
+ context: [],
20
+ contents: ['The dataset under investigation'],
21
+ connections: [],
22
+ rules: [],
23
+ permissions: {
24
+ spaces: [],
25
+ computer: {
26
+ browse: false,
27
+ files: false,
28
+ shell: false,
29
+ },
30
+ },
31
+ interface: {
32
+ layout: 'page',
33
+ accent: 'green',
34
+ welcome: '',
35
+ starters: [],
36
+ settings: [],
37
+ components: [
38
+ 'Card',
39
+ 'Column',
40
+ 'Row',
41
+ 'List',
42
+ 'Tabs',
43
+ 'Text',
44
+ 'Slider',
45
+ 'ChoicePicker',
46
+ 'TextField',
47
+ 'Button',
48
+ ],
49
+ },
50
+ tests: {
51
+ readyAt: 0.8,
52
+ evalset: '',
53
+ cases: [],
54
+ },
55
+ record: {
56
+ keepFor: '1_years',
57
+ include: ['decisions', 'sources', 'checks'],
58
+ retentionDays: 365,
59
+ },
60
+ checks: {
61
+ guards: [],
62
+ gates: [],
63
+ track: '',
64
+ },
65
+ deployment: {
66
+ hosted: {
67
+ visibility: 'private',
68
+ slug: '',
69
+ },
70
+ },
71
+ goal: '',
72
+ triggers: [],
73
+ memory: '',
74
+ notifications: [],
75
+ decision: {
76
+ question: 'Which anomalies in this dataset should we fix first?',
77
+ alternatives: [],
78
+ criteria: [
79
+ {
80
+ name: 'Rows affected',
81
+ kind: 'metric',
82
+ weight: 2.0,
83
+ instructions: 'How many rows the anomaly touches, from a validation run in the sandbox.',
84
+ options: [],
85
+ direction: 'higher',
86
+ measure: '',
87
+ },
88
+ {
89
+ name: 'Effect on the result',
90
+ kind: 'metric',
91
+ weight: 3.0,
92
+ instructions: 'How far the headline figures move when the anomaly is corrected.',
93
+ options: [],
94
+ direction: 'higher',
95
+ measure: '',
96
+ },
97
+ {
98
+ name: 'Kind of anomaly',
99
+ kind: 'choice',
100
+ weight: 0.0,
101
+ instructions: 'What is this anomaly?',
102
+ options: [
103
+ 'Genuine: a real extreme, to keep',
104
+ 'Outlier: a value far from the rest, to check',
105
+ 'Unit: a unit mismatch',
106
+ 'Missing: a missing value',
107
+ 'Duplicate: the same row twice',
108
+ ],
109
+ direction: 'higher',
110
+ measure: '',
111
+ },
112
+ {
113
+ name: 'Safe to correct automatically',
114
+ kind: 'noul',
115
+ weight: 1.0,
116
+ instructions: 'Can the proposed correction be applied without a person checking each row?',
117
+ options: [],
118
+ direction: 'higher',
119
+ measure: '',
120
+ },
121
+ ],
122
+ minConfidence: 0.0,
123
+ scenarios: [],
124
+ judgmentModel: 'cloudflare:gtw/typesafe/jev',
125
+ },
126
+ enabled: true,
127
+ tags: ['example', 'decision', 'data-quality'],
128
+ icon: 'filter',
129
+ emoji: '🧹',
130
+ avatar: '',
131
+ banner: '',
132
+ setup: ["The agent 'jupyter-data-analyst:0.0.1' is not enabled."],
133
+ };
5
134
  export const INBOX_TRIAGE_APP_0_0_1 = {
6
135
  schema: 'loop.app/v1',
7
136
  id: 'inbox-triage',
@@ -161,6 +290,191 @@ export const INBOX_TRIAGE_APP_0_0_1 = {
161
290
  "The MCP server 'google-workspace:0.0.1' is not enabled.",
162
291
  ],
163
292
  };
293
+ export const MODEL_CHOICE_APP_0_0_1 = {
294
+ schema: 'loop.app/v1',
295
+ id: 'model-choice',
296
+ version: '0.0.1',
297
+ name: 'Model Choice',
298
+ kind: 'decision',
299
+ description: 'Which chat model should this use case run on? For a team choosing a model for one job — a summarizer, a classifier, an agent — from what a benchmark run measured and what a judge reads in the answers.',
300
+ owner: 'Datalayer <info@datalayer.io>',
301
+ agent: 'jupyter-data-analyst:0.0.1',
302
+ team: '',
303
+ instructions: '',
304
+ model: '',
305
+ skills: [],
306
+ tools: [],
307
+ context: [],
308
+ contents: [
309
+ 'The benchmark run: one configuration per model, its task results, cost and latency',
310
+ 'The Models page: how each model is billed (standard or credits) and who hosts it',
311
+ ],
312
+ connections: [],
313
+ rules: [],
314
+ permissions: {
315
+ spaces: [],
316
+ computer: {
317
+ browse: false,
318
+ files: false,
319
+ shell: false,
320
+ },
321
+ },
322
+ interface: {
323
+ layout: 'page',
324
+ accent: 'green',
325
+ welcome: '',
326
+ starters: [],
327
+ settings: [],
328
+ components: [
329
+ 'Card',
330
+ 'Column',
331
+ 'Row',
332
+ 'List',
333
+ 'Tabs',
334
+ 'Text',
335
+ 'Slider',
336
+ 'ChoicePicker',
337
+ 'TextField',
338
+ 'Button',
339
+ ],
340
+ },
341
+ tests: {
342
+ readyAt: 0.8,
343
+ evalset: '',
344
+ cases: [],
345
+ },
346
+ record: {
347
+ keepFor: '1_years',
348
+ include: ['decisions', 'sources', 'checks'],
349
+ retentionDays: 365,
350
+ },
351
+ checks: {
352
+ guards: [],
353
+ gates: [],
354
+ track: '',
355
+ },
356
+ deployment: {
357
+ hosted: {
358
+ visibility: 'private',
359
+ slug: '',
360
+ },
361
+ },
362
+ goal: '',
363
+ triggers: [],
364
+ memory: '',
365
+ notifications: [],
366
+ decision: {
367
+ question: 'Which chat model should this use case run on?',
368
+ alternatives: [],
369
+ criteria: [
370
+ {
371
+ name: 'Pass rate',
372
+ kind: 'metric',
373
+ weight: 3.0,
374
+ instructions: 'Share of tasks passed, from the run.',
375
+ options: [],
376
+ direction: 'higher',
377
+ measure: 'pass_rate',
378
+ },
379
+ {
380
+ name: 'Cost per task',
381
+ kind: 'metric',
382
+ weight: 2.0,
383
+ instructions: 'Credits spent per task, from the run; lower is better.',
384
+ options: [],
385
+ direction: 'lower',
386
+ measure: 'cost_per_task',
387
+ },
388
+ {
389
+ name: 'Latency',
390
+ kind: 'metric',
391
+ weight: 2.0,
392
+ instructions: 'Median seconds per task, from the run; lower is better.',
393
+ options: [],
394
+ direction: 'lower',
395
+ measure: 'seconds_per_task',
396
+ },
397
+ {
398
+ name: 'Answer quality',
399
+ kind: 'score',
400
+ weight: 3.0,
401
+ instructions: 'Reading the failures and what the run recorded, how good are this model’s answers for the use case?',
402
+ options: [
403
+ 'Unusable: wrong or off-task answers',
404
+ 'Rough: usable with rework',
405
+ 'Good: usable as they are',
406
+ 'Excellent: better than the reference',
407
+ ],
408
+ direction: 'higher',
409
+ measure: '',
410
+ },
411
+ {
412
+ name: 'Follows the format',
413
+ kind: 'noul',
414
+ weight: 1.0,
415
+ instructions: 'Does this model keep to the output format the use case asks for?',
416
+ options: [],
417
+ direction: 'higher',
418
+ measure: '',
419
+ },
420
+ {
421
+ name: 'Missing information',
422
+ kind: 'choice',
423
+ weight: 0.0,
424
+ instructions: 'What is missing to choose this model?',
425
+ options: [
426
+ 'Price: the billing of this model is not known',
427
+ 'Traces: the failures have no trajectory to read',
428
+ 'Cases: the run is too small to tell',
429
+ 'Nothing: everything needed is there',
430
+ ],
431
+ direction: 'higher',
432
+ measure: '',
433
+ },
434
+ ],
435
+ minConfidence: 0.6,
436
+ scenarios: [
437
+ {
438
+ name: 'Quality first',
439
+ weights: {
440
+ 'Pass rate': 4.0,
441
+ 'Cost per task': 0.0,
442
+ Latency: 1.0,
443
+ 'Answer quality': 4.0,
444
+ 'Follows the format': 1.0,
445
+ },
446
+ },
447
+ {
448
+ name: 'Cheapest that works',
449
+ weights: {
450
+ 'Pass rate': 3.0,
451
+ 'Cost per task': 4.0,
452
+ Latency: 1.0,
453
+ 'Answer quality': 1.0,
454
+ 'Follows the format': 1.0,
455
+ },
456
+ },
457
+ {
458
+ name: 'Fastest that works',
459
+ weights: {
460
+ 'Pass rate': 3.0,
461
+ 'Cost per task': 1.0,
462
+ Latency: 4.0,
463
+ 'Answer quality': 1.0,
464
+ 'Follows the format': 1.0,
465
+ },
466
+ },
467
+ ],
468
+ judgmentModel: 'cloudflare:gtw/typesafe/jev',
469
+ },
470
+ enabled: true,
471
+ tags: ['example', 'decision', 'benchmarks', 'models'],
472
+ icon: 'cpu',
473
+ emoji: '🧠',
474
+ avatar: '',
475
+ banner: '',
476
+ setup: ["The agent 'jupyter-data-analyst:0.0.1' is not enabled."],
477
+ };
164
478
  export const QUOTE_CALCULATOR_APP_0_0_1 = {
165
479
  schema: 'loop.app/v1',
166
480
  id: 'quote-calculator',
@@ -336,25 +650,182 @@ export const QUOTE_CALCULATOR_APP_0_0_1 = {
336
650
  origins: [],
337
651
  },
338
652
  },
339
- goal: '',
340
- triggers: [],
341
- memory: '',
342
- notifications: [],
343
- enabled: false,
344
- tags: ['example', 'widget'],
345
- icon: 'number',
346
- emoji: '🧮',
653
+ goal: '',
654
+ triggers: [],
655
+ memory: '',
656
+ notifications: [],
657
+ enabled: false,
658
+ tags: ['example', 'widget'],
659
+ icon: 'number',
660
+ emoji: '🧮',
661
+ avatar: '',
662
+ banner: '',
663
+ setup: ["The agent 'jupyter-data-analyst:0.0.1' is not enabled."],
664
+ };
665
+ export const SHIP_OR_FIX_APP_0_0_1 = {
666
+ schema: 'loop.app/v1',
667
+ id: 'ship-or-fix',
668
+ version: '0.0.1',
669
+ name: 'Ship or Fix',
670
+ kind: 'decision',
671
+ description: 'Which agent configuration should we ship, on the evidence of a benchmark run? For an AI platform team, after every run.',
672
+ owner: 'Datalayer <info@datalayer.io>',
673
+ agent: 'jupyter-data-analyst:0.0.1',
674
+ team: '',
675
+ instructions: '',
676
+ model: '',
677
+ skills: [],
678
+ tools: [],
679
+ context: [],
680
+ contents: ['The benchmark run: task results, traces, cost and latency'],
681
+ connections: [],
682
+ rules: [],
683
+ permissions: {
684
+ spaces: [],
685
+ computer: {
686
+ browse: false,
687
+ files: false,
688
+ shell: false,
689
+ },
690
+ },
691
+ interface: {
692
+ layout: 'page',
693
+ accent: 'green',
694
+ welcome: '',
695
+ starters: [],
696
+ settings: [],
697
+ components: [
698
+ 'Card',
699
+ 'Column',
700
+ 'Row',
701
+ 'List',
702
+ 'Tabs',
703
+ 'Text',
704
+ 'Slider',
705
+ 'ChoicePicker',
706
+ 'TextField',
707
+ 'Button',
708
+ ],
709
+ },
710
+ tests: {
711
+ readyAt: 0.8,
712
+ evalset: '',
713
+ cases: [],
714
+ },
715
+ record: {
716
+ keepFor: '1_years',
717
+ include: ['decisions', 'sources', 'checks'],
718
+ retentionDays: 365,
719
+ },
720
+ checks: {
721
+ guards: [],
722
+ gates: [],
723
+ track: '',
724
+ },
725
+ deployment: {
726
+ hosted: {
727
+ visibility: 'private',
728
+ slug: '',
729
+ },
730
+ },
731
+ goal: '',
732
+ triggers: [],
733
+ memory: '',
734
+ notifications: [],
735
+ decision: {
736
+ question: 'Which agent configuration should we ship?',
737
+ alternatives: [],
738
+ criteria: [
739
+ {
740
+ name: 'Pass rate',
741
+ kind: 'metric',
742
+ weight: 3.0,
743
+ instructions: 'Share of tasks passed, from the run.',
744
+ options: [],
745
+ direction: 'higher',
746
+ measure: 'pass_rate',
747
+ },
748
+ {
749
+ name: 'Cost per task',
750
+ kind: 'metric',
751
+ weight: 1.0,
752
+ instructions: 'Credits spent per task, from the run; lower is better.',
753
+ options: [],
754
+ direction: 'lower',
755
+ measure: 'cost_per_task',
756
+ },
757
+ {
758
+ name: 'Latency',
759
+ kind: 'metric',
760
+ weight: 1.0,
761
+ instructions: 'Median time per task, from the run; lower is better.',
762
+ options: [],
763
+ direction: 'lower',
764
+ measure: 'seconds_per_task',
765
+ },
766
+ {
767
+ name: 'Failure severity',
768
+ kind: 'score',
769
+ weight: 2.0,
770
+ instructions: 'How bad are the failures of this configuration?',
771
+ options: [
772
+ 'Blocking: a wrong number somebody would act on',
773
+ 'Degraded: a usable answer with a flaw to work around',
774
+ 'Cosmetic: a format, a label, nothing that changes the answer',
775
+ ],
776
+ direction: 'higher',
777
+ measure: '',
778
+ },
779
+ {
780
+ name: 'Formatting failures block shipping',
781
+ kind: 'noul',
782
+ weight: 1.0,
783
+ instructions: 'Are the formatting failures of this configuration blocking for the people who read its answers?',
784
+ options: [],
785
+ direction: 'lower',
786
+ measure: '',
787
+ },
788
+ ],
789
+ minConfidence: 0.6,
790
+ scenarios: [
791
+ {
792
+ name: 'Quality first',
793
+ weights: {
794
+ 'Pass rate': 4.0,
795
+ 'Cost per task': 0.0,
796
+ Latency: 0.0,
797
+ 'Failure severity': 3.0,
798
+ 'Formatting failures block shipping': 1.0,
799
+ },
800
+ },
801
+ {
802
+ name: 'Cost first',
803
+ weights: {
804
+ 'Pass rate': 2.0,
805
+ 'Cost per task': 4.0,
806
+ Latency: 2.0,
807
+ 'Failure severity': 1.0,
808
+ 'Formatting failures block shipping': 0.0,
809
+ },
810
+ },
811
+ ],
812
+ judgmentModel: 'cloudflare:gtw/typesafe/jev',
813
+ },
814
+ enabled: true,
815
+ tags: ['example', 'decision', 'benchmarks'],
816
+ icon: 'checklist',
817
+ emoji: '🚢',
347
818
  avatar: '',
348
819
  banner: '',
349
820
  setup: ["The agent 'jupyter-data-analyst:0.0.1' is not enabled."],
350
821
  };
351
- export const SHIP_OR_FIX_APP_0_0_1 = {
822
+ export const SUPPLIER_COMPARISON_APP_0_0_1 = {
352
823
  schema: 'loop.app/v1',
353
- id: 'ship-or-fix',
824
+ id: 'supplier-comparison',
354
825
  version: '0.0.1',
355
- name: 'Ship or Fix',
826
+ name: 'Supplier Comparison',
356
827
  kind: 'decision',
357
- description: 'Which agent configuration should we ship, on the evidence of a benchmark run? For an AI platform team, after every run.',
828
+ description: 'Which supplier should we choose for these orders? For an operations or procurement lead, at each sourcing round.',
358
829
  owner: 'Datalayer <info@datalayer.io>',
359
830
  agent: 'jupyter-data-analyst:0.0.1',
360
831
  team: '',
@@ -363,7 +834,7 @@ export const SHIP_OR_FIX_APP_0_0_1 = {
363
834
  skills: [],
364
835
  tools: [],
365
836
  context: [],
366
- contents: ['The benchmark run: task results, traces, cost and latency'],
837
+ contents: ['Order history', 'Supplier price lists', 'Delivery records'],
367
838
  connections: [],
368
839
  rules: [],
369
840
  permissions: {
@@ -419,88 +890,73 @@ export const SHIP_OR_FIX_APP_0_0_1 = {
419
890
  memory: '',
420
891
  notifications: [],
421
892
  decision: {
422
- question: 'Which agent configuration should we ship?',
893
+ question: 'Which supplier should we choose for these orders?',
423
894
  alternatives: [],
424
895
  criteria: [
425
896
  {
426
- name: 'Pass rate',
897
+ name: 'Price',
427
898
  kind: 'metric',
428
- weight: 3.0,
429
- instructions: 'Share of tasks passed, from the run.',
899
+ weight: 2.0,
900
+ instructions: 'Total cost of the orders at each supplier’s prices.',
430
901
  options: [],
431
- direction: 'higher',
432
- measure: 'pass_rate',
902
+ direction: 'lower',
903
+ measure: '',
433
904
  },
434
905
  {
435
- name: 'Cost per task',
906
+ name: 'Delivery reliability',
436
907
  kind: 'metric',
437
- weight: 1.0,
438
- instructions: 'Credits spent per task, from the run; lower is better.',
908
+ weight: 2.0,
909
+ instructions: 'Share of past deliveries on time, from the delivery records.',
439
910
  options: [],
440
- direction: 'lower',
441
- measure: 'cost_per_task',
911
+ direction: 'higher',
912
+ measure: '',
442
913
  },
443
914
  {
444
- name: 'Latency',
915
+ name: 'Capacity',
445
916
  kind: 'metric',
446
917
  weight: 1.0,
447
- instructions: 'Median time per task, from the run; lower is better.',
918
+ instructions: 'Whether the supplier’s capacity covers the ordered volume.',
448
919
  options: [],
449
- direction: 'lower',
450
- measure: 'seconds_per_task',
920
+ direction: 'higher',
921
+ measure: '',
451
922
  },
452
923
  {
453
- name: 'Failure severity',
924
+ name: 'Fit with requirements',
454
925
  kind: 'score',
455
926
  weight: 2.0,
456
- instructions: 'How bad are the failures of this configuration?',
927
+ instructions: 'How well does this supplier fit the stated requirements?',
457
928
  options: [
458
- 'Blocking: a wrong number somebody would act on',
459
- 'Degraded: a usable answer with a flaw to work around',
460
- 'Cosmetic: a format, a label, nothing that changes the answer',
929
+ 'None: meets none of the stated requirements',
930
+ 'Some: meets a few, misses the important ones',
931
+ 'Most: meets the important ones, misses a few',
932
+ 'All: meets every stated requirement',
461
933
  ],
462
934
  direction: 'higher',
463
935
  measure: '',
464
936
  },
465
937
  {
466
- name: 'Formatting failures block shipping',
467
- kind: 'noul',
468
- weight: 1.0,
469
- instructions: 'Are the formatting failures of this configuration blocking for the people who read its answers?',
470
- options: [],
471
- direction: 'lower',
938
+ name: 'Missing information',
939
+ kind: 'choice',
940
+ weight: 0.0,
941
+ instructions: 'What is missing to decide on this supplier?',
942
+ options: [
943
+ 'Capacity: a capacity figure is missing',
944
+ 'Delivery: a delivery record is missing',
945
+ 'Price: a price is missing',
946
+ 'Nothing: everything needed is there',
947
+ ],
948
+ direction: 'higher',
472
949
  measure: '',
473
950
  },
474
951
  ],
475
- minConfidence: 0.6,
476
- scenarios: [
477
- {
478
- name: 'Quality first',
479
- weights: {
480
- 'Pass rate': 4.0,
481
- 'Cost per task': 0.0,
482
- Latency: 0.0,
483
- 'Failure severity': 3.0,
484
- 'Formatting failures block shipping': 1.0,
485
- },
486
- },
487
- {
488
- name: 'Cost first',
489
- weights: {
490
- 'Pass rate': 2.0,
491
- 'Cost per task': 4.0,
492
- Latency: 2.0,
493
- 'Failure severity': 1.0,
494
- 'Formatting failures block shipping': 0.0,
495
- },
496
- },
497
- ],
952
+ minConfidence: 0.0,
953
+ scenarios: [],
498
954
  judgmentModel: 'cloudflare:gtw/typesafe/jev',
499
955
  },
500
956
  enabled: true,
501
- tags: ['example', 'decision', 'benchmarks'],
502
- icon: 'checklist',
503
- emoji: '🚢',
957
+ tags: ['example', 'decision', 'procurement'],
958
+ icon: 'package',
959
+ emoji: '🚚',
504
960
  avatar: '',
505
961
  banner: '',
506
962
  setup: ["The agent 'jupyter-data-analyst:0.0.1' is not enabled."],
@@ -614,9 +1070,12 @@ export const WEB_RESEARCH_APP_0_0_1 = {
614
1070
  setup: [],
615
1071
  };
616
1072
  export const APP_CATALOGUE = {
1073
+ 'data-quality': DATA_QUALITY_APP_0_0_1,
617
1074
  'inbox-triage': INBOX_TRIAGE_APP_0_0_1,
1075
+ 'model-choice': MODEL_CHOICE_APP_0_0_1,
618
1076
  'quote-calculator': QUOTE_CALCULATOR_APP_0_0_1,
619
1077
  'ship-or-fix': SHIP_OR_FIX_APP_0_0_1,
1078
+ 'supplier-comparison': SUPPLIER_COMPARISON_APP_0_0_1,
620
1079
  'web-research': WEB_RESEARCH_APP_0_0_1,
621
1080
  };
622
1081
  /** An application, by `id` or `id:version`, or undefined. */
@@ -637,6 +1096,73 @@ export function getApp(ref) {
637
1096
  * Appspec have to give back.
638
1097
  */
639
1098
  export const APP_SOURCES = {
1099
+ 'data-quality': {
1100
+ schema: 'loop.app/v1',
1101
+ id: 'data-quality',
1102
+ name: 'Data Quality Investigation',
1103
+ kind: 'decision',
1104
+ description: 'Which anomalies in this dataset should we fix first? For a data team, before a dataset is used for a decision.',
1105
+ owner: 'Datalayer <info@datalayer.io>',
1106
+ agent: 'jupyter-data-analyst:0.0.1',
1107
+ contents: ['The dataset under investigation'],
1108
+ interface: {
1109
+ components: [
1110
+ 'Card',
1111
+ 'Column',
1112
+ 'Row',
1113
+ 'List',
1114
+ 'Tabs',
1115
+ 'Text',
1116
+ 'Slider',
1117
+ 'ChoicePicker',
1118
+ 'TextField',
1119
+ 'Button',
1120
+ ],
1121
+ },
1122
+ record: {
1123
+ include: ['decisions', 'sources', 'checks'],
1124
+ },
1125
+ deployment: {
1126
+ hosted: {},
1127
+ },
1128
+ decision: {
1129
+ question: 'Which anomalies in this dataset should we fix first?',
1130
+ criteria: [
1131
+ {
1132
+ name: 'Rows affected',
1133
+ weight: 2.0,
1134
+ instructions: 'How many rows the anomaly touches, from a validation run in the sandbox.',
1135
+ },
1136
+ {
1137
+ name: 'Effect on the result',
1138
+ weight: 3.0,
1139
+ instructions: 'How far the headline figures move when the anomaly is corrected.',
1140
+ },
1141
+ {
1142
+ name: 'Kind of anomaly',
1143
+ kind: 'choice',
1144
+ weight: 0.0,
1145
+ instructions: 'What is this anomaly?',
1146
+ options: [
1147
+ 'Genuine: a real extreme, to keep',
1148
+ 'Outlier: a value far from the rest, to check',
1149
+ 'Unit: a unit mismatch',
1150
+ 'Missing: a missing value',
1151
+ 'Duplicate: the same row twice',
1152
+ ],
1153
+ },
1154
+ {
1155
+ name: 'Safe to correct automatically',
1156
+ kind: 'noul',
1157
+ instructions: 'Can the proposed correction be applied without a person checking each row?',
1158
+ },
1159
+ ],
1160
+ judgment_model: 'cloudflare:gtw/typesafe/jev',
1161
+ },
1162
+ tags: ['example', 'decision', 'data-quality'],
1163
+ icon: 'filter',
1164
+ emoji: '🧹',
1165
+ },
640
1166
  'inbox-triage': {
641
1167
  schema: 'loop.app/v1',
642
1168
  id: 'inbox-triage',
@@ -755,6 +1281,130 @@ export const APP_SOURCES = {
755
1281
  icon: 'mail',
756
1282
  emoji: '📬',
757
1283
  },
1284
+ 'model-choice': {
1285
+ schema: 'loop.app/v1',
1286
+ id: 'model-choice',
1287
+ name: 'Model Choice',
1288
+ kind: 'decision',
1289
+ description: 'Which chat model should this use case run on? For a team choosing a model for one job — a summarizer, a classifier, an agent — from what a benchmark run measured and what a judge reads in the answers.',
1290
+ owner: 'Datalayer <info@datalayer.io>',
1291
+ agent: 'jupyter-data-analyst:0.0.1',
1292
+ contents: [
1293
+ 'The benchmark run: one configuration per model, its task results, cost and latency',
1294
+ 'The Models page: how each model is billed (standard or credits) and who hosts it',
1295
+ ],
1296
+ interface: {
1297
+ components: [
1298
+ 'Card',
1299
+ 'Column',
1300
+ 'Row',
1301
+ 'List',
1302
+ 'Tabs',
1303
+ 'Text',
1304
+ 'Slider',
1305
+ 'ChoicePicker',
1306
+ 'TextField',
1307
+ 'Button',
1308
+ ],
1309
+ },
1310
+ record: {
1311
+ include: ['decisions', 'sources', 'checks'],
1312
+ },
1313
+ deployment: {
1314
+ hosted: {},
1315
+ },
1316
+ decision: {
1317
+ question: 'Which chat model should this use case run on?',
1318
+ criteria: [
1319
+ {
1320
+ name: 'Pass rate',
1321
+ weight: 3.0,
1322
+ instructions: 'Share of tasks passed, from the run.',
1323
+ measure: 'pass_rate',
1324
+ },
1325
+ {
1326
+ name: 'Cost per task',
1327
+ weight: 2.0,
1328
+ instructions: 'Credits spent per task, from the run; lower is better.',
1329
+ direction: 'lower',
1330
+ measure: 'cost_per_task',
1331
+ },
1332
+ {
1333
+ name: 'Latency',
1334
+ weight: 2.0,
1335
+ instructions: 'Median seconds per task, from the run; lower is better.',
1336
+ direction: 'lower',
1337
+ measure: 'seconds_per_task',
1338
+ },
1339
+ {
1340
+ name: 'Answer quality',
1341
+ kind: 'score',
1342
+ weight: 3.0,
1343
+ instructions: 'Reading the failures and what the run recorded, how good are this model’s answers for the use case?',
1344
+ options: [
1345
+ 'Unusable: wrong or off-task answers',
1346
+ 'Rough: usable with rework',
1347
+ 'Good: usable as they are',
1348
+ 'Excellent: better than the reference',
1349
+ ],
1350
+ },
1351
+ {
1352
+ name: 'Follows the format',
1353
+ kind: 'noul',
1354
+ instructions: 'Does this model keep to the output format the use case asks for?',
1355
+ },
1356
+ {
1357
+ name: 'Missing information',
1358
+ kind: 'choice',
1359
+ weight: 0.0,
1360
+ instructions: 'What is missing to choose this model?',
1361
+ options: [
1362
+ 'Price: the billing of this model is not known',
1363
+ 'Traces: the failures have no trajectory to read',
1364
+ 'Cases: the run is too small to tell',
1365
+ 'Nothing: everything needed is there',
1366
+ ],
1367
+ },
1368
+ ],
1369
+ min_confidence: 0.6,
1370
+ scenarios: [
1371
+ {
1372
+ name: 'Quality first',
1373
+ weights: {
1374
+ 'Answer quality': 4.0,
1375
+ 'Cost per task': 0.0,
1376
+ 'Follows the format': 1.0,
1377
+ Latency: 1.0,
1378
+ 'Pass rate': 4.0,
1379
+ },
1380
+ },
1381
+ {
1382
+ name: 'Cheapest that works',
1383
+ weights: {
1384
+ 'Answer quality': 1.0,
1385
+ 'Cost per task': 4.0,
1386
+ 'Follows the format': 1.0,
1387
+ Latency: 1.0,
1388
+ 'Pass rate': 3.0,
1389
+ },
1390
+ },
1391
+ {
1392
+ name: 'Fastest that works',
1393
+ weights: {
1394
+ 'Answer quality': 1.0,
1395
+ 'Cost per task': 1.0,
1396
+ 'Follows the format': 1.0,
1397
+ Latency: 4.0,
1398
+ 'Pass rate': 3.0,
1399
+ },
1400
+ },
1401
+ ],
1402
+ judgment_model: 'cloudflare:gtw/typesafe/jev',
1403
+ },
1404
+ tags: ['example', 'decision', 'benchmarks', 'models'],
1405
+ icon: 'cpu',
1406
+ emoji: '🧠',
1407
+ },
758
1408
  'quote-calculator': {
759
1409
  schema: 'loop.app/v1',
760
1410
  id: 'quote-calculator',
@@ -997,6 +1647,84 @@ export const APP_SOURCES = {
997
1647
  icon: 'checklist',
998
1648
  emoji: '🚢',
999
1649
  },
1650
+ 'supplier-comparison': {
1651
+ schema: 'loop.app/v1',
1652
+ id: 'supplier-comparison',
1653
+ name: 'Supplier Comparison',
1654
+ kind: 'decision',
1655
+ description: 'Which supplier should we choose for these orders? For an operations or procurement lead, at each sourcing round.',
1656
+ owner: 'Datalayer <info@datalayer.io>',
1657
+ agent: 'jupyter-data-analyst:0.0.1',
1658
+ contents: ['Order history', 'Supplier price lists', 'Delivery records'],
1659
+ interface: {
1660
+ components: [
1661
+ 'Card',
1662
+ 'Column',
1663
+ 'Row',
1664
+ 'List',
1665
+ 'Tabs',
1666
+ 'Text',
1667
+ 'Slider',
1668
+ 'ChoicePicker',
1669
+ 'TextField',
1670
+ 'Button',
1671
+ ],
1672
+ },
1673
+ record: {
1674
+ include: ['decisions', 'sources', 'checks'],
1675
+ },
1676
+ deployment: {
1677
+ hosted: {},
1678
+ },
1679
+ decision: {
1680
+ question: 'Which supplier should we choose for these orders?',
1681
+ criteria: [
1682
+ {
1683
+ name: 'Price',
1684
+ weight: 2.0,
1685
+ instructions: 'Total cost of the orders at each supplier’s prices.',
1686
+ direction: 'lower',
1687
+ },
1688
+ {
1689
+ name: 'Delivery reliability',
1690
+ weight: 2.0,
1691
+ instructions: 'Share of past deliveries on time, from the delivery records.',
1692
+ },
1693
+ {
1694
+ name: 'Capacity',
1695
+ instructions: 'Whether the supplier’s capacity covers the ordered volume.',
1696
+ },
1697
+ {
1698
+ name: 'Fit with requirements',
1699
+ kind: 'score',
1700
+ weight: 2.0,
1701
+ instructions: 'How well does this supplier fit the stated requirements?',
1702
+ options: [
1703
+ 'None: meets none of the stated requirements',
1704
+ 'Some: meets a few, misses the important ones',
1705
+ 'Most: meets the important ones, misses a few',
1706
+ 'All: meets every stated requirement',
1707
+ ],
1708
+ },
1709
+ {
1710
+ name: 'Missing information',
1711
+ kind: 'choice',
1712
+ weight: 0.0,
1713
+ instructions: 'What is missing to decide on this supplier?',
1714
+ options: [
1715
+ 'Capacity: a capacity figure is missing',
1716
+ 'Delivery: a delivery record is missing',
1717
+ 'Price: a price is missing',
1718
+ 'Nothing: everything needed is there',
1719
+ ],
1720
+ },
1721
+ ],
1722
+ judgment_model: 'cloudflare:gtw/typesafe/jev',
1723
+ },
1724
+ tags: ['example', 'decision', 'procurement'],
1725
+ icon: 'package',
1726
+ emoji: '🚚',
1727
+ },
1000
1728
  'web-research': {
1001
1729
  schema: 'loop.app/v1',
1002
1730
  id: 'web-research',