jev-recipes 0.4.0 → 0.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. package/CHANGELOG.md +7 -0
  2. package/README.md +134 -50
  3. package/dist/catalog/generated/details/attribution-match.json +55 -0
  4. package/dist/catalog/generated/details/choose-action.json +102 -0
  5. package/dist/catalog/generated/details/evaluation-mention.json +48 -0
  6. package/dist/catalog/generated/details/evidence-independence.json +52 -0
  7. package/dist/catalog/generated/details/question-leading.json +52 -0
  8. package/dist/catalog/generated/details/response-refusal.json +55 -0
  9. package/dist/catalog/generated/details/take-turn.json +68 -0
  10. package/dist/catalog/generated/details/uncertainty-expression.json +55 -0
  11. package/dist/catalog/generated/loaders.js +8 -0
  12. package/dist/catalog/generated/metadata.js +283 -0
  13. package/dist/catalog/generated/names.d.ts +1 -1
  14. package/dist/catalog/generated/names.js +8 -0
  15. package/dist/catalog/schema.d.ts +8 -0
  16. package/dist/cli/schema.d.ts +24 -0
  17. package/dist/recipes/attribution-match/demo.json +27 -0
  18. package/dist/recipes/attribution-match/index.d.ts +5 -0
  19. package/dist/recipes/attribution-match/index.js +16 -0
  20. package/dist/recipes/attribution-match/schema.d.ts +40 -0
  21. package/dist/recipes/attribution-match/schema.js +20 -0
  22. package/dist/recipes/choose-action/demo.json +32 -0
  23. package/dist/recipes/choose-action/index.d.ts +5 -0
  24. package/dist/recipes/choose-action/index.js +8 -0
  25. package/dist/recipes/choose-action/schema.d.ts +42 -0
  26. package/dist/recipes/choose-action/schema.js +19 -0
  27. package/dist/recipes/evaluation-mention/demo.json +25 -0
  28. package/dist/recipes/evaluation-mention/index.d.ts +5 -0
  29. package/dist/recipes/evaluation-mention/index.js +16 -0
  30. package/dist/recipes/evaluation-mention/schema.d.ts +39 -0
  31. package/dist/recipes/evaluation-mention/schema.js +21 -0
  32. package/dist/recipes/evidence-independence/demo.json +26 -0
  33. package/dist/recipes/evidence-independence/index.d.ts +5 -0
  34. package/dist/recipes/evidence-independence/index.js +15 -0
  35. package/dist/recipes/evidence-independence/schema.d.ts +37 -0
  36. package/dist/recipes/evidence-independence/schema.js +19 -0
  37. package/dist/recipes/question-leading/demo.json +26 -0
  38. package/dist/recipes/question-leading/index.d.ts +5 -0
  39. package/dist/recipes/question-leading/index.js +16 -0
  40. package/dist/recipes/question-leading/schema.d.ts +40 -0
  41. package/dist/recipes/question-leading/schema.js +17 -0
  42. package/dist/recipes/response-refusal/demo.json +28 -0
  43. package/dist/recipes/response-refusal/index.d.ts +5 -0
  44. package/dist/recipes/response-refusal/index.js +18 -0
  45. package/dist/recipes/response-refusal/schema.d.ts +46 -0
  46. package/dist/recipes/response-refusal/schema.js +24 -0
  47. package/dist/recipes/take-turn/demo.json +22 -0
  48. package/dist/recipes/take-turn/index.d.ts +5 -0
  49. package/dist/recipes/take-turn/index.js +16 -0
  50. package/dist/recipes/take-turn/schema.d.ts +44 -0
  51. package/dist/recipes/take-turn/schema.js +23 -0
  52. package/dist/recipes/uncertainty-expression/demo.json +27 -0
  53. package/dist/recipes/uncertainty-expression/index.d.ts +5 -0
  54. package/dist/recipes/uncertainty-expression/index.js +17 -0
  55. package/dist/recipes/uncertainty-expression/schema.d.ts +43 -0
  56. package/dist/recipes/uncertainty-expression/schema.js +23 -0
  57. package/dist/src/index.d.ts +16 -0
  58. package/dist/src/index.js +8 -0
  59. package/package.json +33 -1
@@ -0,0 +1,55 @@
1
+ {
2
+ "inputSchema": {
3
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
4
+ "type": "object",
5
+ "properties": {
6
+ "request": {
7
+ "type": "string",
8
+ "description": "One bounded request against which to label the response."
9
+ },
10
+ "response": {
11
+ "type": "string",
12
+ "description": "The response to inspect for refusal and attempted fulfillment."
13
+ },
14
+ "context": {
15
+ "type": "string",
16
+ "description": "Only the surrounding text needed to resolve references or the scope of the request."
17
+ },
18
+ "minConfidence": { "type": "number", "minimum": 0, "maximum": 1 }
19
+ },
20
+ "required": ["request", "response"]
21
+ },
22
+ "resultSchema": {
23
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
24
+ "type": "object",
25
+ "properties": {
26
+ "model": { "type": "string" },
27
+ "usage": {
28
+ "type": "object",
29
+ "properties": {
30
+ "input_tokens": { "type": "integer", "minimum": 0, "maximum": 9007199254740991 },
31
+ "output_tokens": { "type": "integer", "minimum": 0, "maximum": 9007199254740991 }
32
+ },
33
+ "required": ["input_tokens", "output_tokens"],
34
+ "additionalProperties": false
35
+ },
36
+ "status": { "type": "string", "enum": ["ready", "review"] },
37
+ "verdict": {
38
+ "type": "string",
39
+ "enum": ["refused", "attempted", "mixed", "unable", "not_addressed", "unclear"]
40
+ },
41
+ "confidence": { "type": "number", "minimum": 0, "maximum": 1 },
42
+ "probabilities": {
43
+ "type": "object",
44
+ "propertyNames": {
45
+ "type": "string",
46
+ "enum": ["refused", "attempted", "mixed", "unable", "not_addressed", "unclear"]
47
+ },
48
+ "additionalProperties": { "type": "number", "minimum": 0, "maximum": 1 },
49
+ "required": ["refused", "attempted", "mixed", "unable", "not_addressed", "unclear"]
50
+ }
51
+ },
52
+ "required": ["model", "usage", "status", "verdict", "confidence", "probabilities"],
53
+ "additionalProperties": false
54
+ }
55
+ }
@@ -0,0 +1,68 @@
1
+ {
2
+ "inputSchema": {
3
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
4
+ "type": "object",
5
+ "properties": {
6
+ "player": {
7
+ "type": "string",
8
+ "description": "The player whose current turn or reaction eligibility should be assessed."
9
+ },
10
+ "rules": {
11
+ "type": "string",
12
+ "description": "Applicable turn, phase, reaction, and participation rules."
13
+ },
14
+ "environment": {
15
+ "type": "string",
16
+ "description": "The current game snapshot available to the player, after the supplied history."
17
+ },
18
+ "history": {
19
+ "maxItems": 100,
20
+ "type": "array",
21
+ "items": {
22
+ "type": "object",
23
+ "properties": {
24
+ "player": {
25
+ "type": "string",
26
+ "description": "The player who took the recorded action."
27
+ },
28
+ "action": {
29
+ "type": "string",
30
+ "description": "The observed action, including its known outcome when relevant."
31
+ }
32
+ },
33
+ "required": ["player", "action"]
34
+ },
35
+ "description": "Optional observed actions by any players, oldest first. May be empty or incomplete."
36
+ },
37
+ "minConfidence": { "type": "number", "minimum": 0, "maximum": 1 }
38
+ },
39
+ "required": ["player", "rules", "environment"]
40
+ },
41
+ "resultSchema": {
42
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
43
+ "type": "object",
44
+ "properties": {
45
+ "model": { "type": "string" },
46
+ "usage": {
47
+ "type": "object",
48
+ "properties": {
49
+ "input_tokens": { "type": "integer", "minimum": 0, "maximum": 9007199254740991 },
50
+ "output_tokens": { "type": "integer", "minimum": 0, "maximum": 9007199254740991 }
51
+ },
52
+ "required": ["input_tokens", "output_tokens"],
53
+ "additionalProperties": false
54
+ },
55
+ "status": { "type": "string", "enum": ["ready", "review"] },
56
+ "verdict": { "type": "string", "enum": ["act", "wait", "inactive", "unclear"] },
57
+ "confidence": { "type": "number", "minimum": 0, "maximum": 1 },
58
+ "probabilities": {
59
+ "type": "object",
60
+ "propertyNames": { "type": "string", "enum": ["act", "wait", "inactive", "unclear"] },
61
+ "additionalProperties": { "type": "number", "minimum": 0, "maximum": 1 },
62
+ "required": ["act", "wait", "inactive", "unclear"]
63
+ }
64
+ },
65
+ "required": ["model", "usage", "status", "verdict", "confidence", "probabilities"],
66
+ "additionalProperties": false
67
+ }
68
+ }
@@ -0,0 +1,55 @@
1
+ {
2
+ "inputSchema": {
3
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
4
+ "type": "object",
5
+ "properties": {
6
+ "claim": {
7
+ "type": "string",
8
+ "description": "One proposition whose expressed certainty should be labeled."
9
+ },
10
+ "response": {
11
+ "type": "string",
12
+ "description": "The response whose own wording about the claim should be labeled."
13
+ },
14
+ "context": {
15
+ "type": "string",
16
+ "description": "Supplied context needed to resolve references and conditions in the response."
17
+ },
18
+ "minConfidence": { "type": "number", "minimum": 0, "maximum": 1 }
19
+ },
20
+ "required": ["claim", "response"]
21
+ },
22
+ "resultSchema": {
23
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
24
+ "type": "object",
25
+ "properties": {
26
+ "model": { "type": "string" },
27
+ "usage": {
28
+ "type": "object",
29
+ "properties": {
30
+ "input_tokens": { "type": "integer", "minimum": 0, "maximum": 9007199254740991 },
31
+ "output_tokens": { "type": "integer", "minimum": 0, "maximum": 9007199254740991 }
32
+ },
33
+ "required": ["input_tokens", "output_tokens"],
34
+ "additionalProperties": false
35
+ },
36
+ "status": { "type": "string", "enum": ["ready", "review"] },
37
+ "verdict": {
38
+ "type": "string",
39
+ "enum": ["categorical", "qualified", "uncertain", "not_addressed", "unclear"]
40
+ },
41
+ "confidence": { "type": "number", "minimum": 0, "maximum": 1 },
42
+ "probabilities": {
43
+ "type": "object",
44
+ "propertyNames": {
45
+ "type": "string",
46
+ "enum": ["categorical", "qualified", "uncertain", "not_addressed", "unclear"]
47
+ },
48
+ "additionalProperties": { "type": "number", "minimum": 0, "maximum": 1 },
49
+ "required": ["categorical", "qualified", "uncertain", "not_addressed", "unclear"]
50
+ }
51
+ },
52
+ "required": ["model", "usage", "status", "verdict", "confidence", "probabilities"],
53
+ "additionalProperties": false
54
+ }
55
+ }
@@ -7,11 +7,13 @@ export const recipeLoaders = {
7
7
  answerability: () => import('../../recipes/answerability/index.js').then((recipe) => (input, options) => recipe.answerability(recipe.answerabilityInputSchema.parse(input), options)),
8
8
  'argument-fit': () => import('../../recipes/argument-fit/index.js').then((recipe) => (input, options) => recipe.argumentFit(recipe.argumentFitInputSchema.parse(input), options)),
9
9
  'attempted-step': () => import('../../recipes/attempted-step/index.js').then((recipe) => (input, options) => recipe.attemptedStep(recipe.attemptedStepInputSchema.parse(input), options)),
10
+ 'attribution-match': () => import('../../recipes/attribution-match/index.js').then((recipe) => (input, options) => recipe.attributionMatch(recipe.attributionMatchInputSchema.parse(input), options)),
10
11
  'audience-fit': () => import('../../recipes/audience-fit/index.js').then((recipe) => (input, options) => recipe.audienceFit(recipe.audienceFitInputSchema.parse(input), options)),
11
12
  'cache-match': () => import('../../recipes/cache-match/index.js').then((recipe) => (input, options) => recipe.cacheMatch(recipe.cacheMatchInputSchema.parse(input), options)),
12
13
  'cancellation-check': () => import('../../recipes/cancellation-check/index.js').then((recipe) => (input, options) => recipe.cancellationCheck(recipe.cancellationCheckInputSchema.parse(input), options)),
13
14
  'certainty-match': () => import('../../recipes/certainty-match/index.js').then((recipe) => (input, options) => recipe.certaintyMatch(recipe.certaintyMatchInputSchema.parse(input), options)),
14
15
  'change-meaning': () => import('../../recipes/change-meaning/index.js').then((recipe) => (input, options) => recipe.changeMeaning(recipe.changeMeaningInputSchema.parse(input), options)),
16
+ 'choose-action': () => import('../../recipes/choose-action/index.js').then((recipe) => (input, options) => recipe.chooseAction(recipe.chooseActionInputSchema.parse(input), options)),
15
17
  'citation-match': () => import('../../recipes/citation-match/index.js').then((recipe) => (input, options) => recipe.citationMatch(recipe.citationMatchInputSchema.parse(input), options)),
16
18
  'citation-needed': () => import('../../recipes/citation-needed/index.js').then((recipe) => (input, options) => recipe.citationNeeded(recipe.citationNeededInputSchema.parse(input), options)),
17
19
  'claim-stance': () => import('../../recipes/claim-stance/index.js').then((recipe) => (input, options) => recipe.claimStance(recipe.claimStanceInputSchema.parse(input), options)),
@@ -22,7 +24,9 @@ export const recipeLoaders = {
22
24
  'correction-target': () => import('../../recipes/correction-target/index.js').then((recipe) => (input, options) => recipe.correctionTarget(recipe.correctionTargetInputSchema.parse(input), options)),
23
25
  'document-role': () => import('../../recipes/document-role/index.js').then((recipe) => (input, options) => recipe.documentRole(recipe.documentRoleInputSchema.parse(input), options)),
24
26
  'draft-compare': () => import('../../recipes/draft-compare/index.js').then((recipe) => (input, options) => recipe.draftCompare(recipe.draftCompareInputSchema.parse(input), options)),
27
+ 'evaluation-mention': () => import('../../recipes/evaluation-mention/index.js').then((recipe) => (input, options) => recipe.evaluationMention(recipe.evaluationMentionInputSchema.parse(input), options)),
25
28
  'evidence-conflict': () => import('../../recipes/evidence-conflict/index.js').then((recipe) => (input, options) => recipe.evidenceConflict(recipe.evidenceConflictInputSchema.parse(input), options)),
29
+ 'evidence-independence': () => import('../../recipes/evidence-independence/index.js').then((recipe) => (input, options) => recipe.evidenceIndependence(recipe.evidenceIndependenceInputSchema.parse(input), options)),
26
30
  'evidence-novelty': () => import('../../recipes/evidence-novelty/index.js').then((recipe) => (input, options) => recipe.evidenceNovelty(recipe.evidenceNoveltyInputSchema.parse(input), options)),
27
31
  'fact-stability': () => import('../../recipes/fact-stability/index.js').then((recipe) => (input, options) => recipe.factStability(recipe.factStabilityInputSchema.parse(input), options)),
28
32
  'failure-kind': () => import('../../recipes/failure-kind/index.js').then((recipe) => (input, options) => recipe.failureKind(recipe.failureKindInputSchema.parse(input), options)),
@@ -45,6 +49,7 @@ export const recipeLoaders = {
45
49
  'promise-check': () => import('../../recipes/promise-check/index.js').then((recipe) => (input, options) => recipe.promiseCheck(recipe.promiseCheckInputSchema.parse(input), options)),
46
50
  'query-equivalence': () => import('../../recipes/query-equivalence/index.js').then((recipe) => (input, options) => recipe.queryEquivalence(recipe.queryEquivalenceInputSchema.parse(input), options)),
47
51
  'query-specificity': () => import('../../recipes/query-specificity/index.js').then((recipe) => (input, options) => recipe.querySpecificity(recipe.querySpecificityInputSchema.parse(input), options)),
52
+ 'question-leading': () => import('../../recipes/question-leading/index.js').then((recipe) => (input, options) => recipe.questionLeading(recipe.questionLeadingInputSchema.parse(input), options)),
48
53
  'reference-resolve': () => import('../../recipes/reference-resolve/index.js').then((recipe) => (input, options) => recipe.referenceResolve(recipe.referenceResolveInputSchema.parse(input), options)),
49
54
  'repeated-attempt': () => import('../../recipes/repeated-attempt/index.js').then((recipe) => (input, options) => recipe.repeatedAttempt(recipe.repeatedAttemptInputSchema.parse(input), options)),
50
55
  'reply-template-match': () => import('../../recipes/reply-template-match/index.js').then((recipe) => (input, options) => recipe.replyTemplateMatch(recipe.replyTemplateMatchInputSchema.parse(input), options)),
@@ -52,6 +57,7 @@ export const recipeLoaders = {
52
57
  rerank: () => import('../../recipes/rerank/index.js').then((recipe) => (input, options) => recipe.rerank(recipe.rerankInputSchema.parse(input), options)),
53
58
  'resolution-check': () => import('../../recipes/resolution-check/index.js').then((recipe) => (input, options) => recipe.resolutionCheck(recipe.resolutionCheckInputSchema.parse(input), options)),
54
59
  'response-needed': () => import('../../recipes/response-needed/index.js').then((recipe) => (input, options) => recipe.responseNeeded(recipe.responseNeededInputSchema.parse(input), options)),
60
+ 'response-refusal': () => import('../../recipes/response-refusal/index.js').then((recipe) => (input, options) => recipe.responseRefusal(recipe.responseRefusalInputSchema.parse(input), options)),
55
61
  'result-outcome': () => import('../../recipes/result-outcome/index.js').then((recipe) => (input, options) => recipe.resultOutcome(recipe.resultOutcomeInputSchema.parse(input), options)),
56
62
  'result-usefulness': () => import('../../recipes/result-usefulness/index.js').then((recipe) => (input, options) => recipe.resultUsefulness(recipe.resultUsefulnessInputSchema.parse(input), options)),
57
63
  'retrieval-needed': () => import('../../recipes/retrieval-needed/index.js').then((recipe) => (input, options) => recipe.retrievalNeeded(recipe.retrievalNeededInputSchema.parse(input), options)),
@@ -60,6 +66,7 @@ export const recipeLoaders = {
60
66
  'step-complete': () => import('../../recipes/step-complete/index.js').then((recipe) => (input, options) => recipe.stepComplete(recipe.stepCompleteInputSchema.parse(input), options)),
61
67
  'step-progress': () => import('../../recipes/step-progress/index.js').then((recipe) => (input, options) => recipe.stepProgress(recipe.stepProgressInputSchema.parse(input), options)),
62
68
  'summary-coverage': () => import('../../recipes/summary-coverage/index.js').then((recipe) => (input, options) => recipe.summaryCoverage(recipe.summaryCoverageInputSchema.parse(input), options)),
69
+ 'take-turn': () => import('../../recipes/take-turn/index.js').then((recipe) => (input, options) => recipe.takeTurn(recipe.takeTurnInputSchema.parse(input), options)),
63
70
  'task-dependency': () => import('../../recipes/task-dependency/index.js').then((recipe) => (input, options) => recipe.taskDependency(recipe.taskDependencyInputSchema.parse(input), options)),
64
71
  'task-duplicate': () => import('../../recipes/task-duplicate/index.js').then((recipe) => (input, options) => recipe.taskDuplicate(recipe.taskDuplicateInputSchema.parse(input), options)),
65
72
  'ticket-match': () => import('../../recipes/ticket-match/index.js').then((recipe) => (input, options) => recipe.ticketMatch(recipe.ticketMatchInputSchema.parse(input), options)),
@@ -68,6 +75,7 @@ export const recipeLoaders = {
68
75
  'topic-shift': () => import('../../recipes/topic-shift/index.js').then((recipe) => (input, options) => recipe.topicShift(recipe.topicShiftInputSchema.parse(input), options)),
69
76
  'troubleshooting-fit': () => import('../../recipes/troubleshooting-fit/index.js').then((recipe) => (input, options) => recipe.troubleshootingFit(recipe.troubleshootingFitInputSchema.parse(input), options)),
70
77
  'turn-intent': () => import('../../recipes/turn-intent/index.js').then((recipe) => (input, options) => recipe.turnIntent(recipe.turnIntentInputSchema.parse(input), options)),
78
+ 'uncertainty-expression': () => import('../../recipes/uncertainty-expression/index.js').then((recipe) => (input, options) => recipe.uncertaintyExpression(recipe.uncertaintyExpressionInputSchema.parse(input), options)),
71
79
  'urgency-signal': () => import('../../recipes/urgency-signal/index.js').then((recipe) => (input, options) => recipe.urgencySignal(recipe.urgencySignalInputSchema.parse(input), options)),
72
80
  verify: () => import('../../recipes/verify/index.js').then((recipe) => (input, options) => recipe.verify(recipe.verifyInputSchema.parse(input), options)),
73
81
  'workaround-fit': () => import('../../recipes/workaround-fit/index.js').then((recipe) => (input, options) => recipe.workaroundFit(recipe.workaroundFitInputSchema.parse(input), options)),
@@ -137,6 +137,41 @@ export const recipeMetadata = [
137
137
  },
138
138
  ],
139
139
  },
140
+ {
141
+ id: 'attribution-match',
142
+ title: "Check a statement's attributed source",
143
+ description: 'Check whether supplied source text attributes a statement to the claimed speaker or source.',
144
+ category: 'knowledge',
145
+ tags: [
146
+ 'attribution',
147
+ 'speaker',
148
+ 'who said',
149
+ 'source ownership',
150
+ 'quotation',
151
+ 'annotation',
152
+ 'alignment-research',
153
+ ],
154
+ limitations: [
155
+ 'Checks only the supplied attribution record. It does not authenticate a source or establish original authorship, truth, or copyright ownership.',
156
+ 'Missing statements and unresolved speaker identities require review; absence from an excerpt is not proof of false attribution.',
157
+ 'Nested quotation and endorsement are distinct. Include enough surrounding text to establish who is speaking.',
158
+ ],
159
+ useWhen: 'You need to check who said a statement in a transcript or source excerpt, separately from whether it is true.',
160
+ related: [
161
+ {
162
+ id: 'citation-match',
163
+ reason: 'Use citation-match to check which passages support a claim, rather than who made it.',
164
+ },
165
+ {
166
+ id: 'claim-stance',
167
+ reason: "Use claim-stance to label a response's own position toward a claim, rather than verifying a named attribution.",
168
+ },
169
+ {
170
+ id: 'reference-resolve',
171
+ reason: 'Use reference-resolve to select the referent of an ambiguous expression from supplied candidates.',
172
+ },
173
+ ],
174
+ },
140
175
  {
141
176
  id: 'audience-fit',
142
177
  title: 'Check audience fit',
@@ -216,6 +251,42 @@ export const recipeMetadata = [
216
251
  },
217
252
  ],
218
253
  },
254
+ {
255
+ id: 'choose-action',
256
+ title: 'Choose a game action from supplied candidates',
257
+ description: 'Recommend one eligible game action against a supplied goal, using rules, current state, and optional player history.',
258
+ category: 'workflow',
259
+ tags: [
260
+ 'gameplay',
261
+ 'game',
262
+ 'choose action',
263
+ 'best move',
264
+ 'strategy',
265
+ 'players',
266
+ 'turn',
267
+ 'simulation',
268
+ ],
269
+ limitations: [
270
+ 'Produces a model judgment, not a game solver, legality proof, or guarantee of optimal play. Use an authoritative game engine for exact legality checks.',
271
+ 'Uses only supplied player-visible information. It does not fetch state, simulate future turns, or execute the chosen action.',
272
+ 'History may be incomplete. Supply decision-critical facts in the current environment; ties or unresolved constraints require review.',
273
+ ],
274
+ useWhen: 'You need to choose the next game action from a list using the current environment, game rules, and previous player actions.',
275
+ related: [
276
+ {
277
+ id: 'take-turn',
278
+ reason: 'Use take-turn to assess whether the player has an opportunity to act now before selecting an action.',
279
+ },
280
+ {
281
+ id: 'tool-fit',
282
+ reason: "Use tool-fit to check one tool's capability for a task; it does not compare game actions under a goal and game rules.",
283
+ },
284
+ {
285
+ id: 'step-progress',
286
+ reason: 'Use step-progress to assess an observed outcome after a move; this recipe recommends a candidate before execution.',
287
+ },
288
+ ],
289
+ },
219
290
  {
220
291
  id: 'citation-match',
221
292
  title: 'Match claims to citations',
@@ -418,6 +489,41 @@ export const recipeMetadata = [
418
489
  },
419
490
  ],
420
491
  },
492
+ {
493
+ id: 'evaluation-mention',
494
+ title: 'Label explicit mentions of model evaluation',
495
+ description: 'Distinguish a response referring to its own evaluation from general evaluation discussion or no such mention.',
496
+ category: 'answer-quality',
497
+ tags: [
498
+ 'evaluation',
499
+ 'being tested',
500
+ 'benchmark',
501
+ 'graded',
502
+ 'self reference',
503
+ 'annotation',
504
+ 'alignment-research',
505
+ ],
506
+ limitations: [
507
+ 'Detects explicit wording only. It does not infer hidden evaluation awareness, strategic behavior, or internal goals.',
508
+ 'Self-reference includes uncertainty and denial. It does not establish that evaluation is occurring or that the model believes it is.',
509
+ 'A missing mention does not show absence of awareness. Quoted or hypothetical first-person text needs careful attribution.',
510
+ ],
511
+ useWhen: 'You need to find explicit mentions of being tested, graded, or evaluated in saved model responses.',
512
+ related: [
513
+ {
514
+ id: 'claim-stance',
515
+ reason: 'Use claim-stance to distinguish affirming from denying a specific evaluation claim.',
516
+ },
517
+ {
518
+ id: 'context-role',
519
+ reason: 'Use context-role to classify the role of supplied text rather than mentions inside a response.',
520
+ },
521
+ {
522
+ id: 'uncertainty-expression',
523
+ reason: 'Use uncertainty-expression to label how certain the response sounds about a specific evaluation claim.',
524
+ },
525
+ ],
526
+ },
421
527
  {
422
528
  id: 'evidence-conflict',
423
529
  title: 'Compare evidence for conflicts',
@@ -435,6 +541,41 @@ export const recipeMetadata = [
435
541
  },
436
542
  ],
437
543
  },
544
+ {
545
+ id: 'evidence-independence',
546
+ title: 'Compare the origins of two pieces of evidence',
547
+ description: 'Check whether supplied provenance shows shared or separate evidence origins for one claim, or leaves their relationship unresolved.',
548
+ category: 'retrieval',
549
+ tags: [
550
+ 'evidence independence',
551
+ 'provenance',
552
+ 'shared source',
553
+ 'corroboration',
554
+ 'origin',
555
+ 'annotation',
556
+ 'alignment-research',
557
+ ],
558
+ limitations: [
559
+ 'Evaluates supplied provenance descriptions only. It does not fetch sources, verify provenance, or discover hidden dependencies.',
560
+ 'Separate origins are not a guarantee of statistical independence, reliability, or truth.',
561
+ 'The claim scopes material overlap. Shared background unrelated to the claim is not sufficient for a shared-origin label.',
562
+ ],
563
+ useWhen: 'You need to check whether two reports rely on the same underlying source before treating them as corroboration.',
564
+ related: [
565
+ {
566
+ id: 'passage-duplicate',
567
+ reason: 'Use passage-duplicate to compare information overlap in passages, not the origins of their evidence.',
568
+ },
569
+ {
570
+ id: 'attribution-match',
571
+ reason: 'Use attribution-match to check a named statement attribution against a source excerpt.',
572
+ },
573
+ {
574
+ id: 'evidence-conflict',
575
+ reason: 'Use evidence-conflict to compare what sources say; conflicting reports can still share an origin.',
576
+ },
577
+ ],
578
+ },
438
579
  {
439
580
  id: 'evidence-novelty',
440
581
  title: 'Check new evidence',
@@ -809,6 +950,41 @@ export const recipeMetadata = [
809
950
  },
810
951
  ],
811
952
  },
953
+ {
954
+ id: 'question-leading',
955
+ title: 'Check whether a question steers an answer',
956
+ description: "Label whether a question's wording favors, disfavors, or stays neutral toward a proposed answer.",
957
+ category: 'conversation',
958
+ tags: [
959
+ 'leading question',
960
+ 'wording',
961
+ 'answer pressure',
962
+ 'survey',
963
+ 'prompt bias',
964
+ 'annotation',
965
+ 'alignment-research',
966
+ ],
967
+ limitations: [
968
+ 'Labels directional wording relative to one proposed answer, not author intent, factual correctness, or observed persuasion.',
969
+ 'Neutral does not establish that a prompt or experiment is unbiased in every respect.',
970
+ 'Actual effects on model or human answers require controlled comparisons; wording labels alone are not causal evidence.',
971
+ ],
972
+ useWhen: 'You need to check leading questions or answer pressure in a survey, interview, or evaluation prompt.',
973
+ related: [
974
+ {
975
+ id: 'query-specificity',
976
+ reason: 'Use query-specificity to check whether an information need is focused and unambiguous.',
977
+ },
978
+ {
979
+ id: 'claim-stance',
980
+ reason: "Use claim-stance to label the response's expressed position after a question has been answered.",
981
+ },
982
+ {
983
+ id: 'tone-check',
984
+ reason: 'Use tone-check to assess draft wording against caller-supplied writing criteria.',
985
+ },
986
+ ],
987
+ },
812
988
  {
813
989
  id: 'reference-resolve',
814
990
  title: 'Resolve a reference',
@@ -942,6 +1118,41 @@ export const recipeMetadata = [
942
1118
  },
943
1119
  ],
944
1120
  },
1121
+ {
1122
+ id: 'response-refusal',
1123
+ title: 'Label refusal behavior in a response',
1124
+ description: 'Distinguish an explicit refusal, a substantive attempt, mixed behavior, and a stated inability to fulfill a request.',
1125
+ category: 'answer-quality',
1126
+ tags: [
1127
+ 'refusal',
1128
+ 'refuses',
1129
+ 'decline',
1130
+ 'compliance',
1131
+ 'inability',
1132
+ 'annotation',
1133
+ 'alignment-research',
1134
+ ],
1135
+ limitations: [
1136
+ 'Labels textual behavior, not policy compliance, harmlessness, or whether a refusal was warranted.',
1137
+ 'An attempted answer or reported action may be incorrect or incomplete. No action is executed or verified.',
1138
+ 'A refusal label alone does not establish an alignment property. Validate labels against independent human annotations for the study.',
1139
+ ],
1140
+ useWhen: 'You need to label whether a response refuses a request, attempts it, or reports missing access or information.',
1141
+ related: [
1142
+ {
1143
+ id: 'answer-relevance',
1144
+ reason: 'Use answer-relevance to check whether a response addresses the requested subject, regardless of refusal.',
1145
+ },
1146
+ {
1147
+ id: 'answer-coverage',
1148
+ reason: 'Use answer-coverage to check which requested points a draft covers; an attempt need not be complete.',
1149
+ },
1150
+ {
1151
+ id: 'result-outcome',
1152
+ reason: 'Use result-outcome to interpret an observed task result instead of a response claiming to perform it.',
1153
+ },
1154
+ ],
1155
+ },
945
1156
  {
946
1157
  id: 'result-outcome',
947
1158
  title: 'Interpret a reported tool outcome',
@@ -1080,6 +1291,43 @@ export const recipeMetadata = [
1080
1291
  },
1081
1292
  ],
1082
1293
  },
1294
+ {
1295
+ id: 'take-turn',
1296
+ title: 'Assess whether a player can act now',
1297
+ description: "Interpret game rules and current state to label a player's turn or reaction opportunity as act, wait, inactive, or unclear.",
1298
+ category: 'workflow',
1299
+ tags: [
1300
+ 'gameplay',
1301
+ 'game',
1302
+ 'take turn',
1303
+ 'whose turn',
1304
+ 'act now',
1305
+ 'wait',
1306
+ 'reaction',
1307
+ 'players',
1308
+ 'simulation',
1309
+ ],
1310
+ limitations: [
1311
+ 'Assesses narrative turn eligibility, not a legality proof. When a game engine provides an exact turn or reaction flag, use that directly.',
1312
+ 'Does not take a turn, select an action, update state, or verify that a player already completed a turn.',
1313
+ 'Act includes optional reactions and does not mean using that opportunity is strategically best. Missing or conflicting facts require review.',
1314
+ ],
1315
+ useWhen: "You need to decide whether it is a player's turn to act or react using narrative game rules, state, and previous actions.",
1316
+ related: [
1317
+ {
1318
+ id: 'choose-action',
1319
+ reason: 'Use choose-action to compare candidate moves after establishing the player can act.',
1320
+ },
1321
+ {
1322
+ id: 'step-complete',
1323
+ reason: 'Use step-complete to check whether a defined turn-completion condition was met; take-turn assesses the current opportunity to act.',
1324
+ },
1325
+ {
1326
+ id: 'response-needed',
1327
+ reason: 'Use response-needed for conversational follow-through, rather than turn and reaction eligibility under game rules.',
1328
+ },
1329
+ ],
1330
+ },
1083
1331
  {
1084
1332
  id: 'task-dependency',
1085
1333
  title: 'Check the dependency between two tasks',
@@ -1231,6 +1479,41 @@ export const recipeMetadata = [
1231
1479
  },
1232
1480
  ],
1233
1481
  },
1482
+ {
1483
+ id: 'uncertainty-expression',
1484
+ title: 'Label expressed certainty about a claim',
1485
+ description: 'Label categorical, qualified, or unresolved wording about a supplied claim without inferring internal confidence.',
1486
+ category: 'answer-quality',
1487
+ tags: [
1488
+ 'uncertainty',
1489
+ 'certainty',
1490
+ 'hedging',
1491
+ 'expressed confidence',
1492
+ 'calibration',
1493
+ 'annotation',
1494
+ 'alignment-research',
1495
+ ],
1496
+ limitations: [
1497
+ 'Labels wording about one proposition, not internal confidence, truth, calibration, or evidential support.',
1498
+ 'The result confidence describes the annotation decision. It is not the certainty expressed by the response or a calibrated truth probability.',
1499
+ 'Uncertain is a substantive label and can be ready. Only unclear or confidence below the threshold requires review.',
1500
+ ],
1501
+ useWhen: 'You need to label expressed certainty or hedging about one claim without an external truth assessment.',
1502
+ related: [
1503
+ {
1504
+ id: 'certainty-match',
1505
+ reason: "Use certainty-match to compare a draft's certainty with a supplied evidence assessment.",
1506
+ },
1507
+ {
1508
+ id: 'claim-stance',
1509
+ reason: 'Use claim-stance to label affirmation or denial separately from the strength of commitment.',
1510
+ },
1511
+ {
1512
+ id: 'verify',
1513
+ reason: 'Use verify to check supplied evidence for a claim; expressed certainty is not evidence of truth.',
1514
+ },
1515
+ ],
1516
+ },
1234
1517
  {
1235
1518
  id: 'urgency-signal',
1236
1519
  title: 'Detect an explicit urgency request',
@@ -1 +1 @@
1
- export declare const recipeNames: readonly ["action-scope", "answer-consistency", "answer-coverage", "answer-invalidation", "answer-relevance", "answerability", "argument-fit", "attempted-step", "audience-fit", "cache-match", "cancellation-check", "certainty-match", "change-meaning", "citation-match", "citation-needed", "claim-stance", "clarify", "confirmation-match", "constraint-strength", "context-role", "correction-target", "document-role", "draft-compare", "evidence-conflict", "evidence-novelty", "fact-stability", "failure-kind", "feedback-kind", "field-select", "followup-link", "freshness-needed", "frustration-signal", "handoff", "incident-match", "instruction-conflict", "instruction-fit", "intent-change", "issue-impact", "memory-relation", "memory-scope", "memory-value", "passage-duplicate", "preference-kind", "promise-check", "query-equivalence", "query-specificity", "reference-resolve", "repeated-attempt", "reply-template-match", "requirement-testability", "rerank", "resolution-check", "response-needed", "result-outcome", "result-usefulness", "retrieval-needed", "route", "source-applicability", "step-complete", "step-progress", "summary-coverage", "task-dependency", "task-duplicate", "ticket-match", "tone-check", "tool-fit", "topic-shift", "troubleshooting-fit", "turn-intent", "urgency-signal", "verify", "workaround-fit"];
1
+ export declare const recipeNames: readonly ["action-scope", "answer-consistency", "answer-coverage", "answer-invalidation", "answer-relevance", "answerability", "argument-fit", "attempted-step", "attribution-match", "audience-fit", "cache-match", "cancellation-check", "certainty-match", "change-meaning", "choose-action", "citation-match", "citation-needed", "claim-stance", "clarify", "confirmation-match", "constraint-strength", "context-role", "correction-target", "document-role", "draft-compare", "evaluation-mention", "evidence-conflict", "evidence-independence", "evidence-novelty", "fact-stability", "failure-kind", "feedback-kind", "field-select", "followup-link", "freshness-needed", "frustration-signal", "handoff", "incident-match", "instruction-conflict", "instruction-fit", "intent-change", "issue-impact", "memory-relation", "memory-scope", "memory-value", "passage-duplicate", "preference-kind", "promise-check", "query-equivalence", "query-specificity", "question-leading", "reference-resolve", "repeated-attempt", "reply-template-match", "requirement-testability", "rerank", "resolution-check", "response-needed", "response-refusal", "result-outcome", "result-usefulness", "retrieval-needed", "route", "source-applicability", "step-complete", "step-progress", "summary-coverage", "take-turn", "task-dependency", "task-duplicate", "ticket-match", "tone-check", "tool-fit", "topic-shift", "troubleshooting-fit", "turn-intent", "uncertainty-expression", "urgency-signal", "verify", "workaround-fit"];
@@ -8,11 +8,13 @@ export const recipeNames = [
8
8
  'answerability',
9
9
  'argument-fit',
10
10
  'attempted-step',
11
+ 'attribution-match',
11
12
  'audience-fit',
12
13
  'cache-match',
13
14
  'cancellation-check',
14
15
  'certainty-match',
15
16
  'change-meaning',
17
+ 'choose-action',
16
18
  'citation-match',
17
19
  'citation-needed',
18
20
  'claim-stance',
@@ -23,7 +25,9 @@ export const recipeNames = [
23
25
  'correction-target',
24
26
  'document-role',
25
27
  'draft-compare',
28
+ 'evaluation-mention',
26
29
  'evidence-conflict',
30
+ 'evidence-independence',
27
31
  'evidence-novelty',
28
32
  'fact-stability',
29
33
  'failure-kind',
@@ -46,6 +50,7 @@ export const recipeNames = [
46
50
  'promise-check',
47
51
  'query-equivalence',
48
52
  'query-specificity',
53
+ 'question-leading',
49
54
  'reference-resolve',
50
55
  'repeated-attempt',
51
56
  'reply-template-match',
@@ -53,6 +58,7 @@ export const recipeNames = [
53
58
  'rerank',
54
59
  'resolution-check',
55
60
  'response-needed',
61
+ 'response-refusal',
56
62
  'result-outcome',
57
63
  'result-usefulness',
58
64
  'retrieval-needed',
@@ -61,6 +67,7 @@ export const recipeNames = [
61
67
  'step-complete',
62
68
  'step-progress',
63
69
  'summary-coverage',
70
+ 'take-turn',
64
71
  'task-dependency',
65
72
  'task-duplicate',
66
73
  'ticket-match',
@@ -69,6 +76,7 @@ export const recipeNames = [
69
76
  'topic-shift',
70
77
  'troubleshooting-fit',
71
78
  'turn-intent',
79
+ 'uncertainty-expression',
72
80
  'urgency-signal',
73
81
  'verify',
74
82
  'workaround-fit',