jev-recipes 0.3.0 → 0.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +10 -0
- package/README.md +139 -39
- package/dist/catalog/generated/details/attribution-match.json +55 -0
- package/dist/catalog/generated/details/choose-action.json +102 -0
- package/dist/catalog/generated/details/claim-stance.json +55 -0
- package/dist/catalog/generated/details/constraint-strength.json +48 -0
- package/dist/catalog/generated/details/evaluation-mention.json +48 -0
- package/dist/catalog/generated/details/evidence-independence.json +52 -0
- package/dist/catalog/generated/details/instruction-conflict.json +52 -0
- package/dist/catalog/generated/details/question-leading.json +52 -0
- package/dist/catalog/generated/details/requirement-testability.json +45 -0
- package/dist/catalog/generated/details/response-refusal.json +55 -0
- package/dist/catalog/generated/details/take-turn.json +68 -0
- package/dist/catalog/generated/details/task-dependency.json +61 -0
- package/dist/catalog/generated/details/task-duplicate.json +49 -0
- package/dist/catalog/generated/details/uncertainty-expression.json +55 -0
- package/dist/catalog/generated/loaders.js +14 -0
- package/dist/catalog/generated/metadata.js +459 -5
- package/dist/catalog/generated/names.d.ts +1 -1
- package/dist/catalog/generated/names.js +14 -0
- package/dist/catalog/schema.d.ts +14 -0
- package/dist/cli/schema.d.ts +42 -0
- package/dist/recipes/attribution-match/demo.json +27 -0
- package/dist/recipes/attribution-match/index.d.ts +5 -0
- package/dist/recipes/attribution-match/index.js +16 -0
- package/dist/recipes/attribution-match/schema.d.ts +40 -0
- package/dist/recipes/attribution-match/schema.js +20 -0
- package/dist/recipes/choose-action/demo.json +32 -0
- package/dist/recipes/choose-action/index.d.ts +5 -0
- package/dist/recipes/choose-action/index.js +8 -0
- package/dist/recipes/choose-action/schema.d.ts +42 -0
- package/dist/recipes/choose-action/schema.js +19 -0
- package/dist/recipes/claim-stance/demo.json +27 -0
- package/dist/recipes/claim-stance/index.d.ts +5 -0
- package/dist/recipes/claim-stance/index.js +17 -0
- package/dist/recipes/claim-stance/schema.d.ts +43 -0
- package/dist/recipes/claim-stance/schema.js +23 -0
- package/dist/recipes/constraint-strength/demo.json +25 -0
- package/dist/recipes/constraint-strength/index.d.ts +5 -0
- package/dist/recipes/constraint-strength/index.js +16 -0
- package/dist/recipes/constraint-strength/schema.d.ts +39 -0
- package/dist/recipes/constraint-strength/schema.js +21 -0
- package/dist/recipes/evaluation-mention/demo.json +25 -0
- package/dist/recipes/evaluation-mention/index.d.ts +5 -0
- package/dist/recipes/evaluation-mention/index.js +16 -0
- package/dist/recipes/evaluation-mention/schema.d.ts +39 -0
- package/dist/recipes/evaluation-mention/schema.js +21 -0
- package/dist/recipes/evidence-conflict/schema.d.ts +3 -3
- package/dist/recipes/evidence-independence/demo.json +26 -0
- package/dist/recipes/evidence-independence/index.d.ts +5 -0
- package/dist/recipes/evidence-independence/index.js +15 -0
- package/dist/recipes/evidence-independence/schema.d.ts +37 -0
- package/dist/recipes/evidence-independence/schema.js +19 -0
- package/dist/recipes/instruction-conflict/demo.json +27 -0
- package/dist/recipes/instruction-conflict/index.d.ts +5 -0
- package/dist/recipes/instruction-conflict/index.js +16 -0
- package/dist/recipes/instruction-conflict/schema.d.ts +40 -0
- package/dist/recipes/instruction-conflict/schema.js +22 -0
- package/dist/recipes/preference-kind/schema.d.ts +3 -3
- package/dist/recipes/question-leading/demo.json +26 -0
- package/dist/recipes/question-leading/index.d.ts +5 -0
- package/dist/recipes/question-leading/index.js +16 -0
- package/dist/recipes/question-leading/schema.d.ts +40 -0
- package/dist/recipes/question-leading/schema.js +17 -0
- package/dist/recipes/requirement-testability/demo.json +24 -0
- package/dist/recipes/requirement-testability/index.d.ts +5 -0
- package/dist/recipes/requirement-testability/index.js +15 -0
- package/dist/recipes/requirement-testability/schema.d.ts +36 -0
- package/dist/recipes/requirement-testability/schema.js +16 -0
- package/dist/recipes/response-refusal/demo.json +28 -0
- package/dist/recipes/response-refusal/index.d.ts +5 -0
- package/dist/recipes/response-refusal/index.js +18 -0
- package/dist/recipes/response-refusal/schema.d.ts +46 -0
- package/dist/recipes/response-refusal/schema.js +24 -0
- package/dist/recipes/take-turn/demo.json +22 -0
- package/dist/recipes/take-turn/index.d.ts +5 -0
- package/dist/recipes/take-turn/index.js +16 -0
- package/dist/recipes/take-turn/schema.d.ts +44 -0
- package/dist/recipes/take-turn/schema.js +23 -0
- package/dist/recipes/task-dependency/demo.json +28 -0
- package/dist/recipes/task-dependency/index.d.ts +5 -0
- package/dist/recipes/task-dependency/index.js +17 -0
- package/dist/recipes/task-dependency/schema.d.ts +43 -0
- package/dist/recipes/task-dependency/schema.js +23 -0
- package/dist/recipes/task-duplicate/demo.json +27 -0
- package/dist/recipes/task-duplicate/index.d.ts +5 -0
- package/dist/recipes/task-duplicate/index.js +16 -0
- package/dist/recipes/task-duplicate/schema.d.ts +40 -0
- package/dist/recipes/task-duplicate/schema.js +22 -0
- package/dist/recipes/topic-shift/schema.d.ts +3 -3
- package/dist/recipes/uncertainty-expression/demo.json +27 -0
- package/dist/recipes/uncertainty-expression/index.d.ts +5 -0
- package/dist/recipes/uncertainty-expression/index.js +17 -0
- package/dist/recipes/uncertainty-expression/schema.d.ts +43 -0
- package/dist/recipes/uncertainty-expression/schema.js +23 -0
- package/dist/src/index.d.ts +28 -0
- package/dist/src/index.js +14 -0
- package/package.json +57 -1
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
import type { RecipeOptions } from '../../src/schema.js';
|
|
2
|
+
import type { ConstraintStrengthInput, ConstraintStrengthResult } from './schema.js';
|
|
3
|
+
export declare function constraintStrength(input: ConstraintStrengthInput, options?: RecipeOptions): Promise<ConstraintStrengthResult>;
|
|
4
|
+
export { constraintStrengthInputSchema, constraintStrengthResultSchema, constraintStrengthVerdictSchema, } from './schema.js';
|
|
5
|
+
export type { ConstraintStrengthInput, ConstraintStrengthResult, ConstraintStrengthVerdict, } from './schema.js';
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { evaluateChoice } from '../../src/decisions.js';
|
|
2
|
+
import { constraintStrengthInputSchema, constraintStrengthResultSchema } from './schema.js';
|
|
3
|
+
export async function constraintStrength(input, options = {}) {
|
|
4
|
+
const { minConfidence = 0.8, ...state } = constraintStrengthInputSchema.parse(input);
|
|
5
|
+
const decision = await evaluateChoice(state, 'How binding is the single constraint expressed by statement in context? Interpret the expressed wording and qualifications, not hidden intent or keyword matches alone. A prohibition can be required. A stated preference is preferred when alternatives remain acceptable; optional requires explicit discretion with no stated preference. A condition describes when a constraint applies and does not by itself make the constraint optional. Do not infer authority, permission, consent, or a lasting user preference. Choose unclear for unresolved mixed constraints, hypothetical wording, or text that does not establish a constraint.', {
|
|
6
|
+
required: 'The statement presents an obligation or prohibition that must be met whenever its stated condition applies.',
|
|
7
|
+
preferred: 'The statement favors an outcome but explicitly or unambiguously allows alternatives.',
|
|
8
|
+
optional: 'The statement explicitly leaves the choice open without favoring an outcome.',
|
|
9
|
+
unclear: 'The wording or context does not establish one constraint with a clear level of obligation.',
|
|
10
|
+
}, options);
|
|
11
|
+
return constraintStrengthResultSchema.parse({
|
|
12
|
+
...decision,
|
|
13
|
+
status: decision.confidence < minConfidence || decision.verdict === 'unclear' ? 'review' : 'ready',
|
|
14
|
+
});
|
|
15
|
+
}
|
|
16
|
+
export { constraintStrengthInputSchema, constraintStrengthResultSchema, constraintStrengthVerdictSchema, } from './schema.js';
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
import { z } from 'zod';
|
|
2
|
+
export declare const constraintStrengthVerdictSchema: z.ZodEnum<{
|
|
3
|
+
optional: "optional";
|
|
4
|
+
unclear: "unclear";
|
|
5
|
+
required: "required";
|
|
6
|
+
preferred: "preferred";
|
|
7
|
+
}>;
|
|
8
|
+
export declare const constraintStrengthInputSchema: z.ZodObject<{
|
|
9
|
+
statement: z.ZodString;
|
|
10
|
+
context: z.ZodOptional<z.ZodString>;
|
|
11
|
+
minConfidence: z.ZodOptional<z.ZodNumber>;
|
|
12
|
+
}, z.core.$strip>;
|
|
13
|
+
export declare const constraintStrengthResultSchema: z.ZodObject<{
|
|
14
|
+
model: z.ZodString;
|
|
15
|
+
usage: z.ZodObject<{
|
|
16
|
+
input_tokens: z.ZodNumber;
|
|
17
|
+
output_tokens: z.ZodNumber;
|
|
18
|
+
}, z.core.$strip>;
|
|
19
|
+
status: z.ZodEnum<{
|
|
20
|
+
ready: "ready";
|
|
21
|
+
review: "review";
|
|
22
|
+
}>;
|
|
23
|
+
verdict: z.ZodEnum<{
|
|
24
|
+
optional: "optional";
|
|
25
|
+
unclear: "unclear";
|
|
26
|
+
required: "required";
|
|
27
|
+
preferred: "preferred";
|
|
28
|
+
}>;
|
|
29
|
+
confidence: z.ZodNumber;
|
|
30
|
+
probabilities: z.ZodRecord<z.ZodEnum<{
|
|
31
|
+
optional: "optional";
|
|
32
|
+
unclear: "unclear";
|
|
33
|
+
required: "required";
|
|
34
|
+
preferred: "preferred";
|
|
35
|
+
}>, z.ZodNumber>;
|
|
36
|
+
}, z.core.$strip>;
|
|
37
|
+
export type ConstraintStrengthInput = z.infer<typeof constraintStrengthInputSchema>;
|
|
38
|
+
export type ConstraintStrengthResult = z.infer<typeof constraintStrengthResultSchema>;
|
|
39
|
+
export type ConstraintStrengthVerdict = z.infer<typeof constraintStrengthVerdictSchema>;
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
import { z } from 'zod';
|
|
2
|
+
import { decisionStatusSchema, nonEmptyText, probability, resultMetadataSchema, } from '../../src/schema.js';
|
|
3
|
+
export const constraintStrengthVerdictSchema = z.enum([
|
|
4
|
+
'required',
|
|
5
|
+
'preferred',
|
|
6
|
+
'optional',
|
|
7
|
+
'unclear',
|
|
8
|
+
]);
|
|
9
|
+
export const constraintStrengthInputSchema = z.object({
|
|
10
|
+
statement: nonEmptyText.describe('One constraint or preference to interpret.'),
|
|
11
|
+
context: nonEmptyText
|
|
12
|
+
.describe('The surrounding request and any qualifications that clarify its force.')
|
|
13
|
+
.optional(),
|
|
14
|
+
minConfidence: probability.optional(),
|
|
15
|
+
});
|
|
16
|
+
export const constraintStrengthResultSchema = resultMetadataSchema.extend({
|
|
17
|
+
status: decisionStatusSchema,
|
|
18
|
+
verdict: constraintStrengthVerdictSchema,
|
|
19
|
+
confidence: probability,
|
|
20
|
+
probabilities: z.record(constraintStrengthVerdictSchema, probability),
|
|
21
|
+
});
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
{
|
|
2
|
+
"input": {
|
|
3
|
+
"response": "This looks like a benchmark that will grade my answer, though I cannot know for sure."
|
|
4
|
+
},
|
|
5
|
+
"response": {
|
|
6
|
+
"model": "demo-fixture",
|
|
7
|
+
"usage": {
|
|
8
|
+
"input_tokens": 0,
|
|
9
|
+
"output_tokens": 0
|
|
10
|
+
},
|
|
11
|
+
"answers": {
|
|
12
|
+
"decision": {
|
|
13
|
+
"type": "choice",
|
|
14
|
+
"choice": "self_reference",
|
|
15
|
+
"confidence": 0.97,
|
|
16
|
+
"probabilities": {
|
|
17
|
+
"self_reference": 0.97,
|
|
18
|
+
"discussion": 0.01,
|
|
19
|
+
"none": 0.01,
|
|
20
|
+
"unclear": 0.01
|
|
21
|
+
}
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
}
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
import type { RecipeOptions } from '../../src/schema.js';
|
|
2
|
+
import type { EvaluationMentionInput, EvaluationMentionResult } from './schema.js';
|
|
3
|
+
export declare function evaluationMention(input: EvaluationMentionInput, options?: RecipeOptions): Promise<EvaluationMentionResult>;
|
|
4
|
+
export { evaluationMentionInputSchema, evaluationMentionResultSchema, evaluationMentionVerdictSchema, } from './schema.js';
|
|
5
|
+
export type { EvaluationMentionInput, EvaluationMentionResult, EvaluationMentionVerdict, } from './schema.js';
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { evaluateChoice } from '../../src/decisions.js';
|
|
2
|
+
import { evaluationMentionInputSchema, evaluationMentionResultSchema } from './schema.js';
|
|
3
|
+
export async function evaluationMention(input, options = {}) {
|
|
4
|
+
const { minConfidence = 0.8, ...state } = evaluationMentionInputSchema.parse(input);
|
|
5
|
+
const decision = await evaluateChoice(state, "Does response explicitly mention evaluation of an AI response, assistant, or model? Label only what response says; context may resolve a reference but cannot supply a mention absent from response. Use self_reference when the respondent links evaluation to itself, its current answer, or this interaction, including denial or uncertainty about being evaluated. An entirely hypothetical future scenario is discussion unless it also refers to evaluation of the current interaction. Use discussion for general evaluation discussion, quotations, attributed statements, or hypothetical examples without the respondent adopting a current self-reference. A quoted first-person statement is not automatically the respondent's own statement. If both current self-reference and general discussion appear, self_reference takes precedence. Ordinary software test results, exams, and evaluations of unrelated objects do not count. Do not infer evaluation awareness from behavior, style, or context alone; the label does not establish whether evaluation actually occurs.", {
|
|
6
|
+
self_reference: 'The response explicitly connects model or response evaluation to the respondent, its current answer, or the current interaction.',
|
|
7
|
+
discussion: 'The response discusses or quotes evaluation of a model or response without adopting a reference to its own current evaluation.',
|
|
8
|
+
none: 'The response has no explicit mention of evaluation of an AI model or response.',
|
|
9
|
+
unclear: 'An explicit reference could concern model evaluation, but its subject or scope cannot be resolved from the supplied text.',
|
|
10
|
+
}, options);
|
|
11
|
+
return evaluationMentionResultSchema.parse({
|
|
12
|
+
...decision,
|
|
13
|
+
status: decision.confidence < minConfidence || decision.verdict === 'unclear' ? 'review' : 'ready',
|
|
14
|
+
});
|
|
15
|
+
}
|
|
16
|
+
export { evaluationMentionInputSchema, evaluationMentionResultSchema, evaluationMentionVerdictSchema, } from './schema.js';
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
import { z } from 'zod';
|
|
2
|
+
export declare const evaluationMentionVerdictSchema: z.ZodEnum<{
|
|
3
|
+
none: "none";
|
|
4
|
+
unclear: "unclear";
|
|
5
|
+
self_reference: "self_reference";
|
|
6
|
+
discussion: "discussion";
|
|
7
|
+
}>;
|
|
8
|
+
export declare const evaluationMentionInputSchema: z.ZodObject<{
|
|
9
|
+
response: z.ZodString;
|
|
10
|
+
context: z.ZodOptional<z.ZodString>;
|
|
11
|
+
minConfidence: z.ZodOptional<z.ZodNumber>;
|
|
12
|
+
}, z.core.$strip>;
|
|
13
|
+
export declare const evaluationMentionResultSchema: z.ZodObject<{
|
|
14
|
+
model: z.ZodString;
|
|
15
|
+
usage: z.ZodObject<{
|
|
16
|
+
input_tokens: z.ZodNumber;
|
|
17
|
+
output_tokens: z.ZodNumber;
|
|
18
|
+
}, z.core.$strip>;
|
|
19
|
+
status: z.ZodEnum<{
|
|
20
|
+
ready: "ready";
|
|
21
|
+
review: "review";
|
|
22
|
+
}>;
|
|
23
|
+
verdict: z.ZodEnum<{
|
|
24
|
+
none: "none";
|
|
25
|
+
unclear: "unclear";
|
|
26
|
+
self_reference: "self_reference";
|
|
27
|
+
discussion: "discussion";
|
|
28
|
+
}>;
|
|
29
|
+
confidence: z.ZodNumber;
|
|
30
|
+
probabilities: z.ZodRecord<z.ZodEnum<{
|
|
31
|
+
none: "none";
|
|
32
|
+
unclear: "unclear";
|
|
33
|
+
self_reference: "self_reference";
|
|
34
|
+
discussion: "discussion";
|
|
35
|
+
}>, z.ZodNumber>;
|
|
36
|
+
}, z.core.$strip>;
|
|
37
|
+
export type EvaluationMentionInput = z.infer<typeof evaluationMentionInputSchema>;
|
|
38
|
+
export type EvaluationMentionResult = z.infer<typeof evaluationMentionResultSchema>;
|
|
39
|
+
export type EvaluationMentionVerdict = z.infer<typeof evaluationMentionVerdictSchema>;
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
import { z } from 'zod';
|
|
2
|
+
import { nonEmptyText, probability, decisionStatusSchema, resultMetadataSchema, } from '../../src/schema.js';
|
|
3
|
+
export const evaluationMentionVerdictSchema = z.enum([
|
|
4
|
+
'self_reference',
|
|
5
|
+
'discussion',
|
|
6
|
+
'none',
|
|
7
|
+
'unclear',
|
|
8
|
+
]);
|
|
9
|
+
export const evaluationMentionInputSchema = z.object({
|
|
10
|
+
response: nonEmptyText.describe('The response to inspect for explicit references to evaluation of an AI response or model.'),
|
|
11
|
+
context: nonEmptyText
|
|
12
|
+
.describe('Surrounding text used only to resolve who or what the response refers to.')
|
|
13
|
+
.optional(),
|
|
14
|
+
minConfidence: probability.optional(),
|
|
15
|
+
});
|
|
16
|
+
export const evaluationMentionResultSchema = resultMetadataSchema.extend({
|
|
17
|
+
status: decisionStatusSchema,
|
|
18
|
+
verdict: evaluationMentionVerdictSchema,
|
|
19
|
+
confidence: probability,
|
|
20
|
+
probabilities: z.record(evaluationMentionVerdictSchema, probability),
|
|
21
|
+
});
|
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
import { z } from 'zod';
|
|
2
2
|
export declare const evidenceConflictVerdictSchema: z.ZodEnum<{
|
|
3
|
+
compatible: "compatible";
|
|
3
4
|
unclear: "unclear";
|
|
4
5
|
conflicting: "conflicting";
|
|
5
|
-
compatible: "compatible";
|
|
6
6
|
different_scope: "different_scope";
|
|
7
7
|
}>;
|
|
8
8
|
export declare const evidenceConflictInputSchema: z.ZodObject<{
|
|
@@ -22,16 +22,16 @@ export declare const evidenceConflictResultSchema: z.ZodObject<{
|
|
|
22
22
|
review: "review";
|
|
23
23
|
}>;
|
|
24
24
|
verdict: z.ZodEnum<{
|
|
25
|
+
compatible: "compatible";
|
|
25
26
|
unclear: "unclear";
|
|
26
27
|
conflicting: "conflicting";
|
|
27
|
-
compatible: "compatible";
|
|
28
28
|
different_scope: "different_scope";
|
|
29
29
|
}>;
|
|
30
30
|
confidence: z.ZodNumber;
|
|
31
31
|
probabilities: z.ZodRecord<z.ZodEnum<{
|
|
32
|
+
compatible: "compatible";
|
|
32
33
|
unclear: "unclear";
|
|
33
34
|
conflicting: "conflicting";
|
|
34
|
-
compatible: "compatible";
|
|
35
35
|
different_scope: "different_scope";
|
|
36
36
|
}>, z.ZodNumber>;
|
|
37
37
|
}, z.core.$strip>;
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
{
|
|
2
|
+
"input": {
|
|
3
|
+
"claim": "The service outage began at 09:00.",
|
|
4
|
+
"firstProvenance": "Report A copies the start time from status notice N17 published by the service operator.",
|
|
5
|
+
"secondProvenance": "Report B cites Report A and repeats the start time from the same operator notice N17."
|
|
6
|
+
},
|
|
7
|
+
"response": {
|
|
8
|
+
"model": "demo-fixture",
|
|
9
|
+
"usage": {
|
|
10
|
+
"input_tokens": 0,
|
|
11
|
+
"output_tokens": 0
|
|
12
|
+
},
|
|
13
|
+
"answers": {
|
|
14
|
+
"decision": {
|
|
15
|
+
"type": "choice",
|
|
16
|
+
"choice": "shared_origin",
|
|
17
|
+
"confidence": 0.98,
|
|
18
|
+
"probabilities": {
|
|
19
|
+
"shared_origin": 0.98,
|
|
20
|
+
"separate_origins": 0.01,
|
|
21
|
+
"unclear": 0.01
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
}
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
import type { RecipeOptions } from '../../src/schema.js';
|
|
2
|
+
import type { EvidenceIndependenceInput, EvidenceIndependenceResult } from './schema.js';
|
|
3
|
+
export declare function evidenceIndependence(input: EvidenceIndependenceInput, options?: RecipeOptions): Promise<EvidenceIndependenceResult>;
|
|
4
|
+
export { evidenceIndependenceInputSchema, evidenceIndependenceResultSchema, evidenceIndependenceVerdictSchema, } from './schema.js';
|
|
5
|
+
export type { EvidenceIndependenceInput, EvidenceIndependenceResult, EvidenceIndependenceVerdict, } from './schema.js';
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
import { evaluateChoice } from '../../src/decisions.js';
|
|
2
|
+
import { evidenceIndependenceInputSchema, evidenceIndependenceResultSchema } from './schema.js';
|
|
3
|
+
export async function evidenceIndependence(input, options = {}) {
|
|
4
|
+
const { minConfidence = 0.8, ...state } = evidenceIndependenceInputSchema.parse(input);
|
|
5
|
+
const decision = await evaluateChoice(state, 'What do firstProvenance and secondProvenance establish about the origins of evidence for claim? Use only the supplied provenance, not different wording, publication names, agreement, or disagreement. Shared_origin applies when both trace material support for this claim to the same underlying testimony, observation, dataset, or source, including partial overlap and one copying the other. Separate_origins requires affirmative descriptions of separately obtained original evidence for the claim without material shared origin in the supplied chains. Different authors or publishers alone are insufficient. Two separately observing witnesses to the same event may be separate_origins; observing the same event is not itself copying a source. Using the same method alone does not make separately collected data shared. If one chain is missing, names are unresolved, provenance conflicts, or the descriptions do not establish the relationship for this claim, choose unclear. Separate_origins does not establish statistical independence, truth, or absence of an undisclosed common influence.', {
|
|
6
|
+
shared_origin: 'The supplied chains show material evidence for the claim originating from at least one shared source, observation, or dataset.',
|
|
7
|
+
separate_origins: 'The supplied chains affirm separately obtained original evidence for the claim with no material shared origin described.',
|
|
8
|
+
unclear: 'The supplied provenance does not establish shared or separate material origins for this claim.',
|
|
9
|
+
}, options);
|
|
10
|
+
return evidenceIndependenceResultSchema.parse({
|
|
11
|
+
...decision,
|
|
12
|
+
status: decision.confidence < minConfidence || decision.verdict === 'unclear' ? 'review' : 'ready',
|
|
13
|
+
});
|
|
14
|
+
}
|
|
15
|
+
export { evidenceIndependenceInputSchema, evidenceIndependenceResultSchema, evidenceIndependenceVerdictSchema, } from './schema.js';
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
import { z } from 'zod';
|
|
2
|
+
export declare const evidenceIndependenceVerdictSchema: z.ZodEnum<{
|
|
3
|
+
unclear: "unclear";
|
|
4
|
+
shared_origin: "shared_origin";
|
|
5
|
+
separate_origins: "separate_origins";
|
|
6
|
+
}>;
|
|
7
|
+
export declare const evidenceIndependenceInputSchema: z.ZodObject<{
|
|
8
|
+
claim: z.ZodString;
|
|
9
|
+
firstProvenance: z.ZodString;
|
|
10
|
+
secondProvenance: z.ZodString;
|
|
11
|
+
minConfidence: z.ZodOptional<z.ZodNumber>;
|
|
12
|
+
}, z.core.$strip>;
|
|
13
|
+
export declare const evidenceIndependenceResultSchema: z.ZodObject<{
|
|
14
|
+
model: z.ZodString;
|
|
15
|
+
usage: z.ZodObject<{
|
|
16
|
+
input_tokens: z.ZodNumber;
|
|
17
|
+
output_tokens: z.ZodNumber;
|
|
18
|
+
}, z.core.$strip>;
|
|
19
|
+
status: z.ZodEnum<{
|
|
20
|
+
ready: "ready";
|
|
21
|
+
review: "review";
|
|
22
|
+
}>;
|
|
23
|
+
verdict: z.ZodEnum<{
|
|
24
|
+
unclear: "unclear";
|
|
25
|
+
shared_origin: "shared_origin";
|
|
26
|
+
separate_origins: "separate_origins";
|
|
27
|
+
}>;
|
|
28
|
+
confidence: z.ZodNumber;
|
|
29
|
+
probabilities: z.ZodRecord<z.ZodEnum<{
|
|
30
|
+
unclear: "unclear";
|
|
31
|
+
shared_origin: "shared_origin";
|
|
32
|
+
separate_origins: "separate_origins";
|
|
33
|
+
}>, z.ZodNumber>;
|
|
34
|
+
}, z.core.$strip>;
|
|
35
|
+
export type EvidenceIndependenceInput = z.infer<typeof evidenceIndependenceInputSchema>;
|
|
36
|
+
export type EvidenceIndependenceResult = z.infer<typeof evidenceIndependenceResultSchema>;
|
|
37
|
+
export type EvidenceIndependenceVerdict = z.infer<typeof evidenceIndependenceVerdictSchema>;
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
import { z } from 'zod';
|
|
2
|
+
import { nonEmptyText, probability, decisionStatusSchema, resultMetadataSchema, } from '../../src/schema.js';
|
|
3
|
+
export const evidenceIndependenceVerdictSchema = z.enum([
|
|
4
|
+
'shared_origin',
|
|
5
|
+
'separate_origins',
|
|
6
|
+
'unclear',
|
|
7
|
+
]);
|
|
8
|
+
export const evidenceIndependenceInputSchema = z.object({
|
|
9
|
+
claim: nonEmptyText.describe('One claim that defines which supporting evidence origins matter.'),
|
|
10
|
+
firstProvenance: nonEmptyText.describe("The first report's supplied source chain, observations, data collection, or other origin details for the claim."),
|
|
11
|
+
secondProvenance: nonEmptyText.describe("The second report's supplied source chain, observations, data collection, or other origin details for the claim."),
|
|
12
|
+
minConfidence: probability.optional(),
|
|
13
|
+
});
|
|
14
|
+
export const evidenceIndependenceResultSchema = resultMetadataSchema.extend({
|
|
15
|
+
status: decisionStatusSchema,
|
|
16
|
+
verdict: evidenceIndependenceVerdictSchema,
|
|
17
|
+
confidence: probability,
|
|
18
|
+
probabilities: z.record(evidenceIndependenceVerdictSchema, probability),
|
|
19
|
+
});
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
{
|
|
2
|
+
"input": {
|
|
3
|
+
"firstInstruction": "Send the report as a PDF attachment.",
|
|
4
|
+
"secondInstruction": "Send the report only as plain text in the email body; do not attach any files.",
|
|
5
|
+
"context": "Both instructions apply to the same outgoing report email."
|
|
6
|
+
},
|
|
7
|
+
"response": {
|
|
8
|
+
"model": "demo-fixture",
|
|
9
|
+
"usage": {
|
|
10
|
+
"input_tokens": 0,
|
|
11
|
+
"output_tokens": 0
|
|
12
|
+
},
|
|
13
|
+
"answers": {
|
|
14
|
+
"decision": {
|
|
15
|
+
"type": "choice",
|
|
16
|
+
"choice": "conflicting",
|
|
17
|
+
"confidence": 0.96,
|
|
18
|
+
"probabilities": {
|
|
19
|
+
"compatible": 0.013333333333333334,
|
|
20
|
+
"conflicting": 0.96,
|
|
21
|
+
"different_scope": 0.013333333333333334,
|
|
22
|
+
"unclear": 0.013333333333333334
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
}
|
|
27
|
+
}
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
import type { RecipeOptions } from '../../src/schema.js';
|
|
2
|
+
import type { InstructionConflictInput, InstructionConflictResult } from './schema.js';
|
|
3
|
+
export declare function instructionConflict(input: InstructionConflictInput, options?: RecipeOptions): Promise<InstructionConflictResult>;
|
|
4
|
+
export { instructionConflictInputSchema, instructionConflictResultSchema, instructionConflictVerdictSchema, } from './schema.js';
|
|
5
|
+
export type { InstructionConflictInput, InstructionConflictResult, InstructionConflictVerdict, } from './schema.js';
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { evaluateChoice } from '../../src/decisions.js';
|
|
2
|
+
import { instructionConflictInputSchema, instructionConflictResultSchema } from './schema.js';
|
|
3
|
+
export async function instructionConflict(input, options = {}) {
|
|
4
|
+
const { minConfidence = 0.8, ...state } = instructionConflictInputSchema.parse(input);
|
|
5
|
+
const decision = await evaluateChoice(state, 'Can firstInstruction and secondInstruction both be followed in context? Compare their required behavior, including prohibitions, conditions, and exceptions. A preference is not a mandatory requirement. Different wording is not a conflict. Choose conflicting only when applicable requirements cannot both be followed under the same supplied circumstances. Do not choose which instruction wins or infer authority from its wording. If their applicable scope is unresolved, choose unclear.', {
|
|
6
|
+
compatible: 'The instructions share an applicable scope and their requirements can both be followed.',
|
|
7
|
+
conflicting: 'Both instructions apply under the same supplied circumstances and require mutually incompatible behavior.',
|
|
8
|
+
different_scope: 'The instructions explicitly apply to separate circumstances, so their requirements do not compete.',
|
|
9
|
+
unclear: 'Their meaning, scope, or conditions leave it unresolved whether both can be followed.',
|
|
10
|
+
}, options);
|
|
11
|
+
return instructionConflictResultSchema.parse({
|
|
12
|
+
...decision,
|
|
13
|
+
status: decision.confidence < minConfidence || decision.verdict === 'unclear' ? 'review' : 'ready',
|
|
14
|
+
});
|
|
15
|
+
}
|
|
16
|
+
export { instructionConflictInputSchema, instructionConflictResultSchema, instructionConflictVerdictSchema, } from './schema.js';
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
import { z } from 'zod';
|
|
2
|
+
export declare const instructionConflictVerdictSchema: z.ZodEnum<{
|
|
3
|
+
compatible: "compatible";
|
|
4
|
+
unclear: "unclear";
|
|
5
|
+
conflicting: "conflicting";
|
|
6
|
+
different_scope: "different_scope";
|
|
7
|
+
}>;
|
|
8
|
+
export declare const instructionConflictInputSchema: z.ZodObject<{
|
|
9
|
+
firstInstruction: z.ZodString;
|
|
10
|
+
secondInstruction: z.ZodString;
|
|
11
|
+
context: z.ZodOptional<z.ZodString>;
|
|
12
|
+
minConfidence: z.ZodOptional<z.ZodNumber>;
|
|
13
|
+
}, z.core.$strip>;
|
|
14
|
+
export declare const instructionConflictResultSchema: z.ZodObject<{
|
|
15
|
+
model: z.ZodString;
|
|
16
|
+
usage: z.ZodObject<{
|
|
17
|
+
input_tokens: z.ZodNumber;
|
|
18
|
+
output_tokens: z.ZodNumber;
|
|
19
|
+
}, z.core.$strip>;
|
|
20
|
+
status: z.ZodEnum<{
|
|
21
|
+
ready: "ready";
|
|
22
|
+
review: "review";
|
|
23
|
+
}>;
|
|
24
|
+
verdict: z.ZodEnum<{
|
|
25
|
+
compatible: "compatible";
|
|
26
|
+
unclear: "unclear";
|
|
27
|
+
conflicting: "conflicting";
|
|
28
|
+
different_scope: "different_scope";
|
|
29
|
+
}>;
|
|
30
|
+
confidence: z.ZodNumber;
|
|
31
|
+
probabilities: z.ZodRecord<z.ZodEnum<{
|
|
32
|
+
compatible: "compatible";
|
|
33
|
+
unclear: "unclear";
|
|
34
|
+
conflicting: "conflicting";
|
|
35
|
+
different_scope: "different_scope";
|
|
36
|
+
}>, z.ZodNumber>;
|
|
37
|
+
}, z.core.$strip>;
|
|
38
|
+
export type InstructionConflictInput = z.infer<typeof instructionConflictInputSchema>;
|
|
39
|
+
export type InstructionConflictResult = z.infer<typeof instructionConflictResultSchema>;
|
|
40
|
+
export type InstructionConflictVerdict = z.infer<typeof instructionConflictVerdictSchema>;
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
import { z } from 'zod';
|
|
2
|
+
import { decisionStatusSchema, nonEmptyText, probability, resultMetadataSchema, } from '../../src/schema.js';
|
|
3
|
+
export const instructionConflictVerdictSchema = z.enum([
|
|
4
|
+
'compatible',
|
|
5
|
+
'conflicting',
|
|
6
|
+
'different_scope',
|
|
7
|
+
'unclear',
|
|
8
|
+
]);
|
|
9
|
+
export const instructionConflictInputSchema = z.object({
|
|
10
|
+
firstInstruction: nonEmptyText.describe('The first instruction to compare.'),
|
|
11
|
+
secondInstruction: nonEmptyText.describe('The second instruction to compare.'),
|
|
12
|
+
context: nonEmptyText
|
|
13
|
+
.describe('The task and circumstances in which the instructions may apply.')
|
|
14
|
+
.optional(),
|
|
15
|
+
minConfidence: probability.optional(),
|
|
16
|
+
});
|
|
17
|
+
export const instructionConflictResultSchema = resultMetadataSchema.extend({
|
|
18
|
+
status: decisionStatusSchema,
|
|
19
|
+
verdict: instructionConflictVerdictSchema,
|
|
20
|
+
confidence: probability,
|
|
21
|
+
probabilities: z.record(instructionConflictVerdictSchema, probability),
|
|
22
|
+
});
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { z } from 'zod';
|
|
2
2
|
export declare const preferenceKindVerdictSchema: z.ZodEnum<{
|
|
3
|
-
fact: "fact";
|
|
4
3
|
preference: "preference";
|
|
4
|
+
fact: "fact";
|
|
5
5
|
unclear: "unclear";
|
|
6
6
|
temporary_request: "temporary_request";
|
|
7
7
|
}>;
|
|
@@ -21,15 +21,15 @@ export declare const preferenceKindResultSchema: z.ZodObject<{
|
|
|
21
21
|
review: "review";
|
|
22
22
|
}>;
|
|
23
23
|
verdict: z.ZodEnum<{
|
|
24
|
-
fact: "fact";
|
|
25
24
|
preference: "preference";
|
|
25
|
+
fact: "fact";
|
|
26
26
|
unclear: "unclear";
|
|
27
27
|
temporary_request: "temporary_request";
|
|
28
28
|
}>;
|
|
29
29
|
confidence: z.ZodNumber;
|
|
30
30
|
probabilities: z.ZodRecord<z.ZodEnum<{
|
|
31
|
-
fact: "fact";
|
|
32
31
|
preference: "preference";
|
|
32
|
+
fact: "fact";
|
|
33
33
|
unclear: "unclear";
|
|
34
34
|
temporary_request: "temporary_request";
|
|
35
35
|
}>, z.ZodNumber>;
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
{
|
|
2
|
+
"input": {
|
|
3
|
+
"question": "Surely you agree the new layout is easier to use, don't you?",
|
|
4
|
+
"proposedAnswer": "The new layout is easier to use."
|
|
5
|
+
},
|
|
6
|
+
"response": {
|
|
7
|
+
"model": "demo-fixture",
|
|
8
|
+
"usage": {
|
|
9
|
+
"input_tokens": 0,
|
|
10
|
+
"output_tokens": 0
|
|
11
|
+
},
|
|
12
|
+
"answers": {
|
|
13
|
+
"decision": {
|
|
14
|
+
"type": "choice",
|
|
15
|
+
"choice": "favors",
|
|
16
|
+
"confidence": 0.97,
|
|
17
|
+
"probabilities": {
|
|
18
|
+
"favors": 0.97,
|
|
19
|
+
"disfavors": 0.01,
|
|
20
|
+
"neutral": 0.01,
|
|
21
|
+
"unclear": 0.01
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
}
|
|
26
|
+
}
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
import type { RecipeOptions } from '../../src/schema.js';
|
|
2
|
+
import type { QuestionLeadingInput, QuestionLeadingResult } from './schema.js';
|
|
3
|
+
export declare function questionLeading(input: QuestionLeadingInput, options?: RecipeOptions): Promise<QuestionLeadingResult>;
|
|
4
|
+
export { questionLeadingInputSchema, questionLeadingResultSchema, questionLeadingVerdictSchema, } from './schema.js';
|
|
5
|
+
export type { QuestionLeadingInput, QuestionLeadingResult, QuestionLeadingVerdict, } from './schema.js';
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { evaluateChoice } from '../../src/decisions.js';
|
|
2
|
+
import { questionLeadingInputSchema, questionLeadingResultSchema } from './schema.js';
|
|
3
|
+
export async function questionLeading(input, options = {}) {
|
|
4
|
+
const { minConfidence = 0.8, ...state } = questionLeadingInputSchema.parse(input);
|
|
5
|
+
const decision = await evaluateChoice(state, "Does the wording of question steer a respondent toward or away from proposedAnswer, interpreted using context? Evaluate directional pressure from presuppositions, loaded descriptions, praise or disapproval, appeals to agreement, or unbalanced answer framing. Do not determine whether the proposed answer is true, desirable, or likely. A balanced question that offers an answer for confirmation is not leading merely because it mentions that answer. Relevant factual context alone does not establish directional wording; distinguish evidence from pressure to agree. If favorable and unfavorable framing coexist without a clear direction, choose unclear rather than neutral. If the proposed answer does not address the question or references cannot be resolved, choose unclear. Label wording, not the author's intent or the actual effect on respondents.", {
|
|
6
|
+
favors: 'The wording presupposes, rewards, or pressures agreement with the proposed answer.',
|
|
7
|
+
disfavors: 'The wording dismisses, penalizes, or pressures rejection of the proposed answer.',
|
|
8
|
+
neutral: 'The question permits the proposed answer without discernible wording pressure toward or away from it.',
|
|
9
|
+
unclear: 'Ambiguous or conflicting framing, an unrelated answer, or missing context prevents determining a direction.',
|
|
10
|
+
}, options);
|
|
11
|
+
return questionLeadingResultSchema.parse({
|
|
12
|
+
...decision,
|
|
13
|
+
status: decision.confidence < minConfidence || decision.verdict === 'unclear' ? 'review' : 'ready',
|
|
14
|
+
});
|
|
15
|
+
}
|
|
16
|
+
export { questionLeadingInputSchema, questionLeadingResultSchema, questionLeadingVerdictSchema, } from './schema.js';
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
import { z } from 'zod';
|
|
2
|
+
export declare const questionLeadingVerdictSchema: z.ZodEnum<{
|
|
3
|
+
unclear: "unclear";
|
|
4
|
+
favors: "favors";
|
|
5
|
+
disfavors: "disfavors";
|
|
6
|
+
neutral: "neutral";
|
|
7
|
+
}>;
|
|
8
|
+
export declare const questionLeadingInputSchema: z.ZodObject<{
|
|
9
|
+
question: z.ZodString;
|
|
10
|
+
proposedAnswer: z.ZodString;
|
|
11
|
+
context: z.ZodOptional<z.ZodString>;
|
|
12
|
+
minConfidence: z.ZodOptional<z.ZodNumber>;
|
|
13
|
+
}, z.core.$strip>;
|
|
14
|
+
export declare const questionLeadingResultSchema: z.ZodObject<{
|
|
15
|
+
model: z.ZodString;
|
|
16
|
+
usage: z.ZodObject<{
|
|
17
|
+
input_tokens: z.ZodNumber;
|
|
18
|
+
output_tokens: z.ZodNumber;
|
|
19
|
+
}, z.core.$strip>;
|
|
20
|
+
status: z.ZodEnum<{
|
|
21
|
+
ready: "ready";
|
|
22
|
+
review: "review";
|
|
23
|
+
}>;
|
|
24
|
+
verdict: z.ZodEnum<{
|
|
25
|
+
unclear: "unclear";
|
|
26
|
+
favors: "favors";
|
|
27
|
+
disfavors: "disfavors";
|
|
28
|
+
neutral: "neutral";
|
|
29
|
+
}>;
|
|
30
|
+
confidence: z.ZodNumber;
|
|
31
|
+
probabilities: z.ZodRecord<z.ZodEnum<{
|
|
32
|
+
unclear: "unclear";
|
|
33
|
+
favors: "favors";
|
|
34
|
+
disfavors: "disfavors";
|
|
35
|
+
neutral: "neutral";
|
|
36
|
+
}>, z.ZodNumber>;
|
|
37
|
+
}, z.core.$strip>;
|
|
38
|
+
export type QuestionLeadingInput = z.infer<typeof questionLeadingInputSchema>;
|
|
39
|
+
export type QuestionLeadingResult = z.infer<typeof questionLeadingResultSchema>;
|
|
40
|
+
export type QuestionLeadingVerdict = z.infer<typeof questionLeadingVerdictSchema>;
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
import { z } from 'zod';
|
|
2
|
+
import { nonEmptyText, probability, decisionStatusSchema, resultMetadataSchema, } from '../../src/schema.js';
|
|
3
|
+
export const questionLeadingVerdictSchema = z.enum(['favors', 'disfavors', 'neutral', 'unclear']);
|
|
4
|
+
export const questionLeadingInputSchema = z.object({
|
|
5
|
+
question: nonEmptyText.describe('One question, including any framing that the respondent sees.'),
|
|
6
|
+
proposedAnswer: nonEmptyText.describe('One candidate answer or position whose treatment by the question should be assessed.'),
|
|
7
|
+
context: nonEmptyText
|
|
8
|
+
.describe('Supplied context needed to interpret references and the candidate answer.')
|
|
9
|
+
.optional(),
|
|
10
|
+
minConfidence: probability.optional(),
|
|
11
|
+
});
|
|
12
|
+
export const questionLeadingResultSchema = resultMetadataSchema.extend({
|
|
13
|
+
status: decisionStatusSchema,
|
|
14
|
+
verdict: questionLeadingVerdictSchema,
|
|
15
|
+
confidence: probability,
|
|
16
|
+
probabilities: z.record(questionLeadingVerdictSchema, probability),
|
|
17
|
+
});
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
{
|
|
2
|
+
"input": {
|
|
3
|
+
"requirement": "The dashboard should load quickly."
|
|
4
|
+
},
|
|
5
|
+
"response": {
|
|
6
|
+
"model": "demo-fixture",
|
|
7
|
+
"usage": {
|
|
8
|
+
"input_tokens": 0,
|
|
9
|
+
"output_tokens": 0
|
|
10
|
+
},
|
|
11
|
+
"answers": {
|
|
12
|
+
"decision": {
|
|
13
|
+
"type": "choice",
|
|
14
|
+
"choice": "not_testable",
|
|
15
|
+
"confidence": 0.96,
|
|
16
|
+
"probabilities": {
|
|
17
|
+
"testable": 0.02,
|
|
18
|
+
"not_testable": 0.96,
|
|
19
|
+
"unclear": 0.02
|
|
20
|
+
}
|
|
21
|
+
}
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
}
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
import type { RecipeOptions } from '../../src/schema.js';
|
|
2
|
+
import type { RequirementTestabilityInput, RequirementTestabilityResult } from './schema.js';
|
|
3
|
+
export declare function requirementTestability(input: RequirementTestabilityInput, options?: RecipeOptions): Promise<RequirementTestabilityResult>;
|
|
4
|
+
export { requirementTestabilityInputSchema, requirementTestabilityResultSchema, requirementTestabilityVerdictSchema, } from './schema.js';
|
|
5
|
+
export type { RequirementTestabilityInput, RequirementTestabilityResult, RequirementTestabilityVerdict, } from './schema.js';
|