@inneranimalmedia/agentsam-sdk 2.6.3 → 2.6.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTSAM.md +6 -0
- package/README.md +16 -1
- package/bin/agentsam +16 -1
- package/docs/BRAND_INTELLIGENCE.md +1 -1
- package/docs/architecture/AGENTSAM_DISTRIBUTION_OWNERSHIP.md +176 -0
- package/docs/architecture/AGENTSAM_GO_RUNTIME.md +282 -0
- package/docs/architecture/CODEBASEINDEX_GUIDED_PATH.md +202 -0
- package/docs/architecture/FS_E2E_CLOSURE_RECEIPT.md +59 -0
- package/docs/architecture/LOCAL_FS_AUTHORITY.md +38 -0
- package/docs/architecture/LOCAL_STUDIO_CLOUDFLARE_OAUTH.md +28 -0
- package/docs/architecture/PLAN_CLI_AND_LOCAL_STUDIO_DESKTOP.md +96 -0
- package/docs/architecture/SAM_ACTIVITY_RECOVERY_RECEIPT.md +39 -0
- package/docs/architecture/SAM_DECISION_WORK_RECEIPT.md +54 -0
- package/docs/architecture/SAM_KERNEL.md +181 -0
- package/docs/architecture/SAM_MACHINE_NORMALIZATION_PRECOMMIT_REPORT.md +208 -0
- package/docs/architecture/SLASH_SKILLS_PORTABLE.md +116 -0
- package/docs/architecture/fs-e2e-receipt.latest.json +42 -0
- package/docs/architecture/previews/codebaseindex-guided-path/CODEBASEINDEX_GUIDED_PATH.md +202 -0
- package/docs/architecture/previews/codebaseindex-guided-path/index.html +321 -0
- package/migrations/d1/0010_portable_tickets_memory.sql +140 -0
- package/migrations/d1/0011_agentsam_skill_v2.sql +185 -0
- package/migrations/d1/0011b_agentsam_skill_v2_cutover.sql +22 -0
- package/migrations/d1/0011c_agentsam_skill_v2_backfill.sql +76 -0
- package/migrations/d1/0011d_agentsam_skill_v2_retrieval_revisions.sql +50 -0
- package/migrations/d1/0012_agentsam_tools_required_seed.sql +67 -0
- package/migrations/d1/0013_identity_oauth_states.sql +15 -0
- package/migrations/d1/0014_auth_event_log.sql +20 -0
- package/migrations/d1/0015_identity_oauth_state_app_id.sql +3 -0
- package/migrations/d1/README_PORTABLE_CONTROL_PLANE.md +16 -0
- package/migrations/sqlite/agentsam_skill_retrieval.portable.sql +42 -0
- package/package.json +19 -4
- package/packages/agentsam-contracts/src/errors.ts +16 -0
- package/packages/agentsam-errors/src/envelope.js +53 -0
- package/packages/agentsam-errors/src/index.js +1 -0
- package/packages/agentsam-errors/src/recovery.js +301 -0
- package/packages/agentsam-knowledge/src/providers/index.js +18 -6
- package/packages/connectors/cloudflare/src/routes.js +9 -0
- package/packages/connectors/cloudflare/tests/connector.test.mjs +26 -1
- package/packages/identity/.agentsam/features/oauth-login-portal/agentsam.feature.json +1 -1
- package/packages/identity/.agentsam/features/oauth-login-portal/routes.json +11 -2
- package/packages/identity/docs/PORTABLE_IDENTITY_ARCHITECTURE.md +50 -0
- package/packages/identity/migrations/D1_SCHEMA_MAPPING.md +31 -0
- package/packages/identity/migrations/sqlite/001_identity_core.sql +99 -0
- package/packages/identity/migrations/sqlite/002_identity_oauth_client.sql +38 -0
- package/packages/identity/migrations/sqlite/003_identity_oauth_server.sql +64 -0
- package/packages/identity/package.json +2 -2
- package/packages/identity/src/adapters/cloudflare-d1/index.js +122 -22
- package/packages/identity/src/adapters/sqlite/index.js +319 -0
- package/packages/identity/src/app/verify-app.js +95 -0
- package/packages/identity/src/contracts/identity-store.js +115 -0
- package/packages/identity/src/contracts/route-ids.js +30 -0
- package/packages/identity/src/contracts/route-projection.js +218 -0
- package/packages/identity/src/contracts/routes.js +11 -0
- package/packages/identity/src/core/browser-paths.js +4 -5
- package/packages/identity/src/core/constants.js +13 -8
- package/packages/identity/src/core/session-policy.js +32 -0
- package/packages/identity/src/frontend/auth-portal/pages/login.html +10 -10
- package/packages/identity/src/frontend/auth-portal/pages/reset.html +3 -3
- package/packages/identity/src/frontend/auth-portal/pages/signup.html +3 -3
- package/packages/identity/src/frontend/auth-portal/preview/dashboard-stub.html +1 -1
- package/packages/identity/src/index.js +25 -0
- package/packages/identity/src/oauth/README.md +10 -4
- package/packages/identity/src/oauth/credentials.js +20 -13
- package/packages/identity/src/oauth/finalize-inbound.js +1 -1
- package/packages/identity/src/oauth/iam-platform.js +8 -7
- package/packages/identity/src/oauth/redirect-paths.js +27 -36
- package/packages/identity/src/server/identity-service.js +30 -13
- package/packages/identity/src/server/mount-policy.js +30 -0
- package/packages/identity/src/server/post-auth.js +79 -0
- package/packages/identity/src/server/worker-router.js +104 -72
- package/packages/identity/tests/finalize-inbound-oauth.test.mjs +6 -6
- package/packages/identity/tests/iam-provider.test.mjs +1 -1
- package/packages/identity/tests/identity-service.test.mjs +36 -5
- package/packages/identity/tests/oauth-credentials.test.mjs +3 -1
- package/packages/identity/tests/portable-identity-architecture.test.mjs +157 -0
- package/packages/identity/tests/session-routes-policy.test.mjs +21 -0
- package/packages/theme-church-site/package.json +2 -1
- package/packages/theme-church-site/src/index.js +1 -0
- package/packages/theme-companions-site/package.json +2 -1
- package/packages/theme-companions-site/src/index.js +1 -0
- package/packages/theme-floors-site/package.json +2 -1
- package/packages/theme-floors-site/src/index.js +1 -0
- package/packages/theme-fuelnfree-site/package.json +2 -1
- package/packages/theme-fuelnfree-site/src/index.js +1 -0
- package/packages/theme-handyman-site/package.json +2 -1
- package/packages/theme-handyman-site/src/index.js +1 -0
- package/packages/theme-insurance-site/package.json +2 -1
- package/packages/theme-insurance-site/src/index.js +1 -0
- package/packages/theme-shinshu-site/package.json +2 -1
- package/packages/theme-shinshu-site/src/index.js +1 -0
- package/protocol/apps/agentsam.app.v1.schema.json +51 -0
- package/protocol/brand/brandpack.v1.schema.json +43 -0
- package/protocol/credentials/issue.v1.schema.json +38 -0
- package/protocol/database/connection.v1.schema.json +41 -0
- package/protocol/embeddings/embedding-profile.v1.schema.json +20 -0
- package/protocol/errors/error-envelope.schema.json +135 -1
- package/protocol/errors/recovery.v1.schema.json +50 -0
- package/protocol/runtime/workspace-fs.v1.schema.json +71 -0
- package/protocol/sam/activity.v1.schema.json +48 -0
- package/protocol/sam/answer.v1.schema.json +35 -0
- package/protocol/sam/calibration.v1.schema.json +21 -0
- package/protocol/sam/decision-receipt.v1.schema.json +29 -0
- package/protocol/sam/evaluation.v1.schema.json +19 -0
- package/protocol/sam/operation.schema.json +66 -0
- package/protocol/sam/outcome.v1.schema.json +36 -0
- package/protocol/sam/question.v1.schema.json +26 -0
- package/protocol/sam/registry.seed.json +153 -0
- package/protocol/sam/result.schema.json +53 -0
- package/protocol/sam/state.v1.schema.json +23 -0
- package/protocol/skills/agentsam.interaction.v1.schema.json +50 -0
- package/protocol/skills/agentsam.skill.v1.schema.json +57 -0
- package/protocol/ui/icon-registry.mjs +226 -0
- package/protocol/ui/icon.v1.schema.json +51 -0
- package/skills/README.md +22 -9
- package/skills/agentsam-codebaseindex/SKILL.md +225 -0
- package/skills/catalog.json +14 -0
- package/src/cli/command-catalog.js +130 -0
- package/src/cli/dispatch.js +48 -0
- package/src/cli.js +43 -1
- package/src/commands/api-key.js +244 -0
- package/src/commands/app.js +60 -28
- package/src/commands/brand.js +17 -19
- package/src/commands/codebaseindex.js +688 -0
- package/src/commands/env.js +152 -25
- package/src/commands/go.js +366 -53
- package/src/commands/interaction-clack.js +115 -0
- package/src/commands/models.js +1 -1
- package/src/commands/providers.js +62 -14
- package/src/commands/shell.js +43 -2
- package/src/commands/skill.js +248 -0
- package/src/commands/skills.js +1 -1
- package/src/commands/start-local.js +4 -0
- package/src/commands/whoami.js +90 -18
- package/src/go/build.js +229 -39
- package/src/go/cloudflare.js +506 -105
- package/src/go/container.js +120 -0
- package/src/go/contract.js +9 -4
- package/src/go/discover.js +149 -33
- package/src/go/index.js +15 -3
- package/src/go/native-probe-runner.mjs +119 -0
- package/src/go/{registry.js → official-registry.js} +64 -7
- package/src/go/official-release.js +10 -0
- package/src/go/receipts.js +77 -10
- package/src/go/verify.js +9 -2
- package/src/index.js +27 -0
- package/src/indexing/ingest/discover-models.js +298 -0
- package/src/indexing/ingest/inventory.js +243 -0
- package/src/indexing/ingest/job-graph.js +181 -0
- package/src/indexing/ingest/materials.js +210 -0
- package/src/lib/provider-credentials.js +63 -21
- package/src/lib/slash-commands.js +1 -0
- package/src/local-fs/capability.js +121 -0
- package/src/local-fs/freshness.js +75 -0
- package/src/local-fs/index.js +385 -0
- package/src/local-fs/paths.js +100 -0
- package/src/local-pty/server.js +295 -19
- package/src/mcp/client.js +2 -2
- package/src/models/ai-access-onboarding.js +112 -0
- package/src/models/discovery.js +10 -2
- package/src/models/inventory-core.js +9 -1
- package/src/sam/activity/index.js +183 -0
- package/src/sam/client.js +252 -0
- package/src/sam/decision/calibration.js +109 -0
- package/src/sam/decision/confidence.js +126 -0
- package/src/sam/decision/evaluate.js +157 -0
- package/src/sam/decision/evaluators/deterministic.js +341 -0
- package/src/sam/decision/evaluators/heuristic.js +61 -0
- package/src/sam/decision/evaluators/select.js +50 -0
- package/src/sam/decision/evaluators/semantic.js +149 -0
- package/src/sam/decision/hierarchical.js +61 -0
- package/src/sam/decision/index.js +53 -0
- package/src/sam/decision/policy.js +86 -0
- package/src/sam/decision/questions.js +120 -0
- package/src/sam/decision/receipt.js +148 -0
- package/src/sam/decision/state.js +117 -0
- package/src/sam/decision/types.js +22 -0
- package/src/sam/decision/validate.js +200 -0
- package/src/sam/define.js +51 -0
- package/src/sam/index.js +65 -0
- package/src/sam/operations/brand-scan.js +62 -0
- package/src/sam/operations/cad-blender-inspect.js +36 -0
- package/src/sam/operations/codebaseindex-ingest.js +49 -0
- package/src/sam/operations/decision-evaluate.js +59 -0
- package/src/sam/operations/planning-astar.js +77 -0
- package/src/sam/operations/planning-goap.js +60 -0
- package/src/sam/operations/repository-inspect.js +72 -0
- package/src/sam/operations/security-scan.js +33 -0
- package/src/sam/operations/terminal-exec.js +29 -0
- package/src/sam/planning/astar.js +311 -0
- package/src/sam/planning/goap.js +177 -0
- package/src/sam/planning/index.js +21 -0
- package/src/sam/planning/state.js +84 -0
- package/src/sam/registry.js +48 -0
- package/src/sam/result.js +77 -0
- package/src/sam/seed.js +44 -0
- package/src/sam/types.js +91 -0
- package/src/skills/catalog.js +64 -0
- package/src/skills/content-resolver.js +124 -0
- package/src/skills/hosted-store.js +37 -0
- package/src/skills/index.js +29 -64
- package/src/skills/interaction.js +102 -0
- package/src/skills/local-store.js +228 -0
- package/src/skills/manifest.js +104 -0
- package/src/skills/metrics.js +31 -0
- package/src/skills/registry.js +184 -0
- package/src/skills/runtime.js +287 -0
- package/src/skills/slash.js +44 -0
- package/src/ui/cli/help.js +94 -101
- package/test/cli/api-key-env-whoami.test.mjs +129 -0
- package/test/cli/codebaseindex-plan-ux.test.mjs +37 -0
- package/test/cli/go.test.mjs +62 -5
- package/test/cli/skill-npm-and-env.test.mjs +52 -0
- package/test/cli/wireframes-go-registry.test.mjs +87 -2
- package/test/go/build-source-identity.test.mjs +31 -0
- package/test/go/cloudflare-probe.test.mjs +274 -10
- package/test/go/distribution.test.mjs +28 -0
- package/test/integration/ai-access-onboarding.test.mjs +47 -0
- package/test/integration/cli-help.test.mjs +1 -1
- package/test/integration/cms-site-tenancy-contract.test.mjs +6 -6
- package/test/integration/codebaseindex-ingest.test.mjs +166 -0
- package/test/integration/icon-registry.test.mjs +67 -0
- package/test/integration/ingest-discover-models.test.mjs +30 -0
- package/test/integration/install-script.test.mjs +12 -9
- package/test/integration/local-fs.test.mjs +113 -0
- package/test/integration/provider-env-cli.test.mjs +2 -1
- package/test/integration/sam-activity-recovery.test.mjs +147 -0
- package/test/integration/sam-decision.test.mjs +584 -0
- package/test/integration/sam-kernel.test.mjs +99 -0
- package/test/integration/sam-planning-astar.test.mjs +279 -0
- package/test/integration/skill-runtime.test.mjs +240 -0
- package/test/integration/studio-fs-pty-e2e.test.mjs +294 -0
- package/test/models.test.mjs +14 -7
- package/test/shell.test.mjs +4 -4
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Semantic evaluator — optional LLM path via existing AgentSam model routing.
|
|
3
|
+
* Numbers from this evaluator are confidence_estimate / support only —
|
|
4
|
+
* NEVER labeled calibrated_probability without calibration_id.
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
import { buildAnswer, concentrationFromScores, clamp01, softmax } from '../confidence.js';
|
|
8
|
+
|
|
9
|
+
export function createSemanticEvaluator(options = {}) {
|
|
10
|
+
const complete = options.complete;
|
|
11
|
+
|
|
12
|
+
return Object.freeze({
|
|
13
|
+
kind: 'semantic',
|
|
14
|
+
version: '1',
|
|
15
|
+
|
|
16
|
+
supports(question) {
|
|
17
|
+
return question?.evaluator_hint === 'semantic';
|
|
18
|
+
},
|
|
19
|
+
|
|
20
|
+
async evaluate(state, questions) {
|
|
21
|
+
if (typeof complete !== 'function') {
|
|
22
|
+
const err = new Error('semantic_evaluator_unconfigured');
|
|
23
|
+
err.code = 'evaluator_unavailable';
|
|
24
|
+
throw err;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/** @type {Record<string, object>} */
|
|
28
|
+
const answers = {};
|
|
29
|
+
for (const q of questions) {
|
|
30
|
+
const prompt = buildPrompt(state, q);
|
|
31
|
+
const raw = await complete({ prompt, question: q, state });
|
|
32
|
+
answers[q.id] = normalizeSemanticAnswer(q, raw);
|
|
33
|
+
}
|
|
34
|
+
return {
|
|
35
|
+
ok: true,
|
|
36
|
+
evaluator: { kind: 'semantic', version: '1' },
|
|
37
|
+
answers,
|
|
38
|
+
model_used: true,
|
|
39
|
+
deterministic: false,
|
|
40
|
+
};
|
|
41
|
+
},
|
|
42
|
+
});
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
function buildPrompt(state, question) {
|
|
46
|
+
return {
|
|
47
|
+
question_id: question.id,
|
|
48
|
+
type: question.type,
|
|
49
|
+
instructions: question.instructions,
|
|
50
|
+
criteria: question.criteria,
|
|
51
|
+
options: question.options,
|
|
52
|
+
levels: question.levels,
|
|
53
|
+
state_facts: state.facts || {},
|
|
54
|
+
intent: state.intent,
|
|
55
|
+
};
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
function normalizeSemanticAnswer(question, raw) {
|
|
59
|
+
const warnings = ['semantic_values_are_estimates_not_calibrated_probabilities'];
|
|
60
|
+
if (!raw || typeof raw !== 'object') {
|
|
61
|
+
const err = new Error('semantic_answer_invalid');
|
|
62
|
+
err.code = 'evaluator_failed';
|
|
63
|
+
throw err;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
if (question.type === 'choose') {
|
|
67
|
+
const value = String(raw.value ?? raw.choice ?? '');
|
|
68
|
+
if (!question.options.includes(value)) {
|
|
69
|
+
const err = new Error(`semantic_choice_not_in_options:${value}`);
|
|
70
|
+
err.code = 'evaluator_failed';
|
|
71
|
+
throw err;
|
|
72
|
+
}
|
|
73
|
+
const scores = raw.scores && typeof raw.scores === 'object'
|
|
74
|
+
? Object.fromEntries(question.options.map((o) => [o, Number(raw.scores[o]) || 0]))
|
|
75
|
+
: Object.fromEntries(question.options.map((o) => [o, o === value ? 1 : 0]));
|
|
76
|
+
return buildAnswer({
|
|
77
|
+
type: 'choose',
|
|
78
|
+
question_id: question.id,
|
|
79
|
+
value,
|
|
80
|
+
scores,
|
|
81
|
+
confidence_estimate: Number.isFinite(Number(raw.confidence_estimate))
|
|
82
|
+
? clamp01(Number(raw.confidence_estimate))
|
|
83
|
+
: concentrationFromScores(scores),
|
|
84
|
+
// Explicitly omit probabilities unless caller marks probabilistic=true AND supplies dist
|
|
85
|
+
probabilities: raw.probabilistic === true && raw.probabilities
|
|
86
|
+
? raw.probabilities
|
|
87
|
+
: null,
|
|
88
|
+
evaluator: { kind: 'semantic', version: '1' },
|
|
89
|
+
warnings,
|
|
90
|
+
evidence: Array.isArray(raw.evidence) ? raw.evidence : [],
|
|
91
|
+
});
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
if (question.type === 'score') {
|
|
95
|
+
let value = Number(raw.value);
|
|
96
|
+
const scores = raw.scores && typeof raw.scores === 'object'
|
|
97
|
+
? Object.fromEntries(question.levels.map((l) => [String(l.index), Number(raw.scores[l.index] ?? raw.scores[String(l.index)]) || 0]))
|
|
98
|
+
: null;
|
|
99
|
+
if (raw.probabilistic === true && raw.probabilities) {
|
|
100
|
+
const probs = Object.fromEntries(
|
|
101
|
+
question.levels.map((l) => [String(l.index), Number(raw.probabilities[l.index] ?? raw.probabilities[String(l.index)]) || 0]),
|
|
102
|
+
);
|
|
103
|
+
value = question.levels.reduce((n, l) => n + l.index * (probs[String(l.index)] || 0), 0);
|
|
104
|
+
return buildAnswer({
|
|
105
|
+
type: 'score',
|
|
106
|
+
question_id: question.id,
|
|
107
|
+
value,
|
|
108
|
+
levels: Object.fromEntries(question.levels.map((l) => [String(l.index), l.label])),
|
|
109
|
+
scores: scores || probs,
|
|
110
|
+
probabilities: probs,
|
|
111
|
+
confidence_estimate: concentrationFromScores(probs),
|
|
112
|
+
evaluator: { kind: 'semantic', version: '1' },
|
|
113
|
+
warnings,
|
|
114
|
+
});
|
|
115
|
+
}
|
|
116
|
+
if (!Number.isFinite(value)) {
|
|
117
|
+
const err = new Error('semantic_score_invalid');
|
|
118
|
+
err.code = 'evaluator_failed';
|
|
119
|
+
throw err;
|
|
120
|
+
}
|
|
121
|
+
return buildAnswer({
|
|
122
|
+
type: 'score',
|
|
123
|
+
question_id: question.id,
|
|
124
|
+
value,
|
|
125
|
+
levels: Object.fromEntries(question.levels.map((l) => [String(l.index), l.label])),
|
|
126
|
+
scores: scores || softmax(Object.fromEntries(question.levels.map((l) => [String(l.index), -Math.abs(l.index - value)]))),
|
|
127
|
+
confidence_estimate: Number.isFinite(Number(raw.confidence_estimate))
|
|
128
|
+
? clamp01(Number(raw.confidence_estimate))
|
|
129
|
+
: 0.5,
|
|
130
|
+
evaluator: { kind: 'semantic', version: '1' },
|
|
131
|
+
warnings,
|
|
132
|
+
});
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
// check
|
|
136
|
+
const support = clamp01(Number(raw.support ?? raw.confidence_estimate ?? (raw.value ? 0.8 : 0.2)));
|
|
137
|
+
return buildAnswer({
|
|
138
|
+
type: 'check',
|
|
139
|
+
question_id: question.id,
|
|
140
|
+
value: Boolean(raw.value ?? support >= 0.5),
|
|
141
|
+
support,
|
|
142
|
+
evaluator: { kind: 'semantic', version: '1' },
|
|
143
|
+
warnings,
|
|
144
|
+
evidence: Array.isArray(raw.evidence) ? raw.evidence : [],
|
|
145
|
+
});
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
/** Default unconfigured instance — fails closed. */
|
|
149
|
+
export const SEMANTIC_EVALUATOR = createSemanticEvaluator();
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Hierarchical choose + beam retention for large taxonomies (skills, tools, errors).
|
|
3
|
+
* When top options are close, keep multiple paths alive instead of greedy commit.
|
|
4
|
+
*/
|
|
5
|
+
|
|
6
|
+
/**
|
|
7
|
+
* @param {Array<{ id: string, score: number, parent?: string|null }>} candidates
|
|
8
|
+
* @param {object} [opts]
|
|
9
|
+
* @param {number} [opts.beamWidth]
|
|
10
|
+
* @param {number} [opts.closeMargin] keep peers within this absolute score of the best
|
|
11
|
+
*/
|
|
12
|
+
export function beamRetain(candidates, opts = {}) {
|
|
13
|
+
const beamWidth = opts.beamWidth ?? 3;
|
|
14
|
+
const closeMargin = opts.closeMargin ?? 0.08;
|
|
15
|
+
const sorted = [...candidates].sort((a, b) => b.score - a.score || a.id.localeCompare(b.id));
|
|
16
|
+
if (!sorted.length) return [];
|
|
17
|
+
const best = sorted[0].score;
|
|
18
|
+
const close = sorted.filter((c) => best - c.score <= closeMargin);
|
|
19
|
+
const beam = (close.length > 1 ? close : sorted).slice(0, beamWidth);
|
|
20
|
+
return beam;
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* Walk a taxonomy level-by-level with beam retention.
|
|
25
|
+
* @param {object} opts
|
|
26
|
+
* @param {Array<{ id: string, parent: string|null, score: number }>} opts.nodes
|
|
27
|
+
* @param {number} [opts.beamWidth]
|
|
28
|
+
* @param {number} [opts.closeMargin]
|
|
29
|
+
*/
|
|
30
|
+
export function hierarchicalChoose(opts = {}) {
|
|
31
|
+
const nodes = Array.isArray(opts.nodes) ? opts.nodes : [];
|
|
32
|
+
const roots = nodes.filter((n) => n.parent == null || n.parent === '');
|
|
33
|
+
let beam = beamRetain(roots, opts);
|
|
34
|
+
const path = [];
|
|
35
|
+
while (beam.length) {
|
|
36
|
+
// If multiple remain close, surface beam instead of committing.
|
|
37
|
+
if (beam.length > 1) {
|
|
38
|
+
return {
|
|
39
|
+
ok: true,
|
|
40
|
+
committed: false,
|
|
41
|
+
beam: beam.map((b) => b.id),
|
|
42
|
+
path: path.map((p) => p.id),
|
|
43
|
+
reason: 'close_candidates_retained',
|
|
44
|
+
};
|
|
45
|
+
}
|
|
46
|
+
const chosen = beam[0];
|
|
47
|
+
path.push(chosen);
|
|
48
|
+
const children = nodes.filter((n) => n.parent === chosen.id);
|
|
49
|
+
if (!children.length) {
|
|
50
|
+
return {
|
|
51
|
+
ok: true,
|
|
52
|
+
committed: true,
|
|
53
|
+
value: chosen.id,
|
|
54
|
+
path: path.map((p) => p.id),
|
|
55
|
+
beam: [chosen.id],
|
|
56
|
+
};
|
|
57
|
+
}
|
|
58
|
+
beam = beamRetain(children, opts);
|
|
59
|
+
}
|
|
60
|
+
return { ok: false, committed: false, beam: [], path: [], reason: 'empty_taxonomy' };
|
|
61
|
+
}
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* SAM decision module — choose / score / check + batch evaluate.
|
|
3
|
+
*/
|
|
4
|
+
|
|
5
|
+
export {
|
|
6
|
+
DECISION_STATE_SCHEMA,
|
|
7
|
+
DECISION_QUESTION_SCHEMA,
|
|
8
|
+
DECISION_ANSWER_SCHEMA,
|
|
9
|
+
DECISION_EVALUATION_SCHEMA,
|
|
10
|
+
DECISION_RECEIPT_SCHEMA,
|
|
11
|
+
DECISION_OUTCOME_SCHEMA,
|
|
12
|
+
DECISION_CALIBRATION_SCHEMA,
|
|
13
|
+
} from './types.js';
|
|
14
|
+
|
|
15
|
+
export { buildDecisionState, hashDecisionState, projectState } from './state.js';
|
|
16
|
+
export {
|
|
17
|
+
defineChoose,
|
|
18
|
+
defineScore,
|
|
19
|
+
defineCheck,
|
|
20
|
+
validateQuestion,
|
|
21
|
+
validateQuestionBatch,
|
|
22
|
+
} from './validate.js';
|
|
23
|
+
export {
|
|
24
|
+
concentrationFromScores,
|
|
25
|
+
softmax,
|
|
26
|
+
expectedScore,
|
|
27
|
+
assertProbabilitiesSumToOne,
|
|
28
|
+
buildAnswer,
|
|
29
|
+
clamp01,
|
|
30
|
+
} from './confidence.js';
|
|
31
|
+
export { evaluate, evaluateDependentStages } from './evaluate.js';
|
|
32
|
+
export { applyDecisionPolicy, composeUtility } from './policy.js';
|
|
33
|
+
export { createDecisionReceipt, appendOutcome, attachAction, OUTCOME_DISPOSITIONS } from './receipt.js';
|
|
34
|
+
export {
|
|
35
|
+
brierScore,
|
|
36
|
+
logLoss,
|
|
37
|
+
expectedCalibrationError,
|
|
38
|
+
exportCalibrationDataset,
|
|
39
|
+
} from './calibration.js';
|
|
40
|
+
export {
|
|
41
|
+
QUESTION_REGISTRY,
|
|
42
|
+
getRegisteredQuestion,
|
|
43
|
+
listRegisteredQuestions,
|
|
44
|
+
} from './questions.js';
|
|
45
|
+
export {
|
|
46
|
+
selectEvaluator,
|
|
47
|
+
DETERMINISTIC_EVALUATOR,
|
|
48
|
+
HEURISTIC_EVALUATOR,
|
|
49
|
+
SEMANTIC_EVALUATOR,
|
|
50
|
+
DEFAULT_CHAIN,
|
|
51
|
+
} from './evaluators/select.js';
|
|
52
|
+
export { createSemanticEvaluator } from './evaluators/semantic.js';
|
|
53
|
+
export { hierarchicalChoose, beamRetain } from './hierarchical.js';
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Policy owns permission — scores/confidence never grant authority alone.
|
|
3
|
+
*/
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* @param {object} opts
|
|
7
|
+
* @param {Record<string, object>} opts.answers
|
|
8
|
+
* @param {object} [opts.policy]
|
|
9
|
+
* @param {object} [opts.state]
|
|
10
|
+
*/
|
|
11
|
+
export function applyDecisionPolicy(opts = {}) {
|
|
12
|
+
const answers = opts.answers || {};
|
|
13
|
+
const policy = opts.policy || {};
|
|
14
|
+
const state = opts.state || {};
|
|
15
|
+
const required = [];
|
|
16
|
+
const denied = [];
|
|
17
|
+
const notes = [];
|
|
18
|
+
|
|
19
|
+
// Hard policy: production writes always require approval regardless of check support.
|
|
20
|
+
if (
|
|
21
|
+
policy.production_database_write === 'ALWAYS_REQUIRE_APPROVAL'
|
|
22
|
+
|| state.constraints?.production_write
|
|
23
|
+
|| state.facts?.production_write
|
|
24
|
+
) {
|
|
25
|
+
required.push('user_approval');
|
|
26
|
+
notes.push('policy:production_write_always_requires_approval');
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
const approvalAnswer = answers.requires_approval;
|
|
30
|
+
if (approvalAnswer?.type === 'check' && approvalAnswer.value === true) {
|
|
31
|
+
required.push('user_approval');
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
// High confidence NEVER authorizes by itself.
|
|
35
|
+
for (const [id, answer] of Object.entries(answers)) {
|
|
36
|
+
if (
|
|
37
|
+
answer?.confidence_estimate != null
|
|
38
|
+
&& answer.confidence_estimate >= 0.95
|
|
39
|
+
&& policy.confidence_grants_authority
|
|
40
|
+
) {
|
|
41
|
+
notes.push(`ignored_confidence_authority_claim:${id}`);
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
if (policy.deny_sandbox && answers.terminal_lane?.value === 'sandbox') {
|
|
46
|
+
denied.push('terminal_lane:sandbox');
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
const allowed = denied.length === 0;
|
|
50
|
+
return {
|
|
51
|
+
allowed,
|
|
52
|
+
require_approval: required.includes('user_approval'),
|
|
53
|
+
required: [...new Set(required)],
|
|
54
|
+
denied,
|
|
55
|
+
notes,
|
|
56
|
+
// Explicit: confidence is not permission
|
|
57
|
+
authorization_source: 'policy',
|
|
58
|
+
};
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/**
|
|
62
|
+
* Composite utility from separate score/check answers — weights live in code.
|
|
63
|
+
* @param {object} parts
|
|
64
|
+
* @param {object} [weights]
|
|
65
|
+
*/
|
|
66
|
+
export function composeUtility(parts = {}, weights = {}) {
|
|
67
|
+
const w = {
|
|
68
|
+
goal_progress: weights.goal_progress ?? 0.4,
|
|
69
|
+
reversibility: weights.reversibility ?? 0.15,
|
|
70
|
+
risk: weights.risk ?? 0.25,
|
|
71
|
+
cost: weights.cost ?? 0.1,
|
|
72
|
+
confidence: weights.confidence ?? 0.1,
|
|
73
|
+
};
|
|
74
|
+
const goal = Number(parts.goal_progress ?? 0);
|
|
75
|
+
const rev = Number(parts.reversibility ?? 0);
|
|
76
|
+
const risk = Number(parts.risk ?? 0);
|
|
77
|
+
const cost = Number(parts.cost ?? 0);
|
|
78
|
+
const conf = Number(parts.confidence ?? 0);
|
|
79
|
+
return (
|
|
80
|
+
goal * w.goal_progress
|
|
81
|
+
+ rev * w.reversibility
|
|
82
|
+
+ conf * w.confidence
|
|
83
|
+
- risk * w.risk
|
|
84
|
+
- cost * w.cost
|
|
85
|
+
);
|
|
86
|
+
}
|
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Initial AgentSam decision question registry.
|
|
3
|
+
*/
|
|
4
|
+
|
|
5
|
+
import { defineChoose, defineScore, defineCheck } from './validate.js';
|
|
6
|
+
|
|
7
|
+
export const QUESTION_REGISTRY = Object.freeze({
|
|
8
|
+
terminal_lane: defineChoose({
|
|
9
|
+
id: 'terminal_lane',
|
|
10
|
+
version: 1,
|
|
11
|
+
evaluator_hint: 'deterministic',
|
|
12
|
+
instructions: {
|
|
13
|
+
question: 'Which execution lane best fits this task?',
|
|
14
|
+
focus: 'Use isolation, repository locality, availability and cost.',
|
|
15
|
+
},
|
|
16
|
+
criteria: {
|
|
17
|
+
local: {
|
|
18
|
+
use_when: ['task requires unsynced local files', 'local terminal healthy'],
|
|
19
|
+
avoid_when: ['strong isolation required'],
|
|
20
|
+
},
|
|
21
|
+
remote: {
|
|
22
|
+
use_when: ['shared remote environment needed'],
|
|
23
|
+
avoid_when: ['requires local-only paths'],
|
|
24
|
+
},
|
|
25
|
+
sandbox: {
|
|
26
|
+
use_when: ['code is untrusted', 'strong isolation is required'],
|
|
27
|
+
avoid_when: ['task requires direct access to unsynced local files'],
|
|
28
|
+
},
|
|
29
|
+
},
|
|
30
|
+
options: ['local', 'remote', 'sandbox'],
|
|
31
|
+
}),
|
|
32
|
+
|
|
33
|
+
retrieval_mode: defineChoose({
|
|
34
|
+
id: 'retrieval_mode',
|
|
35
|
+
version: 1,
|
|
36
|
+
evaluator_hint: 'deterministic',
|
|
37
|
+
instructions: 'Which retrieval corpus should SAM query first?',
|
|
38
|
+
options: ['ast', 'lexical', 'semantic', 'memory', 'web', 'none'],
|
|
39
|
+
criteria: {
|
|
40
|
+
ast: { use_when: ['symbol / import / call-graph questions'] },
|
|
41
|
+
lexical: { use_when: ['exact string / path lookup'] },
|
|
42
|
+
semantic: { use_when: ['architectural concept questions'] },
|
|
43
|
+
memory: { use_when: ['historical decisions'] },
|
|
44
|
+
web: { use_when: ['fresh provider docs'] },
|
|
45
|
+
none: { use_when: ['no retrieval needed'] },
|
|
46
|
+
},
|
|
47
|
+
}),
|
|
48
|
+
|
|
49
|
+
change_risk: defineScore({
|
|
50
|
+
id: 'change_risk',
|
|
51
|
+
version: 1,
|
|
52
|
+
evaluator_hint: 'deterministic',
|
|
53
|
+
instructions: 'How much operational risk does this proposed change carry?',
|
|
54
|
+
levels: [
|
|
55
|
+
{ index: 0, label: 'Local, reversible change with no runtime impact' },
|
|
56
|
+
{ index: 1, label: 'Bounded implementation change with focused verification' },
|
|
57
|
+
{ index: 2, label: 'Cross-package change with known consumers' },
|
|
58
|
+
{ index: 3, label: 'Runtime/schema/security boundary change' },
|
|
59
|
+
{ index: 4, label: 'High-blast-radius or production-sensitive change' },
|
|
60
|
+
],
|
|
61
|
+
}),
|
|
62
|
+
|
|
63
|
+
requires_approval: defineCheck({
|
|
64
|
+
id: 'requires_approval',
|
|
65
|
+
version: 1,
|
|
66
|
+
evaluator_hint: 'deterministic',
|
|
67
|
+
instructions: {
|
|
68
|
+
question: 'Does this action require explicit user approval?',
|
|
69
|
+
inspect: ['action.reversibility', 'action.production_effect', 'policy'],
|
|
70
|
+
},
|
|
71
|
+
}),
|
|
72
|
+
|
|
73
|
+
security_review_required: defineCheck({
|
|
74
|
+
id: 'security_review_required',
|
|
75
|
+
version: 1,
|
|
76
|
+
evaluator_hint: 'deterministic',
|
|
77
|
+
instructions: {
|
|
78
|
+
question: 'Does this change require a security review before merge?',
|
|
79
|
+
inspect: ['security_boundary', 'auth_change', 'secret_handling'],
|
|
80
|
+
},
|
|
81
|
+
}),
|
|
82
|
+
|
|
83
|
+
verification_scope: defineChoose({
|
|
84
|
+
id: 'verification_scope',
|
|
85
|
+
version: 1,
|
|
86
|
+
evaluator_hint: 'deterministic',
|
|
87
|
+
instructions: 'How broad should verification be for this change?',
|
|
88
|
+
options: ['focused', 'package', 'workspace', 'repository'],
|
|
89
|
+
}),
|
|
90
|
+
});
|
|
91
|
+
|
|
92
|
+
/**
|
|
93
|
+
* @param {string} id
|
|
94
|
+
* @param {object} [overrides] e.g. dynamic options for skill_candidate
|
|
95
|
+
*/
|
|
96
|
+
export function getRegisteredQuestion(id, overrides = {}) {
|
|
97
|
+
if (id === 'skill_candidate') {
|
|
98
|
+
const options = Array.isArray(overrides.options) ? overrides.options : [];
|
|
99
|
+
return defineChoose({
|
|
100
|
+
id: 'skill_candidate',
|
|
101
|
+
version: 1,
|
|
102
|
+
evaluator_hint: overrides.evaluator_hint || 'deterministic',
|
|
103
|
+
instructions: overrides.instructions || 'Which skill best fits this task from the retrieved candidates?',
|
|
104
|
+
options,
|
|
105
|
+
criteria: overrides.criteria || null,
|
|
106
|
+
});
|
|
107
|
+
}
|
|
108
|
+
const base = QUESTION_REGISTRY[id];
|
|
109
|
+
if (!base) {
|
|
110
|
+
const err = new Error(`unknown_question:${id}`);
|
|
111
|
+
err.code = 'unknown_question';
|
|
112
|
+
throw err;
|
|
113
|
+
}
|
|
114
|
+
if (!Object.keys(overrides).length) return base;
|
|
115
|
+
return { ...base, ...overrides, id: base.id, type: base.type };
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
export function listRegisteredQuestions() {
|
|
119
|
+
return Object.keys(QUESTION_REGISTRY);
|
|
120
|
+
}
|
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Decision receipts + outcome reconciliation + run/step/action linkage.
|
|
3
|
+
*/
|
|
4
|
+
|
|
5
|
+
import { randomUUID } from 'node:crypto';
|
|
6
|
+
import { DECISION_RECEIPT_SCHEMA, DECISION_OUTCOME_SCHEMA } from './types.js';
|
|
7
|
+
import { hashDecisionState } from './state.js';
|
|
8
|
+
|
|
9
|
+
export const OUTCOME_DISPOSITIONS = Object.freeze([
|
|
10
|
+
'accepted',
|
|
11
|
+
'rejected',
|
|
12
|
+
'overridden',
|
|
13
|
+
'corrected_to',
|
|
14
|
+
'retry_required',
|
|
15
|
+
'verification_failed',
|
|
16
|
+
'user_intervened',
|
|
17
|
+
'success',
|
|
18
|
+
'failure',
|
|
19
|
+
]);
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* @param {object} opts
|
|
23
|
+
*/
|
|
24
|
+
export function createDecisionReceipt(opts = {}) {
|
|
25
|
+
const decision_id = opts.decision_id || `dec_${randomUUID().replace(/-/g, '').slice(0, 20)}`;
|
|
26
|
+
return {
|
|
27
|
+
schema: DECISION_RECEIPT_SCHEMA,
|
|
28
|
+
decision_id,
|
|
29
|
+
run_id: opts.run_id ?? null,
|
|
30
|
+
step_id: opts.step_id ?? null,
|
|
31
|
+
action_id: opts.action_id ?? null,
|
|
32
|
+
outcome_id: null,
|
|
33
|
+
question_contract: opts.question_contract || null,
|
|
34
|
+
question_ids: Array.isArray(opts.question_ids) ? opts.question_ids : [],
|
|
35
|
+
question_versions: opts.question_versions || {},
|
|
36
|
+
evaluator: opts.evaluator || null,
|
|
37
|
+
evaluators: opts.evaluators || [],
|
|
38
|
+
state_hash: opts.state_hash || (opts.state ? hashDecisionState(opts.state) : null),
|
|
39
|
+
answers: opts.answers || {},
|
|
40
|
+
policy_result: opts.policy_result || null,
|
|
41
|
+
evidence: Array.isArray(opts.evidence) ? opts.evidence : [],
|
|
42
|
+
action: opts.action ?? null,
|
|
43
|
+
outcome: null,
|
|
44
|
+
created_at: new Date().toISOString(),
|
|
45
|
+
why: opts.why || buildWhy(opts.answers, opts.policy_result),
|
|
46
|
+
};
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* Attach a consequential outcome to a receipt (pass/fail + human corrections).
|
|
51
|
+
* Passive /why with no correction is NOT treated as a strong correctness label.
|
|
52
|
+
*
|
|
53
|
+
* @param {object} receipt
|
|
54
|
+
* @param {object} outcome
|
|
55
|
+
*/
|
|
56
|
+
export function appendOutcome(receipt, outcome) {
|
|
57
|
+
if (!receipt || receipt.schema !== DECISION_RECEIPT_SCHEMA) {
|
|
58
|
+
const err = new Error('invalid_receipt');
|
|
59
|
+
err.code = 'invalid_receipt';
|
|
60
|
+
throw err;
|
|
61
|
+
}
|
|
62
|
+
const disposition = normalizeDisposition(outcome);
|
|
63
|
+
const outcome_id = outcome.outcome_id || `out_${randomUUID().replace(/-/g, '').slice(0, 16)}`;
|
|
64
|
+
const normalized = {
|
|
65
|
+
schema: DECISION_OUTCOME_SCHEMA,
|
|
66
|
+
outcome_id,
|
|
67
|
+
success: outcome.success != null ? Boolean(outcome.success) : disposition === 'accepted' || disposition === 'success',
|
|
68
|
+
disposition,
|
|
69
|
+
human_corrected: Boolean(outcome.human_corrected || disposition === 'overridden' || disposition === 'corrected_to' || disposition === 'user_intervened'),
|
|
70
|
+
corrected_to: outcome.corrected_to ?? null,
|
|
71
|
+
user_override: Boolean(outcome.user_override),
|
|
72
|
+
actual_class: outcome.actual_class ?? null,
|
|
73
|
+
duration_ms: Number.isFinite(Number(outcome.duration_ms)) ? Number(outcome.duration_ms) : null,
|
|
74
|
+
retries: Number.isInteger(outcome.retries) ? outcome.retries : null,
|
|
75
|
+
verification: Array.isArray(outcome.verification) ? outcome.verification : [],
|
|
76
|
+
// Passive acceptance is weak evidence — do not treat as calibrated label
|
|
77
|
+
label_strength: outcome.label_strength
|
|
78
|
+
|| (outcome.human_corrected || disposition === 'corrected_to' || disposition === 'overridden'
|
|
79
|
+
? 'supervised'
|
|
80
|
+
: disposition === 'accepted' && outcome.explicit_accept
|
|
81
|
+
? 'explicit_accept'
|
|
82
|
+
: 'weak_passive'),
|
|
83
|
+
recorded_at: new Date().toISOString(),
|
|
84
|
+
};
|
|
85
|
+
return {
|
|
86
|
+
...receipt,
|
|
87
|
+
action_id: outcome.action_id ?? receipt.action_id,
|
|
88
|
+
outcome_id,
|
|
89
|
+
outcome: normalized,
|
|
90
|
+
updated_at: normalized.recorded_at,
|
|
91
|
+
};
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* Link an executed action onto a receipt before outcome is known.
|
|
96
|
+
*/
|
|
97
|
+
export function attachAction(receipt, action = {}) {
|
|
98
|
+
if (!receipt || receipt.schema !== DECISION_RECEIPT_SCHEMA) {
|
|
99
|
+
const err = new Error('invalid_receipt');
|
|
100
|
+
err.code = 'invalid_receipt';
|
|
101
|
+
throw err;
|
|
102
|
+
}
|
|
103
|
+
const action_id = action.action_id || `act_${randomUUID().replace(/-/g, '').slice(0, 16)}`;
|
|
104
|
+
return {
|
|
105
|
+
...receipt,
|
|
106
|
+
action_id,
|
|
107
|
+
action: {
|
|
108
|
+
action_id,
|
|
109
|
+
kind: action.kind || null,
|
|
110
|
+
summary: action.summary || null,
|
|
111
|
+
started_at: action.started_at || new Date().toISOString(),
|
|
112
|
+
...action,
|
|
113
|
+
},
|
|
114
|
+
updated_at: new Date().toISOString(),
|
|
115
|
+
};
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
function normalizeDisposition(outcome = {}) {
|
|
119
|
+
if (outcome.disposition && OUTCOME_DISPOSITIONS.includes(outcome.disposition)) {
|
|
120
|
+
return outcome.disposition;
|
|
121
|
+
}
|
|
122
|
+
if (outcome.human_corrected && outcome.corrected_to != null) return 'corrected_to';
|
|
123
|
+
if (outcome.user_override) return 'overridden';
|
|
124
|
+
if (outcome.success === true) return 'accepted';
|
|
125
|
+
if (outcome.success === false) return 'failure';
|
|
126
|
+
return 'accepted';
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
function buildWhy(answers = {}, policy = null) {
|
|
130
|
+
const decisions = Object.entries(answers).map(([id, a]) => ({
|
|
131
|
+
id,
|
|
132
|
+
type: a.type,
|
|
133
|
+
value: a.value,
|
|
134
|
+
support: a.support ?? null,
|
|
135
|
+
confidence_estimate: a.confidence_estimate ?? null,
|
|
136
|
+
}));
|
|
137
|
+
return {
|
|
138
|
+
surface: 'agentsam.why.v1',
|
|
139
|
+
decisions,
|
|
140
|
+
policy: policy
|
|
141
|
+
? {
|
|
142
|
+
allowed: policy.allowed,
|
|
143
|
+
require_approval: policy.require_approval,
|
|
144
|
+
authorization_source: policy.authorization_source,
|
|
145
|
+
}
|
|
146
|
+
: null,
|
|
147
|
+
};
|
|
148
|
+
}
|