specpi 0.26.0 → 0.27.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +43 -0
- package/README.md +38 -4
- package/SECURITY_MODEL.md +36 -2
- package/THIRD_PARTY.md +10 -1
- package/extensions/jev-advisor/broker.mjs +276 -0
- package/extensions/jev-advisor/client.mjs +182 -0
- package/extensions/jev-advisor/config.mjs +254 -0
- package/extensions/jev-advisor/consent.mjs +133 -0
- package/extensions/jev-advisor/gate.mjs +249 -0
- package/extensions/jev-advisor/guard.mjs +140 -0
- package/extensions/jev-advisor/index.ts +849 -0
- package/extensions/jev-advisor/ledger.mjs +138 -0
- package/extensions/jev-advisor/questions/capabilities.mjs +124 -0
- package/extensions/jev-advisor/questions/compaction.mjs +153 -0
- package/extensions/jev-advisor/questions/gap.mjs +140 -0
- package/extensions/jev-advisor/questions/progress.mjs +195 -0
- package/extensions/jev-advisor/questions/retention.mjs +188 -0
- package/extensions/jev-advisor/questions/sources.mjs +91 -0
- package/extensions/jev-advisor/questions/untrusted.mjs +69 -0
- package/extensions/jev-advisor/sanitize.mjs +0 -0
- package/extensions/jev-advisor/usage.mjs +92 -0
- package/extensions/tool-wishlist/authoring-tools.mjs +42 -0
- package/extensions/tool-wishlist/index.ts +11 -0
- package/extensions/workflow-controls/capabilities.mjs +26 -0
- package/extensions/workflow-controls/index.ts +2 -2
- package/package.json +1 -1
- package/scripts/specpi.mjs +33 -1
- package/templates/settings.json +2 -1
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
// Phase 0 of the Jev advisor plan, which needs no model at all.
|
|
2
|
+
//
|
|
3
|
+
// Measured across the recorded eval runs, six of SpecPi's ten offered tools were offered on 31 of
|
|
4
|
+
// 31 attempts and called zero times. Two of them are these: `record_harness_contract` and
|
|
5
|
+
// `finish_harness_improvement` are *authoring* tools, usable only after a human has selected a
|
|
6
|
+
// candidate through /harness-improvement. Whether a selection exists is a fact in local state, so
|
|
7
|
+
// the answer is a boolean — no latency, no cost, no false positives, and nothing for a classifier
|
|
8
|
+
// to route.
|
|
9
|
+
//
|
|
10
|
+
// `report_capability_gap` is deliberately not in this list. It is the *observation* tool and the
|
|
11
|
+
// reason the improvement loop exists; withdrawing it would silently lose the friction reports the
|
|
12
|
+
// loop is built to capture. It stays offered at all times.
|
|
13
|
+
|
|
14
|
+
/** Only usable while an improvement is selected. */
|
|
15
|
+
export const AUTHORING_TOOL_NAMES = Object.freeze(["record_harness_contract", "finish_harness_improvement"]);
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Add or remove the authoring tools without disturbing any other extension's tools. Mirrors the
|
|
19
|
+
* web-access gate: it only ever touches the names it owns, and it does nothing when the active set
|
|
20
|
+
* already matches, so a no-op never costs the cached prompt prefix.
|
|
21
|
+
*/
|
|
22
|
+
export function syncAuthoringTools(pi, selected) {
|
|
23
|
+
if (typeof pi?.getActiveTools !== "function" || typeof pi?.setActiveTools !== "function") {
|
|
24
|
+
return false;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
const owned = new Set(AUTHORING_TOOL_NAMES);
|
|
28
|
+
const active = pi.getActiveTools();
|
|
29
|
+
const present = active.filter((name) => owned.has(name));
|
|
30
|
+
if (selected && present.length === owned.size) {
|
|
31
|
+
return false;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
if (!selected && present.length === 0) {
|
|
35
|
+
return false;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
const others = active.filter((name) => !owned.has(name));
|
|
39
|
+
pi.setActiveTools(selected ? [...others, ...AUTHORING_TOOL_NAMES] : others);
|
|
40
|
+
|
|
41
|
+
return true;
|
|
42
|
+
}
|
|
@@ -13,6 +13,7 @@ import { getMarkdownTheme, type ExtensionAPI } from "@earendil-works/pi-coding-a
|
|
|
13
13
|
import { StringEnum } from "@earendil-works/pi-ai";
|
|
14
14
|
import { Box, Markdown, Text } from "@earendil-works/pi-tui";
|
|
15
15
|
import { Type } from "typebox";
|
|
16
|
+
import { syncAuthoringTools } from "./authoring-tools.mjs";
|
|
16
17
|
import {
|
|
17
18
|
appendWishlistDecision,
|
|
18
19
|
archiveWishlist,
|
|
@@ -498,6 +499,9 @@ export default function toolWishlist(pi: ExtensionAPI) {
|
|
|
498
499
|
let activeRunId = randomUUID();
|
|
499
500
|
let activeImprovement: ActiveImprovement | undefined;
|
|
500
501
|
let improvementLifecycleGeneration = 0;
|
|
502
|
+
// The authoring tools ride every request whether or not they can be used. Local state knows
|
|
503
|
+
// when they cannot be, so the answer is a boolean rather than a prediction.
|
|
504
|
+
const syncAuthoring = () => syncAuthoringTools(pi, activeImprovement !== undefined);
|
|
501
505
|
let improvementMenuGeneration = 0;
|
|
502
506
|
let finishBusy = false;
|
|
503
507
|
let contractBusy = false;
|
|
@@ -583,15 +587,18 @@ export default function toolWishlist(pi: ExtensionAPI) {
|
|
|
583
587
|
|
|
584
588
|
pi.on("session_start", (_event, ctx) => {
|
|
585
589
|
restoreActiveImprovement(ctx);
|
|
590
|
+
syncAuthoring();
|
|
586
591
|
});
|
|
587
592
|
|
|
588
593
|
pi.on("session_tree", (_event, ctx) => {
|
|
589
594
|
restoreActiveImprovement(ctx);
|
|
595
|
+
syncAuthoring();
|
|
590
596
|
});
|
|
591
597
|
|
|
592
598
|
pi.on("session_shutdown", () => {
|
|
593
599
|
improvementLifecycleGeneration += 1;
|
|
594
600
|
activeImprovement = undefined;
|
|
601
|
+
syncAuthoring();
|
|
595
602
|
});
|
|
596
603
|
|
|
597
604
|
const assertImprovementStillCurrent = (expected: ActiveImprovement, generation: number, ctx: any, signal?: any) => {
|
|
@@ -1131,6 +1138,7 @@ export default function toolWishlist(pi: ExtensionAPI) {
|
|
|
1131
1138
|
});
|
|
1132
1139
|
if (activeImprovement?.selectionId === selectedImprovement.selectionId) {
|
|
1133
1140
|
activeImprovement = undefined;
|
|
1141
|
+
syncAuthoring();
|
|
1134
1142
|
}
|
|
1135
1143
|
|
|
1136
1144
|
return {
|
|
@@ -1312,6 +1320,9 @@ export default function toolWishlist(pi: ExtensionAPI) {
|
|
|
1312
1320
|
...policy,
|
|
1313
1321
|
};
|
|
1314
1322
|
assertSelectionContextCurrent();
|
|
1323
|
+
// Pi offers tools added during a call from the next assistant message, so restoring
|
|
1324
|
+
// them here lands exactly when the selection they belong to becomes usable.
|
|
1325
|
+
syncAuthoring();
|
|
1315
1326
|
pi.appendEntry(TASK_CONTRACT_ENTRY, { kind: "cleared" });
|
|
1316
1327
|
leafId = ctx.sessionManager.getLeafId?.();
|
|
1317
1328
|
assertSelectionContextCurrent();
|
|
@@ -37,11 +37,35 @@ export const BROWSER_TOOL_NAMES = Object.freeze([
|
|
|
37
37
|
* the package's own state when it finishes, so a tool-name activation would undo itself.
|
|
38
38
|
* Delegation needs an activation path inside its own package.
|
|
39
39
|
*/
|
|
40
|
+
/**
|
|
41
|
+
* Restoring a group mid-session costs twice, and only one of those costs was ever stated.
|
|
42
|
+
*
|
|
43
|
+
* `schemaCost` is the standing price: those bytes ride every request until the group is withdrawn
|
|
44
|
+
* again. `activationCost` is the one-off, and it is much larger. Adding tool schemas partway
|
|
45
|
+
* through a session is not an additive change the provider can absorb -- it invalidates the cached
|
|
46
|
+
* prompt prefix, and the next request pays fresh input rates for the whole conversation so far.
|
|
47
|
+
*
|
|
48
|
+
* This was believed and reasoned about here for a long time and never measured. It is measured now.
|
|
49
|
+
* Three attempts on `t3-cascade-ledger` flipped the browser group on at turn 6: in all three,
|
|
50
|
+
* cached tokens collapsed to 3,200 at the next request while the prompt kept climbing, and the
|
|
51
|
+
* re-warm cost 14.6%, 21.6% and 23.9% of the attempt -- 20% on average, against a threshold of 10%
|
|
52
|
+
* fixed before the run. See `evals/runs/cache-probe/` and `scripts/cache-probe.mjs`.
|
|
53
|
+
*
|
|
54
|
+
* The same run says what to do about it: arming the same group from the first request cost 16% more
|
|
55
|
+
* than never arming it at all, against 47% for flipping mid-session. Paying up front is roughly
|
|
56
|
+
* three times cheaper than paying when the need appears.
|
|
57
|
+
*/
|
|
40
58
|
export const CAPABILITIES = Object.freeze({
|
|
41
59
|
web: {
|
|
42
60
|
label: "Web access",
|
|
43
61
|
tools: WEB_TOOL_NAMES,
|
|
44
62
|
schemaCost: "about 11 KB of tool schema per request",
|
|
63
|
+
// Not measured directly, and deliberately not extrapolated into a number. It is the larger
|
|
64
|
+
// schema, and pi-web-access is a third-party package that still carries promptSnippet and
|
|
65
|
+
// promptGuidelines, so activating it rebuilds the system prompt as well as the tool schema
|
|
66
|
+
// -- a second invalidation path Browser QA no longer has.
|
|
67
|
+
activationCost:
|
|
68
|
+
"and discards the cached prompt prefix once, which is not measured for this group but is at least as expensive as Browser QA's 20% of attempt cost, because its schema is larger and activating it also rebuilds the system prompt",
|
|
45
69
|
summary: "search the web and fetch page or source content",
|
|
46
70
|
command: "/webaccess",
|
|
47
71
|
},
|
|
@@ -49,6 +73,8 @@ export const CAPABILITIES = Object.freeze({
|
|
|
49
73
|
label: "Browser QA",
|
|
50
74
|
tools: BROWSER_TOOL_NAMES,
|
|
51
75
|
schemaCost: "about 8.7 KB of tool schema per request",
|
|
76
|
+
activationCost:
|
|
77
|
+
"and discards the cached prompt prefix once, measured at about 20% of a mid-length attempt's cost",
|
|
52
78
|
summary: "open pages in an isolated browser to verify rendering, behavior and accessibility",
|
|
53
79
|
command: "/browser",
|
|
54
80
|
},
|
|
@@ -884,7 +884,7 @@ export default function workflowControls(pi: ExtensionAPI) {
|
|
|
884
884
|
pi.registerTool({
|
|
885
885
|
name: "request_capability",
|
|
886
886
|
label: "Request Capability",
|
|
887
|
-
description: `Ask the user to restore a withdrawn SpecPi tool group for this session. Available groups — ${describeCapabilities()}. Their tools are hidden to keep each request small, so request a group only when the current task actually needs it, and continue without it if the user declines. The restored tools are usable from your next message and stay available until the session ends. Delegation is not requestable here; ask the user to run /delegate on.`,
|
|
887
|
+
description: `Ask the user to restore a withdrawn SpecPi tool group for this session. Available groups — ${describeCapabilities()}. Their tools are hidden to keep each request small, and restoring one mid-session also discards the provider's cached prompt prefix, which measured about 20% of a mid-length attempt's cost in this project's own testing — so request a group only when the current task actually needs it, and continue without it if the user declines. The restored tools are usable from your next message and stay available until the session ends. Delegation is not requestable here; ask the user to run /delegate on.`,
|
|
888
888
|
parameters: Type.Object(
|
|
889
889
|
{
|
|
890
890
|
capability: StringEnum(CAPABILITY_NAMES, {
|
|
@@ -959,7 +959,7 @@ export default function workflowControls(pi: ExtensionAPI) {
|
|
|
959
959
|
standing ||
|
|
960
960
|
(await ctx.ui.confirm(
|
|
961
961
|
`Allow ${capability.label} for this session?`,
|
|
962
|
-
`The agent asked to ${capability.summary}. Reason given: ${safeMessage(params.reason)}\n\nThis offers ${pending.length} tool${pending.length === 1 ? "" : "s"} for the rest of this session
|
|
962
|
+
`The agent asked to ${capability.summary}. Reason given: ${safeMessage(params.reason)}\n\nThis offers ${pending.length} tool${pending.length === 1 ? "" : "s"} for the rest of this session, adds ${capability.schemaCost}, ${capability.activationCost}. Withdraw them again with ${capability.command} off.`,
|
|
963
963
|
));
|
|
964
964
|
if (!accepted) {
|
|
965
965
|
return {
|
package/package.json
CHANGED
package/scripts/specpi.mjs
CHANGED
|
@@ -23,6 +23,7 @@ import { validateCapabilityRegistry } from "../extensions/tool-wishlist/registry
|
|
|
23
23
|
import { runValidator } from "../extensions/tool-wishlist/validators.mjs";
|
|
24
24
|
import { acquireSpecPiLock } from "./lock.mjs";
|
|
25
25
|
import { basePackages, checkBasePackages, installBasePackages, packageChanges, runBrowserQA } from "./packages.mjs";
|
|
26
|
+
import { applyConfig as applyGuardConfig } from "../extensions/jev-advisor/guard.mjs";
|
|
26
27
|
|
|
27
28
|
const repoRoot = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..");
|
|
28
29
|
const VERSION = JSON.parse(fs.readFileSync(path.join(repoRoot, "package.json"), "utf8")).version;
|
|
@@ -46,6 +47,24 @@ const resourcePaths = [
|
|
|
46
47
|
"extensions/tool-wishlist/registry.mjs",
|
|
47
48
|
"extensions/tool-wishlist/validators.mjs",
|
|
48
49
|
"extensions/tool-wishlist/capabilities.json",
|
|
50
|
+
"extensions/tool-wishlist/authoring-tools.mjs",
|
|
51
|
+
"extensions/jev-advisor/index.ts",
|
|
52
|
+
"extensions/jev-advisor/config.mjs",
|
|
53
|
+
"extensions/jev-advisor/consent.mjs",
|
|
54
|
+
"extensions/jev-advisor/ledger.mjs",
|
|
55
|
+
"extensions/jev-advisor/sanitize.mjs",
|
|
56
|
+
"extensions/jev-advisor/usage.mjs",
|
|
57
|
+
"extensions/jev-advisor/client.mjs",
|
|
58
|
+
"extensions/jev-advisor/broker.mjs",
|
|
59
|
+
"extensions/jev-advisor/gate.mjs",
|
|
60
|
+
"extensions/jev-advisor/guard.mjs",
|
|
61
|
+
"extensions/jev-advisor/questions/retention.mjs",
|
|
62
|
+
"extensions/jev-advisor/questions/compaction.mjs",
|
|
63
|
+
"extensions/jev-advisor/questions/gap.mjs",
|
|
64
|
+
"extensions/jev-advisor/questions/sources.mjs",
|
|
65
|
+
"extensions/jev-advisor/questions/progress.mjs",
|
|
66
|
+
"extensions/jev-advisor/questions/untrusted.mjs",
|
|
67
|
+
"extensions/jev-advisor/questions/capabilities.mjs",
|
|
49
68
|
"skills/specpi-improve/SKILL.md",
|
|
50
69
|
];
|
|
51
70
|
|
|
@@ -142,7 +161,9 @@ function validateManifestPaths(manifest) {
|
|
|
142
161
|
assertLocalPath(target, agentDir);
|
|
143
162
|
const relative = path.relative(agentDir, target).replaceAll("\\", "/");
|
|
144
163
|
if (
|
|
145
|
-
|
|
164
|
+
// jev-advisor is the one managed extension with a subdirectory, so it is listed
|
|
165
|
+
// separately rather than loosening the single-segment rule for every other family.
|
|
166
|
+
!/^(extensions\/(?:workflow-controls|tool-wishlist|browser|command-guard|delegation|background-tasks|structural-search|files|spec|specpi-ui-refresh)\/[^/]+|extensions\/jev-advisor\/(?:questions\/)?[^/]+|extensions\/spec\.ts|skills\/(?:specpi-improve|donsetch)\/SKILL\.md|themes\/(?:tea-house|specpi-spec)\.json|specpi\/pi-profiles\.sh)$/.test(
|
|
146
167
|
relative,
|
|
147
168
|
)
|
|
148
169
|
) {
|
|
@@ -333,6 +354,17 @@ async function mutate(options, operation) {
|
|
|
333
354
|
runBrowserQA(agentDir, "setup");
|
|
334
355
|
}
|
|
335
356
|
|
|
357
|
+
// specpi-jev-guard's own default is enabled:true, so a freshly installed base would
|
|
358
|
+
// start gating shell and file calls through a third-party service before anyone asked
|
|
359
|
+
// for it — and with no key it fails closed, which means a first install that refuses to
|
|
360
|
+
// run commands. The advisor rewrites this at every session start, but that only helps
|
|
361
|
+
// if the advisor loads, so the inert posture is established here at install time too.
|
|
362
|
+
try {
|
|
363
|
+
applyGuardConfig(false);
|
|
364
|
+
} catch (error) {
|
|
365
|
+
console.log(`SpecPi: could not write the Jev guard's inert settings: ${error.message}`);
|
|
366
|
+
}
|
|
367
|
+
|
|
336
368
|
packageState = {
|
|
337
369
|
basePackages,
|
|
338
370
|
packagesKeyBeforeExists: Object.hasOwn(before, "packages"),
|
package/templates/settings.json
CHANGED