gentle-pi 2.4.0 → 2.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +292 -25
- package/assets/agents/gentle-ai-worker.md +13 -0
- package/assets/agents/jd-fix-agent.md +18 -0
- package/assets/agents/jd-judge-a.md +1 -1
- package/assets/agents/jd-judge-b.md +1 -1
- package/assets/agents/sdd-apply.md +7 -5
- package/assets/agents/sdd-archive.md +5 -3
- package/assets/agents/sdd-design.md +4 -0
- package/assets/agents/sdd-explore.md +4 -0
- package/assets/agents/sdd-init.md +4 -0
- package/assets/agents/sdd-onboard.md +4 -0
- package/assets/agents/sdd-proposal.md +4 -0
- package/assets/agents/sdd-remediate.md +37 -0
- package/assets/agents/sdd-research.md +26 -3
- package/assets/agents/sdd-spec.md +4 -0
- package/assets/agents/sdd-status.md +9 -75
- package/assets/agents/sdd-sync.md +4 -0
- package/assets/agents/sdd-tasks.md +4 -0
- package/assets/agents/sdd-verify.md +5 -3
- package/assets/chains/sdd-full.chain.md +4 -0
- package/assets/chains/sdd-plan.chain.md +4 -0
- package/assets/chains/sdd-verify.chain.md +4 -0
- package/assets/migrations/managed-assets-v2.5.0.json +7 -0
- package/assets/orchestrator-delegation.md +39 -11
- package/assets/orchestrator.md +5 -5
- package/assets/sdd-orchestrator-workflow.md +54 -21
- package/assets/support/sdd-status-contract.md +34 -90
- package/contracts/telemetry/runtime-aggregate-v1.schema.json +65 -0
- package/docs/delegated-verification.md +25 -0
- package/docs/telemetry.md +94 -0
- package/docs/windows-startup-console-visibility.md +18 -0
- package/extensions/ask-user-choice.ts +159 -25
- package/extensions/codegraph-tools.ts +95 -5
- package/extensions/gentle-agents.ts +1337 -0
- package/extensions/gentle-ai.ts +2916 -386
- package/extensions/gentle-shell.ts +650 -0
- package/extensions/gentle-todo.ts +234 -0
- package/extensions/quiet-tools.ts +2 -1
- package/extensions/runtime-metrics.ts +130 -0
- package/extensions/sdd-init.ts +2 -2
- package/extensions/startup-banner.ts +52 -75
- package/lib/agent-profiles.ts +550 -0
- package/lib/agents-completion-delivery.ts +72 -0
- package/lib/agents-config.ts +315 -0
- package/lib/agents-history.ts +88 -0
- package/lib/agents-messaging.ts +187 -0
- package/lib/agents-protocol.ts +501 -0
- package/lib/agents-runner.ts +1012 -0
- package/lib/agents-thread-view.ts +57 -0
- package/lib/agents-transcript.ts +87 -0
- package/lib/agents-view-layout.ts +40 -0
- package/lib/agents-view.ts +914 -0
- package/lib/agents-widget.ts +241 -0
- package/lib/gentle-ai-binary.ts +3 -1
- package/lib/gentle-ai-renderer.ts +143 -25
- package/lib/native-choice-list.ts +194 -0
- package/lib/native-fullscreen-interaction.ts +47 -0
- package/lib/native-pointer-region.ts +164 -0
- package/lib/native-review-cli.ts +371 -13
- package/lib/orchestrator-presence.ts +337 -0
- package/lib/profiles-orchestrator.ts +203 -0
- package/lib/review-candidate-view-owner.ts +427 -0
- package/lib/review-candidate-view.ts +150 -48
- package/lib/review-consent-component.ts +247 -0
- package/lib/review-consent-ui.ts +110 -0
- package/lib/review-host-relay.ts +28 -0
- package/lib/review-integration-v2.ts +243 -11
- package/lib/review-last-event-controller.ts +8 -4
- package/lib/review-relay-contract.ts +11 -0
- package/lib/review-reminder-receipt.ts +74 -0
- package/lib/review-repository.ts +2 -2
- package/lib/review-risk-assessment.ts +339 -0
- package/lib/review-session-standing-permission-ipc.ts +309 -0
- package/lib/review-session-standing-permission.ts +240 -0
- package/lib/runtime-metrics-children.ts +199 -0
- package/lib/runtime-metrics-delivery.ts +68 -0
- package/lib/runtime-metrics-native.ts +166 -0
- package/lib/runtime-metrics-pi-identity.ts +113 -0
- package/lib/runtime-metrics-policy.ts +51 -0
- package/lib/runtime-metrics.ts +255 -0
- package/lib/sdd-preflight.ts +362 -81
- package/lib/sdd-research-capabilities.ts +228 -0
- package/lib/sdd-status.ts +29 -7
- package/lib/session-worktree-registry.ts +118 -0
- package/lib/shell-bar.ts +184 -0
- package/lib/shell-card.ts +133 -0
- package/lib/shell-changes-view.ts +530 -0
- package/lib/shell-changes.ts +290 -0
- package/lib/shell-gauge.ts +40 -0
- package/lib/shell-prompt.ts +115 -0
- package/lib/shell-sidebar-banner.ts +11 -0
- package/lib/shell-sidebar-layout.ts +213 -0
- package/lib/shell-sidebar.ts +41 -0
- package/lib/shell-todo.ts +297 -0
- package/lib/shell-usage-view.ts +76 -0
- package/lib/shell-usage.ts +246 -0
- package/lib/telemetry-trigger.ts +153 -0
- package/package.json +8 -5
- package/runtime/gentle-ai-binary.mjs +3 -1
- package/runtime/native-review-cli.mjs +370 -12
- package/runtime/review-integration-v2.mjs +243 -11
- package/runtime/review-relay-contract.mjs +11 -0
- package/runtime/review-risk-assessment.mjs +340 -0
- package/runtime/telemetry-trigger.mjs +154 -0
- package/scripts/build-runtime-modules.mjs +11 -1
- package/scripts/check-types.mjs +125 -0
- package/scripts/gentle-ai-installer.mjs +10 -10
- package/scripts/install-gentle-ai.mjs +12 -0
- package/scripts/install-tui-mode-setting.mjs +114 -0
- package/scripts/test-packed-runner.mjs +38 -2
- package/scripts/types-baseline.json +99 -0
- package/scripts/verify-package-files.mjs +8 -2
- package/skills/_shared/review-ledger-contract.md +20 -2
- package/skills/issue-creation/SKILL.md +3 -3
- package/skills/judgment-day/SKILL.md +17 -3
- package/skills/judgment-day/references/prompts-and-formats.md +14 -3
- package/tests/agent-profiles.test.ts +722 -0
- package/tests/agents-completion-delivery.test.ts +94 -0
- package/tests/agents-config.test.ts +205 -0
- package/tests/agents-fake-child.ts +66 -0
- package/tests/agents-grouping.test.ts +179 -0
- package/tests/agents-history.test.ts +54 -0
- package/tests/agents-integration.test.ts +100 -0
- package/tests/agents-messaging.test.ts +94 -0
- package/tests/agents-protocol.test.ts +198 -0
- package/tests/agents-queries.test.ts +190 -0
- package/tests/agents-responsive.test.ts +43 -0
- package/tests/agents-runner-process.test.ts +111 -0
- package/tests/agents-runner.test.ts +959 -0
- package/tests/agents-thread-view.test.ts +45 -0
- package/tests/agents-transcript.test.ts +30 -0
- package/tests/agents-view.test.ts +685 -0
- package/tests/agents-widget.test.ts +141 -0
- package/tests/artifact-language.test.ts +25 -2
- package/tests/ask-user-choice.test.ts +325 -5
- package/tests/asset-installation-runtime.test.ts +108 -0
- package/tests/autonomous-guard.test.ts +116 -1
- package/tests/codegraph-tools.test.ts +112 -2
- package/tests/delegated-key-learnings-contract.test.ts +1 -1
- package/tests/devbinary/native-review-parity.devtest.ts +110 -0
- package/tests/feature-request-form.test.ts +67 -0
- package/tests/fixtures/agents-messaging-child.mjs +5 -0
- package/tests/fixtures/agents-process-child.mjs +23 -0
- package/tests/fixtures/runtime-metrics-native-batches.json +6 -0
- package/tests/gentle-agents.test.ts +2168 -0
- package/tests/gentle-ai-binary.test.ts +7 -2
- package/tests/gentle-ai-installer.test.ts +47 -47
- package/tests/gentle-ai-renderer.test.ts +103 -0
- package/tests/gentle-ai.test.ts +971 -15
- package/tests/gentle-card-text.ts +35 -0
- package/tests/gentle-shell.test.ts +818 -0
- package/tests/gentle-todo.test.ts +226 -0
- package/tests/install-tui-mode-setting.test.ts +324 -0
- package/tests/issue-creation-skill.test.ts +22 -0
- package/tests/model-routing-authority.test.ts +12 -0
- package/tests/native-choice-list.test.ts +202 -0
- package/tests/native-fullscreen-interaction.test.ts +125 -0
- package/tests/native-pointer-region.test.ts +245 -0
- package/tests/native-review-capability-contract.test.ts +27 -1
- package/tests/native-review-cli.test.ts +317 -3
- package/tests/native-review-consent.test.ts +91 -0
- package/tests/native-review-parity-runtime.test.ts +8 -2
- package/tests/native-review-parity.test.ts +43 -29
- package/tests/native-sdd-attempt-authority.test.ts +7 -2
- package/tests/orchestrator-budget.test.ts +69 -0
- package/tests/orchestrator-presence.test.ts +389 -0
- package/tests/orchestrator-rdd-ownership.test.ts +9 -0
- package/tests/package-manifest.test.ts +243 -7
- package/tests/profiles-orchestrator.test.ts +208 -0
- package/tests/quiet-tool-rendering.test.ts +97 -37
- package/tests/rdd-aware-verification-contract.test.ts +226 -0
- package/tests/rdd-status-line.test.ts +286 -0
- package/tests/review-agent-end-preflight.test.ts +332 -24
- package/tests/review-candidate-view.test.ts +751 -7
- package/tests/review-consent-ui.test.ts +352 -0
- package/tests/review-contract-prompt.test.ts +17 -0
- package/tests/review-controller-native-recovery.test.ts +29 -4
- package/tests/review-controller-native-routing.test.ts +884 -7
- package/tests/review-controller-workspace-root.test.ts +45 -2
- package/tests/review-controller.test.ts +26 -1
- package/tests/review-host-relay-restart-parity.test.ts +142 -1
- package/tests/review-host-relay-routing.test.ts +384 -8
- package/tests/review-host-relay.test.ts +29 -0
- package/tests/review-integration-v2-forward.test.ts +44 -0
- package/tests/review-integration-v2.test.ts +276 -0
- package/tests/review-last-event-closure.test.ts +112 -3
- package/tests/review-ledger-contract.test.ts +61 -6
- package/tests/review-relay-contract.test.ts +26 -0
- package/tests/review-reminder-receipt.test.ts +62 -0
- package/tests/review-repository.test.ts +28 -1
- package/tests/review-risk-assessment.test.ts +626 -0
- package/tests/review-session-standing-permission-controller.test.ts +656 -0
- package/tests/review-session-standing-permission-ipc.test.ts +233 -0
- package/tests/review-session-standing-permission-runtime.test.ts +212 -0
- package/tests/review-session-standing-permission.test.ts +156 -0
- package/tests/runtime-harness.mjs +447 -39
- package/tests/runtime-metrics-children.test.ts +206 -0
- package/tests/runtime-metrics-delivery.test.ts +85 -0
- package/tests/runtime-metrics-extension.test.ts +187 -0
- package/tests/runtime-metrics-native.test.ts +209 -0
- package/tests/runtime-metrics-pi-identity.test.ts +113 -0
- package/tests/runtime-metrics-policy.test.ts +62 -0
- package/tests/runtime-metrics.test.ts +184 -0
- package/tests/sdd-agent-tools.test.ts +10 -1
- package/tests/sdd-execution-routing-contract.test.ts +28 -0
- package/tests/sdd-managed-runtime-settlement.test.ts +331 -0
- package/tests/sdd-native-managed-uptake.test.ts +253 -0
- package/tests/sdd-planning-routing-contract.test.ts +45 -0
- package/tests/sdd-preflight.test.ts +252 -8
- package/tests/sdd-research-capabilities.test.ts +256 -0
- package/tests/sdd-research-live.test.ts +241 -0
- package/tests/sdd-selection-transport.test.ts +504 -0
- package/tests/sdd-status.test.ts +51 -0
- package/tests/session-worktree-registry.test.ts +135 -0
- package/tests/shell-bar.test.ts +176 -0
- package/tests/shell-card.test.ts +139 -0
- package/tests/shell-changes-view.test.ts +609 -0
- package/tests/shell-changes.test.ts +350 -0
- package/tests/shell-prompt.test.ts +140 -0
- package/tests/shell-sidebar-banner.test.ts +23 -0
- package/tests/shell-sidebar-layout.test.ts +387 -0
- package/tests/shell-sidebar.test.ts +50 -0
- package/tests/shell-todo.test.ts +259 -0
- package/tests/shell-usage-view.test.ts +62 -0
- package/tests/shell-usage.test.ts +197 -0
- package/tests/startup-banner.test.ts +126 -0
- package/tests/telemetry-trigger.test.ts +351 -0
|
@@ -0,0 +1,241 @@
|
|
|
1
|
+
import assert from "node:assert/strict";
|
|
2
|
+
import { spawn } from "node:child_process";
|
|
3
|
+
import { createHash } from "node:crypto";
|
|
4
|
+
import { appendFileSync, existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
|
|
5
|
+
import { tmpdir } from "node:os";
|
|
6
|
+
import { dirname, isAbsolute, join } from "node:path";
|
|
7
|
+
import { fileURLToPath } from "node:url";
|
|
8
|
+
import test from "node:test";
|
|
9
|
+
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
|
10
|
+
import { AgentRunner } from "../lib/agents-runner.ts";
|
|
11
|
+
import { TaskStore } from "../lib/agents-protocol.ts";
|
|
12
|
+
import { researchAgent, RESEARCH_CHILD_TOOLS_ENV } from "../lib/sdd-research-capabilities.ts";
|
|
13
|
+
|
|
14
|
+
// Runtime-owned authentication references the existing profile; no credentials
|
|
15
|
+
// are copied, extracted or symlinked. Only native OAuth refresh may persist there.
|
|
16
|
+
// All resource discovery, settings, model caches and sessions remain isolated.
|
|
17
|
+
const enabled = process.env.GENTLE_PI_LIVE_RESEARCH_TEST === "1";
|
|
18
|
+
const role = process.env.GENTLE_PI_LIVE_RESEARCH_ROLE;
|
|
19
|
+
const tools = ["web_search", "source_check", "fetch_content", "get_search_content"];
|
|
20
|
+
const candidate = fileURLToPath(new URL("../extensions/gentle-agents.ts", import.meta.url));
|
|
21
|
+
const self = fileURLToPath(import.meta.url);
|
|
22
|
+
const question = `Generic runtime capability probe, not an SDD workflow or proposal admission. Artifact store: none. Do not write files or launch agents. Use ALL FOUR tools web_search, source_check, fetch_content and get_search_content to answer: What does the Node.js fs module provide? Set web_search workflow to none. Search only public Node.js documentation (site:nodejs.org). Check and retrieve the original public documentation. Return ONLY JSON with source_url (an https://nodejs.org/ URL) and passage (a verbatim 40-300 character passage from retrieved documentation). Do not use remembered text as evidence. If any tool fails, report inability rather than inventing evidence.`;
|
|
23
|
+
|
|
24
|
+
function record(value: Record<string, unknown>): void {
|
|
25
|
+
appendFileSync(process.env.GENTLE_PI_LIVE_RESEARCH_TRACE!, `${JSON.stringify({ role, ...value })}\n`, { mode: 0o600 });
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
// Only the eval entry invokes launchRpc; loading this file as an extension does
|
|
29
|
+
// not construct another SDK runtime or extension registry.
|
|
30
|
+
const launcher = { command: process.execPath, args: ["--experimental-strip-types", "--input-type=module", "--eval", `import(${JSON.stringify(import.meta.url)}).then(m => m.launchRpc())`, "--"] };
|
|
31
|
+
|
|
32
|
+
export async function launchRpc(): Promise<void> {
|
|
33
|
+
let stage = "arguments";
|
|
34
|
+
try {
|
|
35
|
+
assert.ok(enabled && (role === "host" || role === "child"));
|
|
36
|
+
const args = process.argv.slice(1), values = new Map<string, string>();
|
|
37
|
+
for (let i = 0; i < args.length; i += 2) {
|
|
38
|
+
assert.ok(["--mode", "--session-dir", "--tools", "--append-system-prompt"].includes(args[i]) && args[i + 1] !== undefined, "Unsupported runner argument");
|
|
39
|
+
assert.ok(!values.has(args[i]), "Duplicate runner argument");
|
|
40
|
+
values.set(args[i], args[i + 1]);
|
|
41
|
+
}
|
|
42
|
+
assert.equal(values.get("--mode"), "rpc");
|
|
43
|
+
stage = "sdk_load";
|
|
44
|
+
const { ModelRuntime, SettingsManager, SessionManager, createAgentSessionServices, createAgentSessionFromServices, createAgentSessionRuntime, runRpcMode } = await import("@earendil-works/pi-coding-agent");
|
|
45
|
+
const agentDir = process.env.PI_CODING_AGENT_DIR!, authPath = process.env.GENTLE_PI_LIVE_RESEARCH_AUTH_PATH!;
|
|
46
|
+
assert.ok(isAbsolute(authPath) && existsSync(authPath), "Existing runtime auth storage required");
|
|
47
|
+
stage = "native_auth_and_model_catalog";
|
|
48
|
+
const modelRuntime = await ModelRuntime.create({ authPath, modelsPath: join(agentDir, "models.json"), modelsStorePath: join(agentDir, "models-store.json"), allowModelNetwork: false, signal: AbortSignal.timeout(30_000) });
|
|
49
|
+
const model = modelRuntime.getModel(process.env.PI_PROVIDER!, process.env.PI_MODEL!);
|
|
50
|
+
assert.ok(model, "Inherited model unavailable; no fallback permitted");
|
|
51
|
+
stage = "single_extension_loader";
|
|
52
|
+
const runtime = await createAgentSessionRuntime(async ({ cwd, sessionManager, sessionStartEvent }) => {
|
|
53
|
+
const settingsManager = SettingsManager.create(cwd, agentDir);
|
|
54
|
+
settingsManager.applyOverrides({ retry: { enabled: false }, compaction: { enabled: false } });
|
|
55
|
+
const services = await createAgentSessionServices({ cwd, agentDir, modelRuntime, settingsManager, resourceLoaderOptions: {
|
|
56
|
+
noExtensions: true, noSkills: true, noPromptTemplates: true, noThemes: true, noContextFiles: true,
|
|
57
|
+
additionalExtensionPaths: [process.env.GENTLE_PI_LIVE_RESEARCH_WEB_EXTENSION!, candidate, self],
|
|
58
|
+
appendSystemPrompt: values.has("--append-system-prompt") ? [values.get("--append-system-prompt")!] : [],
|
|
59
|
+
} });
|
|
60
|
+
assert.equal(services.resourceLoader.getExtensions().errors.length, 0, "Extension load failed");
|
|
61
|
+
assert.ok(!services.diagnostics.some(item => item.type === "error"), "Runtime service diagnostics failed");
|
|
62
|
+
return { ...(await createAgentSessionFromServices({ services, sessionManager, sessionStartEvent, model,
|
|
63
|
+
thinkingLevel: settingsManager.getDefaultThinkingLevel(), tools: values.get("--tools")!.split(","),
|
|
64
|
+
})), services, diagnostics: services.diagnostics };
|
|
65
|
+
}, { cwd: process.cwd(), agentDir, sessionManager: SessionManager.create(process.cwd(), values.get("--session-dir")) });
|
|
66
|
+
stage = "native_rpc";
|
|
67
|
+
runtime.session.subscribe(event => {
|
|
68
|
+
if (role === "child" && event.type === "agent_settled") record({ event: "child_settled" });
|
|
69
|
+
});
|
|
70
|
+
await runRpcMode(runtime);
|
|
71
|
+
} catch (error) {
|
|
72
|
+
// Never serialize provider exceptions: they may contain credentials or bodies.
|
|
73
|
+
record({ event: "launcher_failed", stage, assertionFailed: error instanceof assert.AssertionError });
|
|
74
|
+
process.exitCode = 1;
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
// The launcher-owned ResourceLoader registers the real installed extensions once.
|
|
79
|
+
export default function liveProbe(pi: ExtensionAPI): void {
|
|
80
|
+
if (!enabled || (role !== "host" && role !== "child")) return;
|
|
81
|
+
const results: string[] = [];
|
|
82
|
+
const completed = new Set<string>();
|
|
83
|
+
pi.on("session_start", (_event, ctx) => {
|
|
84
|
+
const loaded = pi.getAllTools().filter(tool => tools.includes(tool.name));
|
|
85
|
+
record({ event: "loaded", candidate, candidateHash: createHash("sha256").update(readFileSync(candidate)).digest("hex"), tools: loaded.map(tool => ({ name: tool.name, source: tool.sourceInfo?.path })), active: pi.getActiveTools().filter(name => tools.includes(name)), modelMatches: ctx.model?.id === process.env.PI_MODEL && ctx.model?.provider === process.env.PI_PROVIDER });
|
|
86
|
+
});
|
|
87
|
+
if (role === "child") {
|
|
88
|
+
pi.on("before_agent_start", event => {
|
|
89
|
+
// One block comes from researchAgent; the second is injected by the
|
|
90
|
+
// candidate extension's real child-local capability hook loaded first.
|
|
91
|
+
record({ event: "candidate_hook", observed: (event.systemPrompt.match(/## SDD Research Capabilities/g) ?? []).length >= 2 });
|
|
92
|
+
});
|
|
93
|
+
pi.on("tool_call", event => {
|
|
94
|
+
if (!tools.includes(event.toolName)) return { block: true, reason: "Public web probe permits only its four evidence tools." };
|
|
95
|
+
if (event.toolName === "web_search" && event.input.workflow !== "none") return { block: true, reason: "Public probe requires web_search workflow none." };
|
|
96
|
+
const urls = JSON.stringify(event.input).match(/https?:\/\/[^\s"<>\\]+/g) ?? [];
|
|
97
|
+
if (urls.some(url => { try { const parsed = new URL(url); return parsed.protocol !== "https:" || parsed.hostname !== "nodejs.org"; } catch { return true; } })) {
|
|
98
|
+
return { block: true, reason: "This probe retrieves only public nodejs.org URLs." };
|
|
99
|
+
}
|
|
100
|
+
});
|
|
101
|
+
pi.on("tool_result", event => {
|
|
102
|
+
record({ event: "tool_result", tool: event.toolName, success: !event.isError });
|
|
103
|
+
if (!event.isError && tools.includes(event.toolName)) {
|
|
104
|
+
completed.add(event.toolName);
|
|
105
|
+
if (event.toolName === "fetch_content" || event.toolName === "get_search_content") {
|
|
106
|
+
results.push(event.content.filter(part => part.type === "text").map(part => part.text).join("\n"));
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
});
|
|
110
|
+
pi.on("agent_end", event => {
|
|
111
|
+
const last = [...event.messages].reverse().find(message => message.role === "assistant");
|
|
112
|
+
const text = last?.content.filter(part => part.type === "text").map(part => part.text).join("\n") ?? "";
|
|
113
|
+
try {
|
|
114
|
+
const parsed = JSON.parse(text);
|
|
115
|
+
const url = new URL(parsed.source_url);
|
|
116
|
+
const passage = parsed.passage;
|
|
117
|
+
const valid = url.protocol === "https:" && url.hostname === "nodejs.org" && typeof passage === "string" && passage.length >= 40 && passage.length <= 300 && results.some(result => result.includes(passage) && result.includes(parsed.source_url)) && tools.every(name => completed.has(name));
|
|
118
|
+
// Persist only a passage proven present alongside the public citation,
|
|
119
|
+
// never raw provider errors, tool arguments or arbitrary model output.
|
|
120
|
+
record({ event: "evidence", valid, ...(valid ? { source_url: parsed.source_url, passage } : {}) });
|
|
121
|
+
} catch { record({ event: "evidence", valid: false }); }
|
|
122
|
+
});
|
|
123
|
+
return;
|
|
124
|
+
}
|
|
125
|
+
pi.registerCommand("gentle-live-research-probe", {
|
|
126
|
+
description: "Run the opt-in generic public web capability integration probe.",
|
|
127
|
+
handler: async (_args, ctx) => {
|
|
128
|
+
if (ctx.model?.id !== process.env.PI_MODEL || ctx.model?.provider !== process.env.PI_PROVIDER) {
|
|
129
|
+
record({ event: "prerequisite_failed", reason: "inherited model is unavailable in the isolated profile" });
|
|
130
|
+
ctx.shutdown(); return;
|
|
131
|
+
}
|
|
132
|
+
const selection = { "open-web": { tools, extensions: Object.fromEntries(tools.map(name => [name, process.env.GENTLE_PI_LIVE_RESEARCH_WEB_EXTENSION!])) } };
|
|
133
|
+
const mapped = researchAgent({ name: "runtime-research-probe", description: "Public-only generic capability probe", tools, instructions: question } as never, pi, selection);
|
|
134
|
+
if (mapped.capabilities["open-web"].status !== "available") {
|
|
135
|
+
record({ event: "prerequisite_failed", reason: "four active approved installed web tools required" });
|
|
136
|
+
ctx.shutdown(); return;
|
|
137
|
+
}
|
|
138
|
+
record({ event: "grants", capabilities: mapped.capabilities, childTools: mapped.agent.tools });
|
|
139
|
+
const runner = new AgentRunner(new TaskStore(), { maxConcurrency: 1, stallTimeoutMs: 120_000 }, {
|
|
140
|
+
pi: launcher,
|
|
141
|
+
now: Date.now,
|
|
142
|
+
schedule: (fn, ms) => { const timer = setTimeout(fn, ms); return () => clearTimeout(timer); },
|
|
143
|
+
spawn: (command, args, options) => {
|
|
144
|
+
const child = spawn(command, args, { ...options, stdio: ["pipe", "pipe", "pipe"] });
|
|
145
|
+
record({ event: "spawn", pid: child.pid, candidate });
|
|
146
|
+
child.on("exit", (code, signal) => record({ event: "child_exit", pid: child.pid, code, signal }));
|
|
147
|
+
return child;
|
|
148
|
+
},
|
|
149
|
+
}, { askUser: async () => ({ cancelled: true }) });
|
|
150
|
+
const timer = setTimeout(() => runner.cancelAll(), 145_000);
|
|
151
|
+
try {
|
|
152
|
+
const task = runner.run({ agent: mapped.agent, prompt: question, label: "Public Node.js docs probe", context: undefined, mode: "task", cwd: ctx.cwd, parentSessionId: ctx.sessionManager.getSessionId(), model: undefined, thinking: undefined, sessionDir: join(ctx.cwd, "sessions"), resumeSessionPath: undefined, env: { ...process.env, GENTLE_PI_LIVE_RESEARCH_ROLE: "child", [RESEARCH_CHILD_TOOLS_ENV]: JSON.stringify(mapped.agent.tools), GENTLE_PI_RESEARCH_SELECTION: JSON.stringify(selection) } });
|
|
153
|
+
const outcome = await runner.waitFor(task.id);
|
|
154
|
+
record({ event: "runner_result", status: outcome.status });
|
|
155
|
+
} finally { clearTimeout(timer); runner.cancelAll(); ctx.shutdown(); }
|
|
156
|
+
},
|
|
157
|
+
});
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
if (role === undefined) test("LIVE installed web tools execute through the candidate research runner", { skip: !enabled && "opt in with GENTLE_PI_LIVE_RESEARCH_TEST=1; no live execution performed", timeout: 180_000 }, async t => {
|
|
161
|
+
const web = process.env.GENTLE_PI_LIVE_RESEARCH_WEB_EXTENSION;
|
|
162
|
+
assert.ok(web && isAbsolute(web) && existsSync(web), "Prerequisite: GENTLE_PI_LIVE_RESEARCH_WEB_EXTENSION must identify an already-installed web-tool extension.");
|
|
163
|
+
assert.ok(process.env.PI_MODEL && process.env.PI_PROVIDER, "Prerequisites: inherit PI_MODEL and PI_PROVIDER from the selected installed runtime; do not choose an ad hoc model.");
|
|
164
|
+
const { getAgentDir } = await import("@earendil-works/pi-coding-agent");
|
|
165
|
+
const authPath = join(getAgentDir(), "auth.json");
|
|
166
|
+
assert.ok(existsSync(authPath), "Existing runtime-owned auth storage required; no login is performed.");
|
|
167
|
+
const root = mkdtempSync(join(tmpdir(), "gentle-live-research-"));
|
|
168
|
+
const profile = join(root, "profile"), trace = join(root, "logs", "trace.jsonl");
|
|
169
|
+
mkdirSync(profile, { recursive: true }); mkdirSync(dirname(trace)); mkdirSync(join(root, "sessions"));
|
|
170
|
+
writeFileSync(trace, "", { mode: 0o600 });
|
|
171
|
+
writeFileSync(join(profile, "settings.json"), JSON.stringify({ defaultProvider: process.env.PI_PROVIDER, defaultModel: process.env.PI_MODEL, ...(process.env.PI_REASONING_LEVEL ? { defaultThinkingLevel: process.env.PI_REASONING_LEVEL } : {}), packages: [] }), { mode: 0o600 });
|
|
172
|
+
const child = spawn(launcher.command, [...launcher.args, "--mode", "rpc", "--session-dir", join(root, "sessions"), "--tools", tools.join(",")], {
|
|
173
|
+
cwd: root, detached: process.platform !== "win32", stdio: ["pipe", "pipe", "pipe"],
|
|
174
|
+
env: { ...process.env, HOME: root, USERPROFILE: root, XDG_CONFIG_HOME: profile, XDG_CACHE_HOME: join(root, "cache"), TMPDIR: root, PI_CODING_AGENT_DIR: profile, GENTLE_PI_AGENT_HOME: profile, GENTLE_PI_AGENTS_CHILD: "0", GENTLE_PI_LIVE_RESEARCH_ROLE: "host", GENTLE_PI_LIVE_RESEARCH_TRACE: trace, GENTLE_PI_LIVE_RESEARCH_AUTH_PATH: authPath },
|
|
175
|
+
});
|
|
176
|
+
child.stderr.resume(); child.stdin.on("error", () => {});
|
|
177
|
+
let hostCommandCompleted = false, rpcBuffer = "";
|
|
178
|
+
child.stdout.setEncoding("utf8");
|
|
179
|
+
child.stdout.on("data", (chunk: string) => {
|
|
180
|
+
rpcBuffer += chunk;
|
|
181
|
+
let newline: number;
|
|
182
|
+
while ((newline = rpcBuffer.indexOf("\n")) !== -1) {
|
|
183
|
+
const line = rpcBuffer.slice(0, newline); rpcBuffer = rpcBuffer.slice(newline + 1);
|
|
184
|
+
try {
|
|
185
|
+
const response = JSON.parse(line);
|
|
186
|
+
if (response.type !== "response" || response.id !== "probe" || response.command !== "prompt") continue;
|
|
187
|
+
hostCommandCompleted = response.success === true;
|
|
188
|
+
// AgentSession acknowledges extension prompts only after the handler returns.
|
|
189
|
+
// ctx.shutdown merely sets a flag; RPC EOF invokes native disposal even
|
|
190
|
+
// when this host never produces an agent_settled event. Errors still fail.
|
|
191
|
+
child.stdin.end();
|
|
192
|
+
} catch { /* Ignore non-JSON output without persisting potentially private text. */ }
|
|
193
|
+
}
|
|
194
|
+
});
|
|
195
|
+
const rows = () => {
|
|
196
|
+
const bytes = readFileSync(trace, "utf8");
|
|
197
|
+
return bytes.slice(0, bytes.lastIndexOf("\n") + 1).split("\n").filter(Boolean).map(line => JSON.parse(line));
|
|
198
|
+
};
|
|
199
|
+
let diagnosed = false;
|
|
200
|
+
const reportDiagnostics = () => {
|
|
201
|
+
if (diagnosed) return;
|
|
202
|
+
diagnosed = true;
|
|
203
|
+
// Every retained trace event is authored by this harness. Evidence contains
|
|
204
|
+
// public text only after passage/citation validation; provider output is excluded.
|
|
205
|
+
t.diagnostic(JSON.stringify({ candidate, hostCommandCompleted, hostExit: child.exitCode, hostSignal: child.signalCode, trace: rows() }));
|
|
206
|
+
};
|
|
207
|
+
const kill = (pid: number) => { try { process.kill(process.platform === "win32" ? pid : -pid, "SIGKILL"); } catch { /* Already exited. */ } };
|
|
208
|
+
const stopLiveProcesses = () => {
|
|
209
|
+
const observed = rows();
|
|
210
|
+
for (const row of observed) if (row.event === "spawn" && row.pid && !observed.some(exit => exit.event === "child_exit" && exit.pid === row.pid)) kill(row.pid);
|
|
211
|
+
if (child.pid && child.exitCode === null && child.signalCode === null) kill(child.pid);
|
|
212
|
+
};
|
|
213
|
+
const timer = setTimeout(stopLiveProcesses, 170_000);
|
|
214
|
+
try {
|
|
215
|
+
const exited = new Promise<number | null>((resolve, reject) => { child.once("exit", resolve); child.once("error", () => reject(new Error("Installed Pi launcher could not start."))); });
|
|
216
|
+
child.stdin.write(`${JSON.stringify({ id: "probe", type: "prompt", message: "/gentle-live-research-probe" })}\n`);
|
|
217
|
+
const exit = await exited;
|
|
218
|
+
const observed = rows();
|
|
219
|
+
reportDiagnostics();
|
|
220
|
+
assert.ok(hostCommandCompleted, "Host extension command did not acknowledge successful completion.");
|
|
221
|
+
assert.ok(!observed.some(row => row.event === "launcher_failed"), "SDK launcher failed; see sanitized stage diagnostics.");
|
|
222
|
+
assert.equal(exit, 0, "Live host did not exit cleanly within the bound.");
|
|
223
|
+
assert.ok(!observed.some(row => row.event === "prerequisite_failed"), "Live prerequisites failed: check installed web tools and existing runtime-owned model authentication; no evidence claimed.");
|
|
224
|
+
assert.ok(observed.some(row => row.role === "child" && row.event === "loaded" && row.candidate === candidate && row.modelMatches && row.candidateHash === createHash("sha256").update(readFileSync(candidate)).digest("hex")), "Child did not load the expected candidate and inherited model.");
|
|
225
|
+
assert.ok(observed.some(row => row.event === "candidate_hook" && row.observed), "Candidate child-local capability hook did not execute.");
|
|
226
|
+
for (const name of tools) assert.ok(observed.some(row => row.role === "child" && row.event === "tool_result" && row.tool === name && row.success), `No successful real execution of ${name}.`);
|
|
227
|
+
assert.ok(observed.some(row => row.event === "runner_result" && row.status === "completed"), "Candidate AgentRunner did not complete.");
|
|
228
|
+
assert.ok(observed.some(row => row.event === "child_settled"), "Native child settlement was not observed.");
|
|
229
|
+
// AgentRunner requestStop sends SIGTERM even for completed tasks; native
|
|
230
|
+
// runRpcMode translates SIGTERM to exit 143. Neither code proves completion.
|
|
231
|
+
assert.ok(observed.some(row => row.event === "child_exit" && (row.code === 0 || row.code === 143 || row.signal === "SIGTERM")), "Child did not exit through normal or runner terminal cleanup.");
|
|
232
|
+
const evidence = observed.find(row => row.event === "evidence" && row.valid);
|
|
233
|
+
assert.ok(evidence, "No source-backed public passage and citation were verified.");
|
|
234
|
+
t.diagnostic(JSON.stringify({ candidate, executions: observed.filter(row => row.role === "child" && row.event === "tool_result").map(row => ({ tool: row.tool, success: row.success })), source_url: evidence.source_url, passage: evidence.passage, hostExit: exit, childExit: observed.find(row => row.event === "child_exit")?.code, childSignal: observed.find(row => row.event === "child_exit")?.signal }));
|
|
235
|
+
} finally {
|
|
236
|
+
clearTimeout(timer);
|
|
237
|
+
reportDiagnostics();
|
|
238
|
+
stopLiveProcesses();
|
|
239
|
+
rmSync(root, { recursive: true, force: true });
|
|
240
|
+
}
|
|
241
|
+
});
|