pomerado 0.1.2 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +45 -0
- package/LICENSE +21 -661
- package/README.md +12 -9
- package/dist/typescript/authoring/auth/SKILL.md +21 -159
- package/dist/typescript/authoring/caller-input/SKILL.md +7 -68
- package/dist/typescript/authoring/core/SKILL.md +40 -330
- package/dist/typescript/authoring/forms/SKILL.md +7 -78
- package/dist/typescript/authoring/pagination/SKILL.md +5 -22
- package/dist/typescript/authoring/workspace/AGENTS.md +53 -293
- package/dist/typescript/authoring/workspace/README.md +2 -10
- package/dist/typescript/authoring/writes/SKILL.md +12 -140
- package/dist/typescript/src/execution/sign-in-diagnostics.d.ts +19 -20
- package/dist/typescript/src/guardian/openai.js +3 -1
- package/dist/typescript/src/mint/contracts.d.ts +36 -9
- package/dist/typescript/src/mint/harness.js +80 -8
- package/dist/typescript/src/mint/openai.js +4 -2
- package/dist/typescript/src/mint/skills.d.ts +4 -0
- package/dist/typescript/src/mint/skills.js +60 -61
- package/dist/typescript/src/runtime/input-request.d.ts +1 -0
- package/dist/typescript/src/runtime/input-request.js +2 -1
- package/dist/typescript/src/runtime/provider-metadata.d.ts +7 -6
- package/dist/typescript/src/runtime/script-input.d.ts +2 -2
- package/dist/typescript/src/runtime/script-input.js +13 -3
- package/dist/typescript/src/standalone/mcp-cli.js +13 -2
- package/dist/typescript/src/standalone/mcp-package.js +73 -23
- package/package.json +3 -2
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
import type { ExecutionBoundaryError } from "./boundary.js";
|
|
2
2
|
import type { IdentityErrorCode } from "../runtime/runner-codes.js";
|
|
3
3
|
import type { ManagedAuthErrorCode } from "../destinations/sign-in-provider-codes.js";
|
|
4
|
-
import type { RetainedProviderCode, RetainedProviderStage, RetainedProviderSchemaField,
|
|
4
|
+
import type { RetainedProviderCode, RetainedProviderStage, RetainedProviderSchemaField, RetainedNetworkError } from "../runtime/provider-metadata.js";
|
|
5
5
|
import type { NoResponseError } from "../destinations/navigation-failure.js";
|
|
6
6
|
import type { CaptureCollectionDiagnostic, CaptureFailureReason, CaptureScreeningDiagnostic } from "../runtime/capture-diagnostic.js";
|
|
7
|
-
type
|
|
7
|
+
type ProviderAuthErrorCode = typeof ManagedAuthErrorCode.Type | "unknown";
|
|
8
8
|
type SignInFailureCode = "AccountMismatch" | "AuthenticationFailed" | "CredentialsRejected" | "CaptureUnavailable"
|
|
9
9
|
/**
|
|
10
10
|
* The sign-in went where credentials may not go: off the site's registrable domain and every
|
|
@@ -44,31 +44,30 @@ export interface SignInDiagnostic {
|
|
|
44
44
|
readonly providerStatus?: number;
|
|
45
45
|
readonly providerStage?: RetainedProviderStage;
|
|
46
46
|
readonly providerSchemaField?: RetainedProviderSchemaField;
|
|
47
|
-
readonly providerAuthCode?:
|
|
47
|
+
readonly providerAuthCode?: ProviderAuthErrorCode;
|
|
48
48
|
/** The site answered the failed login's latest page load 429, a rate limit. */
|
|
49
49
|
readonly siteRateLimited?: true;
|
|
50
50
|
/**
|
|
51
|
-
* The failed login's latest page load failed
|
|
52
|
-
* evidence from the login browser's capture
|
|
53
|
-
* login's cleanup.
|
|
51
|
+
* The failed login's latest page load failed in the provider's network layer: typed network or
|
|
52
|
+
* provider evidence from the login browser's capture, read once before the login's cleanup.
|
|
54
53
|
*/
|
|
55
|
-
readonly loginProxyFailure?:
|
|
54
|
+
readonly loginProxyFailure?: RetainedNetworkError | NoResponseError;
|
|
56
55
|
/**
|
|
57
|
-
* An iframe document of the login that
|
|
58
|
-
* sign-in started it moves nothing: context for the agent, never a verdict.
|
|
56
|
+
* An iframe document of the login that the provider's network layer failed with provider
|
|
57
|
+
* evidence. Once a sign-in started it moves nothing: context for the agent, never a verdict.
|
|
59
58
|
*/
|
|
60
59
|
readonly loginFrameRefusal?: {
|
|
61
60
|
readonly origin: string;
|
|
62
|
-
readonly cause:
|
|
61
|
+
readonly cause: RetainedNetworkError | NoResponseError;
|
|
63
62
|
};
|
|
64
63
|
/**
|
|
65
|
-
* Set by the mint host once
|
|
66
|
-
*
|
|
64
|
+
* Set by the mint host once it replaced the browser after this failure, so the agent's answer
|
|
65
|
+
* says the browser moved only when it did.
|
|
67
66
|
*/
|
|
68
67
|
readonly hostMovedBrowser?: true;
|
|
69
|
-
/**
|
|
68
|
+
/** The provider's account of a failed or expired login flow; see `RetainedLoginFailureEvidence`. */
|
|
70
69
|
readonly providerEvidence?: RetainedLoginFailureEvidence;
|
|
71
|
-
/** One line for the agent and the caller:
|
|
70
|
+
/** One line for the agent and the caller: the provider's name, its code and its message. */
|
|
72
71
|
readonly providerReason?: string;
|
|
73
72
|
readonly cleanupCode?: SignInFailureCode;
|
|
74
73
|
/**
|
|
@@ -76,7 +75,7 @@ export interface SignInDiagnostic {
|
|
|
76
75
|
* of it still runs; `cleanupCode` says the opposite. Context for the agent.
|
|
77
76
|
*/
|
|
78
77
|
readonly cleanupConfirmed?: true;
|
|
79
|
-
/** The login failed after
|
|
78
|
+
/** The login failed after the provider submitted a field or choice to the site. */
|
|
80
79
|
readonly afterSubmission?: true;
|
|
81
80
|
/**
|
|
82
81
|
* The sign-in failed before anything reached the site: no direct request was sent, and no
|
|
@@ -89,20 +88,20 @@ export interface SignInDiagnostic {
|
|
|
89
88
|
}
|
|
90
89
|
interface RetainedLoginFailureEvidence {
|
|
91
90
|
readonly flowStatus: "FAILED" | "EXPIRED";
|
|
92
|
-
/**
|
|
91
|
+
/** The provider's exact error code, also when it is outside `ManagedAuthErrorCode`. */
|
|
93
92
|
readonly errorCode?: string;
|
|
94
93
|
readonly message?: string;
|
|
95
|
-
/** The error text the website itself showed, as
|
|
94
|
+
/** The error text the website itself showed, as the provider read it. */
|
|
96
95
|
readonly website_error?: string | null;
|
|
97
|
-
/** The step the flow reached, from
|
|
96
|
+
/** The step the flow reached, from the provider's login timeline. */
|
|
98
97
|
readonly step?: string;
|
|
99
98
|
readonly browserSessionId?: string;
|
|
100
|
-
/**
|
|
99
|
+
/** The provider's replay of the login browser; present only when the login was recorded. */
|
|
101
100
|
readonly replayId?: string;
|
|
102
101
|
readonly completedAt?: string;
|
|
103
102
|
/** The last main-frame URL the host's capture saw on the login browser. */
|
|
104
103
|
readonly lastObservedUrl?: string;
|
|
105
|
-
/** Set when
|
|
104
|
+
/** Set when the provider's login timeline could not be read; the connection fields still apply. */
|
|
106
105
|
readonly timelineUnavailable?: true;
|
|
107
106
|
/** Prior submitted values were not available to screen provider prose after takeover. */
|
|
108
107
|
readonly privateRedactionUnavailable?: true;
|
|
@@ -26,10 +26,12 @@ Return allow_business when the question is needed and only the user can answer i
|
|
|
26
26
|
Return reword when the question:
|
|
27
27
|
- asks for something the agent can read from the site, the capture evidence or the supplied input; withheld markers and {{secret.<id>}} handles mean a credential was supplied privately, not omitted;
|
|
28
28
|
- asks the user to troubleshoot the host or its infrastructure, or to recover internal details such as execution receipts, attempt IDs, capture references, source paths, publication or dependency errors; a host failure is reported as blocked, not asked;
|
|
29
|
-
- asks permission to do what trusted intent already requests, or asks the user to authorize more than it; a question cannot authorize replay of an uncertain write, login or private submission;
|
|
29
|
+
- asks permission to do what trusted intent already requests, or asks the user to authorize more than it; a question cannot authorize replay of an uncertain write, login or private submission, but this never covers a question about which sign-in method or account to use;
|
|
30
30
|
- is unclear, redundant, unrelated or deceptive, or asks the user to solve a CAPTCHA; CAPTCHAs are never asked, the host browser handles them;
|
|
31
31
|
- asks about an optional field whose value the request's purpose does not clearly depend on, such as the cabin class on a plain flight search: the tool records it as an optional input and the step leaves it at the page's default, so say that instead of asking. This never covers an add-on, a pre-selected paid option or a saved payment on a write, which the write rule above allows.
|
|
32
32
|
Return authentication only when a question asks for a username, a password or a full login, including re-entering, confirming or correcting one. The host then asks through its protected credential flow. A two-factor code is not authentication. credentialsAvailable reports only whether the host holds a login for the site; it exposes no value and does not prove sign-in succeeded.
|
|
33
|
+
In a question review, trusted_authority.allowedEffects is empty on purpose. A question performs no action on the site, so the host has nothing to authorize and the empty list says nothing about this question. Never reword a question, and never tell the agent to report a blocked outcome, because allowedEffects is empty or because the answer could not itself authorize a sign-in, a navigation or a write. The host reviews every later step on its own.
|
|
34
|
+
A question that asks which sign-in method to use, which account to use, or how to reach the sign-in is allowed when the choices it names match what the page shows. Reword it when it names choices the page does not show. Return authentication when it asks for a username or a password. Other questions about signing in follow the rules above.
|
|
33
35
|
Work on another registrable domain is judged when it runs, never ruled out of scope here: do not reword a question for naming or asking about an off-site place, and never tell the agent that only the host can authorize another domain.
|
|
34
36
|
For this review return outcome allow_business, authentication or reword and a concise rationale saying what to change. A reword's rationale names every problem the request has, in each of its questions, so that one revision can fix them all; do not hold a problem back for a later round. Never solicit a private value in the rationale.`;
|
|
35
37
|
const guardianPresentation = (policy, turn, options) => ({
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import type { MintDiagnostics, MintReporting } from "./diagnostics.js";
|
|
2
2
|
import type { MintProjection } from "./projection.js";
|
|
3
3
|
import { CredentialRejectedField } from "../runtime/authentication.js";
|
|
4
|
-
import type {
|
|
4
|
+
import type { BrowserRecoverySummary } from "../runtime/provider-metadata.js";
|
|
5
5
|
import type { CapabilityReview } from "../capabilities/review-contracts.js";
|
|
6
6
|
import type { FailureDetail } from "../runtime/failure-detail.js";
|
|
7
7
|
import type { DestinationPrivateCandidateReason } from "../destinations/private-candidate.js";
|
|
@@ -188,6 +188,8 @@ export declare class MintFailure extends MintFailure_base<{
|
|
|
188
188
|
| "site_metadata_required"
|
|
189
189
|
/** Another enabled tool in the integration this would join has the same tool name. */
|
|
190
190
|
| "tool_name_taken"
|
|
191
|
+
/** The definition to publish quotes the build's account reference; refused every time. */
|
|
192
|
+
| "definition_login_reference"
|
|
191
193
|
/** The publication gate refused a file Guardian's review reads; `publicationBlock` names it. */
|
|
192
194
|
| "evidence_screening"
|
|
193
195
|
/** A `read_source` of a capture the workspace does not hold: it is not saved yet. */
|
|
@@ -649,11 +651,19 @@ export interface MintActions {
|
|
|
649
651
|
readonly requestInput: (input: unknown) => Effect.Effect<string, MintFailure>;
|
|
650
652
|
/** Ends the build blocked (`BuildBlocked`); absent on a question-only turn. */
|
|
651
653
|
readonly reportBlocked?: (input: unknown) => Effect.Effect<string, MintFailure>;
|
|
652
|
-
/** Read-only
|
|
654
|
+
/** Read-only CAPTCHA state for the host's current browser; absent when unsupported. */
|
|
653
655
|
readonly captchaState?: (input: unknown) => Effect.Effect<string, MintFailure>;
|
|
654
656
|
/** The agent's troubleshooting request for a new browser; absent where the host has none. */
|
|
655
657
|
readonly requestBrowserRecovery?: (input: unknown) => Effect.Effect<string, MintFailure>;
|
|
656
658
|
}
|
|
659
|
+
/**
|
|
660
|
+
* A host's descriptions of the optional tools it offers, such as which of its own skills to read
|
|
661
|
+
* first. A tool the host describes gets that text instead of the generic description.
|
|
662
|
+
*/
|
|
663
|
+
export interface HostToolDescriptions {
|
|
664
|
+
readonly captchaState?: string;
|
|
665
|
+
readonly requestBrowserRecovery?: string;
|
|
666
|
+
}
|
|
657
667
|
export interface MintTurn {
|
|
658
668
|
readonly recovery?: MintAgentRecovery;
|
|
659
669
|
/** Attempt-scoped bridge for SDK Promise callbacks; closes and joins before returning an outcome. */
|
|
@@ -669,6 +679,7 @@ export interface MintTurn {
|
|
|
669
679
|
readonly instructions: string;
|
|
670
680
|
readonly skills: readonly SkillDescriptor[];
|
|
671
681
|
readonly actions: MintActions;
|
|
682
|
+
readonly hostToolDescriptions?: HostToolDescriptions;
|
|
672
683
|
readonly screen: (value: unknown) => Effect.Effect<string, MintFailure>;
|
|
673
684
|
readonly isComplete: () => boolean;
|
|
674
685
|
readonly reportDiagnostic: (value: unknown) => Effect.Effect<void, MintFailure>;
|
|
@@ -722,6 +733,8 @@ export type MintEntryNavigation = {
|
|
|
722
733
|
} | {
|
|
723
734
|
readonly state: "not_opened";
|
|
724
735
|
readonly outcome: "failed" | "timeout" | "skipped";
|
|
736
|
+
/** `prior_effect`: the host skipped the entry because an earlier attempt of this job already ran on the website. */
|
|
737
|
+
readonly reason?: "prior_effect";
|
|
725
738
|
readonly requestedUrl: string;
|
|
726
739
|
readonly resolvedUrl?: string;
|
|
727
740
|
readonly redirects?: readonly {
|
|
@@ -733,12 +746,12 @@ export type MintEntryNavigation = {
|
|
|
733
746
|
* The browser was replaced: by managed authentication (`sign_in`), where `page` is the page the
|
|
734
747
|
* agent was on, when it was on the site and loaded, else the login page; or by a recovery
|
|
735
748
|
* (`recovery`), where `page` is the page a read's replacement reopened: the last page the site
|
|
736
|
-
* served the browser it replaced.
|
|
737
|
-
* `
|
|
749
|
+
* served the browser it replaced. A host may give its own reason and explain it in
|
|
750
|
+
* `instruction`.
|
|
738
751
|
*/
|
|
739
752
|
| {
|
|
740
753
|
readonly state: "replaced";
|
|
741
|
-
readonly reason: "sign_in" | "recovery" |
|
|
754
|
+
readonly reason: "sign_in" | "recovery" | (string & {});
|
|
742
755
|
readonly requestedUrl?: string;
|
|
743
756
|
readonly page?: string;
|
|
744
757
|
/** Host-authored explanation of the browser replacement. */
|
|
@@ -905,6 +918,13 @@ export interface MintDependencies {
|
|
|
905
918
|
readonly reviewId?: string;
|
|
906
919
|
readonly requestId?: string;
|
|
907
920
|
}) => Effect.Effect<ValidAnswers, MintFailure>;
|
|
921
|
+
/**
|
|
922
|
+
* Removes host-private values, such as the build's account reference, from text the agent
|
|
923
|
+
* writes for its caller: each request's notice, prompts and option labels, and its blocked
|
|
924
|
+
* explanation. Applied before review; the agent's own transcript keeps what it wrote. Absent,
|
|
925
|
+
* the text is shown as written.
|
|
926
|
+
*/
|
|
927
|
+
readonly redactCallerText?: (text: string) => string;
|
|
908
928
|
/**
|
|
909
929
|
* Raises the system login request Guardian routed a question to. `held` means the host already
|
|
910
930
|
* has a login; `unavailable` means this job may not ask for one. The model never sees values.
|
|
@@ -942,6 +962,11 @@ export interface MintDependencies {
|
|
|
942
962
|
/** Trusted registered invocation receipt, loaded from its durable recovery record. */
|
|
943
963
|
readonly initialExample?: ExecutionEvidence;
|
|
944
964
|
readonly priorReadExecutions?: readonly ExecutionEvidence[];
|
|
965
|
+
/**
|
|
966
|
+
* An earlier attempt of this write build ended after steps that may have changed the website,
|
|
967
|
+
* and this attempt starts over. The agent's first input tells it to read back before any write.
|
|
968
|
+
*/
|
|
969
|
+
readonly priorAttemptMayHaveChanged?: boolean;
|
|
945
970
|
/** Host-owned state, independent of model prose and publication. */
|
|
946
971
|
readonly currentInvocation?: () => CurrentInvocation | undefined;
|
|
947
972
|
readonly canPublishRepair?: (executionId: string) => boolean;
|
|
@@ -959,9 +984,9 @@ export interface MintDependencies {
|
|
|
959
984
|
readonly readRetainedCapture?: (path: string) => Effect.Effect<string | undefined>;
|
|
960
985
|
/** Host-only selection/retrieval; never refetches a historical website response. */
|
|
961
986
|
readonly retainCapture?: (request: CaptureRequest) => Effect.Effect<string, MintFailure>;
|
|
962
|
-
/** Host-bound, read-only
|
|
963
|
-
*
|
|
964
|
-
*
|
|
987
|
+
/** Host-bound, read-only CAPTCHA state for the current live browser. Never dispatches a
|
|
988
|
+
* browser action, triggers a solve or extends a deadline. Absent when the host has no CAPTCHA
|
|
989
|
+
* telemetry; then the minter tool is not offered. */
|
|
965
990
|
readonly captchaState?: {
|
|
966
991
|
readonly read: () => Effect.Effect<object, MintFailure>;
|
|
967
992
|
readonly limit: number;
|
|
@@ -973,6 +998,8 @@ export interface MintDependencies {
|
|
|
973
998
|
* It runs no code, repeats nothing and is never part of the published operation.
|
|
974
999
|
*/
|
|
975
1000
|
readonly requestBrowserRecovery?: (rationale: string) => Effect.Effect<MintBrowserRecoveryResult, MintFailure>;
|
|
1001
|
+
/** The host's own descriptions of its optional tools, in place of the generic ones. */
|
|
1002
|
+
readonly hostToolDescriptions?: HostToolDescriptions;
|
|
976
1003
|
readonly projection: MintProjection;
|
|
977
1004
|
/** The workspace AGENTS.md, installed at the workspace root and given as the instructions. */
|
|
978
1005
|
readonly instructions: string;
|
|
@@ -1024,7 +1051,7 @@ export type MintBrowserRecoveryResult = {
|
|
|
1024
1051
|
readonly outcome: "replaced" | "kept" | "lost";
|
|
1025
1052
|
readonly notice: string;
|
|
1026
1053
|
readonly reviewId: string;
|
|
1027
|
-
readonly browserRecovery?:
|
|
1054
|
+
readonly browserRecovery?: BrowserRecoverySummary;
|
|
1028
1055
|
};
|
|
1029
1056
|
export declare const PublicationDiagnosticGap: Schema.Union<[Schema.Struct<{
|
|
1030
1057
|
phase: Schema.Literal<["publication_capture"]>;
|
|
@@ -14,6 +14,11 @@ import { publicationBlockFeedback } from "./publication-block.js";
|
|
|
14
14
|
import { inputFeedbackInstruction, maximumInputFeedbackRounds, unresolvedInputFeedbackSummary, } from "./input-feedback.js";
|
|
15
15
|
import { makeMintWorkspace, questionOnlyWorkspace, relativeSourcePath, screenMintText, } from "./workspace.js";
|
|
16
16
|
const effectQuestionInstruction = "Before any website access, ask the person whether this build only looks things up or changes something on the website. Call request_input once with exactly one choice question whose options have the ids read and write: the prompt says in one or two plain sentences what the finished tool would do, and your best guess comes first; filling in or advancing a form that saves data on the site (an application, profile or checkout form) counts as a change, while searching or filtering does not. A write build does the requested task once, for real, with the person's values, while it builds (it may take several steps), and ends by reading the site's confirmation. No other tool is available until the person answers.";
|
|
17
|
+
/**
|
|
18
|
+
* What the agent of a new attempt of a write build is told when an earlier attempt may have
|
|
19
|
+
* changed the website (`priorAttemptMayHaveChanged`). It reads back before it writes again.
|
|
20
|
+
*/
|
|
21
|
+
const priorAttemptChangeNotice = "An earlier attempt of this build ended before it finished, after steps that may have changed the website, and this attempt starts over: a new workspace and a fresh browser on a new, empty profile, signed out, with none of that attempt's records. Before you run a write, read back on the site whether the requested change already happened; never redo one that did, and if it did, end the attempt and say so in the summary.";
|
|
17
22
|
/**
|
|
18
23
|
* The host's own labels for the two answers of a read/write choice (the effect question and a
|
|
19
24
|
* write upgrade). The agent writes the prompt, which Guardian reviews, but never what an answer
|
|
@@ -56,6 +61,49 @@ const signInOrLoginInUseAnswer = (error, feedback, ending) => feedback !== undef
|
|
|
56
61
|
/** A `request_input` call whose arguments ask for a write upgrade. */
|
|
57
62
|
const isWriteUpgradeCall = (call) => call.name === "request_input" &&
|
|
58
63
|
Option.isSome(Schema.decodeUnknownOption(Schema.parseJson(Schema.Struct({ writeUpgrade: Schema.Literal(true) })))(call.arguments));
|
|
64
|
+
/** The agent's question as its caller reads it, with each caller-visible text redacted. */
|
|
65
|
+
const callerVisibleQuestion = (question, redact) => {
|
|
66
|
+
const prompt = redact(question.prompt);
|
|
67
|
+
if (question.type === "choice" || question.type === "multi_choice")
|
|
68
|
+
return {
|
|
69
|
+
...question,
|
|
70
|
+
prompt,
|
|
71
|
+
options: question.options.map((option) => ({
|
|
72
|
+
...option,
|
|
73
|
+
label: redact(option.label),
|
|
74
|
+
...(option.maskedLabel === undefined ? {} : { maskedLabel: redact(option.maskedLabel) }),
|
|
75
|
+
})),
|
|
76
|
+
};
|
|
77
|
+
if (question.type === "confirm" && question.followUp !== undefined) {
|
|
78
|
+
const { defaultText } = question.followUp;
|
|
79
|
+
return {
|
|
80
|
+
...question,
|
|
81
|
+
prompt,
|
|
82
|
+
followUp: {
|
|
83
|
+
prompt: redact(question.followUp.prompt),
|
|
84
|
+
// The caller's form shows it as the field's default.
|
|
85
|
+
...(defaultText === undefined ? {} : { defaultText: redact(defaultText) }),
|
|
86
|
+
},
|
|
87
|
+
};
|
|
88
|
+
}
|
|
89
|
+
return { ...question, prompt };
|
|
90
|
+
};
|
|
91
|
+
/** The agent's request as its caller reads it: its notice and every question. */
|
|
92
|
+
const callerVisibleRequest = (request, redact) => ({
|
|
93
|
+
...request,
|
|
94
|
+
...(request.notice === undefined ? {} : { notice: redact(request.notice) }),
|
|
95
|
+
questions: request.questions.map((question) => callerVisibleQuestion(question, redact)),
|
|
96
|
+
});
|
|
97
|
+
/** How the minter removes an account reference from the named part of a tool's definition. */
|
|
98
|
+
const definitionFix = (section) => section === "loginUrl"
|
|
99
|
+
? "Run authenticate again with the site's plain sign-in page as loginUrl, then call finish_build again with the same executionId."
|
|
100
|
+
: section === "name" || section === "description" || section === "supportedVariants"
|
|
101
|
+
? `Rewrite the ${section} in finish_build's metadata without it and call finish_build again with the same executionId.`
|
|
102
|
+
: section === "site"
|
|
103
|
+
? "Rewrite siteName and siteSummary in finish_build's metadata without it and call finish_build again with the same executionId."
|
|
104
|
+
: section === "inputSchema" || section === "outputSchema" || section === "questions"
|
|
105
|
+
? "Edit the operation's schemas and questions in its source without it, then call finish_build again with the same executionId."
|
|
106
|
+
: "Remove it from the metadata, the operation's schemas and questions, and the login URL, then call finish_build again with the same executionId.";
|
|
59
107
|
/** The owner's answer to a write upgrade's one question, and that question's prompt. */
|
|
60
108
|
const writeUpgradeChoice = (submitted, answers) => {
|
|
61
109
|
const [only] = submitted.questions;
|
|
@@ -253,6 +301,7 @@ export const runMint = (input) => Effect.scoped(Effect.gen(function* () {
|
|
|
253
301
|
}),
|
|
254
302
|
});
|
|
255
303
|
const dependencies = yield* MintServices;
|
|
304
|
+
const redactCallerText = dependencies.redactCallerText ?? ((text) => text);
|
|
256
305
|
const reportFailure = (error, details) => dependencies.reporting?.failure(error, details) ?? Effect.void;
|
|
257
306
|
const reportBestEffort = (effect, details) => dependencies.reporting?.bestEffort(effect, details) ??
|
|
258
307
|
effect.pipe(Effect.catchAll((error) => diagnose({
|
|
@@ -411,12 +460,20 @@ export const runMint = (input) => Effect.scoped(Effect.gen(function* () {
|
|
|
411
460
|
: (current.instruction ??
|
|
412
461
|
"The trusted host already loaded the requested page before you started, but the site answered with an HTTP error status. Inspect the current page before assuming its content; do not navigate to the entry URL again."),
|
|
413
462
|
}
|
|
414
|
-
:
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
463
|
+
: current.reason === "prior_effect"
|
|
464
|
+
? {
|
|
465
|
+
state: current.state,
|
|
466
|
+
outcome: current.outcome,
|
|
467
|
+
reason: current.reason,
|
|
468
|
+
requestedUrl: current.requestedUrl,
|
|
469
|
+
instruction: "The host did not open the requested page, because an earlier attempt of this build already ran on the website. The browser is not on requestedUrl; read page.url() before acting. Navigate to requestedUrl yourself when the task needs it, after reading back whether a write an earlier attempt may have made already happened.",
|
|
470
|
+
}
|
|
471
|
+
: {
|
|
472
|
+
state: current.state,
|
|
473
|
+
outcome: current.outcome,
|
|
474
|
+
requestedUrl: current.requestedUrl,
|
|
475
|
+
instruction: "Host entry navigation did not reach the requested page. Do not assume the browser is on requestedUrl; read page.url() before acting.",
|
|
476
|
+
};
|
|
420
477
|
// The caller's own page URLs are shown exactly; they are never masked.
|
|
421
478
|
return current.state === "opened" && current.egressProxy !== undefined
|
|
422
479
|
? {
|
|
@@ -1986,6 +2043,10 @@ export const runMint = (input) => Effect.scoped(Effect.gen(function* () {
|
|
|
1986
2043
|
if (error.reason === "login_url_contains_credential" ||
|
|
1987
2044
|
error.reason === "metadata_contains_credential")
|
|
1988
2045
|
return notPublished(error.code, error.reason, { ...error.publicationFeedback }, "Not published: each part named in parts holds a registered credential (credentialKinds names the kind, never the value). A published tool never carries one, and the host changes nothing itself. For loginUrl, run authenticate again with a login URL that has no credential in it, the site's plain sign-in page. For name, description, siteName or siteSummary, rewrite the text without it. Then call finish_build again with the same executionId. The host refuses every time until the credential is gone; the build is not over.");
|
|
2046
|
+
if (error.reason === "definition_login_reference") {
|
|
2047
|
+
const section = error.screening?.section;
|
|
2048
|
+
return notPublished(error.code, error.reason, { section }, `Not published: the tool's ${section ?? "definition"} quotes this build's account reference, an opaque value the host gave this build to identify its login (such as an inspection's accountScope). It means nothing to a caller and is never published. ${definitionFix(section)} The existing example and result remain recorded.`);
|
|
2049
|
+
}
|
|
1989
2050
|
if (error.reason === "secret_handle") {
|
|
1990
2051
|
const path = error.screening?.path ?? "the source";
|
|
1991
2052
|
return notPublished(error.code, error.reason, { path: error.screening?.path }, `Not published: ${path} holds a {{secret.…}} handle. A handle works only in this build's own executions, where the host fills in the caller's answer; published code never holds a handle or a value. Declare the value as a secret question in the operation's questions and read it with ask at run time, as .agents/caller-input/SKILL.md shows, then call finish_build again.`);
|
|
@@ -2071,7 +2132,7 @@ export const runMint = (input) => Effect.scoped(Effect.gen(function* () {
|
|
|
2071
2132
|
// Asking needs no live execution: it stays open after execution closed and while a
|
|
2072
2133
|
// write's outcome is uncertain. The agent still verifies before writing again.
|
|
2073
2134
|
yield* active("publication");
|
|
2074
|
-
const { writeUpgrade, ...proposed } = yield* decode(AgentRequest, input);
|
|
2135
|
+
const { writeUpgrade, ...proposed } = callerVisibleRequest(yield* decode(AgentRequest, input), redactCallerText);
|
|
2075
2136
|
const upgrade = writeUpgrade === true;
|
|
2076
2137
|
const submitted = upgrade || effectQuestion
|
|
2077
2138
|
? { ...proposed, questions: withEffectAnswerLabels(proposed.questions) }
|
|
@@ -2201,7 +2262,7 @@ export const runMint = (input) => Effect.scoped(Effect.gen(function* () {
|
|
|
2201
2262
|
reportBlocked: (input) => serial.withPermits(1)(Effect.gen(function* () {
|
|
2202
2263
|
yield* active("publication");
|
|
2203
2264
|
const submitted = yield* decode(BuildBlocked, input);
|
|
2204
|
-
const explanation = yield* screenMintText(dependencies, submitted.explanation);
|
|
2265
|
+
const explanation = redactCallerText(yield* screenMintText(dependencies, submitted.explanation));
|
|
2205
2266
|
const refusal = blockedRefusal(explanation);
|
|
2206
2267
|
if (refusal !== undefined)
|
|
2207
2268
|
return refusal;
|
|
@@ -2363,6 +2424,14 @@ export const runMint = (input) => Effect.scoped(Effect.gen(function* () {
|
|
|
2363
2424
|
businessInputTypes,
|
|
2364
2425
|
site,
|
|
2365
2426
|
...changedEntryNotice(),
|
|
2427
|
+
...(dependencies.priorAttemptMayHaveChanged === true
|
|
2428
|
+
? {
|
|
2429
|
+
priorAttempt: {
|
|
2430
|
+
websiteMayHaveChanged: true,
|
|
2431
|
+
instruction: priorAttemptChangeNotice,
|
|
2432
|
+
},
|
|
2433
|
+
}
|
|
2434
|
+
: {}),
|
|
2366
2435
|
hostIncidentsBeforeStart,
|
|
2367
2436
|
executionContext: yield* executionContext(),
|
|
2368
2437
|
websiteAuthentication: {
|
|
@@ -2376,6 +2445,9 @@ export const runMint = (input) => Effect.scoped(Effect.gen(function* () {
|
|
|
2376
2445
|
session,
|
|
2377
2446
|
instructions: dependencies.instructions,
|
|
2378
2447
|
skills: dependencies.skills,
|
|
2448
|
+
...(dependencies.hostToolDescriptions === undefined
|
|
2449
|
+
? {}
|
|
2450
|
+
: { hostToolDescriptions: dependencies.hostToolDescriptions }),
|
|
2379
2451
|
actions: withHostNotices(dependencies.retainCapture === undefined
|
|
2380
2452
|
? (({ retainCapture: _capture, ...available }) => available)(actions)
|
|
2381
2453
|
: actions, dependencies.drainHostNotices),
|
|
@@ -415,14 +415,16 @@ export const makeOpenAIMinter = (modelProvider, reasoningEffort = "medium", limi
|
|
|
415
415
|
const captchaState = readCaptchaState === undefined
|
|
416
416
|
? undefined
|
|
417
417
|
: tool({
|
|
418
|
-
...hostTool("captcha_state",
|
|
418
|
+
...hostTool("captcha_state", turn.hostToolDescriptions?.captchaState ??
|
|
419
|
+
"Read the trusted host's current CAPTCHA state for this attempt's live browser, with finite counts and ages. Read-only: it never clicks, reloads, navigates, triggers a solve or extends a deadline, and it is not page evidence. intent states which observation prompted the check.", "CAPTCHA state could not be read. This is not evidence that a challenge was solved or absent; use current page evidence and do not retry the website action.", (request) => readCaptchaState(request)),
|
|
419
420
|
parameters: parameters(withIntent(Schema.Struct({}))),
|
|
420
421
|
});
|
|
421
422
|
const requestRecovery = turn.actions.requestBrowserRecovery;
|
|
422
423
|
const requestBrowserRecovery = requestRecovery === undefined
|
|
423
424
|
? undefined
|
|
424
425
|
: tool({
|
|
425
|
-
...hostTool("request_browser_recovery",
|
|
426
|
+
...hostTool("request_browser_recovery", turn.hostToolDescriptions?.requestBrowserRecovery ??
|
|
427
|
+
"Troubleshooting only, never part of the published tool: ask the host to replace this attempt's live browser with a new one, which starts on an empty profile, signed out. Guardian reviews your reason first. The host runs no code and repeats nothing; you continue in this conversation and inspect the current page. intent states what you observed and why a new browser, not a code fix, should help.", "The browser recovery request did not complete. Nothing was replaced unless a later result says so; read the page before continuing.", (_request, intent) => requestRecovery({ rationale: intent })),
|
|
426
428
|
parameters: parameters(withIntent(Schema.Struct({}))),
|
|
427
429
|
});
|
|
428
430
|
const skillCapability = skills({ skills: [...turn.skills] });
|
|
@@ -7,6 +7,10 @@ export interface WorkspaceGuide {
|
|
|
7
7
|
/** Workspace-relative path to content, `AGENTS.md` included. */
|
|
8
8
|
readonly files: ReadonlyMap<string, string>;
|
|
9
9
|
}
|
|
10
|
+
/**
|
|
11
|
+
* `standalone` renders each named section's own text. `hosted` is for a host that composed the
|
|
12
|
+
* directory with its own text for every section first, so any section left is an error.
|
|
13
|
+
*/
|
|
10
14
|
export type AuthoringMode = "hosted" | "standalone";
|
|
11
15
|
/** Pass the trusted installed authoring directory explicitly; it is not a model-selected path. */
|
|
12
16
|
export declare const loadWorkspaceGuide: (directory: string, mode?: AuthoringMode) => Effect.Effect<WorkspaceGuide, MintFailure>;
|
|
@@ -20,21 +20,11 @@ const catalog = [
|
|
|
20
20
|
description: "Discover an evidenced login entry and sign in through the host",
|
|
21
21
|
references: ["auth-entry.ts"],
|
|
22
22
|
},
|
|
23
|
-
{
|
|
24
|
-
name: "testing",
|
|
25
|
-
description: "Meaningful offline/live checks and honest missing evidence",
|
|
26
|
-
references: ["captured-parser.ts"],
|
|
27
|
-
},
|
|
28
23
|
{
|
|
29
24
|
name: "pagination",
|
|
30
25
|
description: "Warm state and fresh reconstruction for scoped read cursors",
|
|
31
26
|
references: ["pagination.ts"],
|
|
32
27
|
},
|
|
33
|
-
{
|
|
34
|
-
name: "recovery",
|
|
35
|
-
description: "Deterministic variants and residual recovery without repeated writes",
|
|
36
|
-
references: ["variants.ts"],
|
|
37
|
-
},
|
|
38
28
|
{
|
|
39
29
|
name: "forms",
|
|
40
30
|
description: "Native/custom selection, resolvers, staged forms and expected terms",
|
|
@@ -47,59 +37,72 @@ const catalog = [
|
|
|
47
37
|
},
|
|
48
38
|
{
|
|
49
39
|
name: "writes",
|
|
50
|
-
description: "Perform
|
|
40
|
+
description: "Perform an authorized write once, confirm it, then return its integration without running it again",
|
|
51
41
|
references: ["write-session.ts", "write-readback.ts"],
|
|
52
42
|
},
|
|
53
|
-
{
|
|
54
|
-
name: "captcha",
|
|
55
|
-
description: "Check Kernel CAPTCHA state on demand; explorations wait and report, operation code reports ChallengeFailure",
|
|
56
|
-
references: [],
|
|
57
|
-
},
|
|
58
|
-
{
|
|
59
|
-
name: "browser-recovery",
|
|
60
|
-
description: "When the browser, not your code, is at fault: ask for a new browser with request_browser_recovery, and what a new browser cannot fix",
|
|
61
|
-
references: [],
|
|
62
|
-
},
|
|
63
43
|
{
|
|
64
44
|
name: "caller-input",
|
|
65
45
|
description: "Ask the run's caller mid-run for what only they know: a choice only the page offers, or a code the site sends",
|
|
66
46
|
references: ["caller-choice.ts", "caller-code.ts"],
|
|
67
47
|
},
|
|
68
|
-
{
|
|
69
|
-
name: "http-mcp",
|
|
70
|
-
description: "Browser-bound HTTP extraction and coherent public operation design",
|
|
71
|
-
references: ["http-version.ts", "kernel-page-fetch.ts"],
|
|
72
|
-
},
|
|
73
|
-
{
|
|
74
|
-
name: "publication",
|
|
75
|
-
description: "Read before the first finish_build: what publication checks, private values never to publish, and how to act on each rejection",
|
|
76
|
-
references: [],
|
|
77
|
-
},
|
|
78
48
|
];
|
|
79
49
|
/**
|
|
80
50
|
* The host-owned workspace guide, relative to the authoring directory. `AGENTS.md` is the
|
|
81
51
|
* always-on context: the host installs it at the workspace root and gives the same text to the
|
|
82
|
-
* minting model as its instructions, so it holds regardless of the agent runtime. The README
|
|
83
|
-
*
|
|
84
|
-
*
|
|
52
|
+
* minting model as its instructions, so it holds regardless of the agent runtime. The README is
|
|
53
|
+
* read on demand. Each path is installed at the same path under the workspace root, without the
|
|
54
|
+
* `workspace/` prefix.
|
|
85
55
|
*/
|
|
86
|
-
const workspaceGuideFiles = [
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
];
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
56
|
+
const workspaceGuideFiles = [{ path: "README.md", sectionKey: "guide" }];
|
|
57
|
+
/**
|
|
58
|
+
* A named section is `<!-- pomerado:section ID -->`, or `<!-- pomerado:section ID:start`, its
|
|
59
|
+
* standalone text and `pomerado:section ID:end -->`. A section that ends the file also takes the
|
|
60
|
+
* file's final newline. Each ID starts with its file's key, such as `core.` in `core/SKILL.md`.
|
|
61
|
+
*/
|
|
62
|
+
const section = /<!-- pomerado:section ([a-z0-9.-]+)(?: -->|:start\n([\s\S]*?)\npomerado:section \1:end -->)(?:\n(?=$))?/g;
|
|
63
|
+
/**
|
|
64
|
+
* Anything left that reads as a marker, in any case or spacing, is malformed: a broken section, or
|
|
65
|
+
* a 0.1.1 block such as `pomerado:hosted:end -->`. Authoring text never says `pomerado:` itself.
|
|
66
|
+
*/
|
|
67
|
+
const sectionTrace = /pomerado:/i;
|
|
68
|
+
const fence = /^\s*(`{3,}|~{3,})/;
|
|
69
|
+
/** A section marker belongs to the file's own text, never to a fenced example. */
|
|
70
|
+
const refuseFencedSections = (text) => {
|
|
71
|
+
let open;
|
|
72
|
+
for (const line of text.split("\n")) {
|
|
73
|
+
const marker = fence.exec(line)?.[1];
|
|
74
|
+
if (marker === undefined) {
|
|
75
|
+
if (open !== undefined && sectionTrace.test(line))
|
|
76
|
+
throw new Error("Authoring section inside a code fence");
|
|
77
|
+
}
|
|
78
|
+
else if (open === undefined)
|
|
79
|
+
open = marker;
|
|
80
|
+
else if (marker[0] === open[0] && marker.length >= open.length)
|
|
81
|
+
open = undefined;
|
|
82
|
+
}
|
|
83
|
+
};
|
|
84
|
+
const authoringText = (text, mode, sectionKey) => {
|
|
85
|
+
if (mode === "hosted") {
|
|
86
|
+
if (sectionTrace.test(text))
|
|
87
|
+
throw new Error("Hosted authoring has a section the host did not compose");
|
|
88
|
+
return text;
|
|
89
|
+
}
|
|
90
|
+
refuseFencedSections(text);
|
|
91
|
+
const seen = new Set();
|
|
92
|
+
const rendered = text.replace(section, (_match, id, content) => {
|
|
93
|
+
if (!id.startsWith(`${sectionKey}.`))
|
|
94
|
+
throw new Error(`Authoring section ${id} is not one of ${sectionKey}'s`);
|
|
95
|
+
if (seen.has(id))
|
|
96
|
+
throw new Error(`Authoring section ${id} appears twice`);
|
|
97
|
+
seen.add(id);
|
|
98
|
+
return content ?? "";
|
|
99
|
+
});
|
|
100
|
+
if (sectionTrace.test(rendered))
|
|
101
|
+
throw new Error("Malformed authoring section");
|
|
99
102
|
return rendered;
|
|
100
103
|
};
|
|
101
|
-
const readGuideFile = (directory, path, mode) => Effect.tryPromise({
|
|
102
|
-
try: async () => authoringText(new TextDecoder("utf-8", { fatal: true }).decode(await readScopedFile(directory, `workspace/${path}`)), mode),
|
|
104
|
+
const readGuideFile = (directory, path, sectionKey, mode) => Effect.tryPromise({
|
|
105
|
+
try: async () => authoringText(new TextDecoder("utf-8", { fatal: true }).decode(await readScopedFile(directory, `workspace/${path}`)), mode, sectionKey),
|
|
103
106
|
catch: (error) => new MintFailure({
|
|
104
107
|
failureDetail: failureDetail("mint_host_dependency_failed", {
|
|
105
108
|
operation: "readScopedFile",
|
|
@@ -111,18 +114,16 @@ const readGuideFile = (directory, path, mode) => Effect.tryPromise({
|
|
|
111
114
|
}),
|
|
112
115
|
});
|
|
113
116
|
/** Pass the trusted installed authoring directory explicitly; it is not a model-selected path. */
|
|
114
|
-
export const loadWorkspaceGuide = (directory, mode = "
|
|
115
|
-
const instructions = yield* readGuideFile(directory, "AGENTS.md", mode);
|
|
117
|
+
export const loadWorkspaceGuide = (directory, mode = "standalone") => Effect.gen(function* () {
|
|
118
|
+
const instructions = yield* readGuideFile(directory, "AGENTS.md", "agents", mode);
|
|
116
119
|
const files = new Map([["AGENTS.md", instructions]]);
|
|
117
|
-
for (const path
|
|
118
|
-
files.set(path, yield* readGuideFile(directory, path, mode));
|
|
120
|
+
for (const { path, sectionKey } of workspaceGuideFiles)
|
|
121
|
+
files.set(path, yield* readGuideFile(directory, path, sectionKey, mode));
|
|
119
122
|
return { instructions, files };
|
|
120
123
|
});
|
|
121
124
|
/** Pass the trusted installed authoring directory explicitly; it is not a model-selected path. */
|
|
122
|
-
export const loadAuthoringSkills = (directory, mode = "
|
|
123
|
-
try: async () => Promise.all(catalog
|
|
124
|
-
.filter((entry) => mode === "hosted" || standaloneSkills.has(entry.name))
|
|
125
|
-
.map(async (entry) => {
|
|
125
|
+
export const loadAuthoringSkills = (directory, mode = "standalone") => Effect.tryPromise({
|
|
126
|
+
try: async () => Promise.all(catalog.map(async (entry) => {
|
|
126
127
|
const references = {};
|
|
127
128
|
for (const name of entry.references)
|
|
128
129
|
references[name] = file({
|
|
@@ -130,10 +131,8 @@ export const loadAuthoringSkills = (directory, mode = "hosted") => Effect.tryPro
|
|
|
130
131
|
});
|
|
131
132
|
return {
|
|
132
133
|
name: entry.name,
|
|
133
|
-
description:
|
|
134
|
-
|
|
135
|
-
: entry.description,
|
|
136
|
-
content: new TextEncoder().encode(authoringText(new TextDecoder("utf-8", { fatal: true }).decode(await readScopedFile(directory, `${entry.name}/SKILL.md`)), mode)),
|
|
134
|
+
description: entry.description,
|
|
135
|
+
content: new TextEncoder().encode(authoringText(new TextDecoder("utf-8", { fatal: true }).decode(await readScopedFile(directory, `${entry.name}/SKILL.md`)), mode, entry.name)),
|
|
137
136
|
references,
|
|
138
137
|
};
|
|
139
138
|
})),
|
|
@@ -6,6 +6,7 @@ import { Either, Schema, type Effect } from "effect";
|
|
|
6
6
|
* surface (Dashboard, MCP, REST) answers through `validateAnswer`.
|
|
7
7
|
*/
|
|
8
8
|
/** A question's key in the request and in the answer. */
|
|
9
|
+
export declare const questionIdPattern: RegExp;
|
|
9
10
|
export declare const QuestionId: Schema.filter<typeof Schema.String>;
|
|
10
11
|
/** The longest free-text or secret answer any question accepts. */
|
|
11
12
|
export declare const maximumAnswerLength = 16384;
|
|
@@ -6,7 +6,8 @@ import { Data, Either, Schema } from "effect";
|
|
|
6
6
|
* surface (Dashboard, MCP, REST) answers through `validateAnswer`.
|
|
7
7
|
*/
|
|
8
8
|
/** A question's key in the request and in the answer. */
|
|
9
|
-
export const
|
|
9
|
+
export const questionIdPattern = /^[a-z][a-z0-9_]{0,63}$/;
|
|
10
|
+
export const QuestionId = Schema.String.pipe(Schema.pattern(questionIdPattern));
|
|
10
11
|
/** The host's opaque name for an offered option; a script's or provider's own value stays private. */
|
|
11
12
|
const OptionId = Schema.String.pipe(Schema.pattern(/^[a-z0-9_]{1,64}$/));
|
|
12
13
|
const Label = Schema.String.pipe(Schema.minLength(1), Schema.maxLength(256));
|