orchajs 0.12.2 → 0.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -42
- package/dist/cli-agents-template.d.ts +1 -1
- package/dist/cli-agents-template.d.ts.map +1 -1
- package/dist/cli-agents-template.js +38 -51
- package/dist/cli-agents-template.js.map +1 -1
- package/dist/compiler/compile.d.ts.map +1 -1
- package/dist/compiler/compile.js +45 -96
- package/dist/compiler/compile.js.map +1 -1
- package/dist/index.d.ts +1 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/orcha.js +1 -1
- package/dist/orcha.js.map +1 -1
- package/dist/playground/public/app.js +16 -16
- package/dist/playground/public/app.js.map +3 -3
- package/dist/playground/source.d.ts +1 -0
- package/dist/playground/source.d.ts.map +1 -1
- package/dist/playground/source.js +51 -28
- package/dist/playground/source.js.map +1 -1
- package/dist/runtime/create-orcha.d.ts.map +1 -1
- package/dist/runtime/create-orcha.js +5 -12
- package/dist/runtime/create-orcha.js.map +1 -1
- package/dist/testing.d.ts +1 -2
- package/dist/testing.d.ts.map +1 -1
- package/dist/testing.js +46 -85
- package/dist/testing.js.map +1 -1
- package/dist/types.d.ts +8 -5
- package/dist/types.d.ts.map +1 -1
- package/package.json +1 -9
- package/dist/evaluations.d.ts +0 -3
- package/dist/evaluations.d.ts.map +0 -1
- package/dist/evaluations.js +0 -4
- package/dist/evaluations.js.map +0 -1
- package/dist/skills.d.ts +0 -3
- package/dist/skills.d.ts.map +0 -1
- package/dist/skills.js +0 -4
- package/dist/skills.js.map +0 -1
package/README.md
CHANGED
|
@@ -56,26 +56,16 @@ or hosted dependency.
|
|
|
56
56
|
/evaluations → metrics each run is judged against
|
|
57
57
|
```
|
|
58
58
|
|
|
59
|
-
Placement determines behavior.
|
|
60
|
-
|
|
59
|
+
Placement determines behavior. Direct child folders under `actions`, `skills`,
|
|
60
|
+
`tests`, and `evaluations` are discovered automatically. Set
|
|
61
|
+
`"enabled": false` in an item's `index.json` to keep WIP source inactive.
|
|
61
62
|
|
|
62
63
|
---
|
|
63
64
|
|
|
64
65
|
## Skills
|
|
65
66
|
|
|
66
67
|
Skills keep specialized procedures out of the base prompt until the model
|
|
67
|
-
needs them.
|
|
68
|
-
|
|
69
|
-
```js
|
|
70
|
-
// skills/index.js
|
|
71
|
-
import { defineSkills } from "orchajs/skills";
|
|
72
|
-
|
|
73
|
-
export default defineSkills({
|
|
74
|
-
incompleteEvidenceReview: "./incompleteEvidenceReview",
|
|
75
|
-
});
|
|
76
|
-
```
|
|
77
|
-
|
|
78
|
-
Each registered folder contains compact discovery metadata and the full
|
|
68
|
+
needs them. Each skill folder contains compact discovery metadata and the full
|
|
79
69
|
instructions:
|
|
80
70
|
|
|
81
71
|
```json
|
|
@@ -92,13 +82,13 @@ skills/incompleteEvidenceReview/
|
|
|
92
82
|
instructions.md
|
|
93
83
|
```
|
|
94
84
|
|
|
95
|
-
Orcha initially gives the model only each
|
|
85
|
+
Orcha initially gives the model only each enabled skill's name,
|
|
96
86
|
description, and trigger guidance. When the model calls the internal
|
|
97
87
|
`load_skill` tool, Orcha activates the full instructions without running an
|
|
98
88
|
application action or pausing for the client. Durable `skill.requested`,
|
|
99
89
|
`skill.loaded`, and `skill.failed` events expose the loading lifecycle. Loaded
|
|
100
90
|
skills remain active across `resume()` calls and provider changes.
|
|
101
|
-
|
|
91
|
+
Skill folders with `"enabled": false` are not compiled or exposed.
|
|
102
92
|
|
|
103
93
|
---
|
|
104
94
|
|
|
@@ -114,8 +104,9 @@ actions/
|
|
|
114
104
|
index.js // the code
|
|
115
105
|
```
|
|
116
106
|
|
|
117
|
-
`index.json` tells the model what the action is
|
|
118
|
-
|
|
107
|
+
`index.json` tells the model what the action is and its input/output contracts.
|
|
108
|
+
Actions are local by default; set `"execution": "client"` only when the
|
|
109
|
+
application must execute one:
|
|
119
110
|
|
|
120
111
|
```json
|
|
121
112
|
{
|
|
@@ -181,12 +172,14 @@ Action entrypoints and their imported JavaScript or TypeScript modules are
|
|
|
181
172
|
bundled together. Relative project imports, package imports, transitive
|
|
182
173
|
dependencies, and standard exports work normally.
|
|
183
174
|
|
|
184
|
-
|
|
175
|
+
Local actions use the native Node.js runtime by default, so ordinary projects
|
|
176
|
+
do not need global action configuration. Select the sandbox explicitly when
|
|
177
|
+
isolation is required:
|
|
185
178
|
|
|
186
179
|
```js
|
|
187
180
|
orcha.init({
|
|
188
181
|
actions: {
|
|
189
|
-
runtime: "
|
|
182
|
+
runtime: "sandbox",
|
|
190
183
|
},
|
|
191
184
|
// providers and agents...
|
|
192
185
|
});
|
|
@@ -589,16 +582,6 @@ Agent tests live beside the agent and use the same compiled instructions,
|
|
|
589
582
|
provider, output contract, action schemas, `run()`, and `resume()` behavior as
|
|
590
583
|
the application.
|
|
591
584
|
|
|
592
|
-
Register test folders in `tests/index.js`:
|
|
593
|
-
|
|
594
|
-
```js
|
|
595
|
-
import { defineTests } from "orchajs/testing";
|
|
596
|
-
|
|
597
|
-
export default defineTests({
|
|
598
|
-
delayedOrder: "./delayedOrder",
|
|
599
|
-
});
|
|
600
|
-
```
|
|
601
|
-
|
|
602
585
|
Each test folder contains an `index.json` with its input, one simulated
|
|
603
586
|
response sequence for every action declared by the agent, and deterministic
|
|
604
587
|
expectations:
|
|
@@ -659,7 +642,7 @@ terminal, report, session log, and Playground because they may cause side
|
|
|
659
642
|
effects. Client actions must be mocked because the test runner has no attached
|
|
660
643
|
application client. Unknown mock action names are rejected.
|
|
661
644
|
|
|
662
|
-
Run every
|
|
645
|
+
Run every enabled test programmatically:
|
|
663
646
|
|
|
664
647
|
```js
|
|
665
648
|
import { orcha } from "orchajs";
|
|
@@ -678,17 +661,8 @@ build` itself remains offline and does not require those credentials.
|
|
|
678
661
|
|
|
679
662
|
## Evaluations
|
|
680
663
|
|
|
681
|
-
Evaluations are
|
|
682
|
-
every completed run:
|
|
683
|
-
|
|
684
|
-
```js
|
|
685
|
-
// evaluations/index.js
|
|
686
|
-
import { defineEvaluations } from "orchajs/evaluations";
|
|
687
|
-
|
|
688
|
-
export default defineEvaluations({
|
|
689
|
-
responseQuality: "./responseQuality",
|
|
690
|
-
});
|
|
691
|
-
```
|
|
664
|
+
Evaluations are automatically discovered LLM judges that score the cumulative
|
|
665
|
+
session after every completed run:
|
|
692
666
|
|
|
693
667
|
```json
|
|
694
668
|
{
|
|
@@ -1,2 +1,2 @@
|
|
|
1
|
-
export declare const AGENTS_MD = "# Orcha project guide\n\nThis repository uses OrchaJS, a filesystem-convention framework for durable\nAI agents. Treat the `orcha/` directory as source code. Do not edit generated\nfiles under `.orcha/`.\n\n## Commands\n\n- `orcha init` creates the initial Orcha files without overwriting files.\n- `orcha dev` validates the registry and watches `orcha/**` for changes.\n- `orcha playground` opens the built-in local UI and hot reloads agent source.\n- `orcha run <agent> --input \"\u2026\"` executes one registered agent.\n- `orcha run <agent> --input-file request.json` accepts structured input.\n- `orcha run <agent> --session <id> --input \"\u2026\"` continues a session.\n- `orcha run <agent> --session <id> --tool-results results.json` submits\n pending client-action results.\n- `orcha test` runs every registered agent test.\n- `orcha test <agent>` or `orcha test <agent>/<case>` narrows the run.\n- `orcha build` creates the production Orcha bundle without calling models.\n- Add `--json` to `run` and `test` for machine-readable output.\n\n`run`, `test`, and playground executions use real providers and require\ncredentials. `dev` and `build` are offline. Session logs are JSONL files under\n`.orcha/sessions/<agentName>/ses_<timestamp>_<uuid>.jsonl`.\n\n## Registry\n\n`orcha/index.ts` initializes providers and explicitly registers agents:\n\n```ts\nimport { orcha } from \"orchajs\";\n\norcha.init({\n providers: {\n anthropic: process.env.ANTHROPIC_API_KEY ?? \"\",\n openai: process.env.OPENAI_API_KEY ?? \"\",\n },\n actions: { runtime: \"native\" },\n agents: {\n supportBot: \"./supportBot\",\n },\n});\n```\n\nOnly registered folders are compiled. Agent keys become runtime properties\nsuch as `orcha.supportBot`. Use `actions.runtime: \"sandbox\"` for isolated\nlocal action execution or `\"native\"` when the application intentionally\nallows action modules to execute in its Node.js process.\n\n`orcha.init()` fields:\n\n- `providers` (required): provider configurations keyed by built-in provider\n name.\n- `agents` (required): runtime property names mapped to folders relative to\n `orcha/`. At least one agent is required.\n- `actions` (required only when a registered agent has local actions):\n selects the local execution runtime and its environment/sandbox settings.\n- `storage.strategy` (optional): currently only `\"node-jsonl\"`.\n- `storage.directory` (optional): session directory relative to project\n root; defaults to `.orcha/sessions`.\n- `root` (optional): absolute or working-directory-relative project root;\n defaults to `ORCHA_PROJECT_ROOT` and then `process.cwd()`.\n\nThe Orcha library reads the host application's existing `process.env`.\nStandalone Orcha CLI commands also load `.env` from the project root without\noverriding environment variables already in the process.\n\nProvider configuration shapes:\n\n```ts\nproviders: {\n anthropic: process.env.ANTHROPIC_API_KEY ?? \"\",\n deepseek: process.env.DEEPSEEK_API_KEY ?? \"\",\n googlegenai: process.env.GOOGLE_API_KEY ?? \"\",\n openai: {\n apiKey: process.env.OPENAI_API_KEY ?? \"\",\n baseUrl: \"https://api.openai.com/v1\", // optional override\n },\n vertexai: {\n project: process.env.GOOGLE_CLOUD_PROJECT ?? \"\",\n location: process.env.GOOGLE_CLOUD_LOCATION ?? \"us-central1\",\n // credentials is optional; omit it to use Google ADC.\n credentials: {\n clientEmail: process.env.GOOGLE_CLIENT_EMAIL ?? \"\",\n privateKey: process.env.GOOGLE_PRIVATE_KEY ?? \"\",\n },\n baseUrl: undefined, // optional override\n },\n bedrock: {\n region: process.env.AWS_REGION ?? \"us-east-1\",\n // credentials is optional; omit it to use the AWS credential chain.\n credentials: {\n accessKeyId: process.env.AWS_ACCESS_KEY_ID ?? \"\",\n secretAccessKey: process.env.AWS_SECRET_ACCESS_KEY ?? \"\",\n sessionToken: process.env.AWS_SESSION_TOKEN,\n },\n baseUrl: undefined, // optional override\n },\n}\n```\n\nAPI-key providers accept either a string shorthand or\n`{ apiKey, baseUrl? }`. Vertex AI requires `project` and `location`;\nexplicit service-account credentials are optional. Bedrock requires `region`;\nexplicit AWS credentials are optional. Never place credentials in\n`index.json`, instructions, tests, session metadata, or committed files.\n\n## Agent folders\n\n```text\norcha/\n index.ts\n supportBot/\n index.json\n instructions.md\n actions/\n skills/\n tests/\n evaluations/\n```\n\n`index.json` selects the model:\n\n```json\n{\n \"name\": \"Support Agent\",\n \"description\": \"Resolve customer support questions using confirmed account data.\",\n \"provider\": \"anthropic\",\n \"model\": \"claude-sonnet-4-6\",\n \"region\": \"provider_managed\",\n \"maxTokens\": 10240,\n \"outputType\": \"text\"\n}\n```\n\nOptional fields include `reasoningLevel`, `outputType: \"json\"`, and an\n`outputSchema` JSON Schema. Provider-specific reasoning values are forwarded\nwithout translation. Put the agent's stable role, boundaries, and operating\ninstructions in `instructions.md`.\n\nAgent `index.json` fields:\n\n- `name` (required): concise human-readable agent name.\n- `description` (optional): what the agent does and when it should be used.\n- `provider` (required): `\"anthropic\"`, `\"bedrock\"`, `\"deepseek\"`,\n `\"openai\"`, `\"googlegenai\"`, or `\"vertexai\"`.\n- `model` (required): exact provider model identifier.\n- `region` (optional): provider/model routing hint; defaults in durable\n metadata to `\"provider_managed\"`.\n- `maxTokens` (optional): positive integer. If omitted, the provider adapter\n chooses its default.\n- `reasoningLevel` (optional): non-empty provider-native string. Orcha does\n not translate values between providers.\n- `outputType` (optional): `\"text\"` (default) or `\"json\"`. Image and\n audio are reserved but not implemented.\n- `outputSchema` (required for JSON output): JSON Schema used for provider\n structured output and final validation.\n\n`instructions.md` is required and cannot be empty. At compile time it becomes\nthe base system prompt. Orcha appends the compact available-skill catalog and\nthe full instructions for skills already loaded in this durable session.\n\n## Subagents\n\nRegister private subagents alongside a parent in `orcha/index.ts`:\n\n```js\nagents: {\n coordinator: {\n path: \"./coordinator\",\n subagents: {\n researcher: \"./researcher\",\n },\n },\n researcher: \"./researcher\",\n}\n```\n\nOnly top-level keys become `orcha.<agentName>`. In this example the\nresearcher is both directly accessible and available to the coordinator.\nRemove its top-level entry to make it private.\n\nA parent may configure delegation limits in its `index.json`:\n\n```json\n{\n \"subagents\": {\n \"maxPerRun\": 3\n }\n}\n```\n\n`maxPerRun` limits newly created child sessions in one parent run and\ndefaults to 10.\n\nThe parent receives a compact catalog containing each subagent's registered\nname and optional description. Internal `run_agent`, `resume_agent`,\n`inspect_agent` tools let it start, continue, and inspect only child sessions\ninitiated by its current session. A delegated agent cannot delegate again.\n\nSubagent calls are synchronous. The parent becomes\n`waiting_for_subagent` while the child runs, then receives the child's text,\npause state, client-action request, failure, or completion as a normal tool\nresult. The parent and child keep separate linked JSONL sessions.\n\n## Core execution model\n\nAn **agent** is the compiled definition: model settings, instructions, actions,\nskills, and evaluations. An agent can create many independent sessions.\n\nA **session** is one durable conversation owned by one agent. It has one\n`sessionId`, optional name and metadata, fixed prompt variables, and one\nappend-only JSONL timeline. Completing one response does not close the\nsession\u2014the application can resume it later. A session cannot be transferred\nto another registered agent, but later runs may use a different provider or\nmodel if that same agent's configuration changes.\n\nA **run** is one attempt to advance a session. `run()` creates a session and\nits first run. A conversational `resume(sessionId, { content })` creates the\nnext numbered run in that session. Each run accumulates its own model usage and\nends in exactly one of these states:\n\n- `completed`: the model produced final output.\n- `waiting_for_subagent`: the parent is waiting for a synchronous child\n response and continues automatically when it arrives.\n- `waiting_for_client_action`: the model requested work that only the\n application can perform. The run is paused, not completed.\n- `paused`: active provider work was aborted, streamed output was preserved,\n and a later `resume()` starts a new run.\n- `failed`: validation, provider, storage, or execution failed. The durable\n events remain available for diagnosis.\n\nAn **execution** is the in-process handle returned by one call to `run()` or\n`resume()`. It exposes a cumulative output stream, latest snapshot, final\nresult promise, and evaluation promise. An execution ends when that invocation\ncompletes, pauses, or fails; the durable session may continue through another\nexecution.\n\nA **model round** is one provider request inside a run. One run may contain\nseveral rounds:\n\n```text\nuser input\n \u2192 model round\n \u2192 tool calls\n \u2192 tool results\n \u2192 another model round\n \u2192 final answer\n```\n\nLocal actions and skill loads are handled automatically inside the same\nexecution. Their results are sent back to the model and the model loop\ncontinues without application involvement.\n\nA **client action** deliberately crosses the application boundary. Orcha can\ndescribe the tool to the model but cannot execute it because the operation\nbelongs to a browser, mobile app, approval system, or other caller-owned\nenvironment. The complete pause/continue flow is:\n\n```text\n1. Application calls agent.run(...) or agent.resume(...content).\n2. Model requests one or more client actions.\n3. Orcha stores client_action.requested and run.paused.\n4. execution.result resolves with:\n {\n status: \"waiting_for_client_action\",\n sessionId,\n clientToolCalls: [{ callId, name, arguments }]\n }\n5. Application executes every requested action.\n6. Application calls agent.resume(sessionId, {\n toolResults: [{ callId, output, isError? }]\n }).\n7. Orcha validates every callId and output, stores the results, and continues\n the same paused run from its prior model context.\n8. The resumed execution either completes, requests more client actions, or\n fails.\n```\n\nEvery pending call must be resolved exactly once in one resume operation.\n`callId` links the submitted result to the model's request; the action name\nmust not be substituted for it. Re-submitting the identical resolved result is\nidempotent and returns the prior completed result. Submitting different data\nfor an already-resolved call fails with `action_result_conflict`.\n\n`clientCapabilities` is supplied per invocation because different callers\nmay support different client actions. Orcha exposes only declared client\nactions to that model round. Local actions are always available when compiled.\n\nOnly one execution may mutate a session at a time. Concurrent calls for the\nsame `sessionId` return `session_busy`; different sessions can run\nindependently.\n\n## Running and resuming\n\n```ts\nconst execution = orcha.supportBot.run({\n content: \"Check subscription sub_123.\",\n name: \"Subscription check\",\n metadata: { accountId: \"acct_123\" },\n clientCapabilities: [\"request_human_approval\"],\n});\n\nfor await (const snapshot of execution.stream) {\n console.log(snapshot);\n}\n\nconst result = await execution.result;\nconst evaluations = await execution.evaluations;\n```\n\n`run()` input fields:\n\n- `content` (required): a non-empty string, one content item, or an array.\n Use `{ filePath: \"./document.pdf\" }` for a local file or\n `{ url: \"https://example.com/document.pdf\" }` for a remote file.\n `mimeType` is optional when it can be inferred from the extension. Durable\n events store local paths and MIME types, never encoded file bytes.\n- `name` (optional): trimmed session label from 1 through 200 characters.\n- `metadata` (optional): at most 50 fields with non-empty keys and finite\n string, number, boolean, or null values. Metadata is durable and available\n to local action context; never place secrets in it.\n- `variables` (optional): at most 50 string values whose keys are JavaScript\n identifiers. They replace `{{ variableName }}` placeholders in\n `instructions.md`, are fixed when the session is created, and are reused\n by later resumes. A missing referenced variable fails the run.\n- `clientCapabilities` (optional): action names the current caller can\n execute. Client actions not declared here are withheld from the model.\n\n`run()` always creates a new durable session. Continue one with:\n\n```ts\nconst execution = orcha.supportBot.resume(sessionId, {\n content: \"Continue with the confirmed account.\",\n});\n```\n\nIf a result has `status: \"waiting_for_client_action\"`, execute the requested\nclient actions in the application and submit every result:\n\n```ts\norcha.supportBot.resume(sessionId, {\n toolResults: [\n { callId: \"call_123\", output: { approved: true } }\n ],\n});\n```\n\nNever invent call IDs. Use the IDs returned in `clientToolCalls`.\n\n## Actions\n\nEach action has metadata and, for local actions, executable code:\n\n```text\nactions/\n lookupAccount/\n index.json\n index.js\n```\n\n```json\n{\n \"name\": \"lookup_account\",\n \"description\": \"Look up one account.\",\n \"execution\": \"local\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"accountId\": { \"type\": \"string\" }\n },\n \"required\": [\"accountId\"],\n \"additionalProperties\": false\n },\n \"outputSchema\": {\n \"type\": \"object\",\n \"properties\": {\n \"status\": { \"type\": \"string\" }\n },\n \"required\": [\"status\"],\n \"additionalProperties\": false\n }\n}\n```\n\n```js\nexport default async function lookupAccount({ accountId }) {\n return { status: \"active\" };\n}\n```\n\nAction entrypoints can use ordinary relative project imports, TypeScript\nmodules, npm packages, transitive dependencies, and standard exports. Orcha\nbundles the complete import graph for development and production.\n\nClient actions use `\"execution\": \"client\"` and do not include executable\ncode. Orcha pauses until the caller submits their results. Keep action names,\ndescriptions, schemas, and implementations aligned.\n\nAction `index.json` fields:\n\n- `name` (required): model-facing tool name, 1\u201364 letters, numbers,\n underscores, or hyphens. `load_skill` is reserved.\n- `description` (required): tells the model when and why to call the action.\n- `execution` (required): `\"local\"` executes `index.js`; `\"client\"`\n pauses the run and delegates execution to the application.\n- `parameters` (required): JSON Schema for model-generated arguments.\n- `outputSchema` (optional): JSON Schema validated against local or submitted\n client output before the model receives it.\n- `timeoutMs` (optional): integer from 1 through 120000; defaults to 10000.\n- `permissions.env` (optional): names copied from `orcha.init().actions.env`\n into the action context.\n- `permissions.network` (optional): exact hosts or wildcard subdomains such\n as `\"api.example.com\"` or `\"*.example.com\"` allowed through\n `context.fetch`. Redirects are rejected.\n- `sideEffect` (optional): descriptive metadata for whether the operation\n mutates external state. It does not currently change execution behavior.\n\n`orcha.init().actions.runtime` and an action's `execution` solve different\nproblems:\n\n- `execution: \"client\"`: Orcha never executes code for this action.\n- `execution: \"local\"` + `runtime: \"sandbox\"`: compiled code runs in a\n QuickJS isolate with JSON-only inputs/outputs, default 32 MB memory, default\n 512 KB stack, interruptible timeout, declared environment values, and\n allowlisted network access through the provided context.\n- `execution: \"local\"` + `runtime: \"native\"`: code runs in the host Node.js\n process. It can use host privileges directly. The timeout rejects slow\n asynchronous work but cannot interrupt synchronous blocking code.\n\nGlobal local-action configuration:\n\n```ts\nactions: {\n runtime: \"sandbox\", // required when any registered action is local\n env: {\n BILLING_API_TOKEN: process.env.BILLING_API_TOKEN,\n },\n sandbox: {\n memoryLimitMb: 32,\n stackLimitKb: 512,\n },\n}\n```\n\nThe local action signature is\n`(parameters, context) => output | Promise<output>`. Context contains\n`sessionId`, immutable session `metadata`, a stable `idempotencyKey`,\nallowlisted `env`, guarded `fetch`, and prefixed `log`.\n\n## Skills\n\nSkills are lazy-loaded procedural instructions. Register only intended skills:\n\n```js\n// skills/index.js\nimport { defineSkills } from \"orchajs/skills\";\n\nexport default defineSkills({\n incidentTriage: \"./incidentTriage\",\n});\n```\n\nEach skill folder contains `index.json` metadata and `instructions.md`.\nThe model receives a compact catalog and can call the internal `load_skill`\ntool. Loaded instructions remain active for the durable session. Lifecycle\nevents are `skill.requested`, `skill.loaded`, and `skill.failed`.\n\nSkill `index.json` fields:\n\n- `name` (required): model-facing name, 1\u201364 letters, numbers, underscores,\n or hyphens; unique within the agent.\n- `description` (required): compact catalog description shown before loading.\n- `triggers` (optional): non-empty array of non-empty situations describing\n when the model should load the skill.\n\n`instructions.md` is required and cannot be empty. The key in\n`skills/index.js` is only a registration label; `index.json.name` is the\nname used by the model and durable events. Unregistered folders are ignored.\n\n## Tests\n\nRegister tests in `tests/index.js`:\n\n```js\nimport { defineTests } from \"orchajs/testing\";\n\nexport default defineTests({\n activeAccount: \"./activeAccount\",\n});\n```\n\nEach case's `index.json` defines `input`, optional mocked action responses,\nand `expect`. Tests run the real compiled agent and provider. Actions listed\nunder `actions` are mocked; unlisted local actions execute live. Orcha reports\nlive action use in the terminal, test report, and session trace.\nClient actions must always be mocked because no application client is attached\nto the test runner.\n\n```json\n{\n \"input\": { \"content\": \"Check account acct_123.\" },\n \"actions\": {\n \"lookup_account\": {\n \"responses\": [\n { \"output\": { \"status\": \"active\" } }\n ]\n }\n },\n \"expect\": {\n \"status\": \"completed\",\n \"text\": { \"contains\": [\"active\"] },\n \"actions\": [\n {\n \"name\": \"lookup_account\",\n \"arguments\": { \"equals\": { \"accountId\": \"acct_123\" } }\n }\n ]\n }\n}\n```\n\nTest sessions use the `ses_test_<timestamp>_<uuid>` format and end with a `test.completed`\nevent. Prefer semantic output assertions; verify exact identifiers and values\nthrough action-argument assertions.\n\nTest `index.json` fields:\n\n- `description` (optional): human-readable purpose.\n- `input` (required): an inline agent input object or a path to a JSON file\n containing that object. Relative paths resolve from the test case directory.\n- `input.content` (required for inline input): string or multimodal content\n array.\n- `input.variables` (optional): string map available to the session.\n- `input.metadata` (optional): string, number, boolean, or null values.\n- `actions` (optional): mocked actions keyed by compiled action name. Unlisted\n local actions execute live and may cause side effects. Unknown action names\n are rejected, and unlisted client actions are rejected. Every `responses`\n array is consumed in call order.\n- `responses[].output` (required): mocked action result.\n- `responses[].isError` (optional): marks the mocked result as an error.\n- `expect.status` (optional): `\"completed\"` or `\"failed\"`; defaults to\n `\"completed\"`.\n- `expect.output.equals` / `partial` (optional): exact or recursive partial\n comparison against structured output.\n- `expect.text.contains` / `excludes` (optional): case-sensitive semantic\n text checks.\n- `expect.actions` (optional): ordered expected calls. Each may assert\n `arguments.equals` or `arguments.partial`.\n\nThe registration key in `tests/index.js` is the test selector used by\n`orcha test agent/testName`; its value resolves to the case folder.\n\n## Evaluations\n\nEvaluations are asynchronous LLM judges registered in\n`evaluations/index.js` with `defineEvaluations` from\n`orchajs/evaluations`. Each folder's `index.json` defines its provider,\nmodel, metrics, and thresholds:\n\n```json\n{\n \"name\": \"response_quality\",\n \"description\": \"Grounding of agent responses.\",\n \"instructions\": \"Judge the complete response using only confirmed evidence recorded in the session.\",\n \"enabled\": true,\n \"provider\": \"openai\",\n \"model\": \"gpt-5-mini\",\n \"metrics\": [\n {\n \"name\": \"groundedness\",\n \"description\": \"The answer relies on confirmed session evidence.\",\n \"threshold\": 0.8\n }\n ]\n}\n```\n\n`execution.result` does not wait for judges. Await\n`execution.evaluations` when results must finish before process exit.\nEvaluations always finish during `orcha test`; judge errors and missed\nthresholds fail the test. Lifecycle events are `evaluation.requested`,\n`evaluation.completed`, and `evaluation.failed`.\n\nEvaluation `index.json` fields:\n\n- `name` (required): durable model-facing identifier, 1\u201364 letters, numbers,\n underscores, or hyphens; unique within the agent.\n- `description` (optional): short human-facing summary of the evaluator.\n- `instructions` (optional): detailed prompt supplied to the judge. When\n omitted, `description` is used for backward compatibility.\n- `enabled` (optional): defaults to `true`. Disabled evaluations are\n compiled but do not run.\n- `provider` and `model` (required): independently select the judge. The\n provider must also exist in `orcha.init().providers`.\n- `maxTokens` (optional): positive integer; defaults to 2000 for judges.\n- `reasoningLevel` (optional): non-empty provider-native string forwarded\n without translation.\n- `metrics` (required): non-empty array with unique metric names.\n- `metrics[].name`: 1\u201364 letters, numbers, underscores, or hyphens.\n- `metrics[].description`: exact criterion supplied to the judge.\n- `metrics[].threshold`: inclusive number from 0 to 1. A metric passes when\n the returned score is greater than or equal to this threshold.\n\nThe judge sees a sanitized transcript of user/assistant messages, action\nrequests and outcomes, client-action activity, and loaded skill names. It does\nnot receive internal reasoning blocks, replay metadata, previous evaluation\nresults, or test assertions. It must return exactly one score, reasoning\nstring, and non-empty evidence array for every configured metric.\n\n## Sessions and logs\n\nJSONL is the durable source of truth. Each line is one complete JSON object;\nnever treat the file as one JSON array. Events are append-only and ordered by\n`sequence`.\n\nAll events use this envelope:\n\n```ts\ntype SessionEvent<T> = {\n sequence: number; // starts at 1 and increases across the whole session\n type: SessionEventType;\n timestamp: string; // ISO-8601 UTC timestamp\n run?: number; // present for run-scoped events\n data: T;\n};\n```\n\nSession-scoped events omit `run`. Optional properties whose values are\n`undefined` are omitted from serialized JSON.\n\nShared stored structures:\n\n```ts\ntype Usage = {\n inputTokens: number;\n outputTokens: number;\n reasoningTokens: number | null;\n cacheReadTokens: number;\n cacheWriteTokens: number;\n};\n\ntype ErrorData = {\n code: string;\n message: string;\n retryable?: boolean;\n};\n\ntype UserContent =\n | { type: \"text\"; text: string }\n | { type: \"file\"; filePath: string; mimeType: string }\n | {\n type: \"image\" | \"video\" | \"audio\" | \"url\";\n mimeType: string;\n fileUri: string;\n };\n\ntype AssistantContent =\n | { type: \"text\"; text: string }\n | {\n type: \"reasoning\";\n text: string;\n replay?: { providerId?: string; opaqueData?: string };\n }\n | {\n type: \"tool_call\";\n callId: string;\n name: string;\n arguments: Record<string, unknown>;\n replay?: { providerId?: string; opaqueData?: string };\n };\n\ntype ToolResult = {\n callId: string;\n output: unknown;\n isError?: boolean;\n};\n```\n\nExact event payloads:\n\n```ts\ntype SessionCreated = SessionEvent<{\n schemaVersion: 1;\n sessionId: string;\n agent: string;\n status: \"active\";\n name?: string;\n metadata: Record<string, string | number | boolean | null>;\n variables: Record<string, string>;\n lineage?: {\n origin: \"delegated\";\n parentAgent: string;\n parentSessionId: string;\n parentCallId: string;\n };\n}>; // type \"session.created\", no run\n\ntype SessionUpdated = SessionEvent<{\n name?: string;\n metadata?: Record<string, string | number | boolean | null>;\n}>; // type \"session.updated\", no run\n\ntype RunStarted = SessionEvent<{\n status: \"running\";\n agent: string;\n provider: string;\n model: string;\n region: string;\n reasoningLevel?: string;\n outputType: \"text\" | \"json\";\n clientCapabilities: string[];\n}>; // type \"run.started\"\n\ntype UserMessageCreated = SessionEvent<{\n role: \"user\";\n content: UserContent[];\n}>; // type \"message.created\"\n\ntype AssistantMessageCreated = SessionEvent<{\n status: \"completed\" | \"incomplete\";\n provider: string;\n model: string;\n responseId?: string;\n stopReason?: \"end_turn\" | \"tool_call\" | \"max_tokens\" |\n \"content_filter\" | \"unknown\";\n role: \"assistant\";\n content: AssistantContent[];\n parsedOutput?: unknown; // final JSON output only\n usage: Usage;\n durationMs: number;\n}>; // type \"message.created\"\n\ntype ToolMessageCreated = SessionEvent<{\n role: \"tool\";\n content: ToolResult[];\n}>; // type \"message.created\"\n\ntype ActionRequested = SessionEvent<{\n callId: string;\n name: string;\n arguments: Record<string, unknown>;\n sourceHash?: string;\n idempotencyKey: string; // sessionId:callId\n}>; // type \"action.requested\"\n\ntype ActionCompleted = SessionEvent<{\n callId: string;\n name: string;\n output: unknown;\n sourceHash?: string;\n durationMs: number;\n}>; // type \"action.completed\"\n\ntype ActionFailed = SessionEvent<{\n callId: string;\n name: string;\n sourceHash?: string;\n durationMs: number;\n error: {\n code: \"action_execution_failed\";\n message: string;\n };\n}>; // type \"action.failed\"\n\ntype ClientActionRequested = SessionEvent<{\n status: \"waiting\";\n calls: Array<{\n callId: string;\n name: string;\n arguments: Record<string, unknown>;\n }>;\n localResults: ToolResult[];\n toolCallOrder: string[];\n}>; // type \"client_action.requested\"\n\ntype ClientActionResolved = SessionEvent<{\n status: \"completed\";\n results: ToolResult[];\n}>; // type \"client_action.resolved\"\n\ntype SkillRequested = SessionEvent<{\n callId: string;\n name: unknown;\n}>; // type \"skill.requested\"\n\ntype SkillLoaded = SessionEvent<{\n callId: string;\n name: string;\n alreadyLoaded: boolean;\n}>; // type \"skill.loaded\", no run\n\ntype SkillFailed = SessionEvent<{\n callId: string;\n name: unknown;\n error: {\n code: \"skill_not_found\";\n message: string;\n };\n}>; // type \"skill.failed\"\n\ntype EvaluationRequested = SessionEvent<{\n name: string;\n provider: string;\n model: string;\n evaluatedThroughSequence: number;\n}>; // type \"evaluation.requested\"\n\ntype EvaluationMetric = {\n name: string;\n score: number;\n threshold: number;\n passed: boolean;\n reasoning: string;\n evidence: string[];\n};\n\ntype EvaluationCompleted = SessionEvent<{\n name: string;\n status: \"passed\" | \"failed\";\n metrics: EvaluationMetric[];\n usage?: Usage;\n durationMs: number;\n provider: string;\n model: string;\n evaluatedThroughSequence: number;\n}>; // type \"evaluation.completed\"\n\ntype EvaluationFailed = SessionEvent<{\n name: string;\n status: \"error\";\n metrics: [];\n durationMs: number;\n error: { message: string };\n provider: string;\n model: string;\n evaluatedThroughSequence: number;\n}>; // type \"evaluation.failed\"\n\ntype SubagentInitiated = SessionEvent<{\n callId: string;\n agent: string;\n sessionId: string;\n status: \"running\";\n}>; // type \"subagent.initiated\"\n\ntype SubagentResumed = SessionEvent<{\n callId: string;\n agent: string;\n sessionId: string;\n status: \"running\";\n}>; // type \"subagent.resumed\"\n\ntype SubagentCompletedOrPausedOrFailed = SessionEvent<{\n callId: string;\n agent: string;\n sessionId: string;\n status:\n | \"completed\"\n | \"waiting_for_client_action\"\n | \"paused\"\n | \"failed\";\n output?: unknown;\n usage?: Usage;\n clientToolCalls?: Array<{\n callId: string;\n name: string;\n arguments: Record<string, unknown>;\n }>;\n error?: ErrorData;\n}>; // type \"subagent.completed\" | \"subagent.paused\" | \"subagent.failed\"\n\ntype RunPaused = SessionEvent<{\n status: \"waiting_for_client_action\" | \"paused\";\n clientToolCalls?: Array<{\n callId: string;\n name: string;\n arguments: Record<string, unknown>;\n }>;\n usage?: Usage;\n output?: unknown;\n durationMs?: number;\n}>; // type \"run.paused\"\n\ntype RunCompleted = SessionEvent<{\n status: \"completed\";\n durationMs: number;\n usage: Usage;\n}>; // type \"run.completed\"\n\ntype RunFailed = SessionEvent<{\n status: \"failed\";\n durationMs: number;\n usage?: Usage;\n error: ErrorData;\n}>; // type \"run.failed\"\n\ntype SessionPaused = SessionEvent<{\n status: \"paused\";\n}>; // type \"session.paused\", no run\n\ntype SessionResumed = SessionEvent<{\n status: \"active\";\n}>; // type \"session.resumed\", no run\n\ntype TestCompleted = SessionEvent<{\n suiteId: string;\n agent: string;\n test: string;\n status: \"passed\" | \"failed\";\n durationMs: number;\n assertions: Array<{\n path: string;\n passed: boolean;\n message: string;\n expected?: unknown;\n actual?: unknown;\n }>;\n usage?: Usage;\n evaluations?: Array<{\n name: string;\n status: \"passed\" | \"failed\" | \"error\";\n metrics: EvaluationMetric[];\n usage?: Usage;\n durationMs: number;\n error?: { message: string };\n }>;\n warnings?: Array<{\n code: \"live_actions\";\n message: string;\n actions: string[];\n }>;\n error?: ErrorData;\n}>; // type \"test.completed\", no run\n\ntype TestWarning = SessionEvent<{\n code: \"live_actions\";\n message: string;\n actions: string[];\n}>; // type \"test.warning\", no run\n```\n\nTypical event order:\n\n```text\nsession.created\nrun.started\nmessage.created (user)\nmessage.created (assistant, possibly with tool_call)\naction.requested \u2192 action.completed|action.failed # local action\nsubagent.initiated\n...child session advances independently...\nsubagent.paused\nsubagent.resumed\n...child session advances independently...\nsubagent.completed|subagent.paused|subagent.failed\nmessage.created (tool)\n...additional model/action rounds...\nmessage.created (assistant final)\nrun.completed\nevaluation.requested\nevaluation.completed|evaluation.failed\n```\n\nFor client actions, `client_action.requested` and `run.paused` replace the\nimmediate tool message. A later `resume(...toolResults)` appends\n`client_action.resolved`, the tool message, and continues the same run\nnumber. A conversational `resume(...content)` starts a new run number.\n\n## How Orcha works behind the scenes\n\n### Compilation\n\n1. `orcha/index.ts` calls `orcha.init()` with explicit agent paths.\n2. The compiler reads each registered agent's `index.json` and\n `instructions.md`.\n3. Every directory under `actions/` is compiled. Skills, tests, and\n evaluations are included only through their local `index.js` registry.\n4. Local action source is bundled and SHA-256 hashed. Production bundles keep\n only provider adapters required by agents and enabled evaluations.\n5. Invalid paths, duplicate model-facing names, missing files, unsupported\n configuration values, and missing schema objects fail before execution.\n Concrete action arguments and outputs are validated against their schemas\n when the action is used.\n\n`orcha dev` repeats validation when files change. `orcha build` performs\noffline production compilation. Neither command invokes a provider.\n\n### Run lifecycle\n\n1. `run()` creates a `ses_<timestamp>_<uuid>`, acquires the per-session execution lock,\n appends `session.created`, then starts run 1.\n2. `resume()` reads and validates the existing session. Message continuation\n starts a new run; submitted client results continue the paused run.\n3. The provider receives the base instructions, available-skill catalog,\n loaded skill instructions, normalized conversation messages, action\n schemas, model settings, and current client capabilities.\n4. Provider-specific responses are normalized into text, reasoning, and tool\n call blocks. Opaque replay metadata is stored only when a provider needs it\n to replay its own prior block correctly.\n5. Tool calls are checked against compiled actions and declared client\n capabilities. Arguments and outputs are validated against JSON Schema.\n6. `load_skill` updates durable session instructions. Local actions execute\n through the configured runtime. Client actions pause safely. Tool results\n are reordered to match the model's original call order.\n7. Internal agent tools start or continue linked child sessions. A\n `run_agent` input accepts the same text or multimodal content shape as a\n normal run. Local file paths resolve from the configured Orcha project root.\n The parent reports `waiting_for_subagent` until each synchronous child\n call returns.\n8. The model loop continues until final output, failure, a pause, or the\n maximum of 10 action rounds.\n9. Usage is normalized and aggregated across every model call in the run.\n10. After `run.completed`, enabled evaluations start in the background.\n `execution.result` is already available; `execution.evaluations` waits\n for judge completion and durable persistence.\n\nThe per-session lock prevents two model executions from mutating one session\nat once. Background evaluation writes queue behind active runs so they cannot\ncause `resume()` to fail spuriously or reuse sequence numbers.\n\n### Replay and context\n\nOrcha does not send raw JSONL back to the model. It projects durable events\ninto provider-neutral conversation messages. Completed and explicitly paused\nconversational runs become user, assistant, and tool messages, so a new run\nsees incomplete assistant text preserved by `pause()`. Lifecycle bookkeeping\nsuch as durations, test assertions, and evaluation events is excluded from\nmodel context. Reasoning text and provider replay metadata are retained where\nneeded for faithful continuation but are omitted from evaluation transcripts.\n\n### Reading sessions\n\n- `agent.get(sessionId)` projects the latest status, pending client actions,\n last output, metadata, and aggregate usage.\n- `agent.history(sessionId, { page, pageSize })` returns a safe user-facing\n timeline rather than raw provider bookkeeping.\n- `agent.events(sessionId, { page, pageSize })` returns the canonical durable\n event records for observability, audit, and debugging tools. For subsequent\n pages of a changing session, pass the first response's `throughSequence`\n back in the options to keep pagination on a stable event boundary.\n- `agent.list({ page, pageSize, status, metadata })` lists projected session\n snapshots.\n- `agent.subagentHistory(childSessionId, { page, pageSize })` returns the\n projected history of a child owned by this parent agent. Lineage checks\n prevent access through unrelated agents.\n- `agent.subagentEvents(childSessionId, { page, pageSize })` returns that\n owned child's canonical events for full transcript and debugging interfaces.\n- `agent.pause(sessionId)` aborts active provider work for the session and\n active children, preserving streamed text as an incomplete assistant\n message. `agent.resume(sessionId)` starts a new run with `\"Continue.\"`;\n pass content to give different instructions. Pending client actions still\n require their exact tool results.\n- Never read storage files directly. Use these methods so applications remain\n compatible with JSONL, SQLite, IndexedDB, remote, and future storage adapters.\n\n## Change rules\n\n- Register every new agent, skill, test, and evaluation explicitly.\n- Keep runtime behavior provider-neutral.\n- Mock actions when tests must avoid side effects; document intentional live\n action coverage.\n- Do not commit `.orcha/`; it contains generated output and session data.\n- Run `orcha dev` after filesystem changes and `orcha test` when behavior\n changes.\n- Do not weaken assertions to match incorrect behavior. Remove an assertion\n only when it is stricter than the documented agent contract.\n";
|
|
1
|
+
export declare const AGENTS_MD = "# Orcha project guide\n\nThis repository uses OrchaJS, a filesystem-convention framework for durable\nAI agents. Treat the `orcha/` directory as source code. Do not edit generated\nfiles under `.orcha/`.\n\n## Commands\n\n- `orcha init` creates the initial Orcha files without overwriting files.\n- `orcha dev` validates agent source and watches `orcha/**` for changes.\n- `orcha playground` opens the built-in local UI and hot reloads agent source.\n- `orcha run <agent> --input \"\u2026\"` executes one registered agent.\n- `orcha run <agent> --input-file request.json` accepts structured input.\n- `orcha run <agent> --session <id> --input \"\u2026\"` continues a session.\n- `orcha run <agent> --session <id> --tool-results results.json` submits\n pending client-action results.\n- `orcha test` runs every enabled discovered agent test.\n- `orcha test <agent>` or `orcha test <agent>/<case>` narrows the run.\n- `orcha build` creates the production Orcha bundle without calling models.\n- Add `--json` to `run` and `test` for machine-readable output.\n\n`run`, `test`, and playground executions use real providers and require\ncredentials. `dev` and `build` are offline. Session logs are JSONL files under\n`.orcha/sessions/<agentName>/ses_<timestamp>_<uuid>.jsonl`.\n\n## Registry\n\n`orcha/index.ts` initializes providers and explicitly registers agents:\n\n```ts\nimport { orcha } from \"orchajs\";\n\norcha.init({\n providers: {\n anthropic: process.env.ANTHROPIC_API_KEY ?? \"\",\n openai: process.env.OPENAI_API_KEY ?? \"\",\n },\n agents: {\n supportBot: \"./supportBot\",\n },\n});\n```\n\nOnly registered agents are compiled. Agent keys become runtime properties\nsuch as `orcha.supportBot`. Local actions use the native Node.js runtime by\ndefault. Set `actions.runtime: \"sandbox\"` only when isolated execution is\nrequired.\n\n`orcha.init()` fields:\n\n- `providers` (required): provider configurations keyed by built-in provider\n name.\n- `agents` (required): runtime property names mapped to folders relative to\n `orcha/`. At least one agent is required.\n- `actions` (optional): overrides the native local-action runtime or configures\n its environment and sandbox settings.\n- `storage.strategy` (optional): currently only `\"node-jsonl\"`.\n- `storage.directory` (optional): session directory relative to project\n root; defaults to `.orcha/sessions`.\n- `root` (optional): absolute or working-directory-relative project root;\n defaults to `ORCHA_PROJECT_ROOT` and then `process.cwd()`.\n\nThe Orcha library reads the host application's existing `process.env`.\nStandalone Orcha CLI commands also load `.env` from the project root without\noverriding environment variables already in the process.\n\nProvider configuration shapes:\n\n```ts\nproviders: {\n anthropic: process.env.ANTHROPIC_API_KEY ?? \"\",\n deepseek: process.env.DEEPSEEK_API_KEY ?? \"\",\n googlegenai: process.env.GOOGLE_API_KEY ?? \"\",\n openai: {\n apiKey: process.env.OPENAI_API_KEY ?? \"\",\n baseUrl: \"https://api.openai.com/v1\", // optional override\n },\n vertexai: {\n project: process.env.GOOGLE_CLOUD_PROJECT ?? \"\",\n location: process.env.GOOGLE_CLOUD_LOCATION ?? \"us-central1\",\n // credentials is optional; omit it to use Google ADC.\n credentials: {\n clientEmail: process.env.GOOGLE_CLIENT_EMAIL ?? \"\",\n privateKey: process.env.GOOGLE_PRIVATE_KEY ?? \"\",\n },\n baseUrl: undefined, // optional override\n },\n bedrock: {\n region: process.env.AWS_REGION ?? \"us-east-1\",\n // credentials is optional; omit it to use the AWS credential chain.\n credentials: {\n accessKeyId: process.env.AWS_ACCESS_KEY_ID ?? \"\",\n secretAccessKey: process.env.AWS_SECRET_ACCESS_KEY ?? \"\",\n sessionToken: process.env.AWS_SESSION_TOKEN,\n },\n baseUrl: undefined, // optional override\n },\n}\n```\n\nAPI-key providers accept either a string shorthand or\n`{ apiKey, baseUrl? }`. Vertex AI requires `project` and `location`;\nexplicit service-account credentials are optional. Bedrock requires `region`;\nexplicit AWS credentials are optional. Never place credentials in\n`index.json`, instructions, tests, session metadata, or committed files.\n\n## Agent folders\n\n```text\norcha/\n index.ts\n supportBot/\n index.json\n instructions.md\n actions/\n skills/\n tests/\n evaluations/\n```\n\n`index.json` selects the model:\n\n```json\n{\n \"name\": \"Support Agent\",\n \"description\": \"Resolve customer support questions using confirmed account data.\",\n \"provider\": \"anthropic\",\n \"model\": \"claude-sonnet-4-6\",\n \"region\": \"provider_managed\",\n \"maxTokens\": 10240,\n \"outputType\": \"text\"\n}\n```\n\nOptional fields include `reasoningLevel`, `outputType: \"json\"`, and an\n`outputSchema` JSON Schema. Provider-specific reasoning values are forwarded\nwithout translation. Put the agent's stable role, boundaries, and operating\ninstructions in `instructions.md`.\n\nAgent `index.json` fields:\n\n- `name` (required): concise human-readable agent name.\n- `description` (optional): what the agent does and when it should be used.\n- `provider` (required): `\"anthropic\"`, `\"bedrock\"`, `\"deepseek\"`,\n `\"openai\"`, `\"googlegenai\"`, or `\"vertexai\"`.\n- `model` (required): exact provider model identifier.\n- `region` (optional): provider/model routing hint; defaults in durable\n metadata to `\"provider_managed\"`.\n- `maxTokens` (optional): positive integer. If omitted, the provider adapter\n chooses its default.\n- `reasoningLevel` (optional): non-empty provider-native string. Orcha does\n not translate values between providers.\n- `outputType` (optional): `\"text\"` (default) or `\"json\"`. Image and\n audio are reserved but not implemented.\n- `outputSchema` (required for JSON output): JSON Schema used for provider\n structured output and final validation.\n\n`instructions.md` is required and cannot be empty. At compile time it becomes\nthe base system prompt. Orcha appends the compact available-skill catalog and\nthe full instructions for skills already loaded in this durable session.\n\n## Subagents\n\nRegister private subagents alongside a parent in `orcha/index.ts`:\n\n```js\nagents: {\n coordinator: {\n path: \"./coordinator\",\n subagents: {\n researcher: \"./researcher\",\n },\n },\n researcher: \"./researcher\",\n}\n```\n\nOnly top-level keys become `orcha.<agentName>`. In this example the\nresearcher is both directly accessible and available to the coordinator.\nRemove its top-level entry to make it private.\n\nA parent may configure delegation limits in its `index.json`:\n\n```json\n{\n \"subagents\": {\n \"maxPerRun\": 3\n }\n}\n```\n\n`maxPerRun` limits newly created child sessions in one parent run and\ndefaults to 10.\n\nThe parent receives a compact catalog containing each subagent's registered\nname and optional description. Internal `run_agent`, `resume_agent`,\n`inspect_agent` tools let it start, continue, and inspect only child sessions\ninitiated by its current session. A delegated agent cannot delegate again.\n\nSubagent calls are synchronous. The parent becomes\n`waiting_for_subagent` while the child runs, then receives the child's text,\npause state, client-action request, failure, or completion as a normal tool\nresult. The parent and child keep separate linked JSONL sessions.\n\n## Core execution model\n\nAn **agent** is the compiled definition: model settings, instructions, actions,\nskills, and evaluations. An agent can create many independent sessions.\n\nA **session** is one durable conversation owned by one agent. It has one\n`sessionId`, optional name and metadata, fixed prompt variables, and one\nappend-only JSONL timeline. Completing one response does not close the\nsession\u2014the application can resume it later. A session cannot be transferred\nto another registered agent, but later runs may use a different provider or\nmodel if that same agent's configuration changes.\n\nA **run** is one attempt to advance a session. `run()` creates a session and\nits first run. A conversational `resume(sessionId, { content })` creates the\nnext numbered run in that session. Each run accumulates its own model usage and\nends in exactly one of these states:\n\n- `completed`: the model produced final output.\n- `waiting_for_subagent`: the parent is waiting for a synchronous child\n response and continues automatically when it arrives.\n- `waiting_for_client_action`: the model requested work that only the\n application can perform. The run is paused, not completed.\n- `paused`: active provider work was aborted, streamed output was preserved,\n and a later `resume()` starts a new run.\n- `failed`: validation, provider, storage, or execution failed. The durable\n events remain available for diagnosis.\n\nAn **execution** is the in-process handle returned by one call to `run()` or\n`resume()`. It exposes a cumulative output stream, latest snapshot, final\nresult promise, and evaluation promise. An execution ends when that invocation\ncompletes, pauses, or fails; the durable session may continue through another\nexecution.\n\nA **model round** is one provider request inside a run. One run may contain\nseveral rounds:\n\n```text\nuser input\n \u2192 model round\n \u2192 tool calls\n \u2192 tool results\n \u2192 another model round\n \u2192 final answer\n```\n\nLocal actions and skill loads are handled automatically inside the same\nexecution. Their results are sent back to the model and the model loop\ncontinues without application involvement.\n\nA **client action** deliberately crosses the application boundary. Orcha can\ndescribe the tool to the model but cannot execute it because the operation\nbelongs to a browser, mobile app, approval system, or other caller-owned\nenvironment. The complete pause/continue flow is:\n\n```text\n1. Application calls agent.run(...) or agent.resume(...content).\n2. Model requests one or more client actions.\n3. Orcha stores client_action.requested and run.paused.\n4. execution.result resolves with:\n {\n status: \"waiting_for_client_action\",\n sessionId,\n clientToolCalls: [{ callId, name, arguments }]\n }\n5. Application executes every requested action.\n6. Application calls agent.resume(sessionId, {\n toolResults: [{ callId, output, isError? }]\n }).\n7. Orcha validates every callId and output, stores the results, and continues\n the same paused run from its prior model context.\n8. The resumed execution either completes, requests more client actions, or\n fails.\n```\n\nEvery pending call must be resolved exactly once in one resume operation.\n`callId` links the submitted result to the model's request; the action name\nmust not be substituted for it. Re-submitting the identical resolved result is\nidempotent and returns the prior completed result. Submitting different data\nfor an already-resolved call fails with `action_result_conflict`.\n\n`clientCapabilities` is supplied per invocation because different callers\nmay support different client actions. Orcha exposes only declared client\nactions to that model round. Local actions are always available when compiled.\n\nOnly one execution may mutate a session at a time. Concurrent calls for the\nsame `sessionId` return `session_busy`; different sessions can run\nindependently.\n\n## Running and resuming\n\n```ts\nconst execution = orcha.supportBot.run({\n content: \"Check subscription sub_123.\",\n name: \"Subscription check\",\n metadata: { accountId: \"acct_123\" },\n clientCapabilities: [\"request_human_approval\"],\n});\n\nfor await (const snapshot of execution.stream) {\n console.log(snapshot);\n}\n\nconst result = await execution.result;\nconst evaluations = await execution.evaluations;\n```\n\n`run()` input fields:\n\n- `content` (required): a non-empty string, one content item, or an array.\n Use `{ filePath: \"./document.pdf\" }` for a local file or\n `{ url: \"https://example.com/document.pdf\" }` for a remote file.\n `mimeType` is optional when it can be inferred from the extension. Durable\n events store local paths and MIME types, never encoded file bytes.\n- `name` (optional): trimmed session label from 1 through 200 characters.\n- `metadata` (optional): at most 50 fields with non-empty keys and finite\n string, number, boolean, or null values. Metadata is durable and available\n to local action context; never place secrets in it.\n- `variables` (optional): at most 50 string values whose keys are JavaScript\n identifiers. They replace `{{ variableName }}` placeholders in\n `instructions.md`, are fixed when the session is created, and are reused\n by later resumes. A missing referenced variable fails the run.\n- `clientCapabilities` (optional): action names the current caller can\n execute. Client actions not declared here are withheld from the model.\n\n`run()` always creates a new durable session. Continue one with:\n\n```ts\nconst execution = orcha.supportBot.resume(sessionId, {\n content: \"Continue with the confirmed account.\",\n});\n```\n\nIf a result has `status: \"waiting_for_client_action\"`, execute the requested\nclient actions in the application and submit every result:\n\n```ts\norcha.supportBot.resume(sessionId, {\n toolResults: [\n { callId: \"call_123\", output: { approved: true } }\n ],\n});\n```\n\nNever invent call IDs. Use the IDs returned in `clientToolCalls`.\n\n## Actions\n\nEach action has metadata and, for local actions, executable code:\n\n```text\nactions/\n lookupAccount/\n index.json\n index.js\n```\n\n```json\n{\n \"name\": \"lookup_account\",\n \"description\": \"Look up one account.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"accountId\": { \"type\": \"string\" }\n },\n \"required\": [\"accountId\"],\n \"additionalProperties\": false\n },\n \"outputSchema\": {\n \"type\": \"object\",\n \"properties\": {\n \"status\": { \"type\": \"string\" }\n },\n \"required\": [\"status\"],\n \"additionalProperties\": false\n }\n}\n```\n\n```js\nexport default async function lookupAccount({ accountId }) {\n return { status: \"active\" };\n}\n```\n\nAction entrypoints can use ordinary relative project imports, TypeScript\nmodules, npm packages, transitive dependencies, and standard exports. Orcha\nbundles the complete import graph for development and production.\n\nClient actions use `\"execution\": \"client\"` and do not include executable\ncode. Orcha pauses until the caller submits their results. Keep action names,\ndescriptions, schemas, and implementations aligned.\n\nAction `index.json` fields:\n\n- `name` (required): model-facing tool name, 1\u201364 letters, numbers,\n underscores, or hyphens. `load_skill` is reserved.\n- `description` (required): tells the model when and why to call the action.\n- `enabled` (optional): defaults to `true`. Set to `false` to keep a WIP\n action inactive without deleting its folder.\n- `execution` (optional): defaults to `\"local\"`, which executes `index.js`.\n Set to `\"client\"` to pause and delegate execution to the application.\n- `parameters` (required): JSON Schema for model-generated arguments.\n- `outputSchema` (optional): JSON Schema validated against local or submitted\n client output before the model receives it.\n- `timeoutMs` (optional): integer from 1 through 120000; defaults to 10000.\n- `permissions.env` (optional): names copied from `orcha.init().actions.env`\n into the action context.\n- `permissions.network` (optional): exact hosts or wildcard subdomains such\n as `\"api.example.com\"` or `\"*.example.com\"` allowed through\n `context.fetch`. Redirects are rejected.\n- `sideEffect` (optional): descriptive metadata for whether the operation\n mutates external state. It does not currently change execution behavior.\n\n`orcha.init().actions.runtime` and an action's `execution` solve different\nproblems:\n\n- `execution: \"client\"`: Orcha never executes code for this action.\n- `execution: \"local\"` + `runtime: \"sandbox\"`: compiled code runs in a\n QuickJS isolate with JSON-only inputs/outputs, default 32 MB memory, default\n 512 KB stack, interruptible timeout, declared environment values, and\n allowlisted network access through the provided context.\n- `execution: \"local\"` + `runtime: \"native\"`: code runs in the host Node.js\n process. It can use host privileges directly. The timeout rejects slow\n asynchronous work but cannot interrupt synchronous blocking code.\n\nGlobal local-action configuration:\n\n```ts\nactions: {\n runtime: \"sandbox\", // optional; native is the default\n env: {\n BILLING_API_TOKEN: process.env.BILLING_API_TOKEN,\n },\n sandbox: {\n memoryLimitMb: 32,\n stackLimitKb: 512,\n },\n}\n```\n\nThe local action signature is\n`(parameters, context) => output | Promise<output>`. Context contains\n`sessionId`, immutable session `metadata`, a stable `idempotencyKey`,\nallowlisted `env`, guarded `fetch`, and prefixed `log`.\n\n## Skills\n\nSkills are lazy-loaded procedural instructions. Direct child folders under\n`skills/` are discovered automatically. Each folder contains `index.json`\nmetadata and `instructions.md`.\nThe model receives a compact catalog and can call the internal `load_skill`\ntool. Loaded instructions remain active for the durable session. Lifecycle\nevents are `skill.requested`, `skill.loaded`, and `skill.failed`.\n\nSkill `index.json` fields:\n\n- `name` (required): model-facing name, 1\u201364 letters, numbers, underscores,\n or hyphens; unique within the agent.\n- `description` (required): compact catalog description shown before loading.\n- `enabled` (optional): defaults to `true`. Set to `false` to keep a WIP\n skill inactive.\n- `triggers` (optional): non-empty array of non-empty situations describing\n when the model should load the skill.\n\n`instructions.md` is required and cannot be empty for enabled skills.\n`index.json.name` is the name used by the model and durable events.\n\n## Tests\n\nDirect child folders under `tests/` are discovered automatically. Each\ncase's `index.json` defines `input`, optional mocked action responses, and\n`expect`. Tests run the real compiled agent and provider. Actions listed\nunder `actions` are mocked; unlisted local actions execute live. Orcha reports\nlive action use in the terminal, test report, and session trace.\nClient actions must always be mocked because no application client is attached\nto the test runner.\n\n```json\n{\n \"input\": { \"content\": \"Check account acct_123.\" },\n \"actions\": {\n \"lookup_account\": {\n \"responses\": [\n { \"output\": { \"status\": \"active\" } }\n ]\n }\n },\n \"expect\": {\n \"status\": \"completed\",\n \"text\": { \"contains\": [\"active\"] },\n \"actions\": [\n {\n \"name\": \"lookup_account\",\n \"arguments\": { \"equals\": { \"accountId\": \"acct_123\" } }\n }\n ]\n }\n}\n```\n\nTest sessions use the `ses_test_<timestamp>_<uuid>` format and end with a `test.completed`\nevent. Prefer semantic output assertions; verify exact identifiers and values\nthrough action-argument assertions.\n\nTest `index.json` fields:\n\n- `name` (optional): friendly display name. The folder name remains the CLI\n selector.\n- `description` (optional): human-readable purpose.\n- `enabled` (optional): defaults to `true`. Set to `false` to keep a WIP test\n inactive.\n- `input` (required): an inline agent input object or a path to a JSON file\n containing that object. Relative paths resolve from the test case directory.\n- `input.content` (required for inline input): string or multimodal content\n array.\n- `input.variables` (optional): string map available to the session.\n- `input.metadata` (optional): string, number, boolean, or null values.\n- `actions` (optional): mocked actions keyed by compiled action name. Unlisted\n local actions execute live and may cause side effects. Unknown action names\n are rejected, and unlisted client actions are rejected. Every `responses`\n array is consumed in call order.\n- `responses[].output` (required): mocked action result.\n- `responses[].isError` (optional): marks the mocked result as an error.\n- `expect.status` (optional): `\"completed\"` or `\"failed\"`; defaults to\n `\"completed\"`.\n- `expect.output.equals` / `partial` (optional): exact or recursive partial\n comparison against structured output.\n- `expect.text.contains` / `excludes` (optional): case-sensitive semantic\n text checks.\n- `expect.actions` (optional): ordered expected calls. Each may assert\n `arguments.equals` or `arguments.partial`.\n\nThe test folder name is the selector used by `orcha test agent/testName`.\n\n## Evaluations\n\nEvaluations are asynchronous LLM judges discovered from direct child folders\nunder `evaluations/`. Each folder's `index.json` defines its provider, model,\nmetrics, and thresholds:\n\n```json\n{\n \"name\": \"response_quality\",\n \"description\": \"Grounding of agent responses.\",\n \"instructions\": \"Judge the complete response using only confirmed evidence recorded in the session.\",\n \"enabled\": true,\n \"provider\": \"openai\",\n \"model\": \"gpt-5-mini\",\n \"metrics\": [\n {\n \"name\": \"groundedness\",\n \"description\": \"The answer relies on confirmed session evidence.\",\n \"threshold\": 0.8\n }\n ]\n}\n```\n\n`execution.result` does not wait for judges. Await\n`execution.evaluations` when results must finish before process exit.\nEvaluations always finish during `orcha test`; judge errors and missed\nthresholds fail the test. Lifecycle events are `evaluation.requested`,\n`evaluation.completed`, and `evaluation.failed`.\n\nEvaluation `index.json` fields:\n\n- `name` (required): durable model-facing identifier, 1\u201364 letters, numbers,\n underscores, or hyphens; unique within the agent.\n- `description` (optional): short human-facing summary of the evaluator.\n- `instructions` (optional): detailed prompt supplied to the judge. When\n omitted, `description` is used for backward compatibility.\n- `enabled` (optional): defaults to `true`. Disabled evaluations are ignored\n by compilation and runtime.\n- `provider` and `model` (required): independently select the judge. The\n provider must also exist in `orcha.init().providers`.\n- `maxTokens` (optional): positive integer; defaults to 2000 for judges.\n- `reasoningLevel` (optional): non-empty provider-native string forwarded\n without translation.\n- `metrics` (required): non-empty array with unique metric names.\n- `metrics[].name`: 1\u201364 letters, numbers, underscores, or hyphens.\n- `metrics[].description`: exact criterion supplied to the judge.\n- `metrics[].threshold`: inclusive number from 0 to 1. A metric passes when\n the returned score is greater than or equal to this threshold.\n\nThe judge sees a sanitized transcript of user/assistant messages, action\nrequests and outcomes, client-action activity, and loaded skill names. It does\nnot receive internal reasoning blocks, replay metadata, previous evaluation\nresults, or test assertions. It must return exactly one score, reasoning\nstring, and non-empty evidence array for every configured metric.\n\n## Sessions and logs\n\nJSONL is the durable source of truth. Each line is one complete JSON object;\nnever treat the file as one JSON array. Events are append-only and ordered by\n`sequence`.\n\nAll events use this envelope:\n\n```ts\ntype SessionEvent<T> = {\n sequence: number; // starts at 1 and increases across the whole session\n type: SessionEventType;\n timestamp: string; // ISO-8601 UTC timestamp\n run?: number; // present for run-scoped events\n data: T;\n};\n```\n\nSession-scoped events omit `run`. Optional properties whose values are\n`undefined` are omitted from serialized JSON.\n\nShared stored structures:\n\n```ts\ntype Usage = {\n inputTokens: number;\n outputTokens: number;\n reasoningTokens: number | null;\n cacheReadTokens: number;\n cacheWriteTokens: number;\n};\n\ntype ErrorData = {\n code: string;\n message: string;\n retryable?: boolean;\n};\n\ntype UserContent =\n | { type: \"text\"; text: string }\n | { type: \"file\"; filePath: string; mimeType: string }\n | {\n type: \"image\" | \"video\" | \"audio\" | \"url\";\n mimeType: string;\n fileUri: string;\n };\n\ntype AssistantContent =\n | { type: \"text\"; text: string }\n | {\n type: \"reasoning\";\n text: string;\n replay?: { providerId?: string; opaqueData?: string };\n }\n | {\n type: \"tool_call\";\n callId: string;\n name: string;\n arguments: Record<string, unknown>;\n replay?: { providerId?: string; opaqueData?: string };\n };\n\ntype ToolResult = {\n callId: string;\n output: unknown;\n isError?: boolean;\n};\n```\n\nExact event payloads:\n\n```ts\ntype SessionCreated = SessionEvent<{\n schemaVersion: 1;\n sessionId: string;\n agent: string;\n status: \"active\";\n name?: string;\n metadata: Record<string, string | number | boolean | null>;\n variables: Record<string, string>;\n lineage?: {\n origin: \"delegated\";\n parentAgent: string;\n parentSessionId: string;\n parentCallId: string;\n };\n}>; // type \"session.created\", no run\n\ntype SessionUpdated = SessionEvent<{\n name?: string;\n metadata?: Record<string, string | number | boolean | null>;\n}>; // type \"session.updated\", no run\n\ntype RunStarted = SessionEvent<{\n status: \"running\";\n agent: string;\n provider: string;\n model: string;\n region: string;\n reasoningLevel?: string;\n outputType: \"text\" | \"json\";\n clientCapabilities: string[];\n}>; // type \"run.started\"\n\ntype UserMessageCreated = SessionEvent<{\n role: \"user\";\n content: UserContent[];\n}>; // type \"message.created\"\n\ntype AssistantMessageCreated = SessionEvent<{\n status: \"completed\" | \"incomplete\";\n provider: string;\n model: string;\n responseId?: string;\n stopReason?: \"end_turn\" | \"tool_call\" | \"max_tokens\" |\n \"content_filter\" | \"unknown\";\n role: \"assistant\";\n content: AssistantContent[];\n parsedOutput?: unknown; // final JSON output only\n usage: Usage;\n durationMs: number;\n}>; // type \"message.created\"\n\ntype ToolMessageCreated = SessionEvent<{\n role: \"tool\";\n content: ToolResult[];\n}>; // type \"message.created\"\n\ntype ActionRequested = SessionEvent<{\n callId: string;\n name: string;\n arguments: Record<string, unknown>;\n sourceHash?: string;\n idempotencyKey: string; // sessionId:callId\n}>; // type \"action.requested\"\n\ntype ActionCompleted = SessionEvent<{\n callId: string;\n name: string;\n output: unknown;\n sourceHash?: string;\n durationMs: number;\n}>; // type \"action.completed\"\n\ntype ActionFailed = SessionEvent<{\n callId: string;\n name: string;\n sourceHash?: string;\n durationMs: number;\n error: {\n code: \"action_execution_failed\";\n message: string;\n };\n}>; // type \"action.failed\"\n\ntype ClientActionRequested = SessionEvent<{\n status: \"waiting\";\n calls: Array<{\n callId: string;\n name: string;\n arguments: Record<string, unknown>;\n }>;\n localResults: ToolResult[];\n toolCallOrder: string[];\n}>; // type \"client_action.requested\"\n\ntype ClientActionResolved = SessionEvent<{\n status: \"completed\";\n results: ToolResult[];\n}>; // type \"client_action.resolved\"\n\ntype SkillRequested = SessionEvent<{\n callId: string;\n name: unknown;\n}>; // type \"skill.requested\"\n\ntype SkillLoaded = SessionEvent<{\n callId: string;\n name: string;\n alreadyLoaded: boolean;\n}>; // type \"skill.loaded\", no run\n\ntype SkillFailed = SessionEvent<{\n callId: string;\n name: unknown;\n error: {\n code: \"skill_not_found\";\n message: string;\n };\n}>; // type \"skill.failed\"\n\ntype EvaluationRequested = SessionEvent<{\n name: string;\n provider: string;\n model: string;\n evaluatedThroughSequence: number;\n}>; // type \"evaluation.requested\"\n\ntype EvaluationMetric = {\n name: string;\n score: number;\n threshold: number;\n passed: boolean;\n reasoning: string;\n evidence: string[];\n};\n\ntype EvaluationCompleted = SessionEvent<{\n name: string;\n status: \"passed\" | \"failed\";\n metrics: EvaluationMetric[];\n usage?: Usage;\n durationMs: number;\n provider: string;\n model: string;\n evaluatedThroughSequence: number;\n}>; // type \"evaluation.completed\"\n\ntype EvaluationFailed = SessionEvent<{\n name: string;\n status: \"error\";\n metrics: [];\n durationMs: number;\n error: { message: string };\n provider: string;\n model: string;\n evaluatedThroughSequence: number;\n}>; // type \"evaluation.failed\"\n\ntype SubagentInitiated = SessionEvent<{\n callId: string;\n agent: string;\n sessionId: string;\n status: \"running\";\n}>; // type \"subagent.initiated\"\n\ntype SubagentResumed = SessionEvent<{\n callId: string;\n agent: string;\n sessionId: string;\n status: \"running\";\n}>; // type \"subagent.resumed\"\n\ntype SubagentCompletedOrPausedOrFailed = SessionEvent<{\n callId: string;\n agent: string;\n sessionId: string;\n status:\n | \"completed\"\n | \"waiting_for_client_action\"\n | \"paused\"\n | \"failed\";\n output?: unknown;\n usage?: Usage;\n clientToolCalls?: Array<{\n callId: string;\n name: string;\n arguments: Record<string, unknown>;\n }>;\n error?: ErrorData;\n}>; // type \"subagent.completed\" | \"subagent.paused\" | \"subagent.failed\"\n\ntype RunPaused = SessionEvent<{\n status: \"waiting_for_client_action\" | \"paused\";\n clientToolCalls?: Array<{\n callId: string;\n name: string;\n arguments: Record<string, unknown>;\n }>;\n usage?: Usage;\n output?: unknown;\n durationMs?: number;\n}>; // type \"run.paused\"\n\ntype RunCompleted = SessionEvent<{\n status: \"completed\";\n durationMs: number;\n usage: Usage;\n}>; // type \"run.completed\"\n\ntype RunFailed = SessionEvent<{\n status: \"failed\";\n durationMs: number;\n usage?: Usage;\n error: ErrorData;\n}>; // type \"run.failed\"\n\ntype SessionPaused = SessionEvent<{\n status: \"paused\";\n}>; // type \"session.paused\", no run\n\ntype SessionResumed = SessionEvent<{\n status: \"active\";\n}>; // type \"session.resumed\", no run\n\ntype TestCompleted = SessionEvent<{\n suiteId: string;\n agent: string;\n test: string;\n status: \"passed\" | \"failed\";\n durationMs: number;\n assertions: Array<{\n path: string;\n passed: boolean;\n message: string;\n expected?: unknown;\n actual?: unknown;\n }>;\n usage?: Usage;\n evaluations?: Array<{\n name: string;\n status: \"passed\" | \"failed\" | \"error\";\n metrics: EvaluationMetric[];\n usage?: Usage;\n durationMs: number;\n error?: { message: string };\n }>;\n warnings?: Array<{\n code: \"live_actions\";\n message: string;\n actions: string[];\n }>;\n error?: ErrorData;\n}>; // type \"test.completed\", no run\n\ntype TestWarning = SessionEvent<{\n code: \"live_actions\";\n message: string;\n actions: string[];\n}>; // type \"test.warning\", no run\n```\n\nTypical event order:\n\n```text\nsession.created\nrun.started\nmessage.created (user)\nmessage.created (assistant, possibly with tool_call)\naction.requested \u2192 action.completed|action.failed # local action\nsubagent.initiated\n...child session advances independently...\nsubagent.paused\nsubagent.resumed\n...child session advances independently...\nsubagent.completed|subagent.paused|subagent.failed\nmessage.created (tool)\n...additional model/action rounds...\nmessage.created (assistant final)\nrun.completed\nevaluation.requested\nevaluation.completed|evaluation.failed\n```\n\nFor client actions, `client_action.requested` and `run.paused` replace the\nimmediate tool message. A later `resume(...toolResults)` appends\n`client_action.resolved`, the tool message, and continues the same run\nnumber. A conversational `resume(...content)` starts a new run number.\n\n## How Orcha works behind the scenes\n\n### Compilation\n\n1. `orcha/index.ts` calls `orcha.init()` with explicit agent paths.\n2. The compiler reads each registered agent's `index.json` and\n `instructions.md`.\n3. Direct child folders under `actions/`, `skills/`, `tests/`, and\n `evaluations/` are discovered automatically. Items with `\"enabled\": false`\n remain inactive.\n4. Local action source is bundled and SHA-256 hashed. Production bundles keep\n only provider adapters required by agents and enabled evaluations.\n5. Invalid paths, duplicate model-facing names, missing files, unsupported\n configuration values, and missing schema objects fail before execution.\n Concrete action arguments and outputs are validated against their schemas\n when the action is used.\n\n`orcha dev` repeats validation when files change. `orcha build` performs\noffline production compilation. Neither command invokes a provider.\n\n### Run lifecycle\n\n1. `run()` creates a `ses_<timestamp>_<uuid>`, acquires the per-session execution lock,\n appends `session.created`, then starts run 1.\n2. `resume()` reads and validates the existing session. Message continuation\n starts a new run; submitted client results continue the paused run.\n3. The provider receives the base instructions, available-skill catalog,\n loaded skill instructions, normalized conversation messages, action\n schemas, model settings, and current client capabilities.\n4. Provider-specific responses are normalized into text, reasoning, and tool\n call blocks. Opaque replay metadata is stored only when a provider needs it\n to replay its own prior block correctly.\n5. Tool calls are checked against compiled actions and declared client\n capabilities. Arguments and outputs are validated against JSON Schema.\n6. `load_skill` updates durable session instructions. Local actions execute\n through the configured runtime. Client actions pause safely. Tool results\n are reordered to match the model's original call order.\n7. Internal agent tools start or continue linked child sessions. A\n `run_agent` input accepts the same text or multimodal content shape as a\n normal run. Local file paths resolve from the configured Orcha project root.\n The parent reports `waiting_for_subagent` until each synchronous child\n call returns.\n8. The model loop continues until final output, failure, a pause, or the\n maximum of 10 action rounds.\n9. Usage is normalized and aggregated across every model call in the run.\n10. After `run.completed`, enabled evaluations start in the background.\n `execution.result` is already available; `execution.evaluations` waits\n for judge completion and durable persistence.\n\nThe per-session lock prevents two model executions from mutating one session\nat once. Background evaluation writes queue behind active runs so they cannot\ncause `resume()` to fail spuriously or reuse sequence numbers.\n\n### Replay and context\n\nOrcha does not send raw JSONL back to the model. It projects durable events\ninto provider-neutral conversation messages. Completed and explicitly paused\nconversational runs become user, assistant, and tool messages, so a new run\nsees incomplete assistant text preserved by `pause()`. Lifecycle bookkeeping\nsuch as durations, test assertions, and evaluation events is excluded from\nmodel context. Reasoning text and provider replay metadata are retained where\nneeded for faithful continuation but are omitted from evaluation transcripts.\n\n### Reading sessions\n\n- `agent.get(sessionId)` projects the latest status, pending client actions,\n last output, metadata, and aggregate usage.\n- `agent.history(sessionId, { page, pageSize })` returns a safe user-facing\n timeline rather than raw provider bookkeeping.\n- `agent.events(sessionId, { page, pageSize })` returns the canonical durable\n event records for observability, audit, and debugging tools. For subsequent\n pages of a changing session, pass the first response's `throughSequence`\n back in the options to keep pagination on a stable event boundary.\n- `agent.list({ page, pageSize, status, metadata })` lists projected session\n snapshots.\n- `agent.subagentHistory(childSessionId, { page, pageSize })` returns the\n projected history of a child owned by this parent agent. Lineage checks\n prevent access through unrelated agents.\n- `agent.subagentEvents(childSessionId, { page, pageSize })` returns that\n owned child's canonical events for full transcript and debugging interfaces.\n- `agent.pause(sessionId)` aborts active provider work for the session and\n active children, preserving streamed text as an incomplete assistant\n message. `agent.resume(sessionId)` starts a new run with `\"Continue.\"`;\n pass content to give different instructions. Pending client actions still\n require their exact tool results.\n- Never read storage files directly. Use these methods so applications remain\n compatible with JSONL, SQLite, IndexedDB, remote, and future storage adapters.\n\n## Change rules\n\n- Register every new agent and subagent explicitly. Agent-owned actions, skills,\n tests, and evaluations are auto-discovered; use `\"enabled\": false` for WIP.\n- Keep runtime behavior provider-neutral.\n- Mock actions when tests must avoid side effects; document intentional live\n action coverage.\n- Do not commit `.orcha/`; it contains generated output and session data.\n- Run `orcha dev` after filesystem changes and `orcha test` when behavior\n changes.\n- Do not weaken assertions to match incorrect behavior. Remove an assertion\n only when it is stricter than the documented agent contract.\n";
|
|
2
2
|
//# sourceMappingURL=cli-agents-template.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"cli-agents-template.d.ts","sourceRoot":"","sources":["../src/cli-agents-template.ts"],"names":[],"mappings":"AAAA,eAAO,MAAM,SAAS,
|
|
1
|
+
{"version":3,"file":"cli-agents-template.d.ts","sourceRoot":"","sources":["../src/cli-agents-template.ts"],"names":[],"mappings":"AAAA,eAAO,MAAM,SAAS,qgqCA8hCrB,CAAC"}
|
|
@@ -7,14 +7,14 @@ files under \`.orcha/\`.
|
|
|
7
7
|
## Commands
|
|
8
8
|
|
|
9
9
|
- \`orcha init\` creates the initial Orcha files without overwriting files.
|
|
10
|
-
- \`orcha dev\` validates
|
|
10
|
+
- \`orcha dev\` validates agent source and watches \`orcha/**\` for changes.
|
|
11
11
|
- \`orcha playground\` opens the built-in local UI and hot reloads agent source.
|
|
12
12
|
- \`orcha run <agent> --input "…"\` executes one registered agent.
|
|
13
13
|
- \`orcha run <agent> --input-file request.json\` accepts structured input.
|
|
14
14
|
- \`orcha run <agent> --session <id> --input "…"\` continues a session.
|
|
15
15
|
- \`orcha run <agent> --session <id> --tool-results results.json\` submits
|
|
16
16
|
pending client-action results.
|
|
17
|
-
- \`orcha test\` runs every
|
|
17
|
+
- \`orcha test\` runs every enabled discovered agent test.
|
|
18
18
|
- \`orcha test <agent>\` or \`orcha test <agent>/<case>\` narrows the run.
|
|
19
19
|
- \`orcha build\` creates the production Orcha bundle without calling models.
|
|
20
20
|
- Add \`--json\` to \`run\` and \`test\` for machine-readable output.
|
|
@@ -35,17 +35,16 @@ orcha.init({
|
|
|
35
35
|
anthropic: process.env.ANTHROPIC_API_KEY ?? "",
|
|
36
36
|
openai: process.env.OPENAI_API_KEY ?? "",
|
|
37
37
|
},
|
|
38
|
-
actions: { runtime: "native" },
|
|
39
38
|
agents: {
|
|
40
39
|
supportBot: "./supportBot",
|
|
41
40
|
},
|
|
42
41
|
});
|
|
43
42
|
\`\`\`
|
|
44
43
|
|
|
45
|
-
Only registered
|
|
46
|
-
such as \`orcha.supportBot\`.
|
|
47
|
-
|
|
48
|
-
|
|
44
|
+
Only registered agents are compiled. Agent keys become runtime properties
|
|
45
|
+
such as \`orcha.supportBot\`. Local actions use the native Node.js runtime by
|
|
46
|
+
default. Set \`actions.runtime: "sandbox"\` only when isolated execution is
|
|
47
|
+
required.
|
|
49
48
|
|
|
50
49
|
\`orcha.init()\` fields:
|
|
51
50
|
|
|
@@ -53,8 +52,8 @@ allows action modules to execute in its Node.js process.
|
|
|
53
52
|
name.
|
|
54
53
|
- \`agents\` (required): runtime property names mapped to folders relative to
|
|
55
54
|
\`orcha/\`. At least one agent is required.
|
|
56
|
-
- \`actions\` (
|
|
57
|
-
|
|
55
|
+
- \`actions\` (optional): overrides the native local-action runtime or configures
|
|
56
|
+
its environment and sandbox settings.
|
|
58
57
|
- \`storage.strategy\` (optional): currently only \`"node-jsonl"\`.
|
|
59
58
|
- \`storage.directory\` (optional): session directory relative to project
|
|
60
59
|
root; defaults to \`.orcha/sessions\`.
|
|
@@ -363,7 +362,6 @@ actions/
|
|
|
363
362
|
{
|
|
364
363
|
"name": "lookup_account",
|
|
365
364
|
"description": "Look up one account.",
|
|
366
|
-
"execution": "local",
|
|
367
365
|
"parameters": {
|
|
368
366
|
"type": "object",
|
|
369
367
|
"properties": {
|
|
@@ -402,8 +400,10 @@ Action \`index.json\` fields:
|
|
|
402
400
|
- \`name\` (required): model-facing tool name, 1–64 letters, numbers,
|
|
403
401
|
underscores, or hyphens. \`load_skill\` is reserved.
|
|
404
402
|
- \`description\` (required): tells the model when and why to call the action.
|
|
405
|
-
- \`
|
|
406
|
-
|
|
403
|
+
- \`enabled\` (optional): defaults to \`true\`. Set to \`false\` to keep a WIP
|
|
404
|
+
action inactive without deleting its folder.
|
|
405
|
+
- \`execution\` (optional): defaults to \`"local"\`, which executes \`index.js\`.
|
|
406
|
+
Set to \`"client"\` to pause and delegate execution to the application.
|
|
407
407
|
- \`parameters\` (required): JSON Schema for model-generated arguments.
|
|
408
408
|
- \`outputSchema\` (optional): JSON Schema validated against local or submitted
|
|
409
409
|
client output before the model receives it.
|
|
@@ -432,7 +432,7 @@ Global local-action configuration:
|
|
|
432
432
|
|
|
433
433
|
\`\`\`ts
|
|
434
434
|
actions: {
|
|
435
|
-
runtime: "sandbox", //
|
|
435
|
+
runtime: "sandbox", // optional; native is the default
|
|
436
436
|
env: {
|
|
437
437
|
BILLING_API_TOKEN: process.env.BILLING_API_TOKEN,
|
|
438
438
|
},
|
|
@@ -450,18 +450,9 @@ allowlisted \`env\`, guarded \`fetch\`, and prefixed \`log\`.
|
|
|
450
450
|
|
|
451
451
|
## Skills
|
|
452
452
|
|
|
453
|
-
Skills are lazy-loaded procedural instructions.
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
// skills/index.js
|
|
457
|
-
import { defineSkills } from "orchajs/skills";
|
|
458
|
-
|
|
459
|
-
export default defineSkills({
|
|
460
|
-
incidentTriage: "./incidentTriage",
|
|
461
|
-
});
|
|
462
|
-
\`\`\`
|
|
463
|
-
|
|
464
|
-
Each skill folder contains \`index.json\` metadata and \`instructions.md\`.
|
|
453
|
+
Skills are lazy-loaded procedural instructions. Direct child folders under
|
|
454
|
+
\`skills/\` are discovered automatically. Each folder contains \`index.json\`
|
|
455
|
+
metadata and \`instructions.md\`.
|
|
465
456
|
The model receives a compact catalog and can call the internal \`load_skill\`
|
|
466
457
|
tool. Loaded instructions remain active for the durable session. Lifecycle
|
|
467
458
|
events are \`skill.requested\`, \`skill.loaded\`, and \`skill.failed\`.
|
|
@@ -471,27 +462,19 @@ Skill \`index.json\` fields:
|
|
|
471
462
|
- \`name\` (required): model-facing name, 1–64 letters, numbers, underscores,
|
|
472
463
|
or hyphens; unique within the agent.
|
|
473
464
|
- \`description\` (required): compact catalog description shown before loading.
|
|
465
|
+
- \`enabled\` (optional): defaults to \`true\`. Set to \`false\` to keep a WIP
|
|
466
|
+
skill inactive.
|
|
474
467
|
- \`triggers\` (optional): non-empty array of non-empty situations describing
|
|
475
468
|
when the model should load the skill.
|
|
476
469
|
|
|
477
|
-
\`instructions.md\` is required and cannot be empty
|
|
478
|
-
\`
|
|
479
|
-
name used by the model and durable events. Unregistered folders are ignored.
|
|
470
|
+
\`instructions.md\` is required and cannot be empty for enabled skills.
|
|
471
|
+
\`index.json.name\` is the name used by the model and durable events.
|
|
480
472
|
|
|
481
473
|
## Tests
|
|
482
474
|
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
import { defineTests } from "orchajs/testing";
|
|
487
|
-
|
|
488
|
-
export default defineTests({
|
|
489
|
-
activeAccount: "./activeAccount",
|
|
490
|
-
});
|
|
491
|
-
\`\`\`
|
|
492
|
-
|
|
493
|
-
Each case's \`index.json\` defines \`input\`, optional mocked action responses,
|
|
494
|
-
and \`expect\`. Tests run the real compiled agent and provider. Actions listed
|
|
475
|
+
Direct child folders under \`tests/\` are discovered automatically. Each
|
|
476
|
+
case's \`index.json\` defines \`input\`, optional mocked action responses, and
|
|
477
|
+
\`expect\`. Tests run the real compiled agent and provider. Actions listed
|
|
495
478
|
under \`actions\` are mocked; unlisted local actions execute live. Orcha reports
|
|
496
479
|
live action use in the terminal, test report, and session trace.
|
|
497
480
|
Client actions must always be mocked because no application client is attached
|
|
@@ -526,7 +509,11 @@ through action-argument assertions.
|
|
|
526
509
|
|
|
527
510
|
Test \`index.json\` fields:
|
|
528
511
|
|
|
512
|
+
- \`name\` (optional): friendly display name. The folder name remains the CLI
|
|
513
|
+
selector.
|
|
529
514
|
- \`description\` (optional): human-readable purpose.
|
|
515
|
+
- \`enabled\` (optional): defaults to \`true\`. Set to \`false\` to keep a WIP test
|
|
516
|
+
inactive.
|
|
530
517
|
- \`input\` (required): an inline agent input object or a path to a JSON file
|
|
531
518
|
containing that object. Relative paths resolve from the test case directory.
|
|
532
519
|
- \`input.content\` (required for inline input): string or multimodal content
|
|
@@ -548,15 +535,13 @@ Test \`index.json\` fields:
|
|
|
548
535
|
- \`expect.actions\` (optional): ordered expected calls. Each may assert
|
|
549
536
|
\`arguments.equals\` or \`arguments.partial\`.
|
|
550
537
|
|
|
551
|
-
The
|
|
552
|
-
\`orcha test agent/testName\`; its value resolves to the case folder.
|
|
538
|
+
The test folder name is the selector used by \`orcha test agent/testName\`.
|
|
553
539
|
|
|
554
540
|
## Evaluations
|
|
555
541
|
|
|
556
|
-
Evaluations are asynchronous LLM judges
|
|
557
|
-
\`evaluations
|
|
558
|
-
|
|
559
|
-
model, metrics, and thresholds:
|
|
542
|
+
Evaluations are asynchronous LLM judges discovered from direct child folders
|
|
543
|
+
under \`evaluations/\`. Each folder's \`index.json\` defines its provider, model,
|
|
544
|
+
metrics, and thresholds:
|
|
560
545
|
|
|
561
546
|
\`\`\`json
|
|
562
547
|
{
|
|
@@ -589,8 +574,8 @@ Evaluation \`index.json\` fields:
|
|
|
589
574
|
- \`description\` (optional): short human-facing summary of the evaluator.
|
|
590
575
|
- \`instructions\` (optional): detailed prompt supplied to the judge. When
|
|
591
576
|
omitted, \`description\` is used for backward compatibility.
|
|
592
|
-
- \`enabled\` (optional): defaults to \`true\`. Disabled evaluations are
|
|
593
|
-
|
|
577
|
+
- \`enabled\` (optional): defaults to \`true\`. Disabled evaluations are ignored
|
|
578
|
+
by compilation and runtime.
|
|
594
579
|
- \`provider\` and \`model\` (required): independently select the judge. The
|
|
595
580
|
provider must also exist in \`orcha.init().providers\`.
|
|
596
581
|
- \`maxTokens\` (optional): positive integer; defaults to 2000 for judges.
|
|
@@ -974,8 +959,9 @@ number. A conversational \`resume(...content)\` starts a new run number.
|
|
|
974
959
|
1. \`orcha/index.ts\` calls \`orcha.init()\` with explicit agent paths.
|
|
975
960
|
2. The compiler reads each registered agent's \`index.json\` and
|
|
976
961
|
\`instructions.md\`.
|
|
977
|
-
3.
|
|
978
|
-
evaluations are
|
|
962
|
+
3. Direct child folders under \`actions/\`, \`skills/\`, \`tests/\`, and
|
|
963
|
+
\`evaluations/\` are discovered automatically. Items with \`"enabled": false\`
|
|
964
|
+
remain inactive.
|
|
979
965
|
4. Local action source is bundled and SHA-256 hashed. Production bundles keep
|
|
980
966
|
only provider adapters required by agents and enabled evaluations.
|
|
981
967
|
5. Invalid paths, duplicate model-facing names, missing files, unsupported
|
|
@@ -1056,7 +1042,8 @@ needed for faithful continuation but are omitted from evaluation transcripts.
|
|
|
1056
1042
|
|
|
1057
1043
|
## Change rules
|
|
1058
1044
|
|
|
1059
|
-
- Register every new agent
|
|
1045
|
+
- Register every new agent and subagent explicitly. Agent-owned actions, skills,
|
|
1046
|
+
tests, and evaluations are auto-discovered; use \`"enabled": false\` for WIP.
|
|
1060
1047
|
- Keep runtime behavior provider-neutral.
|
|
1061
1048
|
- Mock actions when tests must avoid side effects; document intentional live
|
|
1062
1049
|
action coverage.
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"cli-agents-template.js","sourceRoot":"","sources":["../src/cli-agents-template.ts"],"names":[],"mappings":"AAAA,MAAM,CAAC,MAAM,SAAS,GAAG
|
|
1
|
+
{"version":3,"file":"cli-agents-template.js","sourceRoot":"","sources":["../src/cli-agents-template.ts"],"names":[],"mappings":"AAAA,MAAM,CAAC,MAAM,SAAS,GAAG;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA8hCxB,CAAC"}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"compile.d.ts","sourceRoot":"","sources":["../../src/compiler/compile.ts"],"names":[],"mappings":"
|
|
1
|
+
{"version":3,"file":"compile.d.ts","sourceRoot":"","sources":["../../src/compiler/compile.ts"],"names":[],"mappings":"AAKA,OAAO,KAAK,EAEV,iBAAiB,EAIjB,cAAc,EAKf,MAAM,aAAa,CAAC;AAErB,wBAAgB,eAAe,CAC7B,aAAa,EAAE,MAAM,CAAC,MAAM,EAAE,iBAAiB,CAAC,EAChD,YAAY,EAAE,MAAM,EACpB,aAAa,GAAE,QAAQ,GAAG,SAAoB,GAC7C,cAAc,CA4DhB;AA0mBD,wBAAgB,eAAe,CAAC,MAAM,EAAE,cAAc,GAAG,MAAM,CAO9D"}
|