orchajs 0.12.2 → 0.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -56,26 +56,16 @@ or hosted dependency.
56
56
  /evaluations → metrics each run is judged against
57
57
  ```
58
58
 
59
- Placement determines behavior. Agent skills and test cases use local index
60
- files to explicitly register which folders belong to the compiled agent.
59
+ Placement determines behavior. Direct child folders under `actions`, `skills`,
60
+ `tests`, and `evaluations` are discovered automatically. Set
61
+ `"enabled": false` in an item's `index.json` to keep WIP source inactive.
61
62
 
62
63
  ---
63
64
 
64
65
  ## Skills
65
66
 
66
67
  Skills keep specialized procedures out of the base prompt until the model
67
- needs them. Register the skill folders available to an agent:
68
-
69
- ```js
70
- // skills/index.js
71
- import { defineSkills } from "orchajs/skills";
72
-
73
- export default defineSkills({
74
- incompleteEvidenceReview: "./incompleteEvidenceReview",
75
- });
76
- ```
77
-
78
- Each registered folder contains compact discovery metadata and the full
68
+ needs them. Each skill folder contains compact discovery metadata and the full
79
69
  instructions:
80
70
 
81
71
  ```json
@@ -92,13 +82,13 @@ skills/incompleteEvidenceReview/
92
82
  instructions.md
93
83
  ```
94
84
 
95
- Orcha initially gives the model only each registered skill's name,
85
+ Orcha initially gives the model only each enabled skill's name,
96
86
  description, and trigger guidance. When the model calls the internal
97
87
  `load_skill` tool, Orcha activates the full instructions without running an
98
88
  application action or pausing for the client. Durable `skill.requested`,
99
89
  `skill.loaded`, and `skill.failed` events expose the loading lifecycle. Loaded
100
90
  skills remain active across `resume()` calls and provider changes.
101
- Unregistered skill folders are not compiled or exposed.
91
+ Skill folders with `"enabled": false` are not compiled or exposed.
102
92
 
103
93
  ---
104
94
 
@@ -114,8 +104,9 @@ actions/
114
104
  index.js // the code
115
105
  ```
116
106
 
117
- `index.json` tells the model what the action is, where it executes, and its
118
- input/output contracts:
107
+ `index.json` tells the model what the action is and its input/output contracts.
108
+ Actions are local by default; set `"execution": "client"` only when the
109
+ application must execute one:
119
110
 
120
111
  ```json
121
112
  {
@@ -181,12 +172,14 @@ Action entrypoints and their imported JavaScript or TypeScript modules are
181
172
  bundled together. Relative project imports, package imports, transitive
182
173
  dependencies, and standard exports work normally.
183
174
 
184
- Projects with local actions must explicitly choose their runtime:
175
+ Local actions use the native Node.js runtime by default, so ordinary projects
176
+ do not need global action configuration. Select the sandbox explicitly when
177
+ isolation is required:
185
178
 
186
179
  ```js
187
180
  orcha.init({
188
181
  actions: {
189
- runtime: "native",
182
+ runtime: "sandbox",
190
183
  },
191
184
  // providers and agents...
192
185
  });
@@ -589,16 +582,6 @@ Agent tests live beside the agent and use the same compiled instructions,
589
582
  provider, output contract, action schemas, `run()`, and `resume()` behavior as
590
583
  the application.
591
584
 
592
- Register test folders in `tests/index.js`:
593
-
594
- ```js
595
- import { defineTests } from "orchajs/testing";
596
-
597
- export default defineTests({
598
- delayedOrder: "./delayedOrder",
599
- });
600
- ```
601
-
602
585
  Each test folder contains an `index.json` with its input, one simulated
603
586
  response sequence for every action declared by the agent, and deterministic
604
587
  expectations:
@@ -659,7 +642,7 @@ terminal, report, session log, and Playground because they may cause side
659
642
  effects. Client actions must be mocked because the test runner has no attached
660
643
  application client. Unknown mock action names are rejected.
661
644
 
662
- Run every registered test programmatically:
645
+ Run every enabled test programmatically:
663
646
 
664
647
  ```js
665
648
  import { orcha } from "orchajs";
@@ -678,17 +661,8 @@ build` itself remains offline and does not require those credentials.
678
661
 
679
662
  ## Evaluations
680
663
 
681
- Evaluations are registered LLM judges that score the cumulative session after
682
- every completed run:
683
-
684
- ```js
685
- // evaluations/index.js
686
- import { defineEvaluations } from "orchajs/evaluations";
687
-
688
- export default defineEvaluations({
689
- responseQuality: "./responseQuality",
690
- });
691
- ```
664
+ Evaluations are automatically discovered LLM judges that score the cumulative
665
+ session after every completed run:
692
666
 
693
667
  ```json
694
668
  {
@@ -1,2 +1,2 @@
1
- export declare const AGENTS_MD = "# Orcha project guide\n\nThis repository uses OrchaJS, a filesystem-convention framework for durable\nAI agents. Treat the `orcha/` directory as source code. Do not edit generated\nfiles under `.orcha/`.\n\n## Commands\n\n- `orcha init` creates the initial Orcha files without overwriting files.\n- `orcha dev` validates the registry and watches `orcha/**` for changes.\n- `orcha playground` opens the built-in local UI and hot reloads agent source.\n- `orcha run <agent> --input \"\u2026\"` executes one registered agent.\n- `orcha run <agent> --input-file request.json` accepts structured input.\n- `orcha run <agent> --session <id> --input \"\u2026\"` continues a session.\n- `orcha run <agent> --session <id> --tool-results results.json` submits\n pending client-action results.\n- `orcha test` runs every registered agent test.\n- `orcha test <agent>` or `orcha test <agent>/<case>` narrows the run.\n- `orcha build` creates the production Orcha bundle without calling models.\n- Add `--json` to `run` and `test` for machine-readable output.\n\n`run`, `test`, and playground executions use real providers and require\ncredentials. `dev` and `build` are offline. Session logs are JSONL files under\n`.orcha/sessions/<agentName>/ses_<timestamp>_<uuid>.jsonl`.\n\n## Registry\n\n`orcha/index.ts` initializes providers and explicitly registers agents:\n\n```ts\nimport { orcha } from \"orchajs\";\n\norcha.init({\n providers: {\n anthropic: process.env.ANTHROPIC_API_KEY ?? \"\",\n openai: process.env.OPENAI_API_KEY ?? \"\",\n },\n actions: { runtime: \"native\" },\n agents: {\n supportBot: \"./supportBot\",\n },\n});\n```\n\nOnly registered folders are compiled. Agent keys become runtime properties\nsuch as `orcha.supportBot`. Use `actions.runtime: \"sandbox\"` for isolated\nlocal action execution or `\"native\"` when the application intentionally\nallows action modules to execute in its Node.js process.\n\n`orcha.init()` fields:\n\n- `providers` (required): provider configurations keyed by built-in provider\n name.\n- `agents` (required): runtime property names mapped to folders relative to\n `orcha/`. At least one agent is required.\n- `actions` (required only when a registered agent has local actions):\n selects the local execution runtime and its environment/sandbox settings.\n- `storage.strategy` (optional): currently only `\"node-jsonl\"`.\n- `storage.directory` (optional): session directory relative to project\n root; defaults to `.orcha/sessions`.\n- `root` (optional): absolute or working-directory-relative project root;\n defaults to `ORCHA_PROJECT_ROOT` and then `process.cwd()`.\n\nThe Orcha library reads the host application's existing `process.env`.\nStandalone Orcha CLI commands also load `.env` from the project root without\noverriding environment variables already in the process.\n\nProvider configuration shapes:\n\n```ts\nproviders: {\n anthropic: process.env.ANTHROPIC_API_KEY ?? \"\",\n deepseek: process.env.DEEPSEEK_API_KEY ?? \"\",\n googlegenai: process.env.GOOGLE_API_KEY ?? \"\",\n openai: {\n apiKey: process.env.OPENAI_API_KEY ?? \"\",\n baseUrl: \"https://api.openai.com/v1\", // optional override\n },\n vertexai: {\n project: process.env.GOOGLE_CLOUD_PROJECT ?? \"\",\n location: process.env.GOOGLE_CLOUD_LOCATION ?? \"us-central1\",\n // credentials is optional; omit it to use Google ADC.\n credentials: {\n clientEmail: process.env.GOOGLE_CLIENT_EMAIL ?? \"\",\n privateKey: process.env.GOOGLE_PRIVATE_KEY ?? \"\",\n },\n baseUrl: undefined, // optional override\n },\n bedrock: {\n region: process.env.AWS_REGION ?? \"us-east-1\",\n // credentials is optional; omit it to use the AWS credential chain.\n credentials: {\n accessKeyId: process.env.AWS_ACCESS_KEY_ID ?? \"\",\n secretAccessKey: process.env.AWS_SECRET_ACCESS_KEY ?? \"\",\n sessionToken: process.env.AWS_SESSION_TOKEN,\n },\n baseUrl: undefined, // optional override\n },\n}\n```\n\nAPI-key providers accept either a string shorthand or\n`{ apiKey, baseUrl? }`. Vertex AI requires `project` and `location`;\nexplicit service-account credentials are optional. Bedrock requires `region`;\nexplicit AWS credentials are optional. Never place credentials in\n`index.json`, instructions, tests, session metadata, or committed files.\n\n## Agent folders\n\n```text\norcha/\n index.ts\n supportBot/\n index.json\n instructions.md\n actions/\n skills/\n tests/\n evaluations/\n```\n\n`index.json` selects the model:\n\n```json\n{\n \"name\": \"Support Agent\",\n \"description\": \"Resolve customer support questions using confirmed account data.\",\n \"provider\": \"anthropic\",\n \"model\": \"claude-sonnet-4-6\",\n \"region\": \"provider_managed\",\n \"maxTokens\": 10240,\n \"outputType\": \"text\"\n}\n```\n\nOptional fields include `reasoningLevel`, `outputType: \"json\"`, and an\n`outputSchema` JSON Schema. Provider-specific reasoning values are forwarded\nwithout translation. Put the agent's stable role, boundaries, and operating\ninstructions in `instructions.md`.\n\nAgent `index.json` fields:\n\n- `name` (required): concise human-readable agent name.\n- `description` (optional): what the agent does and when it should be used.\n- `provider` (required): `\"anthropic\"`, `\"bedrock\"`, `\"deepseek\"`,\n `\"openai\"`, `\"googlegenai\"`, or `\"vertexai\"`.\n- `model` (required): exact provider model identifier.\n- `region` (optional): provider/model routing hint; defaults in durable\n metadata to `\"provider_managed\"`.\n- `maxTokens` (optional): positive integer. If omitted, the provider adapter\n chooses its default.\n- `reasoningLevel` (optional): non-empty provider-native string. Orcha does\n not translate values between providers.\n- `outputType` (optional): `\"text\"` (default) or `\"json\"`. Image and\n audio are reserved but not implemented.\n- `outputSchema` (required for JSON output): JSON Schema used for provider\n structured output and final validation.\n\n`instructions.md` is required and cannot be empty. At compile time it becomes\nthe base system prompt. Orcha appends the compact available-skill catalog and\nthe full instructions for skills already loaded in this durable session.\n\n## Subagents\n\nRegister private subagents alongside a parent in `orcha/index.ts`:\n\n```js\nagents: {\n coordinator: {\n path: \"./coordinator\",\n subagents: {\n researcher: \"./researcher\",\n },\n },\n researcher: \"./researcher\",\n}\n```\n\nOnly top-level keys become `orcha.<agentName>`. In this example the\nresearcher is both directly accessible and available to the coordinator.\nRemove its top-level entry to make it private.\n\nA parent may configure delegation limits in its `index.json`:\n\n```json\n{\n \"subagents\": {\n \"maxPerRun\": 3\n }\n}\n```\n\n`maxPerRun` limits newly created child sessions in one parent run and\ndefaults to 10.\n\nThe parent receives a compact catalog containing each subagent's registered\nname and optional description. Internal `run_agent`, `resume_agent`,\n`inspect_agent` tools let it start, continue, and inspect only child sessions\ninitiated by its current session. A delegated agent cannot delegate again.\n\nSubagent calls are synchronous. The parent becomes\n`waiting_for_subagent` while the child runs, then receives the child's text,\npause state, client-action request, failure, or completion as a normal tool\nresult. The parent and child keep separate linked JSONL sessions.\n\n## Core execution model\n\nAn **agent** is the compiled definition: model settings, instructions, actions,\nskills, and evaluations. An agent can create many independent sessions.\n\nA **session** is one durable conversation owned by one agent. It has one\n`sessionId`, optional name and metadata, fixed prompt variables, and one\nappend-only JSONL timeline. Completing one response does not close the\nsession\u2014the application can resume it later. A session cannot be transferred\nto another registered agent, but later runs may use a different provider or\nmodel if that same agent's configuration changes.\n\nA **run** is one attempt to advance a session. `run()` creates a session and\nits first run. A conversational `resume(sessionId, { content })` creates the\nnext numbered run in that session. Each run accumulates its own model usage and\nends in exactly one of these states:\n\n- `completed`: the model produced final output.\n- `waiting_for_subagent`: the parent is waiting for a synchronous child\n response and continues automatically when it arrives.\n- `waiting_for_client_action`: the model requested work that only the\n application can perform. The run is paused, not completed.\n- `paused`: active provider work was aborted, streamed output was preserved,\n and a later `resume()` starts a new run.\n- `failed`: validation, provider, storage, or execution failed. The durable\n events remain available for diagnosis.\n\nAn **execution** is the in-process handle returned by one call to `run()` or\n`resume()`. It exposes a cumulative output stream, latest snapshot, final\nresult promise, and evaluation promise. An execution ends when that invocation\ncompletes, pauses, or fails; the durable session may continue through another\nexecution.\n\nA **model round** is one provider request inside a run. One run may contain\nseveral rounds:\n\n```text\nuser input\n \u2192 model round\n \u2192 tool calls\n \u2192 tool results\n \u2192 another model round\n \u2192 final answer\n```\n\nLocal actions and skill loads are handled automatically inside the same\nexecution. Their results are sent back to the model and the model loop\ncontinues without application involvement.\n\nA **client action** deliberately crosses the application boundary. Orcha can\ndescribe the tool to the model but cannot execute it because the operation\nbelongs to a browser, mobile app, approval system, or other caller-owned\nenvironment. The complete pause/continue flow is:\n\n```text\n1. Application calls agent.run(...) or agent.resume(...content).\n2. Model requests one or more client actions.\n3. Orcha stores client_action.requested and run.paused.\n4. execution.result resolves with:\n {\n status: \"waiting_for_client_action\",\n sessionId,\n clientToolCalls: [{ callId, name, arguments }]\n }\n5. Application executes every requested action.\n6. Application calls agent.resume(sessionId, {\n toolResults: [{ callId, output, isError? }]\n }).\n7. Orcha validates every callId and output, stores the results, and continues\n the same paused run from its prior model context.\n8. The resumed execution either completes, requests more client actions, or\n fails.\n```\n\nEvery pending call must be resolved exactly once in one resume operation.\n`callId` links the submitted result to the model's request; the action name\nmust not be substituted for it. Re-submitting the identical resolved result is\nidempotent and returns the prior completed result. Submitting different data\nfor an already-resolved call fails with `action_result_conflict`.\n\n`clientCapabilities` is supplied per invocation because different callers\nmay support different client actions. Orcha exposes only declared client\nactions to that model round. Local actions are always available when compiled.\n\nOnly one execution may mutate a session at a time. Concurrent calls for the\nsame `sessionId` return `session_busy`; different sessions can run\nindependently.\n\n## Running and resuming\n\n```ts\nconst execution = orcha.supportBot.run({\n content: \"Check subscription sub_123.\",\n name: \"Subscription check\",\n metadata: { accountId: \"acct_123\" },\n clientCapabilities: [\"request_human_approval\"],\n});\n\nfor await (const snapshot of execution.stream) {\n console.log(snapshot);\n}\n\nconst result = await execution.result;\nconst evaluations = await execution.evaluations;\n```\n\n`run()` input fields:\n\n- `content` (required): a non-empty string, one content item, or an array.\n Use `{ filePath: \"./document.pdf\" }` for a local file or\n `{ url: \"https://example.com/document.pdf\" }` for a remote file.\n `mimeType` is optional when it can be inferred from the extension. Durable\n events store local paths and MIME types, never encoded file bytes.\n- `name` (optional): trimmed session label from 1 through 200 characters.\n- `metadata` (optional): at most 50 fields with non-empty keys and finite\n string, number, boolean, or null values. Metadata is durable and available\n to local action context; never place secrets in it.\n- `variables` (optional): at most 50 string values whose keys are JavaScript\n identifiers. They replace `{{ variableName }}` placeholders in\n `instructions.md`, are fixed when the session is created, and are reused\n by later resumes. A missing referenced variable fails the run.\n- `clientCapabilities` (optional): action names the current caller can\n execute. Client actions not declared here are withheld from the model.\n\n`run()` always creates a new durable session. Continue one with:\n\n```ts\nconst execution = orcha.supportBot.resume(sessionId, {\n content: \"Continue with the confirmed account.\",\n});\n```\n\nIf a result has `status: \"waiting_for_client_action\"`, execute the requested\nclient actions in the application and submit every result:\n\n```ts\norcha.supportBot.resume(sessionId, {\n toolResults: [\n { callId: \"call_123\", output: { approved: true } }\n ],\n});\n```\n\nNever invent call IDs. Use the IDs returned in `clientToolCalls`.\n\n## Actions\n\nEach action has metadata and, for local actions, executable code:\n\n```text\nactions/\n lookupAccount/\n index.json\n index.js\n```\n\n```json\n{\n \"name\": \"lookup_account\",\n \"description\": \"Look up one account.\",\n \"execution\": \"local\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"accountId\": { \"type\": \"string\" }\n },\n \"required\": [\"accountId\"],\n \"additionalProperties\": false\n },\n \"outputSchema\": {\n \"type\": \"object\",\n \"properties\": {\n \"status\": { \"type\": \"string\" }\n },\n \"required\": [\"status\"],\n \"additionalProperties\": false\n }\n}\n```\n\n```js\nexport default async function lookupAccount({ accountId }) {\n return { status: \"active\" };\n}\n```\n\nAction entrypoints can use ordinary relative project imports, TypeScript\nmodules, npm packages, transitive dependencies, and standard exports. Orcha\nbundles the complete import graph for development and production.\n\nClient actions use `\"execution\": \"client\"` and do not include executable\ncode. Orcha pauses until the caller submits their results. Keep action names,\ndescriptions, schemas, and implementations aligned.\n\nAction `index.json` fields:\n\n- `name` (required): model-facing tool name, 1\u201364 letters, numbers,\n underscores, or hyphens. `load_skill` is reserved.\n- `description` (required): tells the model when and why to call the action.\n- `execution` (required): `\"local\"` executes `index.js`; `\"client\"`\n pauses the run and delegates execution to the application.\n- `parameters` (required): JSON Schema for model-generated arguments.\n- `outputSchema` (optional): JSON Schema validated against local or submitted\n client output before the model receives it.\n- `timeoutMs` (optional): integer from 1 through 120000; defaults to 10000.\n- `permissions.env` (optional): names copied from `orcha.init().actions.env`\n into the action context.\n- `permissions.network` (optional): exact hosts or wildcard subdomains such\n as `\"api.example.com\"` or `\"*.example.com\"` allowed through\n `context.fetch`. Redirects are rejected.\n- `sideEffect` (optional): descriptive metadata for whether the operation\n mutates external state. It does not currently change execution behavior.\n\n`orcha.init().actions.runtime` and an action's `execution` solve different\nproblems:\n\n- `execution: \"client\"`: Orcha never executes code for this action.\n- `execution: \"local\"` + `runtime: \"sandbox\"`: compiled code runs in a\n QuickJS isolate with JSON-only inputs/outputs, default 32 MB memory, default\n 512 KB stack, interruptible timeout, declared environment values, and\n allowlisted network access through the provided context.\n- `execution: \"local\"` + `runtime: \"native\"`: code runs in the host Node.js\n process. It can use host privileges directly. The timeout rejects slow\n asynchronous work but cannot interrupt synchronous blocking code.\n\nGlobal local-action configuration:\n\n```ts\nactions: {\n runtime: \"sandbox\", // required when any registered action is local\n env: {\n BILLING_API_TOKEN: process.env.BILLING_API_TOKEN,\n },\n sandbox: {\n memoryLimitMb: 32,\n stackLimitKb: 512,\n },\n}\n```\n\nThe local action signature is\n`(parameters, context) => output | Promise<output>`. Context contains\n`sessionId`, immutable session `metadata`, a stable `idempotencyKey`,\nallowlisted `env`, guarded `fetch`, and prefixed `log`.\n\n## Skills\n\nSkills are lazy-loaded procedural instructions. Register only intended skills:\n\n```js\n// skills/index.js\nimport { defineSkills } from \"orchajs/skills\";\n\nexport default defineSkills({\n incidentTriage: \"./incidentTriage\",\n});\n```\n\nEach skill folder contains `index.json` metadata and `instructions.md`.\nThe model receives a compact catalog and can call the internal `load_skill`\ntool. Loaded instructions remain active for the durable session. Lifecycle\nevents are `skill.requested`, `skill.loaded`, and `skill.failed`.\n\nSkill `index.json` fields:\n\n- `name` (required): model-facing name, 1\u201364 letters, numbers, underscores,\n or hyphens; unique within the agent.\n- `description` (required): compact catalog description shown before loading.\n- `triggers` (optional): non-empty array of non-empty situations describing\n when the model should load the skill.\n\n`instructions.md` is required and cannot be empty. The key in\n`skills/index.js` is only a registration label; `index.json.name` is the\nname used by the model and durable events. Unregistered folders are ignored.\n\n## Tests\n\nRegister tests in `tests/index.js`:\n\n```js\nimport { defineTests } from \"orchajs/testing\";\n\nexport default defineTests({\n activeAccount: \"./activeAccount\",\n});\n```\n\nEach case's `index.json` defines `input`, optional mocked action responses,\nand `expect`. Tests run the real compiled agent and provider. Actions listed\nunder `actions` are mocked; unlisted local actions execute live. Orcha reports\nlive action use in the terminal, test report, and session trace.\nClient actions must always be mocked because no application client is attached\nto the test runner.\n\n```json\n{\n \"input\": { \"content\": \"Check account acct_123.\" },\n \"actions\": {\n \"lookup_account\": {\n \"responses\": [\n { \"output\": { \"status\": \"active\" } }\n ]\n }\n },\n \"expect\": {\n \"status\": \"completed\",\n \"text\": { \"contains\": [\"active\"] },\n \"actions\": [\n {\n \"name\": \"lookup_account\",\n \"arguments\": { \"equals\": { \"accountId\": \"acct_123\" } }\n }\n ]\n }\n}\n```\n\nTest sessions use the `ses_test_<timestamp>_<uuid>` format and end with a `test.completed`\nevent. Prefer semantic output assertions; verify exact identifiers and values\nthrough action-argument assertions.\n\nTest `index.json` fields:\n\n- `description` (optional): human-readable purpose.\n- `input` (required): an inline agent input object or a path to a JSON file\n containing that object. Relative paths resolve from the test case directory.\n- `input.content` (required for inline input): string or multimodal content\n array.\n- `input.variables` (optional): string map available to the session.\n- `input.metadata` (optional): string, number, boolean, or null values.\n- `actions` (optional): mocked actions keyed by compiled action name. Unlisted\n local actions execute live and may cause side effects. Unknown action names\n are rejected, and unlisted client actions are rejected. Every `responses`\n array is consumed in call order.\n- `responses[].output` (required): mocked action result.\n- `responses[].isError` (optional): marks the mocked result as an error.\n- `expect.status` (optional): `\"completed\"` or `\"failed\"`; defaults to\n `\"completed\"`.\n- `expect.output.equals` / `partial` (optional): exact or recursive partial\n comparison against structured output.\n- `expect.text.contains` / `excludes` (optional): case-sensitive semantic\n text checks.\n- `expect.actions` (optional): ordered expected calls. Each may assert\n `arguments.equals` or `arguments.partial`.\n\nThe registration key in `tests/index.js` is the test selector used by\n`orcha test agent/testName`; its value resolves to the case folder.\n\n## Evaluations\n\nEvaluations are asynchronous LLM judges registered in\n`evaluations/index.js` with `defineEvaluations` from\n`orchajs/evaluations`. Each folder's `index.json` defines its provider,\nmodel, metrics, and thresholds:\n\n```json\n{\n \"name\": \"response_quality\",\n \"description\": \"Grounding of agent responses.\",\n \"instructions\": \"Judge the complete response using only confirmed evidence recorded in the session.\",\n \"enabled\": true,\n \"provider\": \"openai\",\n \"model\": \"gpt-5-mini\",\n \"metrics\": [\n {\n \"name\": \"groundedness\",\n \"description\": \"The answer relies on confirmed session evidence.\",\n \"threshold\": 0.8\n }\n ]\n}\n```\n\n`execution.result` does not wait for judges. Await\n`execution.evaluations` when results must finish before process exit.\nEvaluations always finish during `orcha test`; judge errors and missed\nthresholds fail the test. Lifecycle events are `evaluation.requested`,\n`evaluation.completed`, and `evaluation.failed`.\n\nEvaluation `index.json` fields:\n\n- `name` (required): durable model-facing identifier, 1\u201364 letters, numbers,\n underscores, or hyphens; unique within the agent.\n- `description` (optional): short human-facing summary of the evaluator.\n- `instructions` (optional): detailed prompt supplied to the judge. When\n omitted, `description` is used for backward compatibility.\n- `enabled` (optional): defaults to `true`. Disabled evaluations are\n compiled but do not run.\n- `provider` and `model` (required): independently select the judge. The\n provider must also exist in `orcha.init().providers`.\n- `maxTokens` (optional): positive integer; defaults to 2000 for judges.\n- `reasoningLevel` (optional): non-empty provider-native string forwarded\n without translation.\n- `metrics` (required): non-empty array with unique metric names.\n- `metrics[].name`: 1\u201364 letters, numbers, underscores, or hyphens.\n- `metrics[].description`: exact criterion supplied to the judge.\n- `metrics[].threshold`: inclusive number from 0 to 1. A metric passes when\n the returned score is greater than or equal to this threshold.\n\nThe judge sees a sanitized transcript of user/assistant messages, action\nrequests and outcomes, client-action activity, and loaded skill names. It does\nnot receive internal reasoning blocks, replay metadata, previous evaluation\nresults, or test assertions. It must return exactly one score, reasoning\nstring, and non-empty evidence array for every configured metric.\n\n## Sessions and logs\n\nJSONL is the durable source of truth. Each line is one complete JSON object;\nnever treat the file as one JSON array. Events are append-only and ordered by\n`sequence`.\n\nAll events use this envelope:\n\n```ts\ntype SessionEvent<T> = {\n sequence: number; // starts at 1 and increases across the whole session\n type: SessionEventType;\n timestamp: string; // ISO-8601 UTC timestamp\n run?: number; // present for run-scoped events\n data: T;\n};\n```\n\nSession-scoped events omit `run`. Optional properties whose values are\n`undefined` are omitted from serialized JSON.\n\nShared stored structures:\n\n```ts\ntype Usage = {\n inputTokens: number;\n outputTokens: number;\n reasoningTokens: number | null;\n cacheReadTokens: number;\n cacheWriteTokens: number;\n};\n\ntype ErrorData = {\n code: string;\n message: string;\n retryable?: boolean;\n};\n\ntype UserContent =\n | { type: \"text\"; text: string }\n | { type: \"file\"; filePath: string; mimeType: string }\n | {\n type: \"image\" | \"video\" | \"audio\" | \"url\";\n mimeType: string;\n fileUri: string;\n };\n\ntype AssistantContent =\n | { type: \"text\"; text: string }\n | {\n type: \"reasoning\";\n text: string;\n replay?: { providerId?: string; opaqueData?: string };\n }\n | {\n type: \"tool_call\";\n callId: string;\n name: string;\n arguments: Record<string, unknown>;\n replay?: { providerId?: string; opaqueData?: string };\n };\n\ntype ToolResult = {\n callId: string;\n output: unknown;\n isError?: boolean;\n};\n```\n\nExact event payloads:\n\n```ts\ntype SessionCreated = SessionEvent<{\n schemaVersion: 1;\n sessionId: string;\n agent: string;\n status: \"active\";\n name?: string;\n metadata: Record<string, string | number | boolean | null>;\n variables: Record<string, string>;\n lineage?: {\n origin: \"delegated\";\n parentAgent: string;\n parentSessionId: string;\n parentCallId: string;\n };\n}>; // type \"session.created\", no run\n\ntype SessionUpdated = SessionEvent<{\n name?: string;\n metadata?: Record<string, string | number | boolean | null>;\n}>; // type \"session.updated\", no run\n\ntype RunStarted = SessionEvent<{\n status: \"running\";\n agent: string;\n provider: string;\n model: string;\n region: string;\n reasoningLevel?: string;\n outputType: \"text\" | \"json\";\n clientCapabilities: string[];\n}>; // type \"run.started\"\n\ntype UserMessageCreated = SessionEvent<{\n role: \"user\";\n content: UserContent[];\n}>; // type \"message.created\"\n\ntype AssistantMessageCreated = SessionEvent<{\n status: \"completed\" | \"incomplete\";\n provider: string;\n model: string;\n responseId?: string;\n stopReason?: \"end_turn\" | \"tool_call\" | \"max_tokens\" |\n \"content_filter\" | \"unknown\";\n role: \"assistant\";\n content: AssistantContent[];\n parsedOutput?: unknown; // final JSON output only\n usage: Usage;\n durationMs: number;\n}>; // type \"message.created\"\n\ntype ToolMessageCreated = SessionEvent<{\n role: \"tool\";\n content: ToolResult[];\n}>; // type \"message.created\"\n\ntype ActionRequested = SessionEvent<{\n callId: string;\n name: string;\n arguments: Record<string, unknown>;\n sourceHash?: string;\n idempotencyKey: string; // sessionId:callId\n}>; // type \"action.requested\"\n\ntype ActionCompleted = SessionEvent<{\n callId: string;\n name: string;\n output: unknown;\n sourceHash?: string;\n durationMs: number;\n}>; // type \"action.completed\"\n\ntype ActionFailed = SessionEvent<{\n callId: string;\n name: string;\n sourceHash?: string;\n durationMs: number;\n error: {\n code: \"action_execution_failed\";\n message: string;\n };\n}>; // type \"action.failed\"\n\ntype ClientActionRequested = SessionEvent<{\n status: \"waiting\";\n calls: Array<{\n callId: string;\n name: string;\n arguments: Record<string, unknown>;\n }>;\n localResults: ToolResult[];\n toolCallOrder: string[];\n}>; // type \"client_action.requested\"\n\ntype ClientActionResolved = SessionEvent<{\n status: \"completed\";\n results: ToolResult[];\n}>; // type \"client_action.resolved\"\n\ntype SkillRequested = SessionEvent<{\n callId: string;\n name: unknown;\n}>; // type \"skill.requested\"\n\ntype SkillLoaded = SessionEvent<{\n callId: string;\n name: string;\n alreadyLoaded: boolean;\n}>; // type \"skill.loaded\", no run\n\ntype SkillFailed = SessionEvent<{\n callId: string;\n name: unknown;\n error: {\n code: \"skill_not_found\";\n message: string;\n };\n}>; // type \"skill.failed\"\n\ntype EvaluationRequested = SessionEvent<{\n name: string;\n provider: string;\n model: string;\n evaluatedThroughSequence: number;\n}>; // type \"evaluation.requested\"\n\ntype EvaluationMetric = {\n name: string;\n score: number;\n threshold: number;\n passed: boolean;\n reasoning: string;\n evidence: string[];\n};\n\ntype EvaluationCompleted = SessionEvent<{\n name: string;\n status: \"passed\" | \"failed\";\n metrics: EvaluationMetric[];\n usage?: Usage;\n durationMs: number;\n provider: string;\n model: string;\n evaluatedThroughSequence: number;\n}>; // type \"evaluation.completed\"\n\ntype EvaluationFailed = SessionEvent<{\n name: string;\n status: \"error\";\n metrics: [];\n durationMs: number;\n error: { message: string };\n provider: string;\n model: string;\n evaluatedThroughSequence: number;\n}>; // type \"evaluation.failed\"\n\ntype SubagentInitiated = SessionEvent<{\n callId: string;\n agent: string;\n sessionId: string;\n status: \"running\";\n}>; // type \"subagent.initiated\"\n\ntype SubagentResumed = SessionEvent<{\n callId: string;\n agent: string;\n sessionId: string;\n status: \"running\";\n}>; // type \"subagent.resumed\"\n\ntype SubagentCompletedOrPausedOrFailed = SessionEvent<{\n callId: string;\n agent: string;\n sessionId: string;\n status:\n | \"completed\"\n | \"waiting_for_client_action\"\n | \"paused\"\n | \"failed\";\n output?: unknown;\n usage?: Usage;\n clientToolCalls?: Array<{\n callId: string;\n name: string;\n arguments: Record<string, unknown>;\n }>;\n error?: ErrorData;\n}>; // type \"subagent.completed\" | \"subagent.paused\" | \"subagent.failed\"\n\ntype RunPaused = SessionEvent<{\n status: \"waiting_for_client_action\" | \"paused\";\n clientToolCalls?: Array<{\n callId: string;\n name: string;\n arguments: Record<string, unknown>;\n }>;\n usage?: Usage;\n output?: unknown;\n durationMs?: number;\n}>; // type \"run.paused\"\n\ntype RunCompleted = SessionEvent<{\n status: \"completed\";\n durationMs: number;\n usage: Usage;\n}>; // type \"run.completed\"\n\ntype RunFailed = SessionEvent<{\n status: \"failed\";\n durationMs: number;\n usage?: Usage;\n error: ErrorData;\n}>; // type \"run.failed\"\n\ntype SessionPaused = SessionEvent<{\n status: \"paused\";\n}>; // type \"session.paused\", no run\n\ntype SessionResumed = SessionEvent<{\n status: \"active\";\n}>; // type \"session.resumed\", no run\n\ntype TestCompleted = SessionEvent<{\n suiteId: string;\n agent: string;\n test: string;\n status: \"passed\" | \"failed\";\n durationMs: number;\n assertions: Array<{\n path: string;\n passed: boolean;\n message: string;\n expected?: unknown;\n actual?: unknown;\n }>;\n usage?: Usage;\n evaluations?: Array<{\n name: string;\n status: \"passed\" | \"failed\" | \"error\";\n metrics: EvaluationMetric[];\n usage?: Usage;\n durationMs: number;\n error?: { message: string };\n }>;\n warnings?: Array<{\n code: \"live_actions\";\n message: string;\n actions: string[];\n }>;\n error?: ErrorData;\n}>; // type \"test.completed\", no run\n\ntype TestWarning = SessionEvent<{\n code: \"live_actions\";\n message: string;\n actions: string[];\n}>; // type \"test.warning\", no run\n```\n\nTypical event order:\n\n```text\nsession.created\nrun.started\nmessage.created (user)\nmessage.created (assistant, possibly with tool_call)\naction.requested \u2192 action.completed|action.failed # local action\nsubagent.initiated\n...child session advances independently...\nsubagent.paused\nsubagent.resumed\n...child session advances independently...\nsubagent.completed|subagent.paused|subagent.failed\nmessage.created (tool)\n...additional model/action rounds...\nmessage.created (assistant final)\nrun.completed\nevaluation.requested\nevaluation.completed|evaluation.failed\n```\n\nFor client actions, `client_action.requested` and `run.paused` replace the\nimmediate tool message. A later `resume(...toolResults)` appends\n`client_action.resolved`, the tool message, and continues the same run\nnumber. A conversational `resume(...content)` starts a new run number.\n\n## How Orcha works behind the scenes\n\n### Compilation\n\n1. `orcha/index.ts` calls `orcha.init()` with explicit agent paths.\n2. The compiler reads each registered agent's `index.json` and\n `instructions.md`.\n3. Every directory under `actions/` is compiled. Skills, tests, and\n evaluations are included only through their local `index.js` registry.\n4. Local action source is bundled and SHA-256 hashed. Production bundles keep\n only provider adapters required by agents and enabled evaluations.\n5. Invalid paths, duplicate model-facing names, missing files, unsupported\n configuration values, and missing schema objects fail before execution.\n Concrete action arguments and outputs are validated against their schemas\n when the action is used.\n\n`orcha dev` repeats validation when files change. `orcha build` performs\noffline production compilation. Neither command invokes a provider.\n\n### Run lifecycle\n\n1. `run()` creates a `ses_<timestamp>_<uuid>`, acquires the per-session execution lock,\n appends `session.created`, then starts run 1.\n2. `resume()` reads and validates the existing session. Message continuation\n starts a new run; submitted client results continue the paused run.\n3. The provider receives the base instructions, available-skill catalog,\n loaded skill instructions, normalized conversation messages, action\n schemas, model settings, and current client capabilities.\n4. Provider-specific responses are normalized into text, reasoning, and tool\n call blocks. Opaque replay metadata is stored only when a provider needs it\n to replay its own prior block correctly.\n5. Tool calls are checked against compiled actions and declared client\n capabilities. Arguments and outputs are validated against JSON Schema.\n6. `load_skill` updates durable session instructions. Local actions execute\n through the configured runtime. Client actions pause safely. Tool results\n are reordered to match the model's original call order.\n7. Internal agent tools start or continue linked child sessions. A\n `run_agent` input accepts the same text or multimodal content shape as a\n normal run. Local file paths resolve from the configured Orcha project root.\n The parent reports `waiting_for_subagent` until each synchronous child\n call returns.\n8. The model loop continues until final output, failure, a pause, or the\n maximum of 10 action rounds.\n9. Usage is normalized and aggregated across every model call in the run.\n10. After `run.completed`, enabled evaluations start in the background.\n `execution.result` is already available; `execution.evaluations` waits\n for judge completion and durable persistence.\n\nThe per-session lock prevents two model executions from mutating one session\nat once. Background evaluation writes queue behind active runs so they cannot\ncause `resume()` to fail spuriously or reuse sequence numbers.\n\n### Replay and context\n\nOrcha does not send raw JSONL back to the model. It projects durable events\ninto provider-neutral conversation messages. Completed and explicitly paused\nconversational runs become user, assistant, and tool messages, so a new run\nsees incomplete assistant text preserved by `pause()`. Lifecycle bookkeeping\nsuch as durations, test assertions, and evaluation events is excluded from\nmodel context. Reasoning text and provider replay metadata are retained where\nneeded for faithful continuation but are omitted from evaluation transcripts.\n\n### Reading sessions\n\n- `agent.get(sessionId)` projects the latest status, pending client actions,\n last output, metadata, and aggregate usage.\n- `agent.history(sessionId, { page, pageSize })` returns a safe user-facing\n timeline rather than raw provider bookkeeping.\n- `agent.events(sessionId, { page, pageSize })` returns the canonical durable\n event records for observability, audit, and debugging tools. For subsequent\n pages of a changing session, pass the first response's `throughSequence`\n back in the options to keep pagination on a stable event boundary.\n- `agent.list({ page, pageSize, status, metadata })` lists projected session\n snapshots.\n- `agent.subagentHistory(childSessionId, { page, pageSize })` returns the\n projected history of a child owned by this parent agent. Lineage checks\n prevent access through unrelated agents.\n- `agent.subagentEvents(childSessionId, { page, pageSize })` returns that\n owned child's canonical events for full transcript and debugging interfaces.\n- `agent.pause(sessionId)` aborts active provider work for the session and\n active children, preserving streamed text as an incomplete assistant\n message. `agent.resume(sessionId)` starts a new run with `\"Continue.\"`;\n pass content to give different instructions. Pending client actions still\n require their exact tool results.\n- Never read storage files directly. Use these methods so applications remain\n compatible with JSONL, SQLite, IndexedDB, remote, and future storage adapters.\n\n## Change rules\n\n- Register every new agent, skill, test, and evaluation explicitly.\n- Keep runtime behavior provider-neutral.\n- Mock actions when tests must avoid side effects; document intentional live\n action coverage.\n- Do not commit `.orcha/`; it contains generated output and session data.\n- Run `orcha dev` after filesystem changes and `orcha test` when behavior\n changes.\n- Do not weaken assertions to match incorrect behavior. Remove an assertion\n only when it is stricter than the documented agent contract.\n";
1
+ export declare const AGENTS_MD = "# Orcha project guide\n\nThis repository uses OrchaJS, a filesystem-convention framework for durable\nAI agents. Treat the `orcha/` directory as source code. Do not edit generated\nfiles under `.orcha/`.\n\n## Commands\n\n- `orcha init` creates the initial Orcha files without overwriting files.\n- `orcha dev` validates agent source and watches `orcha/**` for changes.\n- `orcha playground` opens the built-in local UI and hot reloads agent source.\n- `orcha run <agent> --input \"\u2026\"` executes one registered agent.\n- `orcha run <agent> --input-file request.json` accepts structured input.\n- `orcha run <agent> --session <id> --input \"\u2026\"` continues a session.\n- `orcha run <agent> --session <id> --tool-results results.json` submits\n pending client-action results.\n- `orcha test` runs every enabled discovered agent test.\n- `orcha test <agent>` or `orcha test <agent>/<case>` narrows the run.\n- `orcha build` creates the production Orcha bundle without calling models.\n- Add `--json` to `run` and `test` for machine-readable output.\n\n`run`, `test`, and playground executions use real providers and require\ncredentials. `dev` and `build` are offline. Session logs are JSONL files under\n`.orcha/sessions/<agentName>/ses_<timestamp>_<uuid>.jsonl`.\n\n## Registry\n\n`orcha/index.ts` initializes providers and explicitly registers agents:\n\n```ts\nimport { orcha } from \"orchajs\";\n\norcha.init({\n providers: {\n anthropic: process.env.ANTHROPIC_API_KEY ?? \"\",\n openai: process.env.OPENAI_API_KEY ?? \"\",\n },\n agents: {\n supportBot: \"./supportBot\",\n },\n});\n```\n\nOnly registered agents are compiled. Agent keys become runtime properties\nsuch as `orcha.supportBot`. Local actions use the native Node.js runtime by\ndefault. Set `actions.runtime: \"sandbox\"` only when isolated execution is\nrequired.\n\n`orcha.init()` fields:\n\n- `providers` (required): provider configurations keyed by built-in provider\n name.\n- `agents` (required): runtime property names mapped to folders relative to\n `orcha/`. At least one agent is required.\n- `actions` (optional): overrides the native local-action runtime or configures\n its environment and sandbox settings.\n- `storage.strategy` (optional): currently only `\"node-jsonl\"`.\n- `storage.directory` (optional): session directory relative to project\n root; defaults to `.orcha/sessions`.\n- `root` (optional): absolute or working-directory-relative project root;\n defaults to `ORCHA_PROJECT_ROOT` and then `process.cwd()`.\n\nThe Orcha library reads the host application's existing `process.env`.\nStandalone Orcha CLI commands also load `.env` from the project root without\noverriding environment variables already in the process.\n\nProvider configuration shapes:\n\n```ts\nproviders: {\n anthropic: process.env.ANTHROPIC_API_KEY ?? \"\",\n deepseek: process.env.DEEPSEEK_API_KEY ?? \"\",\n googlegenai: process.env.GOOGLE_API_KEY ?? \"\",\n openai: {\n apiKey: process.env.OPENAI_API_KEY ?? \"\",\n baseUrl: \"https://api.openai.com/v1\", // optional override\n },\n vertexai: {\n project: process.env.GOOGLE_CLOUD_PROJECT ?? \"\",\n location: process.env.GOOGLE_CLOUD_LOCATION ?? \"us-central1\",\n // credentials is optional; omit it to use Google ADC.\n credentials: {\n clientEmail: process.env.GOOGLE_CLIENT_EMAIL ?? \"\",\n privateKey: process.env.GOOGLE_PRIVATE_KEY ?? \"\",\n },\n baseUrl: undefined, // optional override\n },\n bedrock: {\n region: process.env.AWS_REGION ?? \"us-east-1\",\n // credentials is optional; omit it to use the AWS credential chain.\n credentials: {\n accessKeyId: process.env.AWS_ACCESS_KEY_ID ?? \"\",\n secretAccessKey: process.env.AWS_SECRET_ACCESS_KEY ?? \"\",\n sessionToken: process.env.AWS_SESSION_TOKEN,\n },\n baseUrl: undefined, // optional override\n },\n}\n```\n\nAPI-key providers accept either a string shorthand or\n`{ apiKey, baseUrl? }`. Vertex AI requires `project` and `location`;\nexplicit service-account credentials are optional. Bedrock requires `region`;\nexplicit AWS credentials are optional. Never place credentials in\n`index.json`, instructions, tests, session metadata, or committed files.\n\n## Agent folders\n\n```text\norcha/\n index.ts\n supportBot/\n index.json\n instructions.md\n actions/\n skills/\n tests/\n evaluations/\n```\n\n`index.json` selects the model:\n\n```json\n{\n \"name\": \"Support Agent\",\n \"description\": \"Resolve customer support questions using confirmed account data.\",\n \"provider\": \"anthropic\",\n \"model\": \"claude-sonnet-4-6\",\n \"region\": \"provider_managed\",\n \"maxTokens\": 10240,\n \"outputType\": \"text\"\n}\n```\n\nOptional fields include `reasoningLevel`, `outputType: \"json\"`, and an\n`outputSchema` JSON Schema. Provider-specific reasoning values are forwarded\nwithout translation. Put the agent's stable role, boundaries, and operating\ninstructions in `instructions.md`.\n\nAgent `index.json` fields:\n\n- `name` (required): concise human-readable agent name.\n- `description` (optional): what the agent does and when it should be used.\n- `provider` (required): `\"anthropic\"`, `\"bedrock\"`, `\"deepseek\"`,\n `\"openai\"`, `\"googlegenai\"`, or `\"vertexai\"`.\n- `model` (required): exact provider model identifier.\n- `region` (optional): provider/model routing hint; defaults in durable\n metadata to `\"provider_managed\"`.\n- `maxTokens` (optional): positive integer. If omitted, the provider adapter\n chooses its default.\n- `reasoningLevel` (optional): non-empty provider-native string. Orcha does\n not translate values between providers.\n- `outputType` (optional): `\"text\"` (default) or `\"json\"`. Image and\n audio are reserved but not implemented.\n- `outputSchema` (required for JSON output): JSON Schema used for provider\n structured output and final validation.\n\n`instructions.md` is required and cannot be empty. At compile time it becomes\nthe base system prompt. Orcha appends the compact available-skill catalog and\nthe full instructions for skills already loaded in this durable session.\n\n## Subagents\n\nRegister private subagents alongside a parent in `orcha/index.ts`:\n\n```js\nagents: {\n coordinator: {\n path: \"./coordinator\",\n subagents: {\n researcher: \"./researcher\",\n },\n },\n researcher: \"./researcher\",\n}\n```\n\nOnly top-level keys become `orcha.<agentName>`. In this example the\nresearcher is both directly accessible and available to the coordinator.\nRemove its top-level entry to make it private.\n\nA parent may configure delegation limits in its `index.json`:\n\n```json\n{\n \"subagents\": {\n \"maxPerRun\": 3\n }\n}\n```\n\n`maxPerRun` limits newly created child sessions in one parent run and\ndefaults to 10.\n\nThe parent receives a compact catalog containing each subagent's registered\nname and optional description. Internal `run_agent`, `resume_agent`,\n`inspect_agent` tools let it start, continue, and inspect only child sessions\ninitiated by its current session. A delegated agent cannot delegate again.\n\nSubagent calls are synchronous. The parent becomes\n`waiting_for_subagent` while the child runs, then receives the child's text,\npause state, client-action request, failure, or completion as a normal tool\nresult. The parent and child keep separate linked JSONL sessions.\n\n## Core execution model\n\nAn **agent** is the compiled definition: model settings, instructions, actions,\nskills, and evaluations. An agent can create many independent sessions.\n\nA **session** is one durable conversation owned by one agent. It has one\n`sessionId`, optional name and metadata, fixed prompt variables, and one\nappend-only JSONL timeline. Completing one response does not close the\nsession\u2014the application can resume it later. A session cannot be transferred\nto another registered agent, but later runs may use a different provider or\nmodel if that same agent's configuration changes.\n\nA **run** is one attempt to advance a session. `run()` creates a session and\nits first run. A conversational `resume(sessionId, { content })` creates the\nnext numbered run in that session. Each run accumulates its own model usage and\nends in exactly one of these states:\n\n- `completed`: the model produced final output.\n- `waiting_for_subagent`: the parent is waiting for a synchronous child\n response and continues automatically when it arrives.\n- `waiting_for_client_action`: the model requested work that only the\n application can perform. The run is paused, not completed.\n- `paused`: active provider work was aborted, streamed output was preserved,\n and a later `resume()` starts a new run.\n- `failed`: validation, provider, storage, or execution failed. The durable\n events remain available for diagnosis.\n\nAn **execution** is the in-process handle returned by one call to `run()` or\n`resume()`. It exposes a cumulative output stream, latest snapshot, final\nresult promise, and evaluation promise. An execution ends when that invocation\ncompletes, pauses, or fails; the durable session may continue through another\nexecution.\n\nA **model round** is one provider request inside a run. One run may contain\nseveral rounds:\n\n```text\nuser input\n \u2192 model round\n \u2192 tool calls\n \u2192 tool results\n \u2192 another model round\n \u2192 final answer\n```\n\nLocal actions and skill loads are handled automatically inside the same\nexecution. Their results are sent back to the model and the model loop\ncontinues without application involvement.\n\nA **client action** deliberately crosses the application boundary. Orcha can\ndescribe the tool to the model but cannot execute it because the operation\nbelongs to a browser, mobile app, approval system, or other caller-owned\nenvironment. The complete pause/continue flow is:\n\n```text\n1. Application calls agent.run(...) or agent.resume(...content).\n2. Model requests one or more client actions.\n3. Orcha stores client_action.requested and run.paused.\n4. execution.result resolves with:\n {\n status: \"waiting_for_client_action\",\n sessionId,\n clientToolCalls: [{ callId, name, arguments }]\n }\n5. Application executes every requested action.\n6. Application calls agent.resume(sessionId, {\n toolResults: [{ callId, output, isError? }]\n }).\n7. Orcha validates every callId and output, stores the results, and continues\n the same paused run from its prior model context.\n8. The resumed execution either completes, requests more client actions, or\n fails.\n```\n\nEvery pending call must be resolved exactly once in one resume operation.\n`callId` links the submitted result to the model's request; the action name\nmust not be substituted for it. Re-submitting the identical resolved result is\nidempotent and returns the prior completed result. Submitting different data\nfor an already-resolved call fails with `action_result_conflict`.\n\n`clientCapabilities` is supplied per invocation because different callers\nmay support different client actions. Orcha exposes only declared client\nactions to that model round. Local actions are always available when compiled.\n\nOnly one execution may mutate a session at a time. Concurrent calls for the\nsame `sessionId` return `session_busy`; different sessions can run\nindependently.\n\n## Running and resuming\n\n```ts\nconst execution = orcha.supportBot.run({\n content: \"Check subscription sub_123.\",\n name: \"Subscription check\",\n metadata: { accountId: \"acct_123\" },\n clientCapabilities: [\"request_human_approval\"],\n});\n\nfor await (const snapshot of execution.stream) {\n console.log(snapshot);\n}\n\nconst result = await execution.result;\nconst evaluations = await execution.evaluations;\n```\n\n`run()` input fields:\n\n- `content` (required): a non-empty string, one content item, or an array.\n Use `{ filePath: \"./document.pdf\" }` for a local file or\n `{ url: \"https://example.com/document.pdf\" }` for a remote file.\n `mimeType` is optional when it can be inferred from the extension. Durable\n events store local paths and MIME types, never encoded file bytes.\n- `name` (optional): trimmed session label from 1 through 200 characters.\n- `metadata` (optional): at most 50 fields with non-empty keys and finite\n string, number, boolean, or null values. Metadata is durable and available\n to local action context; never place secrets in it.\n- `variables` (optional): at most 50 string values whose keys are JavaScript\n identifiers. They replace `{{ variableName }}` placeholders in\n `instructions.md`, are fixed when the session is created, and are reused\n by later resumes. A missing referenced variable fails the run.\n- `clientCapabilities` (optional): action names the current caller can\n execute. Client actions not declared here are withheld from the model.\n\n`run()` always creates a new durable session. Continue one with:\n\n```ts\nconst execution = orcha.supportBot.resume(sessionId, {\n content: \"Continue with the confirmed account.\",\n});\n```\n\nIf a result has `status: \"waiting_for_client_action\"`, execute the requested\nclient actions in the application and submit every result:\n\n```ts\norcha.supportBot.resume(sessionId, {\n toolResults: [\n { callId: \"call_123\", output: { approved: true } }\n ],\n});\n```\n\nNever invent call IDs. Use the IDs returned in `clientToolCalls`.\n\n## Actions\n\nEach action has metadata and, for local actions, executable code:\n\n```text\nactions/\n lookupAccount/\n index.json\n index.js\n```\n\n```json\n{\n \"name\": \"lookup_account\",\n \"description\": \"Look up one account.\",\n \"parameters\": {\n \"type\": \"object\",\n \"properties\": {\n \"accountId\": { \"type\": \"string\" }\n },\n \"required\": [\"accountId\"],\n \"additionalProperties\": false\n },\n \"outputSchema\": {\n \"type\": \"object\",\n \"properties\": {\n \"status\": { \"type\": \"string\" }\n },\n \"required\": [\"status\"],\n \"additionalProperties\": false\n }\n}\n```\n\n```js\nexport default async function lookupAccount({ accountId }) {\n return { status: \"active\" };\n}\n```\n\nAction entrypoints can use ordinary relative project imports, TypeScript\nmodules, npm packages, transitive dependencies, and standard exports. Orcha\nbundles the complete import graph for development and production.\n\nClient actions use `\"execution\": \"client\"` and do not include executable\ncode. Orcha pauses until the caller submits their results. Keep action names,\ndescriptions, schemas, and implementations aligned.\n\nAction `index.json` fields:\n\n- `name` (required): model-facing tool name, 1\u201364 letters, numbers,\n underscores, or hyphens. `load_skill` is reserved.\n- `description` (required): tells the model when and why to call the action.\n- `enabled` (optional): defaults to `true`. Set to `false` to keep a WIP\n action inactive without deleting its folder.\n- `execution` (optional): defaults to `\"local\"`, which executes `index.js`.\n Set to `\"client\"` to pause and delegate execution to the application.\n- `parameters` (required): JSON Schema for model-generated arguments.\n- `outputSchema` (optional): JSON Schema validated against local or submitted\n client output before the model receives it.\n- `timeoutMs` (optional): integer from 1 through 120000; defaults to 10000.\n- `permissions.env` (optional): names copied from `orcha.init().actions.env`\n into the action context.\n- `permissions.network` (optional): exact hosts or wildcard subdomains such\n as `\"api.example.com\"` or `\"*.example.com\"` allowed through\n `context.fetch`. Redirects are rejected.\n- `sideEffect` (optional): descriptive metadata for whether the operation\n mutates external state. It does not currently change execution behavior.\n\n`orcha.init().actions.runtime` and an action's `execution` solve different\nproblems:\n\n- `execution: \"client\"`: Orcha never executes code for this action.\n- `execution: \"local\"` + `runtime: \"sandbox\"`: compiled code runs in a\n QuickJS isolate with JSON-only inputs/outputs, default 32 MB memory, default\n 512 KB stack, interruptible timeout, declared environment values, and\n allowlisted network access through the provided context.\n- `execution: \"local\"` + `runtime: \"native\"`: code runs in the host Node.js\n process. It can use host privileges directly. The timeout rejects slow\n asynchronous work but cannot interrupt synchronous blocking code.\n\nGlobal local-action configuration:\n\n```ts\nactions: {\n runtime: \"sandbox\", // optional; native is the default\n env: {\n BILLING_API_TOKEN: process.env.BILLING_API_TOKEN,\n },\n sandbox: {\n memoryLimitMb: 32,\n stackLimitKb: 512,\n },\n}\n```\n\nThe local action signature is\n`(parameters, context) => output | Promise<output>`. Context contains\n`sessionId`, immutable session `metadata`, a stable `idempotencyKey`,\nallowlisted `env`, guarded `fetch`, and prefixed `log`.\n\n## Skills\n\nSkills are lazy-loaded procedural instructions. Direct child folders under\n`skills/` are discovered automatically. Each folder contains `index.json`\nmetadata and `instructions.md`.\nThe model receives a compact catalog and can call the internal `load_skill`\ntool. Loaded instructions remain active for the durable session. Lifecycle\nevents are `skill.requested`, `skill.loaded`, and `skill.failed`.\n\nSkill `index.json` fields:\n\n- `name` (required): model-facing name, 1\u201364 letters, numbers, underscores,\n or hyphens; unique within the agent.\n- `description` (required): compact catalog description shown before loading.\n- `enabled` (optional): defaults to `true`. Set to `false` to keep a WIP\n skill inactive.\n- `triggers` (optional): non-empty array of non-empty situations describing\n when the model should load the skill.\n\n`instructions.md` is required and cannot be empty for enabled skills.\n`index.json.name` is the name used by the model and durable events.\n\n## Tests\n\nDirect child folders under `tests/` are discovered automatically. Each\ncase's `index.json` defines `input`, optional mocked action responses, and\n`expect`. Tests run the real compiled agent and provider. Actions listed\nunder `actions` are mocked; unlisted local actions execute live. Orcha reports\nlive action use in the terminal, test report, and session trace.\nClient actions must always be mocked because no application client is attached\nto the test runner.\n\n```json\n{\n \"input\": { \"content\": \"Check account acct_123.\" },\n \"actions\": {\n \"lookup_account\": {\n \"responses\": [\n { \"output\": { \"status\": \"active\" } }\n ]\n }\n },\n \"expect\": {\n \"status\": \"completed\",\n \"text\": { \"contains\": [\"active\"] },\n \"actions\": [\n {\n \"name\": \"lookup_account\",\n \"arguments\": { \"equals\": { \"accountId\": \"acct_123\" } }\n }\n ]\n }\n}\n```\n\nTest sessions use the `ses_test_<timestamp>_<uuid>` format and end with a `test.completed`\nevent. Prefer semantic output assertions; verify exact identifiers and values\nthrough action-argument assertions.\n\nTest `index.json` fields:\n\n- `name` (optional): friendly display name. The folder name remains the CLI\n selector.\n- `description` (optional): human-readable purpose.\n- `enabled` (optional): defaults to `true`. Set to `false` to keep a WIP test\n inactive.\n- `input` (required): an inline agent input object or a path to a JSON file\n containing that object. Relative paths resolve from the test case directory.\n- `input.content` (required for inline input): string or multimodal content\n array.\n- `input.variables` (optional): string map available to the session.\n- `input.metadata` (optional): string, number, boolean, or null values.\n- `actions` (optional): mocked actions keyed by compiled action name. Unlisted\n local actions execute live and may cause side effects. Unknown action names\n are rejected, and unlisted client actions are rejected. Every `responses`\n array is consumed in call order.\n- `responses[].output` (required): mocked action result.\n- `responses[].isError` (optional): marks the mocked result as an error.\n- `expect.status` (optional): `\"completed\"` or `\"failed\"`; defaults to\n `\"completed\"`.\n- `expect.output.equals` / `partial` (optional): exact or recursive partial\n comparison against structured output.\n- `expect.text.contains` / `excludes` (optional): case-sensitive semantic\n text checks.\n- `expect.actions` (optional): ordered expected calls. Each may assert\n `arguments.equals` or `arguments.partial`.\n\nThe test folder name is the selector used by `orcha test agent/testName`.\n\n## Evaluations\n\nEvaluations are asynchronous LLM judges discovered from direct child folders\nunder `evaluations/`. Each folder's `index.json` defines its provider, model,\nmetrics, and thresholds:\n\n```json\n{\n \"name\": \"response_quality\",\n \"description\": \"Grounding of agent responses.\",\n \"instructions\": \"Judge the complete response using only confirmed evidence recorded in the session.\",\n \"enabled\": true,\n \"provider\": \"openai\",\n \"model\": \"gpt-5-mini\",\n \"metrics\": [\n {\n \"name\": \"groundedness\",\n \"description\": \"The answer relies on confirmed session evidence.\",\n \"threshold\": 0.8\n }\n ]\n}\n```\n\n`execution.result` does not wait for judges. Await\n`execution.evaluations` when results must finish before process exit.\nEvaluations always finish during `orcha test`; judge errors and missed\nthresholds fail the test. Lifecycle events are `evaluation.requested`,\n`evaluation.completed`, and `evaluation.failed`.\n\nEvaluation `index.json` fields:\n\n- `name` (required): durable model-facing identifier, 1\u201364 letters, numbers,\n underscores, or hyphens; unique within the agent.\n- `description` (optional): short human-facing summary of the evaluator.\n- `instructions` (optional): detailed prompt supplied to the judge. When\n omitted, `description` is used for backward compatibility.\n- `enabled` (optional): defaults to `true`. Disabled evaluations are ignored\n by compilation and runtime.\n- `provider` and `model` (required): independently select the judge. The\n provider must also exist in `orcha.init().providers`.\n- `maxTokens` (optional): positive integer; defaults to 2000 for judges.\n- `reasoningLevel` (optional): non-empty provider-native string forwarded\n without translation.\n- `metrics` (required): non-empty array with unique metric names.\n- `metrics[].name`: 1\u201364 letters, numbers, underscores, or hyphens.\n- `metrics[].description`: exact criterion supplied to the judge.\n- `metrics[].threshold`: inclusive number from 0 to 1. A metric passes when\n the returned score is greater than or equal to this threshold.\n\nThe judge sees a sanitized transcript of user/assistant messages, action\nrequests and outcomes, client-action activity, and loaded skill names. It does\nnot receive internal reasoning blocks, replay metadata, previous evaluation\nresults, or test assertions. It must return exactly one score, reasoning\nstring, and non-empty evidence array for every configured metric.\n\n## Sessions and logs\n\nJSONL is the durable source of truth. Each line is one complete JSON object;\nnever treat the file as one JSON array. Events are append-only and ordered by\n`sequence`.\n\nAll events use this envelope:\n\n```ts\ntype SessionEvent<T> = {\n sequence: number; // starts at 1 and increases across the whole session\n type: SessionEventType;\n timestamp: string; // ISO-8601 UTC timestamp\n run?: number; // present for run-scoped events\n data: T;\n};\n```\n\nSession-scoped events omit `run`. Optional properties whose values are\n`undefined` are omitted from serialized JSON.\n\nShared stored structures:\n\n```ts\ntype Usage = {\n inputTokens: number;\n outputTokens: number;\n reasoningTokens: number | null;\n cacheReadTokens: number;\n cacheWriteTokens: number;\n};\n\ntype ErrorData = {\n code: string;\n message: string;\n retryable?: boolean;\n};\n\ntype UserContent =\n | { type: \"text\"; text: string }\n | { type: \"file\"; filePath: string; mimeType: string }\n | {\n type: \"image\" | \"video\" | \"audio\" | \"url\";\n mimeType: string;\n fileUri: string;\n };\n\ntype AssistantContent =\n | { type: \"text\"; text: string }\n | {\n type: \"reasoning\";\n text: string;\n replay?: { providerId?: string; opaqueData?: string };\n }\n | {\n type: \"tool_call\";\n callId: string;\n name: string;\n arguments: Record<string, unknown>;\n replay?: { providerId?: string; opaqueData?: string };\n };\n\ntype ToolResult = {\n callId: string;\n output: unknown;\n isError?: boolean;\n};\n```\n\nExact event payloads:\n\n```ts\ntype SessionCreated = SessionEvent<{\n schemaVersion: 1;\n sessionId: string;\n agent: string;\n status: \"active\";\n name?: string;\n metadata: Record<string, string | number | boolean | null>;\n variables: Record<string, string>;\n lineage?: {\n origin: \"delegated\";\n parentAgent: string;\n parentSessionId: string;\n parentCallId: string;\n };\n}>; // type \"session.created\", no run\n\ntype SessionUpdated = SessionEvent<{\n name?: string;\n metadata?: Record<string, string | number | boolean | null>;\n}>; // type \"session.updated\", no run\n\ntype RunStarted = SessionEvent<{\n status: \"running\";\n agent: string;\n provider: string;\n model: string;\n region: string;\n reasoningLevel?: string;\n outputType: \"text\" | \"json\";\n clientCapabilities: string[];\n}>; // type \"run.started\"\n\ntype UserMessageCreated = SessionEvent<{\n role: \"user\";\n content: UserContent[];\n}>; // type \"message.created\"\n\ntype AssistantMessageCreated = SessionEvent<{\n status: \"completed\" | \"incomplete\";\n provider: string;\n model: string;\n responseId?: string;\n stopReason?: \"end_turn\" | \"tool_call\" | \"max_tokens\" |\n \"content_filter\" | \"unknown\";\n role: \"assistant\";\n content: AssistantContent[];\n parsedOutput?: unknown; // final JSON output only\n usage: Usage;\n durationMs: number;\n}>; // type \"message.created\"\n\ntype ToolMessageCreated = SessionEvent<{\n role: \"tool\";\n content: ToolResult[];\n}>; // type \"message.created\"\n\ntype ActionRequested = SessionEvent<{\n callId: string;\n name: string;\n arguments: Record<string, unknown>;\n sourceHash?: string;\n idempotencyKey: string; // sessionId:callId\n}>; // type \"action.requested\"\n\ntype ActionCompleted = SessionEvent<{\n callId: string;\n name: string;\n output: unknown;\n sourceHash?: string;\n durationMs: number;\n}>; // type \"action.completed\"\n\ntype ActionFailed = SessionEvent<{\n callId: string;\n name: string;\n sourceHash?: string;\n durationMs: number;\n error: {\n code: \"action_execution_failed\";\n message: string;\n };\n}>; // type \"action.failed\"\n\ntype ClientActionRequested = SessionEvent<{\n status: \"waiting\";\n calls: Array<{\n callId: string;\n name: string;\n arguments: Record<string, unknown>;\n }>;\n localResults: ToolResult[];\n toolCallOrder: string[];\n}>; // type \"client_action.requested\"\n\ntype ClientActionResolved = SessionEvent<{\n status: \"completed\";\n results: ToolResult[];\n}>; // type \"client_action.resolved\"\n\ntype SkillRequested = SessionEvent<{\n callId: string;\n name: unknown;\n}>; // type \"skill.requested\"\n\ntype SkillLoaded = SessionEvent<{\n callId: string;\n name: string;\n alreadyLoaded: boolean;\n}>; // type \"skill.loaded\", no run\n\ntype SkillFailed = SessionEvent<{\n callId: string;\n name: unknown;\n error: {\n code: \"skill_not_found\";\n message: string;\n };\n}>; // type \"skill.failed\"\n\ntype EvaluationRequested = SessionEvent<{\n name: string;\n provider: string;\n model: string;\n evaluatedThroughSequence: number;\n}>; // type \"evaluation.requested\"\n\ntype EvaluationMetric = {\n name: string;\n score: number;\n threshold: number;\n passed: boolean;\n reasoning: string;\n evidence: string[];\n};\n\ntype EvaluationCompleted = SessionEvent<{\n name: string;\n status: \"passed\" | \"failed\";\n metrics: EvaluationMetric[];\n usage?: Usage;\n durationMs: number;\n provider: string;\n model: string;\n evaluatedThroughSequence: number;\n}>; // type \"evaluation.completed\"\n\ntype EvaluationFailed = SessionEvent<{\n name: string;\n status: \"error\";\n metrics: [];\n durationMs: number;\n error: { message: string };\n provider: string;\n model: string;\n evaluatedThroughSequence: number;\n}>; // type \"evaluation.failed\"\n\ntype SubagentInitiated = SessionEvent<{\n callId: string;\n agent: string;\n sessionId: string;\n status: \"running\";\n}>; // type \"subagent.initiated\"\n\ntype SubagentResumed = SessionEvent<{\n callId: string;\n agent: string;\n sessionId: string;\n status: \"running\";\n}>; // type \"subagent.resumed\"\n\ntype SubagentCompletedOrPausedOrFailed = SessionEvent<{\n callId: string;\n agent: string;\n sessionId: string;\n status:\n | \"completed\"\n | \"waiting_for_client_action\"\n | \"paused\"\n | \"failed\";\n output?: unknown;\n usage?: Usage;\n clientToolCalls?: Array<{\n callId: string;\n name: string;\n arguments: Record<string, unknown>;\n }>;\n error?: ErrorData;\n}>; // type \"subagent.completed\" | \"subagent.paused\" | \"subagent.failed\"\n\ntype RunPaused = SessionEvent<{\n status: \"waiting_for_client_action\" | \"paused\";\n clientToolCalls?: Array<{\n callId: string;\n name: string;\n arguments: Record<string, unknown>;\n }>;\n usage?: Usage;\n output?: unknown;\n durationMs?: number;\n}>; // type \"run.paused\"\n\ntype RunCompleted = SessionEvent<{\n status: \"completed\";\n durationMs: number;\n usage: Usage;\n}>; // type \"run.completed\"\n\ntype RunFailed = SessionEvent<{\n status: \"failed\";\n durationMs: number;\n usage?: Usage;\n error: ErrorData;\n}>; // type \"run.failed\"\n\ntype SessionPaused = SessionEvent<{\n status: \"paused\";\n}>; // type \"session.paused\", no run\n\ntype SessionResumed = SessionEvent<{\n status: \"active\";\n}>; // type \"session.resumed\", no run\n\ntype TestCompleted = SessionEvent<{\n suiteId: string;\n agent: string;\n test: string;\n status: \"passed\" | \"failed\";\n durationMs: number;\n assertions: Array<{\n path: string;\n passed: boolean;\n message: string;\n expected?: unknown;\n actual?: unknown;\n }>;\n usage?: Usage;\n evaluations?: Array<{\n name: string;\n status: \"passed\" | \"failed\" | \"error\";\n metrics: EvaluationMetric[];\n usage?: Usage;\n durationMs: number;\n error?: { message: string };\n }>;\n warnings?: Array<{\n code: \"live_actions\";\n message: string;\n actions: string[];\n }>;\n error?: ErrorData;\n}>; // type \"test.completed\", no run\n\ntype TestWarning = SessionEvent<{\n code: \"live_actions\";\n message: string;\n actions: string[];\n}>; // type \"test.warning\", no run\n```\n\nTypical event order:\n\n```text\nsession.created\nrun.started\nmessage.created (user)\nmessage.created (assistant, possibly with tool_call)\naction.requested \u2192 action.completed|action.failed # local action\nsubagent.initiated\n...child session advances independently...\nsubagent.paused\nsubagent.resumed\n...child session advances independently...\nsubagent.completed|subagent.paused|subagent.failed\nmessage.created (tool)\n...additional model/action rounds...\nmessage.created (assistant final)\nrun.completed\nevaluation.requested\nevaluation.completed|evaluation.failed\n```\n\nFor client actions, `client_action.requested` and `run.paused` replace the\nimmediate tool message. A later `resume(...toolResults)` appends\n`client_action.resolved`, the tool message, and continues the same run\nnumber. A conversational `resume(...content)` starts a new run number.\n\n## How Orcha works behind the scenes\n\n### Compilation\n\n1. `orcha/index.ts` calls `orcha.init()` with explicit agent paths.\n2. The compiler reads each registered agent's `index.json` and\n `instructions.md`.\n3. Direct child folders under `actions/`, `skills/`, `tests/`, and\n `evaluations/` are discovered automatically. Items with `\"enabled\": false`\n remain inactive.\n4. Local action source is bundled and SHA-256 hashed. Production bundles keep\n only provider adapters required by agents and enabled evaluations.\n5. Invalid paths, duplicate model-facing names, missing files, unsupported\n configuration values, and missing schema objects fail before execution.\n Concrete action arguments and outputs are validated against their schemas\n when the action is used.\n\n`orcha dev` repeats validation when files change. `orcha build` performs\noffline production compilation. Neither command invokes a provider.\n\n### Run lifecycle\n\n1. `run()` creates a `ses_<timestamp>_<uuid>`, acquires the per-session execution lock,\n appends `session.created`, then starts run 1.\n2. `resume()` reads and validates the existing session. Message continuation\n starts a new run; submitted client results continue the paused run.\n3. The provider receives the base instructions, available-skill catalog,\n loaded skill instructions, normalized conversation messages, action\n schemas, model settings, and current client capabilities.\n4. Provider-specific responses are normalized into text, reasoning, and tool\n call blocks. Opaque replay metadata is stored only when a provider needs it\n to replay its own prior block correctly.\n5. Tool calls are checked against compiled actions and declared client\n capabilities. Arguments and outputs are validated against JSON Schema.\n6. `load_skill` updates durable session instructions. Local actions execute\n through the configured runtime. Client actions pause safely. Tool results\n are reordered to match the model's original call order.\n7. Internal agent tools start or continue linked child sessions. A\n `run_agent` input accepts the same text or multimodal content shape as a\n normal run. Local file paths resolve from the configured Orcha project root.\n The parent reports `waiting_for_subagent` until each synchronous child\n call returns.\n8. The model loop continues until final output, failure, a pause, or the\n maximum of 10 action rounds.\n9. Usage is normalized and aggregated across every model call in the run.\n10. After `run.completed`, enabled evaluations start in the background.\n `execution.result` is already available; `execution.evaluations` waits\n for judge completion and durable persistence.\n\nThe per-session lock prevents two model executions from mutating one session\nat once. Background evaluation writes queue behind active runs so they cannot\ncause `resume()` to fail spuriously or reuse sequence numbers.\n\n### Replay and context\n\nOrcha does not send raw JSONL back to the model. It projects durable events\ninto provider-neutral conversation messages. Completed and explicitly paused\nconversational runs become user, assistant, and tool messages, so a new run\nsees incomplete assistant text preserved by `pause()`. Lifecycle bookkeeping\nsuch as durations, test assertions, and evaluation events is excluded from\nmodel context. Reasoning text and provider replay metadata are retained where\nneeded for faithful continuation but are omitted from evaluation transcripts.\n\n### Reading sessions\n\n- `agent.get(sessionId)` projects the latest status, pending client actions,\n last output, metadata, and aggregate usage.\n- `agent.history(sessionId, { page, pageSize })` returns a safe user-facing\n timeline rather than raw provider bookkeeping.\n- `agent.events(sessionId, { page, pageSize })` returns the canonical durable\n event records for observability, audit, and debugging tools. For subsequent\n pages of a changing session, pass the first response's `throughSequence`\n back in the options to keep pagination on a stable event boundary.\n- `agent.list({ page, pageSize, status, metadata })` lists projected session\n snapshots.\n- `agent.subagentHistory(childSessionId, { page, pageSize })` returns the\n projected history of a child owned by this parent agent. Lineage checks\n prevent access through unrelated agents.\n- `agent.subagentEvents(childSessionId, { page, pageSize })` returns that\n owned child's canonical events for full transcript and debugging interfaces.\n- `agent.pause(sessionId)` aborts active provider work for the session and\n active children, preserving streamed text as an incomplete assistant\n message. `agent.resume(sessionId)` starts a new run with `\"Continue.\"`;\n pass content to give different instructions. Pending client actions still\n require their exact tool results.\n- Never read storage files directly. Use these methods so applications remain\n compatible with JSONL, SQLite, IndexedDB, remote, and future storage adapters.\n\n## Change rules\n\n- Register every new agent and subagent explicitly. Agent-owned actions, skills,\n tests, and evaluations are auto-discovered; use `\"enabled\": false` for WIP.\n- Keep runtime behavior provider-neutral.\n- Mock actions when tests must avoid side effects; document intentional live\n action coverage.\n- Do not commit `.orcha/`; it contains generated output and session data.\n- Run `orcha dev` after filesystem changes and `orcha test` when behavior\n changes.\n- Do not weaken assertions to match incorrect behavior. Remove an assertion\n only when it is stricter than the documented agent contract.\n";
2
2
  //# sourceMappingURL=cli-agents-template.d.ts.map
@@ -1 +1 @@
1
- {"version":3,"file":"cli-agents-template.d.ts","sourceRoot":"","sources":["../src/cli-agents-template.ts"],"names":[],"mappings":"AAAA,eAAO,MAAM,SAAS,ohqCA2iCrB,CAAC"}
1
+ {"version":3,"file":"cli-agents-template.d.ts","sourceRoot":"","sources":["../src/cli-agents-template.ts"],"names":[],"mappings":"AAAA,eAAO,MAAM,SAAS,qgqCA8hCrB,CAAC"}
@@ -7,14 +7,14 @@ files under \`.orcha/\`.
7
7
  ## Commands
8
8
 
9
9
  - \`orcha init\` creates the initial Orcha files without overwriting files.
10
- - \`orcha dev\` validates the registry and watches \`orcha/**\` for changes.
10
+ - \`orcha dev\` validates agent source and watches \`orcha/**\` for changes.
11
11
  - \`orcha playground\` opens the built-in local UI and hot reloads agent source.
12
12
  - \`orcha run <agent> --input "…"\` executes one registered agent.
13
13
  - \`orcha run <agent> --input-file request.json\` accepts structured input.
14
14
  - \`orcha run <agent> --session <id> --input "…"\` continues a session.
15
15
  - \`orcha run <agent> --session <id> --tool-results results.json\` submits
16
16
  pending client-action results.
17
- - \`orcha test\` runs every registered agent test.
17
+ - \`orcha test\` runs every enabled discovered agent test.
18
18
  - \`orcha test <agent>\` or \`orcha test <agent>/<case>\` narrows the run.
19
19
  - \`orcha build\` creates the production Orcha bundle without calling models.
20
20
  - Add \`--json\` to \`run\` and \`test\` for machine-readable output.
@@ -35,17 +35,16 @@ orcha.init({
35
35
  anthropic: process.env.ANTHROPIC_API_KEY ?? "",
36
36
  openai: process.env.OPENAI_API_KEY ?? "",
37
37
  },
38
- actions: { runtime: "native" },
39
38
  agents: {
40
39
  supportBot: "./supportBot",
41
40
  },
42
41
  });
43
42
  \`\`\`
44
43
 
45
- Only registered folders are compiled. Agent keys become runtime properties
46
- such as \`orcha.supportBot\`. Use \`actions.runtime: "sandbox"\` for isolated
47
- local action execution or \`"native"\` when the application intentionally
48
- allows action modules to execute in its Node.js process.
44
+ Only registered agents are compiled. Agent keys become runtime properties
45
+ such as \`orcha.supportBot\`. Local actions use the native Node.js runtime by
46
+ default. Set \`actions.runtime: "sandbox"\` only when isolated execution is
47
+ required.
49
48
 
50
49
  \`orcha.init()\` fields:
51
50
 
@@ -53,8 +52,8 @@ allows action modules to execute in its Node.js process.
53
52
  name.
54
53
  - \`agents\` (required): runtime property names mapped to folders relative to
55
54
  \`orcha/\`. At least one agent is required.
56
- - \`actions\` (required only when a registered agent has local actions):
57
- selects the local execution runtime and its environment/sandbox settings.
55
+ - \`actions\` (optional): overrides the native local-action runtime or configures
56
+ its environment and sandbox settings.
58
57
  - \`storage.strategy\` (optional): currently only \`"node-jsonl"\`.
59
58
  - \`storage.directory\` (optional): session directory relative to project
60
59
  root; defaults to \`.orcha/sessions\`.
@@ -363,7 +362,6 @@ actions/
363
362
  {
364
363
  "name": "lookup_account",
365
364
  "description": "Look up one account.",
366
- "execution": "local",
367
365
  "parameters": {
368
366
  "type": "object",
369
367
  "properties": {
@@ -402,8 +400,10 @@ Action \`index.json\` fields:
402
400
  - \`name\` (required): model-facing tool name, 1–64 letters, numbers,
403
401
  underscores, or hyphens. \`load_skill\` is reserved.
404
402
  - \`description\` (required): tells the model when and why to call the action.
405
- - \`execution\` (required): \`"local"\` executes \`index.js\`; \`"client"\`
406
- pauses the run and delegates execution to the application.
403
+ - \`enabled\` (optional): defaults to \`true\`. Set to \`false\` to keep a WIP
404
+ action inactive without deleting its folder.
405
+ - \`execution\` (optional): defaults to \`"local"\`, which executes \`index.js\`.
406
+ Set to \`"client"\` to pause and delegate execution to the application.
407
407
  - \`parameters\` (required): JSON Schema for model-generated arguments.
408
408
  - \`outputSchema\` (optional): JSON Schema validated against local or submitted
409
409
  client output before the model receives it.
@@ -432,7 +432,7 @@ Global local-action configuration:
432
432
 
433
433
  \`\`\`ts
434
434
  actions: {
435
- runtime: "sandbox", // required when any registered action is local
435
+ runtime: "sandbox", // optional; native is the default
436
436
  env: {
437
437
  BILLING_API_TOKEN: process.env.BILLING_API_TOKEN,
438
438
  },
@@ -450,18 +450,9 @@ allowlisted \`env\`, guarded \`fetch\`, and prefixed \`log\`.
450
450
 
451
451
  ## Skills
452
452
 
453
- Skills are lazy-loaded procedural instructions. Register only intended skills:
454
-
455
- \`\`\`js
456
- // skills/index.js
457
- import { defineSkills } from "orchajs/skills";
458
-
459
- export default defineSkills({
460
- incidentTriage: "./incidentTriage",
461
- });
462
- \`\`\`
463
-
464
- Each skill folder contains \`index.json\` metadata and \`instructions.md\`.
453
+ Skills are lazy-loaded procedural instructions. Direct child folders under
454
+ \`skills/\` are discovered automatically. Each folder contains \`index.json\`
455
+ metadata and \`instructions.md\`.
465
456
  The model receives a compact catalog and can call the internal \`load_skill\`
466
457
  tool. Loaded instructions remain active for the durable session. Lifecycle
467
458
  events are \`skill.requested\`, \`skill.loaded\`, and \`skill.failed\`.
@@ -471,27 +462,19 @@ Skill \`index.json\` fields:
471
462
  - \`name\` (required): model-facing name, 1–64 letters, numbers, underscores,
472
463
  or hyphens; unique within the agent.
473
464
  - \`description\` (required): compact catalog description shown before loading.
465
+ - \`enabled\` (optional): defaults to \`true\`. Set to \`false\` to keep a WIP
466
+ skill inactive.
474
467
  - \`triggers\` (optional): non-empty array of non-empty situations describing
475
468
  when the model should load the skill.
476
469
 
477
- \`instructions.md\` is required and cannot be empty. The key in
478
- \`skills/index.js\` is only a registration label; \`index.json.name\` is the
479
- name used by the model and durable events. Unregistered folders are ignored.
470
+ \`instructions.md\` is required and cannot be empty for enabled skills.
471
+ \`index.json.name\` is the name used by the model and durable events.
480
472
 
481
473
  ## Tests
482
474
 
483
- Register tests in \`tests/index.js\`:
484
-
485
- \`\`\`js
486
- import { defineTests } from "orchajs/testing";
487
-
488
- export default defineTests({
489
- activeAccount: "./activeAccount",
490
- });
491
- \`\`\`
492
-
493
- Each case's \`index.json\` defines \`input\`, optional mocked action responses,
494
- and \`expect\`. Tests run the real compiled agent and provider. Actions listed
475
+ Direct child folders under \`tests/\` are discovered automatically. Each
476
+ case's \`index.json\` defines \`input\`, optional mocked action responses, and
477
+ \`expect\`. Tests run the real compiled agent and provider. Actions listed
495
478
  under \`actions\` are mocked; unlisted local actions execute live. Orcha reports
496
479
  live action use in the terminal, test report, and session trace.
497
480
  Client actions must always be mocked because no application client is attached
@@ -526,7 +509,11 @@ through action-argument assertions.
526
509
 
527
510
  Test \`index.json\` fields:
528
511
 
512
+ - \`name\` (optional): friendly display name. The folder name remains the CLI
513
+ selector.
529
514
  - \`description\` (optional): human-readable purpose.
515
+ - \`enabled\` (optional): defaults to \`true\`. Set to \`false\` to keep a WIP test
516
+ inactive.
530
517
  - \`input\` (required): an inline agent input object or a path to a JSON file
531
518
  containing that object. Relative paths resolve from the test case directory.
532
519
  - \`input.content\` (required for inline input): string or multimodal content
@@ -548,15 +535,13 @@ Test \`index.json\` fields:
548
535
  - \`expect.actions\` (optional): ordered expected calls. Each may assert
549
536
  \`arguments.equals\` or \`arguments.partial\`.
550
537
 
551
- The registration key in \`tests/index.js\` is the test selector used by
552
- \`orcha test agent/testName\`; its value resolves to the case folder.
538
+ The test folder name is the selector used by \`orcha test agent/testName\`.
553
539
 
554
540
  ## Evaluations
555
541
 
556
- Evaluations are asynchronous LLM judges registered in
557
- \`evaluations/index.js\` with \`defineEvaluations\` from
558
- \`orchajs/evaluations\`. Each folder's \`index.json\` defines its provider,
559
- model, metrics, and thresholds:
542
+ Evaluations are asynchronous LLM judges discovered from direct child folders
543
+ under \`evaluations/\`. Each folder's \`index.json\` defines its provider, model,
544
+ metrics, and thresholds:
560
545
 
561
546
  \`\`\`json
562
547
  {
@@ -589,8 +574,8 @@ Evaluation \`index.json\` fields:
589
574
  - \`description\` (optional): short human-facing summary of the evaluator.
590
575
  - \`instructions\` (optional): detailed prompt supplied to the judge. When
591
576
  omitted, \`description\` is used for backward compatibility.
592
- - \`enabled\` (optional): defaults to \`true\`. Disabled evaluations are
593
- compiled but do not run.
577
+ - \`enabled\` (optional): defaults to \`true\`. Disabled evaluations are ignored
578
+ by compilation and runtime.
594
579
  - \`provider\` and \`model\` (required): independently select the judge. The
595
580
  provider must also exist in \`orcha.init().providers\`.
596
581
  - \`maxTokens\` (optional): positive integer; defaults to 2000 for judges.
@@ -974,8 +959,9 @@ number. A conversational \`resume(...content)\` starts a new run number.
974
959
  1. \`orcha/index.ts\` calls \`orcha.init()\` with explicit agent paths.
975
960
  2. The compiler reads each registered agent's \`index.json\` and
976
961
  \`instructions.md\`.
977
- 3. Every directory under \`actions/\` is compiled. Skills, tests, and
978
- evaluations are included only through their local \`index.js\` registry.
962
+ 3. Direct child folders under \`actions/\`, \`skills/\`, \`tests/\`, and
963
+ \`evaluations/\` are discovered automatically. Items with \`"enabled": false\`
964
+ remain inactive.
979
965
  4. Local action source is bundled and SHA-256 hashed. Production bundles keep
980
966
  only provider adapters required by agents and enabled evaluations.
981
967
  5. Invalid paths, duplicate model-facing names, missing files, unsupported
@@ -1056,7 +1042,8 @@ needed for faithful continuation but are omitted from evaluation transcripts.
1056
1042
 
1057
1043
  ## Change rules
1058
1044
 
1059
- - Register every new agent, skill, test, and evaluation explicitly.
1045
+ - Register every new agent and subagent explicitly. Agent-owned actions, skills,
1046
+ tests, and evaluations are auto-discovered; use \`"enabled": false\` for WIP.
1060
1047
  - Keep runtime behavior provider-neutral.
1061
1048
  - Mock actions when tests must avoid side effects; document intentional live
1062
1049
  action coverage.
@@ -1 +1 @@
1
- {"version":3,"file":"cli-agents-template.js","sourceRoot":"","sources":["../src/cli-agents-template.ts"],"names":[],"mappings":"AAAA,MAAM,CAAC,MAAM,SAAS,GAAG;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA2iCxB,CAAC"}
1
+ {"version":3,"file":"cli-agents-template.js","sourceRoot":"","sources":["../src/cli-agents-template.ts"],"names":[],"mappings":"AAAA,MAAM,CAAC,MAAM,SAAS,GAAG;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;CA8hCxB,CAAC"}
@@ -1 +1 @@
1
- {"version":3,"file":"compile.d.ts","sourceRoot":"","sources":["../../src/compiler/compile.ts"],"names":[],"mappings":"AAOA,OAAO,KAAK,EAGV,iBAAiB,EAKjB,cAAc,EAKf,MAAM,aAAa,CAAC;AAErB,wBAAgB,eAAe,CAC7B,aAAa,EAAE,MAAM,CAAC,MAAM,EAAE,iBAAiB,CAAC,EAChD,YAAY,EAAE,MAAM,EACpB,aAAa,CAAC,EAAE,QAAQ,GAAG,SAAS,GACnC,cAAc,CA4DhB;AA4uBD,wBAAgB,eAAe,CAAC,MAAM,EAAE,cAAc,GAAG,MAAM,CAO9D"}
1
+ {"version":3,"file":"compile.d.ts","sourceRoot":"","sources":["../../src/compiler/compile.ts"],"names":[],"mappings":"AAKA,OAAO,KAAK,EAEV,iBAAiB,EAIjB,cAAc,EAKf,MAAM,aAAa,CAAC;AAErB,wBAAgB,eAAe,CAC7B,aAAa,EAAE,MAAM,CAAC,MAAM,EAAE,iBAAiB,CAAC,EAChD,YAAY,EAAE,MAAM,EACpB,aAAa,GAAE,QAAQ,GAAG,SAAoB,GAC7C,cAAc,CA4DhB;AA0mBD,wBAAgB,eAAe,CAAC,MAAM,EAAE,cAAc,GAAG,MAAM,CAO9D"}