open-multi-agent-kit 0.99.0 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +24 -0
- package/README.md +3 -3
- package/dist/bun/cli.d.ts.map +1 -1
- package/dist/bun/cli.js +1 -0
- package/dist/bun/cli.js.map +1 -1
- package/dist/bun/register-bundled-coding-agent.d.ts +10 -0
- package/dist/bun/register-bundled-coding-agent.d.ts.map +1 -0
- package/dist/bun/register-bundled-coding-agent.js +12 -0
- package/dist/bun/register-bundled-coding-agent.js.map +1 -0
- package/dist/cli/args.d.ts +1 -1
- package/dist/cli/args.d.ts.map +1 -1
- package/dist/cli/args.js +1 -1
- package/dist/cli/args.js.map +1 -1
- package/dist/cli/help.d.ts.map +1 -1
- package/dist/cli/help.js +1 -1
- package/dist/cli/help.js.map +1 -1
- package/dist/cli.d.ts.map +1 -1
- package/dist/cli.js +14 -2
- package/dist/cli.js.map +1 -1
- package/dist/commands/neo-cli.d.ts +9 -0
- package/dist/commands/neo-cli.d.ts.map +1 -0
- package/dist/commands/neo-cli.js +61 -0
- package/dist/commands/neo-cli.js.map +1 -0
- package/dist/core/agent-session.d.ts +29 -2
- package/dist/core/agent-session.d.ts.map +1 -1
- package/dist/core/agent-session.js +107 -30
- package/dist/core/agent-session.js.map +1 -1
- package/dist/core/bundled-skills.d.ts +4 -0
- package/dist/core/bundled-skills.d.ts.map +1 -0
- package/dist/core/bundled-skills.js +32 -0
- package/dist/core/bundled-skills.js.map +1 -0
- package/dist/core/cli-diagnostics.d.ts +6 -0
- package/dist/core/cli-diagnostics.d.ts.map +1 -0
- package/dist/core/cli-diagnostics.js +20 -0
- package/dist/core/cli-diagnostics.js.map +1 -0
- package/dist/core/context-budget-headroom.d.ts +12 -0
- package/dist/core/context-budget-headroom.d.ts.map +1 -1
- package/dist/core/context-budget-headroom.js +35 -0
- package/dist/core/context-budget-headroom.js.map +1 -1
- package/dist/core/context-budget-v2-input-validation.d.ts +4 -0
- package/dist/core/context-budget-v2-input-validation.d.ts.map +1 -0
- package/dist/core/context-budget-v2-input-validation.js +79 -0
- package/dist/core/context-budget-v2-input-validation.js.map +1 -0
- package/dist/core/context-budget-v2-observability.d.ts +15 -0
- package/dist/core/context-budget-v2-observability.d.ts.map +1 -0
- package/dist/core/context-budget-v2-observability.js +35 -0
- package/dist/core/context-budget-v2-observability.js.map +1 -0
- package/dist/core/context-budget-v2-planned-items.d.ts +5 -0
- package/dist/core/context-budget-v2-planned-items.d.ts.map +1 -0
- package/dist/core/context-budget-v2-planned-items.js +40 -0
- package/dist/core/context-budget-v2-planned-items.js.map +1 -0
- package/dist/core/context-budget-v2-planner.d.ts.map +1 -1
- package/dist/core/context-budget-v2-planner.js +57 -38
- package/dist/core/context-budget-v2-planner.js.map +1 -1
- package/dist/core/context-budget-v2-selection.d.ts +19 -3
- package/dist/core/context-budget-v2-selection.d.ts.map +1 -1
- package/dist/core/context-budget-v2-selection.js +35 -57
- package/dist/core/context-budget-v2-selection.js.map +1 -1
- package/dist/core/context-budget-v2-tiers.d.ts.map +1 -1
- package/dist/core/context-budget-v2-tiers.js +9 -42
- package/dist/core/context-budget-v2-tiers.js.map +1 -1
- package/dist/core/context-budget-v2-types.d.ts +8 -1
- package/dist/core/context-budget-v2-types.d.ts.map +1 -1
- package/dist/core/context-budget-v2-types.js.map +1 -1
- package/dist/core/extensions/bundled-virtual-modules.d.ts +15 -0
- package/dist/core/extensions/bundled-virtual-modules.d.ts.map +1 -0
- package/dist/core/extensions/bundled-virtual-modules.js +63 -0
- package/dist/core/extensions/bundled-virtual-modules.js.map +1 -0
- package/dist/core/extensions/loader.d.ts.map +1 -1
- package/dist/core/extensions/loader.js +2 -42
- package/dist/core/extensions/loader.js.map +1 -1
- package/dist/core/extensions/runner.d.ts +1 -0
- package/dist/core/extensions/runner.d.ts.map +1 -1
- package/dist/core/extensions/runner.js +6 -0
- package/dist/core/extensions/runner.js.map +1 -1
- package/dist/core/extensions/types.d.ts +10 -0
- package/dist/core/extensions/types.d.ts.map +1 -1
- package/dist/core/extensions/types.js.map +1 -1
- package/dist/core/index.d.ts +2 -1
- package/dist/core/index.d.ts.map +1 -1
- package/dist/core/index.js +2 -1
- package/dist/core/index.js.map +1 -1
- package/dist/core/loadout-runtime.d.ts +2 -8
- package/dist/core/loadout-runtime.d.ts.map +1 -1
- package/dist/core/loadout-runtime.js.map +1 -1
- package/dist/core/mcp/client.d.ts +27 -6
- package/dist/core/mcp/client.d.ts.map +1 -1
- package/dist/core/mcp/client.js +80 -20
- package/dist/core/mcp/client.js.map +1 -1
- package/dist/core/mcp/manager.d.ts +10 -1
- package/dist/core/mcp/manager.d.ts.map +1 -1
- package/dist/core/mcp/manager.js +68 -9
- package/dist/core/mcp/manager.js.map +1 -1
- package/dist/core/mcp/protocol.d.ts +9 -3
- package/dist/core/mcp/protocol.d.ts.map +1 -1
- package/dist/core/mcp/protocol.js +61 -15
- package/dist/core/mcp/protocol.js.map +1 -1
- package/dist/core/mcp/stdio-transport.d.ts +12 -1
- package/dist/core/mcp/stdio-transport.d.ts.map +1 -1
- package/dist/core/mcp/stdio-transport.js +30 -2
- package/dist/core/mcp/stdio-transport.js.map +1 -1
- package/dist/core/mcp-descriptor-injection.d.ts +26 -0
- package/dist/core/mcp-descriptor-injection.d.ts.map +1 -0
- package/dist/core/mcp-descriptor-injection.js +25 -0
- package/dist/core/mcp-descriptor-injection.js.map +1 -0
- package/dist/core/mcp-public-presets.d.ts +1 -3
- package/dist/core/mcp-public-presets.d.ts.map +1 -1
- package/dist/core/mcp-public-presets.js +3 -15
- package/dist/core/mcp-public-presets.js.map +1 -1
- package/dist/core/model-registry.d.ts.map +1 -1
- package/dist/core/model-registry.js +24 -3
- package/dist/core/model-registry.js.map +1 -1
- package/dist/core/neo/catalog.d.ts +20 -0
- package/dist/core/neo/catalog.d.ts.map +1 -0
- package/dist/core/neo/catalog.js +50 -0
- package/dist/core/neo/catalog.js.map +1 -0
- package/dist/core/neo/setup.d.ts +3 -0
- package/dist/core/neo/setup.d.ts.map +1 -0
- package/dist/core/neo/setup.js +47 -0
- package/dist/core/neo/setup.js.map +1 -0
- package/dist/core/provider-default-models.d.ts +1 -0
- package/dist/core/provider-default-models.d.ts.map +1 -1
- package/dist/core/provider-default-models.js +1 -0
- package/dist/core/provider-default-models.js.map +1 -1
- package/dist/core/provider-display-names.d.ts.map +1 -1
- package/dist/core/provider-display-names.js +1 -0
- package/dist/core/provider-display-names.js.map +1 -1
- package/dist/core/provider-error-classification.d.ts +57 -0
- package/dist/core/provider-error-classification.d.ts.map +1 -0
- package/dist/core/provider-error-classification.js +103 -0
- package/dist/core/provider-error-classification.js.map +1 -0
- package/dist/core/provider-resilience.d.ts +10 -0
- package/dist/core/provider-resilience.d.ts.map +1 -1
- package/dist/core/provider-resilience.js +36 -3
- package/dist/core/provider-resilience.js.map +1 -1
- package/dist/core/provider-usage-commandcode.d.ts +9 -0
- package/dist/core/provider-usage-commandcode.d.ts.map +1 -0
- package/dist/core/provider-usage-commandcode.js +198 -0
- package/dist/core/provider-usage-commandcode.js.map +1 -0
- package/dist/core/provider-usage-devin.d.ts +5 -3
- package/dist/core/provider-usage-devin.d.ts.map +1 -1
- package/dist/core/provider-usage-devin.js +15 -5
- package/dist/core/provider-usage-devin.js.map +1 -1
- package/dist/core/provider-usage-types.d.ts +1 -1
- package/dist/core/provider-usage-types.d.ts.map +1 -1
- package/dist/core/provider-usage-types.js.map +1 -1
- package/dist/core/provider-usage.d.ts.map +1 -1
- package/dist/core/provider-usage.js +6 -0
- package/dist/core/provider-usage.js.map +1 -1
- package/dist/core/resource-admission.d.ts +36 -0
- package/dist/core/resource-admission.d.ts.map +1 -1
- package/dist/core/resource-admission.js +59 -0
- package/dist/core/resource-admission.js.map +1 -1
- package/dist/core/resource-loader.d.ts.map +1 -1
- package/dist/core/resource-loader.js +2 -2
- package/dist/core/resource-loader.js.map +1 -1
- package/dist/core/run-usage-ledger.d.ts +47 -0
- package/dist/core/run-usage-ledger.d.ts.map +1 -0
- package/dist/core/run-usage-ledger.js +162 -0
- package/dist/core/run-usage-ledger.js.map +1 -0
- package/dist/core/run-usage-operation.d.ts +8 -0
- package/dist/core/run-usage-operation.d.ts.map +1 -0
- package/dist/core/run-usage-operation.js +12 -0
- package/dist/core/run-usage-operation.js.map +1 -0
- package/dist/core/sdk.d.ts.map +1 -1
- package/dist/core/sdk.js +5 -6
- package/dist/core/sdk.js.map +1 -1
- package/dist/core/session-failure-cause.d.ts.map +1 -1
- package/dist/core/session-failure-cause.js +14 -5
- package/dist/core/session-failure-cause.js.map +1 -1
- package/dist/core/session-prompt-lifecycle.d.ts +6 -0
- package/dist/core/session-prompt-lifecycle.d.ts.map +1 -1
- package/dist/core/session-prompt-lifecycle.js +47 -2
- package/dist/core/session-prompt-lifecycle.js.map +1 -1
- package/dist/core/session-termination.d.ts.map +1 -1
- package/dist/core/session-termination.js +4 -2
- package/dist/core/session-termination.js.map +1 -1
- package/dist/core/subagent-lane-authority.d.ts +28 -0
- package/dist/core/subagent-lane-authority.d.ts.map +1 -0
- package/dist/core/subagent-lane-authority.js +119 -0
- package/dist/core/subagent-lane-authority.js.map +1 -0
- package/dist/core/subagent-lane-contract.d.ts +114 -0
- package/dist/core/subagent-lane-contract.d.ts.map +1 -0
- package/dist/core/subagent-lane-contract.js +18 -0
- package/dist/core/subagent-lane-contract.js.map +1 -0
- package/dist/core/subagent-lane-launcher.d.ts +3 -6
- package/dist/core/subagent-lane-launcher.d.ts.map +1 -1
- package/dist/core/subagent-lane-launcher.js +53 -12
- package/dist/core/subagent-lane-launcher.js.map +1 -1
- package/dist/core/subagent-orchestration.d.ts +4 -17
- package/dist/core/subagent-orchestration.d.ts.map +1 -1
- package/dist/core/subagent-orchestration.js.map +1 -1
- package/dist/core/verified-run/broker.d.ts.map +1 -1
- package/dist/core/verified-run/broker.js +4 -1
- package/dist/core/verified-run/broker.js.map +1 -1
- package/dist/core/workload-permit-pool.d.ts +1 -1
- package/dist/core/workload-permit-pool.d.ts.map +1 -1
- package/dist/core/workload-permit-pool.js +3 -0
- package/dist/core/workload-permit-pool.js.map +1 -1
- package/dist/guardrails/strict-evidence-approval-adapter.d.ts +34 -0
- package/dist/guardrails/strict-evidence-approval-adapter.d.ts.map +1 -0
- package/dist/guardrails/strict-evidence-approval-adapter.js +70 -0
- package/dist/guardrails/strict-evidence-approval-adapter.js.map +1 -0
- package/dist/index.d.ts +3 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -0
- package/dist/index.js.map +1 -1
- package/dist/main.d.ts.map +1 -1
- package/dist/main.js +11 -18
- package/dist/main.js.map +1 -1
- package/dist/modes/acp/acp-agent.d.ts +24 -0
- package/dist/modes/acp/acp-agent.d.ts.map +1 -0
- package/dist/modes/acp/acp-agent.js +133 -0
- package/dist/modes/acp/acp-agent.js.map +1 -0
- package/dist/modes/acp/acp-mode.d.ts +4 -0
- package/dist/modes/acp/acp-mode.d.ts.map +1 -0
- package/dist/modes/acp/acp-mode.js +34 -0
- package/dist/modes/acp/acp-mode.js.map +1 -0
- package/dist/modes/acp/acp-session.d.ts +4 -0
- package/dist/modes/acp/acp-session.d.ts.map +1 -0
- package/dist/modes/acp/acp-session.js +76 -0
- package/dist/modes/acp/acp-session.js.map +1 -0
- package/dist/modes/acp/acp-transport.d.ts +5 -0
- package/dist/modes/acp/acp-transport.d.ts.map +1 -0
- package/dist/modes/acp/acp-transport.js +90 -0
- package/dist/modes/acp/acp-transport.js.map +1 -0
- package/dist/modes/interactive/interactive-mode.d.ts.map +1 -1
- package/dist/modes/interactive/interactive-mode.js +1 -0
- package/dist/modes/interactive/interactive-mode.js.map +1 -1
- package/docs/audit-revalidation-2026-09-17.md +77 -0
- package/docs/devin-harness.md +33 -4
- package/docs/ecraf-normalization.md +83 -0
- package/docs/model-catalog-refresh.md +51 -1
- package/docs/models.md +1 -1
- package/docs/neo.md +134 -0
- package/docs/providers.md +72 -6
- package/docs/run-usage-ledger.md +56 -0
- package/docs/runtime-algorithms.md +44 -0
- package/docs/skills.md +1 -1
- package/docs/usage.md +2 -1
- package/examples/extensions/custom-provider-anthropic/package-lock.json +2 -2
- package/examples/extensions/custom-provider-anthropic/package.json +1 -1
- package/examples/extensions/custom-provider-gitlab-duo/package.json +1 -1
- package/examples/extensions/gondolin/package-lock.json +2 -2
- package/examples/extensions/gondolin/package.json +1 -1
- package/examples/extensions/sandbox/package-lock.json +2 -2
- package/examples/extensions/sandbox/package.json +1 -1
- package/examples/extensions/subagent/adaptive-agent-runtime.ts +14 -1
- package/examples/extensions/subagent/graph-result.ts +48 -0
- package/examples/extensions/subagent/index.ts +246 -111
- package/examples/extensions/subagent/managed-process-tree.ts +42 -0
- package/examples/extensions/subagent/managed-process.test.ts +24 -0
- package/examples/extensions/subagent/managed-process.ts +93 -112
- package/examples/extensions/subagent/subagent-runtime-types.ts +17 -1
- package/examples/extensions/subagent/subagent-stream.ts +161 -0
- package/examples/extensions/terminal-browser/README.md +54 -0
- package/examples/extensions/terminal-browser/bridge-protocol.ts +29 -0
- package/examples/extensions/terminal-browser/bridge.ts +219 -0
- package/examples/extensions/terminal-browser/browser-surface.ts +288 -0
- package/examples/extensions/terminal-browser/index.ts +215 -0
- package/examples/extensions/terminal-browser/placeholders.ts +49 -0
- package/examples/extensions/with-deps/package-lock.json +2 -2
- package/examples/extensions/with-deps/package.json +1 -1
- package/npm-shrinkwrap.json +18 -18
- package/package.json +8 -7
- package/resources/neo/skills/omk-browser/SKILL.md +32 -0
- package/resources/neo/skills/omk-code-review/SKILL.md +24 -0
- package/resources/neo/skills/omk-computeruse/SKILL.md +34 -0
- package/resources/neo/skills/omk-mcp-setup/SKILL.md +48 -0
- package/resources/neo/skills/omk-research/SKILL.md +24 -0
- package/resources/neo/skills/omk-site/SKILL.md +28 -0
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
# Audit revalidation — 2026-09-17
|
|
2
|
+
|
|
3
|
+
Baseline: `1e58611d0c28775c8a066b6e913ea6710ccea396`, with unrelated local
|
|
4
|
+
provider/model changes present. This is a scoped local revalidation, not a release
|
|
5
|
+
approval, full security audit, or competitor benchmark.
|
|
6
|
+
|
|
7
|
+
## Inputs and scope
|
|
8
|
+
|
|
9
|
+
The local September 14 reproduction ZIP targets
|
|
10
|
+
`69e0637d8fddd26eb40a7ec3f8ebfb65b500de5b`. All 25 entries listed in its
|
|
11
|
+
SHA256SUMS matched. Its recorded six expected failures describe that historical
|
|
12
|
+
snapshot, not the current checkout. The archived runner was not executed or
|
|
13
|
+
substituted for current repository tests.
|
|
14
|
+
|
|
15
|
+
The September 15 strategy report combines historical findings with proposed
|
|
16
|
+
resource-aware scheduling, recovery, and comparative evaluation. These proposals
|
|
17
|
+
are not measured product benefits. The September 16 GitHub audit targets
|
|
18
|
+
`739bc6f3b6fe1c89bfe058aad7ec17ad252fb10e`; its missing DAG contract is no longer
|
|
19
|
+
present in this baseline. The documents were used to select the checks below;
|
|
20
|
+
not every proposed acceptance criterion was executed.
|
|
21
|
+
|
|
22
|
+
## Executed regression groups
|
|
23
|
+
|
|
24
|
+
Commands use `node ../../node_modules/vitest/dist/cli.js --run` from the named
|
|
25
|
+
workspace. No external provider requests were needed.
|
|
26
|
+
|
|
27
|
+
| Workspace | Test files | Result |
|
|
28
|
+
| --- | --- | --- |
|
|
29
|
+
| agent | `test/tool-dag-*.test.ts` | 83 passed, including 10,000 seeded bounded schedules |
|
|
30
|
+
| coding-agent | `test/mcp/{protocol,protocol-required,manager-lifecycle,manager-health,client,tools}.test.ts` | 61 passed; client tests include a local stdio fixture |
|
|
31
|
+
| protocol | `test/{protocol,evaluation-candidate-binding,validation,run-dag-contract,run-dag-properties}.test.ts` | 44 passed |
|
|
32
|
+
| coding-agent | `test/{context-budget-v2-validation-cache,context-budget-cache,context-budget-cache-policy,context-budget-v2-tier-floor,context-budget-v2-eligibility,context-budget-v2-nonfinite-tokens,context-budget-governor-v2}.test.ts` | 46 passed after the fix below |
|
|
33
|
+
|
|
34
|
+
These are 234 distinct passing test cases, not 234 independent production tasks.
|
|
35
|
+
An initial command included nonexistent `test/mcp/manager.test.ts`; Vitest ran
|
|
36
|
+
other matching files successfully. Manager coverage above comes from the actual
|
|
37
|
+
`manager-lifecycle`, `manager-health`, and `client` files, not that missing path.
|
|
38
|
+
|
|
39
|
+
## Fixed: input diagnostics leaked across plan-cache reuse
|
|
40
|
+
|
|
41
|
+
`planPromptContextBudgetV2()` originally decided cache eligibility before
|
|
42
|
+
`validateBudgetItems()`. Duplicate IDs are dropped and invalid token estimates are
|
|
43
|
+
recomputed. Their sanitized items can produce exactly the same plan key as valid
|
|
44
|
+
input, but their input diagnostics are different.
|
|
45
|
+
|
|
46
|
+
Three regressions failed before the fix:
|
|
47
|
+
|
|
48
|
+
1. A cached valid plan hid a duplicate-ID diagnostic on a later call.
|
|
49
|
+
2. A plan produced from duplicate IDs carried its diagnostic into a later valid call.
|
|
50
|
+
3. A cached plan hid the diagnostic for a recomputed non-finite token estimate.
|
|
51
|
+
|
|
52
|
+
Cache eligibility is now decided after input validation. Calls with input or
|
|
53
|
+
budget diagnostics neither read nor write the plan cache. Representation-cache
|
|
54
|
+
validation and normal valid-input plan reuse remain unchanged. The regression
|
|
55
|
+
uses a required item to isolate plan reuse from the existing
|
|
56
|
+
`cache_dependency_unsafe` rejection for plans containing representation-cache hits.
|
|
57
|
+
|
|
58
|
+
Changed implementation: `src/core/context-budget-v2-planner.ts`.
|
|
59
|
+
Regression: `test/context-budget-v2-validation-cache.test.ts` (3 passed).
|
|
60
|
+
This preserves diagnostics; it does not redesign duplicate-item handling or tier
|
|
61
|
+
allocation policy.
|
|
62
|
+
|
|
63
|
+
## Remaining boundaries
|
|
64
|
+
|
|
65
|
+
- Full `npm run check` has been blocked by unrelated local module-size growth in
|
|
66
|
+
`agent-session.ts` and `model-registry.ts`; those files and the ratchet baseline
|
|
67
|
+
are not changed by this fix. A passing focused compiler/test run does not close
|
|
68
|
+
the full repository gate.
|
|
69
|
+
- General observation evaluation still has existential semantics. Passing
|
|
70
|
+
candidate-binding tests does not establish latest-result or complete-coverage
|
|
71
|
+
semantics, or independently authenticate a verifier.
|
|
72
|
+
- ECRAF remains an internal planner. Resource normalization, fairness, live
|
|
73
|
+
admission integration, and equal-budget performance comparisons were not added.
|
|
74
|
+
- This pass does not validate the complete timeout/ownership fault matrix,
|
|
75
|
+
verified-run crash recovery, all MCP authorization boundaries, remote CI,
|
|
76
|
+
branch protection, published artifacts, or router calibration.
|
|
77
|
+
- No release, commit, push, paid benchmark, or new runtime default is implied.
|
package/docs/devin-harness.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Devin SWE-2 harness
|
|
2
2
|
|
|
3
|
-
This page is the canonical operator guide for the built-in `devin` provider
|
|
3
|
+
This page is the canonical operator guide for the built-in `devin` provider, centered on the logical `swe-2` model. Use `/login devin` for the Devin CLI subscription PKCE flow or `DEVIN_API_KEY` for an already-owned CLI session token. A user-local `~/.omk/agent/devin.md` may add operator notes, but it is not the portable product contract.
|
|
4
4
|
|
|
5
5
|
Authentication, transport, and verification limits are owned by [Providers](providers.md#devin-cli); this page covers how OMK drives SWE-2 as a harness.
|
|
6
6
|
|
|
@@ -44,12 +44,12 @@ Guidance from the [SWE-2 announcement](https://cognition.com/blog/swe-2): `mediu
|
|
|
44
44
|
|
|
45
45
|
## Context budget: 1,000,000 tokens
|
|
46
46
|
|
|
47
|
-
`devin/swe-2` ships with `contextWindow: 1000000` and `maxTokens: 16384`. These are local budgets that drive OMK's context budgeting and compaction, **not published SWE-2 limits**; Cognition has not published a context window for SWE-2. The budget also selects the catalog lane:
|
|
47
|
+
`devin/swe-2` ships with `contextWindow: 1000000` and `maxTokens: 16384`. These are local budgets that drive OMK's context budgeting and compaction, **not published SWE-2 limits**; Cognition has not published a context window for SWE-2. Chat completion settings follow the captured native Devin CLI 3000.6.2 defaults (`maxNewlines` 400, empty stop list). `topP=0.95` is protobuf field 8; field 6 is `firstTemperature` and is omitted. This removes the old 200-newline cap and synthetic stops, but does not prove that every interrupted reply had that cause. The budget also selects the catalog lane:
|
|
48
48
|
|
|
49
49
|
1. Before each turn OMK reads `GetCliModelConfigs`. SWE-2 family entries may carry a `1M Context` axis (order `1`) beside the effort axis. A local budget of 1,000,000 or more asks for that 1M-context lane; below it, the standard lane is used and 1M entries are ignored.
|
|
50
50
|
2. A catalog with no 1M-context lane keeps the standard lane for the selected effort.
|
|
51
51
|
3. If the chosen lane declares a context window smaller than the local budget, the request fails with `... declares a N-token context window; lower the models.json contextWindow before retrying`. OMK never shrinks the budget silently, never invents a wire UID, and never downgrades to another effort.
|
|
52
|
-
4. Fast-lane (`Fast Mode`) entries are
|
|
52
|
+
4. Fast-lane (`Fast Mode`) entries are excluded from effort routing and are reachable only through their own UID models (ids ending in `-fast` or `-priority`). Output is capped against the authenticated catalog's declared maximum.
|
|
53
53
|
|
|
54
54
|
To lower the budget (for example if your account only serves the standard lane at 262,144 tokens), override the built-in model in `~/.omk/agent/models.json`:
|
|
55
55
|
|
|
@@ -87,7 +87,7 @@ General prompt-based domain routing is separate and opt-in through `OMK_DOMAIN_R
|
|
|
87
87
|
|
|
88
88
|
## Model selection
|
|
89
89
|
|
|
90
|
-
The `devin` catalog
|
|
90
|
+
The `devin` catalog leads with the logical `swe-2` model; the server's SWE-2 family metadata supplies each effort's wire UID at request time. Every other lane the account catalog advertises is its own logical model whose id is the wire UID — for example `devin/claude-opus-5-high`, `devin/gpt-5-6-sol-xhigh`, `devin/gemini-3-8-flash-medium`, `devin/kimi-k3-max`, `devin/glm-5-3-high`, `devin/grok-4-6-xhigh`, `devin/deepseek-v4-pro-max`, `devin/swe-1-7`, or `devin/inkling-max`. Flat models pin their declared effort lane, so `/think` levels are fixed per model and `No Thinking`/`None` lanes report `reasoning: false`. Availability is account- and plan-dependent: a lane absent from your catalog fails loudly instead of being remapped. Use `/model` or `omk --list-models devin` for the current list. Image input is unsupported; provide text.
|
|
91
91
|
|
|
92
92
|
## Skill and MCP matrix summary
|
|
93
93
|
|
|
@@ -116,6 +116,35 @@ Relevant evidence hooks for SWE-2 lanes are `pre-shell-guard`, `protect-secrets`
|
|
|
116
116
|
4. If a turn fails with an "unavailable or ambiguous" route or a smaller declared context window, treat it as a configuration signal: check `devin models list`, or lower `contextWindow` as shown above. Do not retry with a guessed wire UID.
|
|
117
117
|
5. Keep credentials out of preset JSON, prompts, and logs: the session token, the exchanged user JWT, and `auth.json` contents are secrets under `protect-secrets`.
|
|
118
118
|
|
|
119
|
+
## Troubleshooting `does not provide an export named`
|
|
120
|
+
|
|
121
|
+
If Devin fails with `The requested module './devin-connect.js' does not provide an export named 'MAX_FRAME_BYTES'`, the loaded stream module is mixed with an older unary module. That is a client load error, not an orphan tool call or a remote protocol trailer.
|
|
122
|
+
|
|
123
|
+
- Current OMK keeps the 16 MiB Connect frame cap inside `devin-connect-stream.ts`, so a stale `devin-connect.js` cannot fail that named import.
|
|
124
|
+
- `/new` does not reload already-imported provider modules. Quit and restart OMK after a rebuild.
|
|
125
|
+
- The failure banner should say to restart OMK, not to sanitize a sticky transcript.
|
|
126
|
+
|
|
127
|
+
## Troubleshooting `invalid_argument`
|
|
128
|
+
|
|
129
|
+
A Connect `invalid_argument` trailer is a provider/request failure, not evidence of
|
|
130
|
+
an orphan tool call or a safety refusal. Keep any reported trace ID for support;
|
|
131
|
+
OMK includes only a bounded hexadecimal trace ID, never the remote error body.
|
|
132
|
+
|
|
133
|
+
- The SWE-2 adapter rejects zero, negative, and non-finite `temperature` values
|
|
134
|
+
before sending credentials. A controlled high-effort probe returned
|
|
135
|
+
`invalid_argument` at `temperature: 0`, while the default temperature completed
|
|
136
|
+
the same prompt. Omit the option to use the native default of `1`; OMK does not
|
|
137
|
+
silently replace an explicit zero.
|
|
138
|
+
- Check model, effort, context budget, and request settings with `/debug`. Repair
|
|
139
|
+
tool history only when a tool/message mismatch is actually identified.
|
|
140
|
+
- After updating or rebuilding the adapter, quit and restart the OMK process.
|
|
141
|
+
`/new` replaces conversation state; it does not reload an imported provider
|
|
142
|
+
module. This is an update-application step, not a guaranteed fix for every
|
|
143
|
+
provider error.
|
|
144
|
+
- If a minimal tool-free request still fails at the default temperature, preserve
|
|
145
|
+
the trace ID and investigate provider compatibility or availability rather than
|
|
146
|
+
repeatedly sanitizing the same transcript.
|
|
147
|
+
|
|
119
148
|
## Local overlay
|
|
120
149
|
|
|
121
150
|
When the `devin` provider is active, OMK appends `~/.omk/agent/devin.md` (capped at 24,000 characters) to the system prompt, mirroring the Grok `grok.md` overlay. Treat that file as optional host configuration for effort defaults, compaction notes, or team conventions; this page and the current provider documentation remain authoritative and it cannot override higher-priority instructions.
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
# ECRAF dimensionless normalization (R11)
|
|
2
|
+
|
|
3
|
+
Status: **pure planner, opt-in, not wired** — the same gate vocabulary as
|
|
4
|
+
[runtime-algorithms](./runtime-algorithms.md) applies. This page documents what is
|
|
5
|
+
implemented and verified, and states explicitly what remains unproven.
|
|
6
|
+
|
|
7
|
+
## Versions
|
|
8
|
+
|
|
9
|
+
`planEcrafAdmissions()` accepts an explicit `algorithmVersion`:
|
|
10
|
+
|
|
11
|
+
| Version | Density formula | When selected |
|
|
12
|
+
| --- | --- | --- |
|
|
13
|
+
| `legacy-v1` (default when omitted) | `P_i / (epsilon + Σ_r weight_r · a_ir)` | No normalization options supplied. |
|
|
14
|
+
| `normalized-v2` | `P_i / (epsilon + slotCost + Σ_r weight_r · a_ir / s_r)` | `algorithmVersion: "normalized-v2"` **or** `referenceScales` supplied without a version (compatibility with the earlier scales-only opt-in). |
|
|
15
|
+
|
|
16
|
+
Compatibility policy:
|
|
17
|
+
|
|
18
|
+
- Omitting `algorithmVersion` and every normalization option reproduces the
|
|
19
|
+
byte-for-byte legacy plan. Legacy regression and property tests are unchanged.
|
|
20
|
+
- Supplying `referenceScales` (or `slotCost`) with `legacy-v1` is rejected with
|
|
21
|
+
`RangeError`; the scales-only shape silently choosing a different algorithm was
|
|
22
|
+
the ambiguity this version policy closes.
|
|
23
|
+
- Unknown versions and unknown future options are rejected, not coerced.
|
|
24
|
+
|
|
25
|
+
## normalized-v2 semantics (spec §13.2)
|
|
26
|
+
|
|
27
|
+
- Every resource with a nonzero demand needs a positive finite reference scale
|
|
28
|
+
`s_r`. In `normalized-v2` omitted entries default to the resource's **positive
|
|
29
|
+
total capacity** — never remaining headroom — so the caller cannot implicitly
|
|
30
|
+
re-scale ranking by scheduling pressure. Unbounded resources (capacity omitted)
|
|
31
|
+
cannot fall back to a scale and require an explicit one.
|
|
32
|
+
- Zero capacity is a **feasibility gate**, not a scale: a positive demand on a
|
|
33
|
+
zero-capacity resource is deferred before scoring; a zero demand still uses the
|
|
34
|
+
resource in the legacy sense (admission fails on held usage), so the documented
|
|
35
|
+
"missing capacity = unbounded" and `capacity: 0` meanings are preserved.
|
|
36
|
+
- `slotCost` (λ_slot) is finite and **positive**; the slot term charges every
|
|
37
|
+
running candidate one execution slot even when its resource vector is empty or
|
|
38
|
+
all-zero, so an empty node cannot dominate on `epsilon` alone.
|
|
39
|
+
- Infinite capacities are rejected in v2 (`RangeError`); the legacy finite-input
|
|
40
|
+
contract is unchanged. Finite inputs can still overflow the denominator,
|
|
41
|
+
density, or reserved usage, and those derived values are rejected with
|
|
42
|
+
`RangeError` as before.
|
|
43
|
+
|
|
44
|
+
## Verified properties (unit invariance, spec §13.3)
|
|
45
|
+
|
|
46
|
+
`a'_ir = c_r·a_ir` with `s'_r = c_r·s_r` (`c_r > 0`) produces the identical plan.
|
|
47
|
+
The seeded property test (`tool-dag-ecraf-normalization.test.ts`) replays the same
|
|
48
|
+
batch under independent power-of-two memory/CPU rescaling, both bounded and
|
|
49
|
+
unbounded, with conflicts, held usage, and slot caps, and additionally asserts
|
|
50
|
+
partition completeness, slot bounds, feasibility, conflict invariants, and input
|
|
51
|
+
immutability. The spec's worked example holds: memory scale 4 GiB / CPU scale 8
|
|
52
|
+
gives A = 0.625 < B = 0.75 in both GiB and byte units.
|
|
53
|
+
|
|
54
|
+
Unit normalization is **not** a fairness guarantee or an optimality proof, and it
|
|
55
|
+
is not equivalent to DRF.
|
|
56
|
+
|
|
57
|
+
## What this is not
|
|
58
|
+
|
|
59
|
+
- **No live path calls the planner.** `runDagFrontier()` in
|
|
60
|
+
`packages/agent/src/agent-loop.ts` admits ready calls by source order and
|
|
61
|
+
settled-claim conflicts; it does not import `tool-dag-ecraf`. There is no
|
|
62
|
+
shadow recording, feature flag, or default change, and **no measured benefit**
|
|
63
|
+
is claimed anywhere.
|
|
64
|
+
- **The conflict predicate covers only this pass.** A live caller must re-check
|
|
65
|
+
unsettled claims, re-validate post-hook arguments, and refuse stale plans
|
|
66
|
+
before granting (spec §13.4). None of that wiring exists.
|
|
67
|
+
- Starvation/fairness accounting (spec §13.5) and same-budget comparisons
|
|
68
|
+
(§13.6 steps 4–6) are separate, unstarted work.
|
|
69
|
+
|
|
70
|
+
## Tests
|
|
71
|
+
|
|
72
|
+
- `packages/agent/test/tool-dag-ecraf-normalization.test.ts` — version policy,
|
|
73
|
+
scale derivation, zero-capacity gating, slot cost, overflow/validation
|
|
74
|
+
boundaries, unit-invariance property test (seed 110917, 300 runs).
|
|
75
|
+
- `packages/agent/test/tool-dag-ecraf.test.ts` — legacy behavior, input
|
|
76
|
+
rejection, admission invariants, and the §13.3 GiB/bytes ranking-flip example.
|
|
77
|
+
- `packages/agent/test/tool-dag-ecraf-arithmetic.test.ts` — finite-arithmetic
|
|
78
|
+
overflow rejection, unchanged.
|
|
79
|
+
|
|
80
|
+
Mutation/negative-control evidence: removing the demand/scale division, weakening
|
|
81
|
+
slot-cost positivity to `< 0`, removing the zero-capacity gate, or removing the
|
|
82
|
+
version whitelist each makes the assertion harness fail (run from a /tmp mutant
|
|
83
|
+
copy; the working tree was not modified during the check).
|
|
@@ -3,6 +3,56 @@
|
|
|
3
3
|
확인일: 2026-09-09. 생성기와 공급자 어댑터를 수정한 뒤 `npm run models:refresh`로
|
|
4
4
|
두 카탈로그를 재생성했다. 생성 파일을 손으로 수정하지 않았다.
|
|
5
5
|
|
|
6
|
+
## 2026-09-17 후속: OpenCode Go DeepSeek V4.1 ID 변경
|
|
7
|
+
|
|
8
|
+
[OpenCode Go 공식 endpoint 목록](https://opencode.ai/docs/go/)의 현재 ID는
|
|
9
|
+
`deepseek-v4.1-flash`다. 직접 DeepSeek의 `deepseek-flash`와 구분한다.
|
|
10
|
+
9월 17일 카탈로그 갱신은 새 ID를 반영했지만, 생성기의 V4.1 메타데이터 보정은
|
|
11
|
+
이전 ID만 인식해 `low/max`와 `max_tokens` 설정이 누락됐다.
|
|
12
|
+
|
|
13
|
+
생성기 조건에 Go의 새 ID를 추가하고 기존 Go ID의 보정도 유지했다. 회귀 테스트는
|
|
14
|
+
공급자별 실제 요청 ID를 사용하며 `off/low/high/max`, 이미지 입력, 출력 상한,
|
|
15
|
+
전송 직전 `thinking`과 `reasoning_effort` 검사를 그대로 유지한다.
|
|
16
|
+
ID만 교체한 상태에서도 5개 실패가 재현됐으며 메타데이터 수정 후 통과했다.
|
|
17
|
+
|
|
18
|
+
`node packages/ai/scripts/generate-models.ts`로 생성한 결과 중 해당 Go 모델의
|
|
19
|
+
메타데이터 변경만 포함했다. 실시간 목록에서 함께 발생한 다른 모델의 추가/삭제,
|
|
20
|
+
가격/상한 변경은 이번 수정에 포함하지 않았다. 생성 파일 값을 수작업으로 만들지 않았다.
|
|
21
|
+
공급자 추론이나 계정별 사용 가능 여부는 검증하지 않았다. 아래 9월 10일 표는 당시 기록이다.
|
|
22
|
+
|
|
23
|
+
## 2026-09-17 갱신: OpenRouter Union Alpha (stealth)
|
|
24
|
+
|
|
25
|
+
`stealth/union-alpha` 추가 요청으로 `npm run models:refresh`를 종료0으로 재생성했다.
|
|
26
|
+
생성 파일은 손대지 않았다. 전 소스가 응답했고 `--allow-partial`은 쓰지 않았다.
|
|
27
|
+
|
|
28
|
+
| 항목 | 값 |
|
|
29
|
+
| --- | --- |
|
|
30
|
+
| 요청 ID | `stealth/union-alpha` (OpenRouter) |
|
|
31
|
+
| context / maxTokens | 262,144 / 131,072 |
|
|
32
|
+
| 입력 | text, image |
|
|
33
|
+
| tool 지원 | `tools`, `tool_choice` |
|
|
34
|
+
| 가격 | prompt/completion 모두 `0` |
|
|
35
|
+
| **thinking** | **없음 — route가 `reasoning`을 선언하지 않음** |
|
|
36
|
+
|
|
37
|
+
목록과 `/api/v1/models/stealth/union-alpha/endpoints` 모두
|
|
38
|
+
`supported_parameters`가 `max_tokens, temperature, top_p, tools, tool_choice,
|
|
39
|
+
response_format`이다. `reasoning`도 `include_reasoning`도 없다. 선언이 없으므로
|
|
40
|
+
수준을 지어내지 않고 `reasoning: false`로 들어갔으며 `thinkingLevelMap`도 없다.
|
|
41
|
+
이름이 frontier 계열을 연상시킨다는 이유로 effort를 이식하지 않는다.
|
|
42
|
+
|
|
43
|
+
`created`는 2026-09-16으로 직전 갱신(09-09) 이후에 생긴 항목이다. 한 모델만
|
|
44
|
+
집어넣는 경로가 없어 카탈로그 전체가 8일치 드리프트를 함께 반영한다.
|
|
45
|
+
OpenRouter 371 → 376, 전체 추가 57 · 제거 28(고유 4)이다. 제거는
|
|
46
|
+
`deepseek.r1-v1:0`과 mistral `devstral-small-2` · `mistral-medium` · `pixtral-12b`로,
|
|
47
|
+
갱신 소스에서 더 이상 선정되지 않았다는 뜻이며 공급자의 폐기 공지나 모든 계정의
|
|
48
|
+
사용 불가를 뜻하지 않는다.
|
|
49
|
+
|
|
50
|
+
가격 `0`은 stealth 공개 기간의 목록 값이다. 무상 사용을 보장하지 않으며 stealth
|
|
51
|
+
해제 시 달라질 수 있다. 실제 청구는 계정에서 따로 확인한다.
|
|
52
|
+
|
|
53
|
+
표적 검사 `latest-model-thinking` · `catalog-thinking` · `latest-thinking-payload`
|
|
54
|
+
51개가 통과했다. 공급자 추론은 호출하지 않았다.
|
|
55
|
+
|
|
6
56
|
## 2026-09-10 재검증: DeepSeek V4.1 Flash 제공 경로
|
|
7
57
|
|
|
8
58
|
공식 문서·공개 API에서 확인한 기존 OMK 공급자 4곳을 반영했다.
|
|
@@ -114,7 +164,7 @@ OpenRouter의 `reasoning.supported_efforts`로 표시할 수준을 만들고,
|
|
|
114
164
|
|
|
115
165
|
| 모델·route | 이번 정합성 규칙 |
|
|
116
166
|
| --- | --- |
|
|
117
|
-
| GPT-6 Astra, OpenAI/Responses 및 OpenRouter | low·medium·high·xhigh·max. off/minimal 미노출 |
|
|
167
|
+
| GPT-6 Astra, OpenAI/Responses 및 OpenRouter | low·medium·high·xhigh·max. OMK `ultra`는 공식 천장 `max`의 선택기 별칭이며 와이어에 `ultra`를 보내지 않음. off/minimal 미노출 |
|
|
118
168
|
| Claude Opus 5, Anthropic Messages | adaptive thinking, low·medium·high·xhigh·max. off는 high 이하에서 가능 |
|
|
119
169
|
| Claude Opus 5, Bedrock | legacy budget 대신 adaptive 및 xhigh 사용. application profile의 표시명 매칭 보존 |
|
|
120
170
|
| Gemini 3.7/3.8 Flash, Google/Vertex | low·medium·high. minimal은 API 오류. SDK의 off 요청도 LOW로 처리하며 완전 비활성화로 주장하지 않음 |
|
package/docs/models.md
CHANGED
|
@@ -233,7 +233,7 @@ Current behavior:
|
|
|
233
233
|
|
|
234
234
|
### Thinking Level Map
|
|
235
235
|
|
|
236
|
-
Use `thinkingLevelMap` on a model to describe model-specific thinking controls. `models.json` accepts `off`, `minimal`, `low`, `medium`, `high`, `xhigh`, and `max`. Built-in metadata can also expose the `ultra` CLI tier, but the current `models.json` schema has no `ultra` key.
|
|
236
|
+
Use `thinkingLevelMap` on a model to describe model-specific thinking controls. `models.json` accepts `off`, `minimal`, `low`, `medium`, `high`, `xhigh`, and `max`. Built-in metadata can also expose the `ultra` CLI tier (GPT-6 Astra maps it to documented `max`; GPT-5.6 Sol/Terra/MoA on Codex map it to backend `xhigh`), but the current `models.json` schema has no `ultra` key.
|
|
237
237
|
|
|
238
238
|
| Value | Meaning |
|
|
239
239
|
| --- | --- |
|
package/docs/neo.md
ADDED
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
# Neo public skills and computer-use setup
|
|
2
|
+
|
|
3
|
+
Status: locally integrated and built on 2026-09-17, not a published release. Neo names the public computer-use workflow in this change; it is not a new model or an independent agent runtime.
|
|
4
|
+
|
|
5
|
+
## What changes
|
|
6
|
+
|
|
7
|
+
The normal `DefaultResourceLoader` now reads six publicly authored skills from `resources/neo/skills` inside the installed package. No private agent-home content is copied. Explicit, project and user skills retain ownership of matching names; bundled skills fill only missing names. `noSkills` disables automatic bundled loading while preserving explicitly supplied skills, and `OMK_BUNDLED_SKILLS=0` disables just the bundled set. Skill overrides continue to run after resource loading.
|
|
8
|
+
|
|
9
|
+
Both the npm package allowlist and the two binary asset paths include the same `resources` tree. Files remain inside the installation; startup does not write skills into the user's home. A missing bundle produces a loader diagnostic, and `omk neo list` returns a nonzero exit code if one of its required skill files is missing.
|
|
10
|
+
|
|
11
|
+
| Default skill | Purpose |
|
|
12
|
+
| --- | --- |
|
|
13
|
+
| `omk-computeruse` | Neo target/capability routing and observe, act, verify workflow |
|
|
14
|
+
| `omk-browser` | DOM/accessibility interaction, bounded retries and side-effect approval |
|
|
15
|
+
| `omk-site` | Existing-project site implementation, preview and responsive validation |
|
|
16
|
+
| `omk-research` | Version-aware primary documentation and evidence handling |
|
|
17
|
+
| `omk-code-review` | Minimal changes, real regression checks and live-call-path review |
|
|
18
|
+
| `omk-mcp-setup` | Inventory, configuration preview, explicit setup and health verification |
|
|
19
|
+
|
|
20
|
+
These instructions are available independently of the selected model family. Tool calling must actually work. Image interpretation additionally requires compatible model and provider input support. Text-only models should use DOM/accessibility observations. This change does not establish Astra-equivalent performance, add a native desktop driver, or turn an incapable model into a computer-use model.
|
|
21
|
+
|
|
22
|
+
## Skills and MCPs are different products
|
|
23
|
+
|
|
24
|
+
| Quantity | Meaning |
|
|
25
|
+
| --- | --- |
|
|
26
|
+
| Packaged skill | Instruction file is present in the installation |
|
|
27
|
+
| Loaded skill | Resource loader admitted the file after overrides and disable settings |
|
|
28
|
+
| Offered MCP | Public configuration preset is available |
|
|
29
|
+
| Configured MCP | User selected a server in a configuration file |
|
|
30
|
+
| Connected MCP | The running process completed its connection/handshake |
|
|
31
|
+
| Verified tool | A discovered tool succeeded against the intended target |
|
|
32
|
+
|
|
33
|
+
`omk neo list` reports packaged files and offered presets. It deliberately returns `connected: null` and `connectionStatus: "not_probed"`; it does not manufacture a nonzero MCP connection count. The normal session MCP status remains the connection authority.
|
|
34
|
+
|
|
35
|
+
## Curated MCP choices
|
|
36
|
+
|
|
37
|
+
| Preset | Distribution | Activation | Scope |
|
|
38
|
+
| --- | --- | --- | --- |
|
|
39
|
+
| `playwright` | Pinned stdio recipe for `@playwright/mcp@0.0.81` | Explicit user setup | Browser actions and accessibility snapshots; compatible browser required |
|
|
40
|
+
| `context7` | Pinned stdio recipe for `@upstash/context7-mcp@4.1.1` | Explicit user setup | External technical documentation queries |
|
|
41
|
+
| Native CUA/desktop driver | Not installed or configured by this change | Separate reviewed setup | Host permissions and host identity must be verified |
|
|
42
|
+
| Chrome personal-profile attachment | Not enabled by default | Separate explicit session approval | May expose existing authenticated state |
|
|
43
|
+
| Stagehand and Browserbase cloud | Not installed or configured by this change | Separate dependency, credentials and cost approval | Optional local library or cloud browser |
|
|
44
|
+
| Account-connected repository/services | Not configured by this change | Per-user authentication and scope approval | No publisher credential is shared |
|
|
45
|
+
|
|
46
|
+
The recipes pin the direct package version, not an entire third-party transitive dependency graph. They do not bundle executables or browser binaries. Installation and tool behavior still need to be tested on the target platform. No remote HTTP URL is advertised as directly loadable by the current stdio-only OMK loader.
|
|
47
|
+
|
|
48
|
+
## Commands
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
omk neo list
|
|
52
|
+
omk neo setup playwright context7
|
|
53
|
+
omk neo mcp-config playwright context7
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
All three are local and do not start a process, download a server or connect a browser. `setup` without `--apply` is a dry run. `mcp-config` emits disabled entries only. Existing configuration is never printed.
|
|
57
|
+
|
|
58
|
+
After approving the exact target and the server execution/network effects:
|
|
59
|
+
|
|
60
|
+
```bash
|
|
61
|
+
# Project-local, in the intended project directory:
|
|
62
|
+
omk neo setup playwright context7 --apply
|
|
63
|
+
|
|
64
|
+
# Alternative: explicitly select the global configuration target:
|
|
65
|
+
omk neo setup playwright context7 --global --apply
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
Project target: `<cwd>/.omk/mcp.json`. Global target: `~/.omk/mcp.json`, not `~/.omk/agent/mcp.json`. On the next ordinary OMK startup `npx` may download and execute the selected pinned servers. The command itself only creates configuration and reports `configured_not_connected`. A project entry can override a global entry with the same server name.
|
|
69
|
+
|
|
70
|
+
Setup is intentionally no-overwrite. It validates all presets before filesystem changes, writes a complete temporary file with mode `0600` and publishes it through an atomic no-replace hard link. An existing target, including malformed data or a symlink, is preserved. A symlinked `.omk` directory is rejected. Filesystems that do not support hard links fail closed. This is not a defense against a hostile process concurrently replacing an ancestor directory; use a trusted user-owned configuration location. Windows ACL and power-loss durability guarantees are outside this implementation.
|
|
71
|
+
|
|
72
|
+
For an existing configuration, generate disabled entries with `mcp-config`, then review a manual merge without exposing secrets. Do not replace the file to work around the refusal. This change does not alter legacy MCP config precedence or overwrite any account settings.
|
|
73
|
+
|
|
74
|
+
## Browser safety and capability
|
|
75
|
+
|
|
76
|
+
The Playwright recipe requests `--headless --isolated --sandbox` and bounded action/navigation timeouts. It does not attach to a personal browser, disable TLS checks or request unrestricted file access. The upstream Linux bundled-browser default can disable its sandbox; the explicit positive flag is therefore intentional. A sandbox launch failure must be diagnosed rather than bypassed.
|
|
77
|
+
|
|
78
|
+
Profile isolation is not OS or network isolation. Origin allowlists are not complete redirect or SSRF boundaries. Operator-managed process/network isolation is required for stronger threat models. The workflow rejects instruction authority from pages, tool output and server descriptions. Skills guide the model; they do not independently enforce every permission boundary in the runtime.
|
|
79
|
+
|
|
80
|
+
After configuration, restart the intended OMK process, inspect the actual connection status and tool roster, and perform a read-only health check. Then use an observation before each action and verify the final requested state. A timeout is not permission to repeat a non-idempotent operation. Native desktop, paid cloud and private-account actions remain separate opt-ins.
|
|
81
|
+
|
|
82
|
+
## Verification and release gates
|
|
83
|
+
|
|
84
|
+
Harness impact: `advance` for distributable capability discovery; `preserve` for existing skill precedence, disable behavior and configuration ownership. Baseline is `739bc6f3b6fe1c89bfe058aad7ec17ad252fb10e` (`0.99.0`).
|
|
85
|
+
|
|
86
|
+
Acceptance: six valid files in the npm and each platform bundle; six default skills in an empty agent/project environment; zero writes on list/dry run; byte-identical preservation of existing MCP config; no false connected count. No performance superiority claim is made.
|
|
87
|
+
|
|
88
|
+
```bash
|
|
89
|
+
node --experimental-strip-types --test scripts/test/neo-distribution.test.mjs
|
|
90
|
+
cd packages/coding-agent
|
|
91
|
+
node ../../node_modules/vitest/dist/cli.js --run test/neo-distribution.test.ts
|
|
92
|
+
cd ../..
|
|
93
|
+
npm run check
|
|
94
|
+
npm run build
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
The Node test covers actual catalog/CLI/setup functions, file preservation, symlinks and asset inclusion declarations. The Vitest test additionally exercises the real skill loader and `DefaultResourceLoader`; it requires the repository dependencies and a buildable base. Source inspection of binary copy commands is not a built-binary smoke test. Each final archive must be extracted and `omk neo list` checked before release.
|
|
98
|
+
|
|
99
|
+
Before publishing, run the targeted tests and complete repository gates, verify a clean npm install and real MCP/browser smoke tests, then follow the existing lockstep release procedure. A green test for a leaf module cannot override a failed monorepo build. Do not tag/publish this candidate until all release surfaces agree and blockers are resolved.
|
|
100
|
+
|
|
101
|
+
## Local installation verification, 2026-09-17
|
|
102
|
+
|
|
103
|
+
Integrated the candidate onto `887152bc5d`. `npm run build` and `tsgo --noEmit`
|
|
104
|
+
passed. Standalone source tests passed 13 checks; Neo, resource-loader and SDK skill
|
|
105
|
+
Vitest suites passed 39 tests. The Neo fixture isolates HOME and the agent directory
|
|
106
|
+
so personal skills cannot contaminate its six-skill assertion.
|
|
107
|
+
|
|
108
|
+
The existing local launcher resolves this checkout's `dist/cli.js`; its `neo list`
|
|
109
|
+
reported all six packaged skills. The built resource loader also selected six skills
|
|
110
|
+
in a clean environment. `npm pack --ignore-scripts --dry-run --json` included all six.
|
|
111
|
+
Run `node --test scripts/test/neo-built-cli.test.mjs` after building for subprocess
|
|
112
|
+
checks of list, write-free preview, apply and no-replace behavior.
|
|
113
|
+
|
|
114
|
+
No live MCP configuration was created or overwritten. The existing Playwright
|
|
115
|
+
connection answered a tab-list request and opened/closed a task-owned blank tab;
|
|
116
|
+
this does not verify the candidate preset's browser or sandbox flags. Existing
|
|
117
|
+
Context7 returned an invalid API-key error, so documentation retrieval is blocked.
|
|
118
|
+
No credentials were changed and no native desktop driver was installed.
|
|
119
|
+
|
|
120
|
+
Full `npm run check` stopped at pre-existing module-size failures in `agent-session.ts`
|
|
121
|
+
and `model-registry.ts`. The documentation link guard also rejects the new untracked
|
|
122
|
+
`neo.md` until it is included in Git; no files were staged to bypass that gate.
|
|
123
|
+
Release-surface and shrinkwrap checks passed. Platform archives, clean dependency
|
|
124
|
+
installation, all-suite tests and published release verification remain unrun.
|
|
125
|
+
Restart OMK to load rebuilt code; changing source does not update an active process.
|
|
126
|
+
|
|
127
|
+
## Primary source check, 2026-09-16
|
|
128
|
+
|
|
129
|
+
- Playwright MCP release `v0.0.81`: https://github.com/microsoft/playwright-mcp/releases/tag/v0.0.81
|
|
130
|
+
- Exact documented flags: https://github.com/microsoft/playwright-mcp/blob/v0.0.81/README.md
|
|
131
|
+
- Context7 release `@upstash/context7-mcp@4.1.1`: https://github.com/upstash/context7/releases/tag/%40upstash/context7-mcp%404.1.1
|
|
132
|
+
- Context7 stdio CLI: https://github.com/upstash/context7/blob/master/packages/mcp/src/index.ts
|
|
133
|
+
|
|
134
|
+
Upstream source/release inspection is not an installed-server integration test. Credentials, provider quality and target-host permissions are not verified by this document.
|
package/docs/providers.md
CHANGED
|
@@ -25,7 +25,7 @@ Run `/login` and choose a configured subscription provider to open its account p
|
|
|
25
25
|
|
|
26
26
|
Use `/logout` to clear all stored accounts for a provider. Tokens are stored in `~/.omk/agent/auth.json` and auto-refresh when expired.
|
|
27
27
|
|
|
28
|
-
When the status sidebar is pinned, its **USAGE** section lists every configured subscription provider, with the active provider first. OMK reads quota windows from fixed provider endpoints for Codex, Claude, Kimi Code, GLM/ZAI Coding Plan,
|
|
28
|
+
When the status sidebar is pinned, its **USAGE** section lists every configured subscription provider, with the active provider first. OMK reads quota windows from fixed provider endpoints for Codex, Claude, Kimi Code, GLM/ZAI Coding Plan, native xAI SuperGrok, Devin CLI, and Command Code, caches the result, and displays each percentage and reset countdown separately. Claude also passively merges the official `anthropic-ratelimit-unified-*` response headers used by Claude Code. If Anthropic's usage endpoint is rate limited and no complete recent snapshot exists, OMK mirrors Claude Code's own startup quota check with one fixed-endpoint Haiku request capped at one output token, no more than once per OAuth credential per hour. This fallback consumes a small amount of Claude plan quota.
|
|
29
29
|
|
|
30
30
|
Codex streaming passively merges `x-codex-primary-*`, `x-codex-secondary-*`, and `codex.rate_limits` signals through the non-blocking `StreamOptions.onRateLimit` observer. These signals supplement missing polling windows only when the Codex service returns them; OMK does not infer a missing 5-hour value from a 7-day value.
|
|
31
31
|
|
|
@@ -33,6 +33,8 @@ Alibaba Model Studio Token Plan is recognized as **QWEN TOKEN PLAN** and reads i
|
|
|
33
33
|
|
|
34
34
|
With a stored native `xai` OAuth credential, OMK reads `GET https://cli-chat-proxy.grok.com/v1/billing?format=credits` and shows the weekly SuperGrok pool from `config.creditUsagePercent` plus its reset from `config.currentPeriod.end`. `XAI_API_KEY` is a separate API-billing credential and does not authorize this subscription endpoint.
|
|
35
35
|
|
|
36
|
+
With a configured `commandcode` API key (`models.json` or auth storage), OMK reads `GET https://api.commandcode.ai/alpha/whoami`, then `/alpha/billing/credits`, `/alpha/billing/subscriptions`, and `/alpha/usage/summary` on that same origin. The rail shows the 5-hour and weekly credit windows plus the monthly pool (`spent / remaining+spent`) and the plan name. OMK never sends the key anywhere except `api.commandcode.ai`, never reads browser cookies, and does not scrape Studio.
|
|
37
|
+
|
|
36
38
|
### Model Studio DeepSeek V4
|
|
37
39
|
|
|
38
40
|
Model Studio uses `enable_thinking`, including for DeepSeek. OMK's Chat Completions
|
|
@@ -83,6 +85,19 @@ Your account's model access and quota still apply. For presets, effort guidance,
|
|
|
83
85
|
the 1M-token context budget, and the `devin-harness` loadout, see the
|
|
84
86
|
[Devin SWE-2 harness](devin-harness.md).
|
|
85
87
|
|
|
88
|
+
The `devin` catalog also carries every other lane `GetCliModelConfigs`
|
|
89
|
+
advertises — each wire UID is its own logical model (`claude-opus-5-high`,
|
|
90
|
+
`gpt-5-6-sol-xhigh`, `gemini-3-8-flash-medium`, `kimi-k3-max`, `glm-5-3-high`,
|
|
91
|
+
`grok-4-6-xhigh`, `deepseek-v4-pro-max`, `swe-1-7`, `inkling-max`, …). Flat
|
|
92
|
+
models pin their declared effort lane, so `/think` is fixed per model; entries
|
|
93
|
+
whose lane is a no-thinking variant report `reasoning: false`. Availability is
|
|
94
|
+
account- and plan-dependent: a lane absent from your catalog fails loudly
|
|
95
|
+
instead of being remapped. Image input is still text-only at the adapter.
|
|
96
|
+
|
|
97
|
+
```bash
|
|
98
|
+
omk --provider devin --model claude-opus-5-high
|
|
99
|
+
```
|
|
100
|
+
|
|
86
101
|
**Authentication.** OMK uses the CLI's PKCE flow at
|
|
87
102
|
`app.devin.ai/auth/cli/continue` and `api.devin.ai/auth/cli/token`. Its callback
|
|
88
103
|
binds only to `127.0.0.1:59653`, validates state, and closes after success,
|
|
@@ -99,10 +114,12 @@ Devin agent, read browser cookies, or import another application's credentials.
|
|
|
99
114
|
|
|
100
115
|
**Transport and limits.** The Node-only `devin-agent` adapter uses Connect/protobuf
|
|
101
116
|
at the fixed HTTPS origin `https://server.codeium.com`. It exchanges the session
|
|
102
|
-
token for a user JWT and reads `GetCliModelConfigs` before each turn. The
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
117
|
+
token for a user JWT and reads `GetCliModelConfigs` before each turn. The logical
|
|
118
|
+
`swe-2` model resolves its effort through the server's SWE-2 family metadata;
|
|
119
|
+
every other catalog model resolves by its own wire UID. OMK never invents a
|
|
120
|
+
`swe-2-max` ID or downgrades an unavailable route. Disabled, internal, and
|
|
121
|
+
ambiguous entries are excluded; fast-lane entries match only when their own UID
|
|
122
|
+
is selected. The family's separate 1M-context entries form
|
|
106
123
|
a second lane that is selected only when the model's local `contextWindow` is
|
|
107
124
|
1,000,000 or more (the bundled default); a smaller budget uses the standard lane.
|
|
108
125
|
|
|
@@ -139,9 +156,49 @@ and the [official manifest](https://static.devin.ai/cli/current/manifest.json)
|
|
|
139
156
|
(observed identity `3000.10.21`). Protocol fields follow the third-party
|
|
140
157
|
[oh-my-pi snapshot](https://github.com/can1357/oh-my-pi/blob/942383f768c5f2f6a620fcab57326c0f59df623a/packages/catalog/src/discovery/devin-proto.ts),
|
|
141
158
|
not a stable public inference contract; attribution is in `packages/ai/DEVIN-NOTICE`.
|
|
142
|
-
Regenerate only
|
|
159
|
+
Regenerate only the Devin models, preserving every other catalog entry, with
|
|
143
160
|
`npm --prefix packages/ai run generate-models -- --devin-only`.
|
|
144
161
|
|
|
162
|
+
### Cursor
|
|
163
|
+
|
|
164
|
+
Run `/login cursor` (PKCE + polling against `cursor.com`/`api2.cursor.sh`), then
|
|
165
|
+
select a `cursor/*` model. `cursor/default` is the Auto router; sibling slugs
|
|
166
|
+
carry their effort and lane (`-low|medium|high|xhigh|max`, `-thinking-*`,
|
|
167
|
+
`-fast`) in the id, mirroring the CLI's flat list. `CURSOR_API_KEY` accepts an
|
|
168
|
+
already-owned access token instead of the OAuth flow.
|
|
169
|
+
|
|
170
|
+
```bash
|
|
171
|
+
omk --provider cursor --model default
|
|
172
|
+
```
|
|
173
|
+
|
|
174
|
+
**Transport and limits.** The Node-only `cursor-agent` adapter runs the
|
|
175
|
+
`agent.v1.AgentService/Run` bidirectional RPC over HTTP/2 at `api2.cursor.sh`
|
|
176
|
+
with Connect-framed protobuf. Each `stream()` call is one `Run`:
|
|
177
|
+
`userMessageAction` for a fresh user turn or `resumeAction` otherwise; history
|
|
178
|
+
is replayed through `conversationState.rootPromptMessagesJson` — SHA256-keyed
|
|
179
|
+
JSON message blobs the server fetches back via the kv channel — and system
|
|
180
|
+
prompts ride `requestContext.rules` in the exec handshake. For OpenAI-family
|
|
181
|
+
sibling slugs the adapter splits the effort into a `reasoning` request
|
|
182
|
+
parameter because the Run endpoint rejects sibling `model_id`s; other ids pass
|
|
183
|
+
through whole. `tokenDelta` frames accumulate output usage; plan-specific
|
|
184
|
+
model availability is enforced server-side (free plans serve Auto only).
|
|
185
|
+
|
|
186
|
+
Text, thinking, and reported usage enter the normal OMK agent loop; images
|
|
187
|
+
attach through `selectedContext.selectedImages`. Server exec frames that ask
|
|
188
|
+
the client to run tools are answered with the protocol's `throw` failure
|
|
189
|
+
channel — OMK tools are deliberately not advertised inside the provider, so a
|
|
190
|
+
cursor model reports the capability gap instead of bypassing tool governance.
|
|
191
|
+
Wiring exec frames through governed OMK tools is a coding-agent bridge
|
|
192
|
+
feature, not provider scope. Hosted search/fetch permission queries are
|
|
193
|
+
approved; interactive ones are rejected.
|
|
194
|
+
|
|
195
|
+
The bundled catalog was captured from `GetUsableModels` (2026-09-18; 223
|
|
196
|
+
entries). Regenerate only the Cursor section with
|
|
197
|
+
`npm --prefix packages/ai run generate-models -- --cursor-only`. Protocol
|
|
198
|
+
fields follow the same third-party snapshot attribution as the Devin adapter.
|
|
199
|
+
Live login, plan gating, and quota behavior remain account-dependent; the
|
|
200
|
+
adapter is covered by loopback HTTP/2 protocol tests.
|
|
201
|
+
|
|
145
202
|
### OpenAI Codex
|
|
146
203
|
|
|
147
204
|
- Requires ChatGPT Plus or Pro subscription
|
|
@@ -153,6 +210,14 @@ Regenerate only this logical model, preserving every other catalog entry, with
|
|
|
153
210
|
omk --provider openai-codex --model gpt-5.6-moa --thinking ultra
|
|
154
211
|
```
|
|
155
212
|
|
|
213
|
+
### GPT-6 Astra
|
|
214
|
+
|
|
215
|
+
OpenAI-shaped Astra routes (`openai`, `azure-openai-responses`, `opencode`, `github-copilot`, OpenRouter) expose OMK `ultra` in the selector and send `reasoning.effort: "max"`. Official Astra effort values remain `low`, `medium`, `high`, `xhigh`, and `max`; there is no native `ultra` wire value. ChatGPT/Codex account catalogs may still reject `gpt-6-astra`.
|
|
216
|
+
|
|
217
|
+
```bash
|
|
218
|
+
omk --provider openai --model gpt-6-astra --thinking ultra
|
|
219
|
+
```
|
|
220
|
+
|
|
156
221
|
### Claude Pro/Max
|
|
157
222
|
|
|
158
223
|
Anthropic subscription auth is active for Claude Pro/Max accounts. Third-party harness usage draws from [extra usage](https://claude.ai/settings/usage) and is billed per token, not against Claude plan limits.
|
|
@@ -187,6 +252,7 @@ omk
|
|
|
187
252
|
| OpenAI | `OPENAI_API_KEY` | `openai` |
|
|
188
253
|
| DeepSeek | `DEEPSEEK_API_KEY` | `deepseek` |
|
|
189
254
|
| Devin CLI session token | `DEVIN_API_KEY` | `devin` |
|
|
255
|
+
| Cursor subscription | `CURSOR_API_KEY` | `cursor` |
|
|
190
256
|
| NVIDIA NIM | `NVIDIA_API_KEY` | `nvidia` |
|
|
191
257
|
| Google Gemini | `GEMINI_API_KEY` | `google` |
|
|
192
258
|
| Mistral | `MISTRAL_API_KEY` | `mistral` |
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
# Run usage ledger
|
|
2
|
+
|
|
3
|
+
Status: implemented as an in-memory accounting module and explicit operation adapter.
|
|
4
|
+
Not automatically connected to AgentSession, provider HTTP retries, MCP, or child CLI.
|
|
5
|
+
Existing `RunBudget` request/concurrency/deadline behavior is unchanged.
|
|
6
|
+
|
|
7
|
+
Source: `src/core/run-usage-ledger.ts` and `src/core/run-usage-operation.ts`.
|
|
8
|
+
|
|
9
|
+
## Contract
|
|
10
|
+
|
|
11
|
+
- `reserve(attemptId, requestId, reservation)` admits an operation only when reported
|
|
12
|
+
consumption + held reservation + proposed reservation fits each supplied cap.
|
|
13
|
+
Retry attempts may share a logical request ID. IDs cannot be reused for attempts.
|
|
14
|
+
- `recordTransport(transportId, attemptId)` records a transport start observed by the
|
|
15
|
+
caller. An application dispatch does not automatically count as an HTTP attempt.
|
|
16
|
+
- `recordUsage(eventId, attemptId, usage)` accepts one cumulative report per attempt.
|
|
17
|
+
Identical redelivery is idempotent; changed payloads and second reports are rejected.
|
|
18
|
+
Reports arriving after settlement or closure still bind to the original attempt.
|
|
19
|
+
- `settle(attemptId)` requires observed operation settlement, not a cancellation request.
|
|
20
|
+
Unknown usage retains its reservation after settlement and blocks new capped admission.
|
|
21
|
+
- `close()` seals admission and transport starts but retains ownership and reservations.
|
|
22
|
+
- Snapshots distinguish the accounted subtotal from `total: null` when usage is missing.
|
|
23
|
+
`estimatedUsd` is an estimate in USD, never invoice-confirmed spending. Token figures
|
|
24
|
+
are caller-normalized reports; the ledger cannot detect provider zero-filled missing usage.
|
|
25
|
+
- Counts and amounts reject negative, non-finite and unsafe arithmetic. Maps are bounded
|
|
26
|
+
by `maxEntries` (default 10,000); there is no automatic eviction of deduplication state.
|
|
27
|
+
|
|
28
|
+
`runUsageOperation` reserves before invoking the callback and settles in `finally`.
|
|
29
|
+
The adapter captures the admitted attempt ID before invoking the callback, so later
|
|
30
|
+
input mutation cannot redirect settlement. Usage reporters must still supply the correct
|
|
31
|
+
attempt ID themselves. The callback promise must represent the operation lifetime. Do not pass a timeout race
|
|
32
|
+
that returns before the underlying work settles. No AbortSignal is treated as proof
|
|
33
|
+
of termination. Failed operations still count as requests/attempts.
|
|
34
|
+
|
|
35
|
+
## Verification and limits
|
|
36
|
+
|
|
37
|
+
From `packages/coding-agent`:
|
|
38
|
+
|
|
39
|
+
```sh
|
|
40
|
+
node ../../node_modules/vitest/dist/cli.js run test/run-usage-ledger.test.ts test/run-usage-ownership.test.ts test/run-budget.test.ts test/run-budget-scope.test.ts
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
The local fixture records main work with two transport retries, continuation, summary,
|
|
44
|
+
and child-labelled work: four logical requests, six transports, eight input tokens.
|
|
45
|
+
These are explicit local adapter events, not real provider, process, or billing evidence.
|
|
46
|
+
Tests cover 6 consumed + 3 reserved + 2 proposed against cap 10, delayed settlement,
|
|
47
|
+
late usage, duplicate/conflicting events, missing usage, bounded input/state, and the
|
|
48
|
+
ownership boundary: denied admission never runs the operation, cancellation retains
|
|
49
|
+
the reservation until observed settlement, and settlement attribution survives caller
|
|
50
|
+
input mutation.
|
|
51
|
+
|
|
52
|
+
No durable replay/restart support, price-table revision binding, invoice corrections,
|
|
53
|
+
work/verification/cleanup partitioning, or automatic whole-run transport instrumentation
|
|
54
|
+
is provided. Actual usage can exceed its reservation: it is recorded honestly and blocks
|
|
55
|
+
later capped admissions rather than being clipped. This is not a hard financial limit.
|
|
56
|
+
R08 end-to-end product integration remains incomplete.
|