open-multi-agent-kit 0.98.5 → 0.99.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (163) hide show
  1. package/CHANGELOG.md +24 -0
  2. package/README.md +3 -2
  3. package/dist/cli/help.d.ts.map +1 -1
  4. package/dist/cli/help.js +1 -0
  5. package/dist/cli/help.js.map +1 -1
  6. package/dist/commands/verified-run-cli.d.ts.map +1 -1
  7. package/dist/commands/verified-run-cli.js +2 -2
  8. package/dist/commands/verified-run-cli.js.map +1 -1
  9. package/dist/core/agent-session.d.ts.map +1 -1
  10. package/dist/core/agent-session.js +19 -2
  11. package/dist/core/agent-session.js.map +1 -1
  12. package/dist/core/devin-harness-dispatch.d.ts +12 -0
  13. package/dist/core/devin-harness-dispatch.d.ts.map +1 -0
  14. package/dist/core/devin-harness-dispatch.js +12 -0
  15. package/dist/core/devin-harness-dispatch.js.map +1 -0
  16. package/dist/core/devin-harness.d.ts +53 -0
  17. package/dist/core/devin-harness.d.ts.map +1 -0
  18. package/dist/core/devin-harness.js +112 -0
  19. package/dist/core/devin-harness.js.map +1 -0
  20. package/dist/core/domain-dispatch.d.ts +4 -1
  21. package/dist/core/domain-dispatch.d.ts.map +1 -1
  22. package/dist/core/domain-dispatch.js +5 -0
  23. package/dist/core/domain-dispatch.js.map +1 -1
  24. package/dist/core/domain-loadouts-provider-harness.d.ts +12 -0
  25. package/dist/core/domain-loadouts-provider-harness.d.ts.map +1 -0
  26. package/dist/core/domain-loadouts-provider-harness.js +122 -0
  27. package/dist/core/domain-loadouts-provider-harness.js.map +1 -0
  28. package/dist/core/domain-loadouts.d.ts +2 -33
  29. package/dist/core/domain-loadouts.d.ts.map +1 -1
  30. package/dist/core/domain-loadouts.js +3 -56
  31. package/dist/core/domain-loadouts.js.map +1 -1
  32. package/dist/core/domain-profile.d.ts +40 -0
  33. package/dist/core/domain-profile.d.ts.map +1 -0
  34. package/dist/core/domain-profile.js +7 -0
  35. package/dist/core/domain-profile.js.map +1 -0
  36. package/dist/core/grok-harness-dispatch.d.ts +6 -20
  37. package/dist/core/grok-harness-dispatch.d.ts.map +1 -1
  38. package/dist/core/grok-harness-dispatch.js +6 -55
  39. package/dist/core/grok-harness-dispatch.js.map +1 -1
  40. package/dist/core/grok-harness.d.ts +7 -9
  41. package/dist/core/grok-harness.d.ts.map +1 -1
  42. package/dist/core/grok-harness.js +9 -23
  43. package/dist/core/grok-harness.js.map +1 -1
  44. package/dist/core/harness-skills.d.ts +20 -0
  45. package/dist/core/harness-skills.d.ts.map +1 -0
  46. package/dist/core/harness-skills.js +36 -0
  47. package/dist/core/harness-skills.js.map +1 -0
  48. package/dist/core/loadout-runtime-state.d.ts +25 -0
  49. package/dist/core/loadout-runtime-state.d.ts.map +1 -0
  50. package/dist/core/loadout-runtime-state.js +39 -0
  51. package/dist/core/loadout-runtime-state.js.map +1 -0
  52. package/dist/core/loadout-runtime.d.ts +2 -12
  53. package/dist/core/loadout-runtime.d.ts.map +1 -1
  54. package/dist/core/loadout-runtime.js +6 -13
  55. package/dist/core/loadout-runtime.js.map +1 -1
  56. package/dist/core/model-resolver.d.ts +1 -41
  57. package/dist/core/model-resolver.d.ts.map +1 -1
  58. package/dist/core/model-resolver.js +2 -49
  59. package/dist/core/model-resolver.js.map +1 -1
  60. package/dist/core/provider-default-models.d.ts +42 -0
  61. package/dist/core/provider-default-models.d.ts.map +1 -0
  62. package/dist/core/provider-default-models.js +50 -0
  63. package/dist/core/provider-default-models.js.map +1 -0
  64. package/dist/core/provider-display-names.d.ts.map +1 -1
  65. package/dist/core/provider-display-names.js +1 -0
  66. package/dist/core/provider-display-names.js.map +1 -1
  67. package/dist/core/provider-harness-dispatch.d.ts +63 -0
  68. package/dist/core/provider-harness-dispatch.d.ts.map +1 -0
  69. package/dist/core/provider-harness-dispatch.js +60 -0
  70. package/dist/core/provider-harness-dispatch.js.map +1 -0
  71. package/dist/core/provider-usage-devin.d.ts +16 -0
  72. package/dist/core/provider-usage-devin.d.ts.map +1 -0
  73. package/dist/core/provider-usage-devin.js +61 -0
  74. package/dist/core/provider-usage-devin.js.map +1 -0
  75. package/dist/core/provider-usage-text.d.ts +5 -0
  76. package/dist/core/provider-usage-text.d.ts.map +1 -0
  77. package/dist/core/provider-usage-text.js +15 -0
  78. package/dist/core/provider-usage-text.js.map +1 -0
  79. package/dist/core/provider-usage-types.d.ts +2 -1
  80. package/dist/core/provider-usage-types.d.ts.map +1 -1
  81. package/dist/core/provider-usage-types.js.map +1 -1
  82. package/dist/core/provider-usage.d.ts +1 -2
  83. package/dist/core/provider-usage.d.ts.map +1 -1
  84. package/dist/core/provider-usage.js +12 -7
  85. package/dist/core/provider-usage.js.map +1 -1
  86. package/dist/core/run-execution-api.d.ts +1 -1
  87. package/dist/core/run-execution-api.d.ts.map +1 -1
  88. package/dist/core/run-execution-api.js.map +1 -1
  89. package/dist/core/sdk.d.ts.map +1 -1
  90. package/dist/core/sdk.js +23 -18
  91. package/dist/core/sdk.js.map +1 -1
  92. package/dist/core/verified-run/dag-phase.d.ts +1 -1
  93. package/dist/core/verified-run/dag-phase.d.ts.map +1 -1
  94. package/dist/core/verified-run/dag-phase.js +46 -8
  95. package/dist/core/verified-run/dag-phase.js.map +1 -1
  96. package/dist/core/verified-run/dag-projection.d.ts.map +1 -1
  97. package/dist/core/verified-run/dag-projection.js +25 -12
  98. package/dist/core/verified-run/dag-projection.js.map +1 -1
  99. package/dist/core/verified-run/dag-types.d.ts +11 -0
  100. package/dist/core/verified-run/dag-types.d.ts.map +1 -1
  101. package/dist/core/verified-run/dag-types.js.map +1 -1
  102. package/dist/core/verified-run/event-parser.d.ts.map +1 -1
  103. package/dist/core/verified-run/event-parser.js +1 -0
  104. package/dist/core/verified-run/event-parser.js.map +1 -1
  105. package/dist/core/verified-run/owned-execution.d.ts +1 -0
  106. package/dist/core/verified-run/owned-execution.d.ts.map +1 -1
  107. package/dist/core/verified-run/owned-execution.js +7 -1
  108. package/dist/core/verified-run/owned-execution.js.map +1 -1
  109. package/dist/core/verified-run/process-projection.d.ts +13 -0
  110. package/dist/core/verified-run/process-projection.d.ts.map +1 -0
  111. package/dist/core/verified-run/process-projection.js +82 -0
  112. package/dist/core/verified-run/process-projection.js.map +1 -0
  113. package/dist/core/verified-run/projection.d.ts.map +1 -1
  114. package/dist/core/verified-run/projection.js +11 -55
  115. package/dist/core/verified-run/projection.js.map +1 -1
  116. package/dist/core/verified-run/run-types.d.ts +1 -0
  117. package/dist/core/verified-run/run-types.d.ts.map +1 -1
  118. package/dist/core/verified-run/run-types.js.map +1 -1
  119. package/dist/core/verified-run/task-execution.d.ts +9 -0
  120. package/dist/core/verified-run/task-execution.d.ts.map +1 -0
  121. package/dist/core/verified-run/task-execution.js +29 -0
  122. package/dist/core/verified-run/task-execution.js.map +1 -0
  123. package/dist/core/verified-run/writer-projection.d.ts.map +1 -1
  124. package/dist/core/verified-run/writer-projection.js +6 -5
  125. package/dist/core/verified-run/writer-projection.js.map +1 -1
  126. package/dist/modes/interactive/interactive-mode.d.ts.map +1 -1
  127. package/dist/modes/interactive/interactive-mode.js +2 -5
  128. package/dist/modes/interactive/interactive-mode.js.map +1 -1
  129. package/dist/modes/interactive/resource-description.d.ts +3 -0
  130. package/dist/modes/interactive/resource-description.d.ts.map +1 -0
  131. package/dist/modes/interactive/resource-description.js +15 -0
  132. package/dist/modes/interactive/resource-description.js.map +1 -0
  133. package/docs/devin-harness.md +121 -0
  134. package/docs/docs.json +4 -0
  135. package/docs/environment-variables.md +2 -1
  136. package/docs/index.md +1 -0
  137. package/docs/loadout-domains/README.md +2 -1
  138. package/docs/loadout-domains/devin-harness.md +72 -0
  139. package/docs/metrics.md +57 -16
  140. package/docs/providers.md +76 -0
  141. package/docs/release-audit-0.98.5.md +17 -0
  142. package/docs/release-audit-0.99.0.md +68 -0
  143. package/docs/run-protocol.md +4 -3
  144. package/docs/runtime-algorithms.md +4 -1
  145. package/docs/sdk.md +9 -5
  146. package/docs/settings.md +1 -1
  147. package/docs/startup-resource-labels-testing.md +47 -0
  148. package/docs/tb21-audit.md +18 -4
  149. package/docs/usage.md +5 -0
  150. package/docs/verified-run-remaining-design.md +881 -0
  151. package/docs/verified-run-testing.md +91 -0
  152. package/docs/verified-run.md +30 -9
  153. package/examples/extensions/custom-provider-anthropic/package-lock.json +2 -2
  154. package/examples/extensions/custom-provider-anthropic/package.json +1 -1
  155. package/examples/extensions/custom-provider-gitlab-duo/package.json +1 -1
  156. package/examples/extensions/gondolin/package-lock.json +2 -2
  157. package/examples/extensions/gondolin/package.json +1 -1
  158. package/examples/extensions/sandbox/package-lock.json +2 -2
  159. package/examples/extensions/sandbox/package.json +1 -1
  160. package/examples/extensions/with-deps/package-lock.json +2 -2
  161. package/examples/extensions/with-deps/package.json +1 -1
  162. package/npm-shrinkwrap.json +18 -18
  163. package/package.json +6 -6
@@ -0,0 +1,3 @@
1
+ /** Keep decorative imported harness markers out of OMK display; source metadata stays unchanged. */
2
+ export declare function formatResourceDescription(description: string | undefined, sourceTag?: string): string | undefined;
3
+ //# sourceMappingURL=resource-description.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"resource-description.d.ts","sourceRoot":"","sources":["../../../src/modes/interactive/resource-description.ts"],"names":[],"mappings":"AAAA,oGAAoG;AACpG,wBAAgB,yBAAyB,CAAC,WAAW,EAAE,MAAM,GAAG,SAAS,EAAE,SAAS,CAAC,EAAE,MAAM,GAAG,MAAM,GAAG,SAAS,CAWjH","sourcesContent":["/** Keep decorative imported harness markers out of OMK display; source metadata stays unchanged. */\nexport function formatResourceDescription(description: string | undefined, sourceTag?: string): string | undefined {\n\tlet text = description;\n\tif (description) {\n\t\tconst marker = /^\\[(OMX|OMO)\\]\\s*/.exec(description);\n\t\tif (marker) {\n\t\t\tconst body = description.slice(marker[0].length);\n\t\t\ttext = body || \"OMK resource\";\n\t\t}\n\t}\n\tif (!sourceTag) return text;\n\treturn text ? `[${sourceTag}] ${text}` : `[${sourceTag}]`;\n}\n"]}
@@ -0,0 +1,15 @@
1
+ /** Keep decorative imported harness markers out of OMK display; source metadata stays unchanged. */
2
+ export function formatResourceDescription(description, sourceTag) {
3
+ let text = description;
4
+ if (description) {
5
+ const marker = /^\[(OMX|OMO)\]\s*/.exec(description);
6
+ if (marker) {
7
+ const body = description.slice(marker[0].length);
8
+ text = body || "OMK resource";
9
+ }
10
+ }
11
+ if (!sourceTag)
12
+ return text;
13
+ return text ? `[${sourceTag}] ${text}` : `[${sourceTag}]`;
14
+ }
15
+ //# sourceMappingURL=resource-description.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"resource-description.js","sourceRoot":"","sources":["../../../src/modes/interactive/resource-description.ts"],"names":[],"mappings":"AAAA,oGAAoG;AACpG,MAAM,UAAU,yBAAyB,CAAC,WAA+B,EAAE,SAAkB,EAAsB;IAClH,IAAI,IAAI,GAAG,WAAW,CAAC;IACvB,IAAI,WAAW,EAAE,CAAC;QACjB,MAAM,MAAM,GAAG,mBAAmB,CAAC,IAAI,CAAC,WAAW,CAAC,CAAC;QACrD,IAAI,MAAM,EAAE,CAAC;YACZ,MAAM,IAAI,GAAG,WAAW,CAAC,KAAK,CAAC,MAAM,CAAC,CAAC,CAAC,CAAC,MAAM,CAAC,CAAC;YACjD,IAAI,GAAG,IAAI,IAAI,cAAc,CAAC;QAC/B,CAAC;IACF,CAAC;IACD,IAAI,CAAC,SAAS;QAAE,OAAO,IAAI,CAAC;IAC5B,OAAO,IAAI,CAAC,CAAC,CAAC,IAAI,SAAS,KAAK,IAAI,EAAE,CAAC,CAAC,CAAC,IAAI,SAAS,GAAG,CAAC;AAAA,CAC1D","sourcesContent":["/** Keep decorative imported harness markers out of OMK display; source metadata stays unchanged. */\nexport function formatResourceDescription(description: string | undefined, sourceTag?: string): string | undefined {\n\tlet text = description;\n\tif (description) {\n\t\tconst marker = /^\\[(OMX|OMO)\\]\\s*/.exec(description);\n\t\tif (marker) {\n\t\t\tconst body = description.slice(marker[0].length);\n\t\t\ttext = body || \"OMK resource\";\n\t\t}\n\t}\n\tif (!sourceTag) return text;\n\treturn text ? `[${sourceTag}] ${text}` : `[${sourceTag}]`;\n}\n"]}
@@ -0,0 +1,121 @@
1
+ # Devin SWE-2 harness
2
+
3
+ This page is the canonical operator guide for the built-in `devin` provider and its only logical model, `swe-2`. Use `/login devin` for the Devin CLI subscription PKCE flow or `DEVIN_API_KEY` for an already-owned CLI session token. A user-local `~/.omk/agent/devin.md` may add operator notes, but it is not the portable product contract.
4
+
5
+ Authentication, transport, and verification limits are owned by [Providers](providers.md#devin-cli); this page covers how OMK drives SWE-2 as a harness.
6
+
7
+ ## Presets
8
+
9
+ Project presets live in `.omk/presets.json` (or `~/.omk/agent/presets.json`) and are consumed by the preset extension from `packages/coding-agent/examples/extensions/preset.ts`. The shared SWE-2 presets intentionally omit the `tools` key so role/domain lane grants keep control of the active tools.
10
+
11
+ | Preset | Provider | Model | Thinking | Use |
12
+ | --- | --- | --- | --- | --- |
13
+ | `swe2-verified` | `devin` | `swe-2` | `high` | Default SWE-2 coding baseline: multi-file edits with tests. |
14
+ | `swe2-max` | `devin` | `swe-2` | `max` | Long-horizon, uncertain, or repository-wide work that should use the 1M budget. |
15
+ | `swe2-fast-edit` | `devin` | `swe-2` | `medium` | Small, well-specified edits where medium acts sooner and costs less. |
16
+
17
+ ```json
18
+ {
19
+ "swe2-verified": { "provider": "devin", "model": "swe-2", "thinkingLevel": "high" },
20
+ "swe2-max": { "provider": "devin", "model": "swe-2", "thinkingLevel": "max" },
21
+ "swe2-fast-edit": { "provider": "devin", "model": "swe-2", "thinkingLevel": "medium" }
22
+ }
23
+ ```
24
+
25
+ For a new session without presets:
26
+
27
+ ```bash
28
+ omk --provider devin --model swe-2 --thinking max
29
+ ```
30
+
31
+ ## Effort tiers
32
+
33
+ SWE-2 exposes exactly three server-declared efforts. OMK maps its thinking tiers as follows and rejects anything else before sending credentials; `max` is a reasoning level, not the Devin Max subscription tier.
34
+
35
+ | OMK tier | `swe-2` |
36
+ | --- | --- |
37
+ | `off`, `minimal`, `low` | unavailable |
38
+ | `medium` | `medium` |
39
+ | `high` | `high` |
40
+ | `xhigh`, `ultra` | unavailable |
41
+ | `max` | `max` |
42
+
43
+ Guidance from the [SWE-2 announcement](https://cognition.com/blog/swe-2): `medium` makes its first real edit sooner and is the cost-efficient choice for simple and intermediate tasks; `high` and `max` plan more, explore more of the codebase, and verify more on complex tasks. `recommendedDevinEffortForIntent()` encodes the same split (`code` → `medium`, `test` → `high`, `debug`/`repo` → `max`).
44
+
45
+ ## Context budget: 1,000,000 tokens
46
+
47
+ `devin/swe-2` ships with `contextWindow: 1000000` and `maxTokens: 16384`. These are local budgets that drive OMK's context budgeting and compaction, **not published SWE-2 limits**; Cognition has not published a context window for SWE-2. The budget also selects the catalog lane:
48
+
49
+ 1. Before each turn OMK reads `GetCliModelConfigs`. SWE-2 family entries may carry a `1M Context` axis (order `1`) beside the effort axis. A local budget of 1,000,000 or more asks for that 1M-context lane; below it, the standard lane is used and 1M entries are ignored.
50
+ 2. A catalog with no 1M-context lane keeps the standard lane for the selected effort.
51
+ 3. If the chosen lane declares a context window smaller than the local budget, the request fails with `... declares a N-token context window; lower the models.json contextWindow before retrying`. OMK never shrinks the budget silently, never invents a wire UID, and never downgrades to another effort.
52
+ 4. Fast-lane (`Fast Mode`) entries are always excluded. Output is capped against the authenticated catalog's declared maximum.
53
+
54
+ To lower the budget (for example if your account only serves the standard lane at 262,144 tokens), override the built-in model in `~/.omk/agent/models.json`:
55
+
56
+ ```json
57
+ {
58
+ "providers": {
59
+ "devin": {
60
+ "modelOverrides": {
61
+ "swe-2": { "contextWindow": 262144 }
62
+ }
63
+ }
64
+ }
65
+ }
66
+ ```
67
+
68
+ Recommended compaction settings for 1M sessions keep the defaults but raise the recent-token window so summaries do not discard the working set:
69
+
70
+ ```json
71
+ {
72
+ "compaction": { "enabled": true, "reserveTokens": 16384, "keepRecentTokens": 60000, "maxUsageRatio": 0.85 }
73
+ }
74
+ ```
75
+
76
+ Context discipline still applies: the budget is room for the repository, not an invitation to dump it. Prefer targeted reads and searches, keep tool output bounded, and let `precompact-checkpoint` snapshot state before compaction rather than restarting sessions. Quota is per account; a larger window consumes more of it per turn.
77
+
78
+ With a Devin credential configured, the status rail's USAGE section shows the account's daily and weekly quota meters from `GetUserStatus` (or the plan name and credit balances when the plan reports no quota windows), matching the CLI's `/usage` surface.
79
+
80
+ ## Domain routing
81
+
82
+ Selecting the `devin` provider auto-applies the `devin-harness` loadout by default; this does not require `OMK_DOMAIN_ROUTING=1`. Set `OMK_DEVIN_HARNESS=0` to disable that provider-specific dispatch. The flag is independent from `OMK_GROK_HARNESS`.
83
+
84
+ Loadout policies reject extension tools that shadow builtins fail-closed. A host that intentionally replaces the builtin `bash` (for example a Landstrip shell provider) should keep `OMK_DEVIN_HARNESS=0` in its launcher until the replacement is removed or bridged; the `devin.md` overlay and the model budget still apply in that case.
85
+
86
+ General prompt-based domain routing is separate and opt-in through `OMK_DOMAIN_ROUTING=1`. It selects one of the profiles under [`loadout-domains/`](loadout-domains/README.md) and composes it with the active role loadout. SWE-2 presets only set provider, model, thinking level, and instruction pointers.
87
+
88
+ ## Model selection
89
+
90
+ The `devin` catalog contains only the logical `swe-2` model; the server's SWE-2 family metadata supplies each effort's wire UID at request time. Use `/model` or `omk --list-models devin` for the current list. Image input is unsupported; provide text.
91
+
92
+ ## Skill and MCP matrix summary
93
+
94
+ Use the normal OMK lane grant model: grant the smallest skill and MCP surface that matches the task.
95
+
96
+ For each non-queued `devin` request started through `AgentSession.prompt()`, OMK calls `selectDevinHarnessSkills()` against the live discovered skill descriptions after ordinary prompt-template expansion. It merges up to three matches with explicit/settings selections and rebuilds that turn's `<active_skills source="devin-harness">` marker. The scorer is the same `selectSkills()` used by the Grok harness: weak 0.35 / strong 0.7 thresholds, deterministic input-order ties, first-name-wins deduplication, explicit-only skills never auto-selected, and `headroom` only for lexical pressure cues or the session's measured context-pressure bucket. A task with no signals yields an empty automatic grant rather than the full allowlist. Queued `steer`, `followUp`, or `prompt(..., { streamingBehavior })` messages retain the active run's system prompt.
97
+
98
+ | Task class | Skills | MCP |
99
+ | --- | --- | --- |
100
+ | Multi-package or repo-context work | `packages`; add `headroom` only under context pressure | none by default |
101
+ | Repo graph or broad comprehension | `understand-anything`; optionally `packages` | `understand-anything` |
102
+ | TypeScript/Rust/Python/Go edits | `programming`; add `lsp` or `ast-grep` only for symbol/structural work | none by default |
103
+ | New behavior or bug fix with regression test | `tdd-workflow`, `programming` | none by default |
104
+ | Runtime failures or broken behavior | `debugging` | task-specific only |
105
+ | Library API lookup | task skill as needed | `context7` |
106
+ | Current public URL or docs lookup | task skill as needed | `fetch` |
107
+ | UI/TUI verification | task skill as needed | `playwright` only when browser/UI evidence is required |
108
+
109
+ Relevant evidence hooks for SWE-2 lanes are `pre-shell-guard`, `protect-secrets`, `typecheck-after-edit`, `stop-verify`, `session-context`, and `precompact-checkpoint`. Hook output is incremental evidence; code changes still need the project's required final verification command before claiming type/lint cleanliness.
110
+
111
+ ## Suggested TUI flow
112
+
113
+ 1. Run `/login devin` once; the account picker stores the CLI session token in `auth.json`.
114
+ 2. Select `/preset swe2-verified` for normal coding work, `/preset swe2-max` for long-horizon or repository-wide tasks, and `/preset swe2-fast-edit` for small edits.
115
+ 3. Use `/think medium`, `/think high`, or `/think max` to change effort mid-session; other levels are rejected.
116
+ 4. If a turn fails with an "unavailable or ambiguous" route or a smaller declared context window, treat it as a configuration signal: check `devin models list`, or lower `contextWindow` as shown above. Do not retry with a guessed wire UID.
117
+ 5. Keep credentials out of preset JSON, prompts, and logs: the session token, the exchanged user JWT, and `auth.json` contents are secrets under `protect-secrets`.
118
+
119
+ ## Local overlay
120
+
121
+ When the `devin` provider is active, OMK appends `~/.omk/agent/devin.md` (capped at 24,000 characters) to the system prompt, mirroring the Grok `grok.md` overlay. Treat that file as optional host configuration for effort defaults, compaction notes, or team conventions; this page and the current provider documentation remain authoritative and it cannot override higher-priority instructions.
package/docs/docs.json CHANGED
@@ -27,6 +27,10 @@
27
27
  "title": "Native xAI Grok",
28
28
  "path": "grok-harness.md"
29
29
  },
30
+ {
31
+ "title": "Devin SWE-2",
32
+ "path": "devin-harness.md"
33
+ },
30
34
  {
31
35
  "title": "Containerization",
32
36
  "path": "containerization.md"
@@ -100,7 +100,8 @@ These variables are read by OMK itself. The four built-in harness flags below ar
100
100
  | `OMK_YOLO`, `OMK_COMMAND_SAFETY`, `OMK_DISABLE_COMMAND_SAFETY` | Disable the command-safety gate entirely (YOLO mode). `OMK_YOLO` and `OMK_DISABLE_COMMAND_SAFETY` accept `1`, `true`, `yes`, or `on`; `OMK_COMMAND_SAFETY` accepts `0`, `false`, `off`, `disable`, or `disabled`. Every verdict, including block-tier and privilege commands, is skipped in interactive and headless runs. Use only when a verified outer sandbox owns the boundary |
101
101
  | `OMK_COMMAND_SAFETY_ASSUME_YES` | `1` or `true` auto-accepts non-privilege confirm-tier commands in interactive **and headless** runs. Privilege confirmation and block-tier commands remain denied. Use only under a trusted outer sandbox when headless auto-accept is intended |
102
102
  | `OMK_GROK_HARNESS` | Default-on native `xai` provider dispatch to the `grok-harness` loadout. `0`, `false`, `off`, or `no` disables it |
103
- | `OMK_DOMAIN_ROUTING` | Set to `1` to enable general prompt-based domain routing. Native xAI harness dispatch does not require it |
103
+ | `OMK_DEVIN_HARNESS` | Default-on `devin` provider dispatch to the `devin-harness` loadout. `0`, `false`, `off`, or `no` disables it; independent from `OMK_GROK_HARNESS` |
104
+ | `OMK_DOMAIN_ROUTING` | Set to `1` to enable general prompt-based domain routing. Native xAI and Devin harness dispatch do not require it |
104
105
  | `VISUAL`, `EDITOR` | External editor fallback when `externalEditor` is unset |
105
106
  | `HTTP_PROXY`, `HTTPS_PROXY` | Proxy outbound HTTP requests |
106
107
  | `OMK_RESOURCE_GOVERNOR` | Resource-governor mode: `off`, `observe` (default), `adaptive`, or `strict`. Feeds `/resource [probe\|policy]` and `omk doctor resources [--json]`; see the resource governor section in [Settings](settings.md) |
package/docs/index.md CHANGED
@@ -39,6 +39,7 @@ For the full first-run flow, see [Quickstart](quickstart.md).
39
39
  - [Providers](providers.md) - subscription and API-key setup for built-in providers.
40
40
  - [Provider Resilience](provider-resilience.md) - retry, failover, quota, and safety-stop recovery.
41
41
  - [Native xAI Grok](grok-harness.md) - authentication, weekly SuperGrok usage, presets, and thinking tiers.
42
+ - [Devin SWE-2](devin-harness.md) - presets, effort tiers, the 1M-token context budget and lane rule, and the `devin-harness` loadout.
42
43
  - [Containerization](containerization.md) - sandbox omk with OpenShell, Gondolin, or Docker.
43
44
  - [Settings](settings.md) - global and project settings.
44
45
  - [Environment Variables](environment-variables.md) - process configuration, harness opt-outs, and bash-tool session environment.
@@ -21,7 +21,7 @@ OMK routes incoming tasks to a **domain capability profile** ("inherited documen
21
21
 
22
22
  Thresholds: `STRONG_THRESHOLD = 8`, `WEAK_THRESHOLD = 4`, `AMBIGUITY_MARGIN = 2`.
23
23
 
24
- ## Domains (13 + 1 fallback)
24
+ ## Domains (14 + 1 fallback)
25
25
 
26
26
  - [`frontend-ui`](frontend-ui.md) — Frontend & UI
27
27
  - [`visual-qa`](visual-qa.md) — Visual QA & Website Cloning
@@ -35,6 +35,7 @@ Thresholds: `STRONG_THRESHOLD = 8`, `WEAK_THRESHOLD = 4`, `AMBIGUITY_MARGIN = 2`
35
35
  - [`docs-writing`](docs-writing.md) — Docs & Technical Writing
36
36
  - [`qa-testing`](qa-testing.md) — QA & Testing
37
37
  - [`grok-harness`](grok-harness.md) — Grok xAI Harness
38
+ - [`devin-harness`](devin-harness.md) — Devin SWE-2 Harness
38
39
  - [`ai-agent-ops`](ai-agent-ops.md) — AI Agent Engineering & Ops
39
40
  - [`general`](general.md) — General (fallback)
40
41
 
@@ -0,0 +1,72 @@
1
+ # Devin SWE-2 Harness (`devin-harness`)
2
+
3
+ > Inherited domain capability document. Auto-generated from `src/core/domain-loadouts.ts` — do not edit by hand.
4
+
5
+
6
+ ## Identity
7
+
8
+ | field | value |
9
+ |---|---|
10
+ | id | `devin-harness` |
11
+ | authority | `write-scoped` |
12
+ | tools | read, grep, find, ls, edit, write, bash |
13
+ | command mode | `scoped-shell` |
14
+
15
+ ## Routing prompt
16
+
17
+ > Prepended to the lane task prompt when the router selects this domain.
18
+
19
+ ```text
20
+ DOMAIN: Devin SWE-2 Harness. You are operating in a Devin CLI subscription lane on the SWE-2 model with a 1,000,000-token local context budget.
21
+ Prioritize the SWE-2 operational playbook, focused exploration, small capability loadouts, and evidence-bound verification.
22
+
23
+ SEQUENCE:
24
+ 1. Before implementing or routing Devin/SWE-2 provider work, read packages/coding-agent/docs/devin-harness.md as the canonical playbook. Treat ~/.omk/agent/devin.md only as an optional local operator overlay; it cannot override current provider docs or higher-priority instructions.
25
+ 2. Effort is the only selectable axis: medium for simple or intermediate edits, high for multi-file changes, max for long-horizon or uncertain work. Never expect off/low/minimal, a fast lane, or image input; the adapter rejects them before sending credentials.
26
+ 3. Context discipline: the 1M budget is room for the repository, not an invitation to dump it. Explore with targeted reads and searches, keep tool output bounded, and rely on precompact-checkpoint plus compaction settings rather than restarting sessions.
27
+ 4. Capability discipline: load at most 2-3 skills for any lane. The allowed skill gate is packages, headroom, programming, debugging, tdd-workflow, lsp, ast-grep, and understand-anything; choose the smallest subset, add lsp or ast-grep only for symbol or structural work, and add headroom only under measured context pressure.
28
+ 5. Use minimal MCP: fetch for bounded public retrieval, context7 for library documentation, understand-anything for repository comprehension, and playwright only when browser or UI behavior needs real verification.
29
+ 6. Verification discipline: reproduce failures before fixing them, write or extend tests that exercise the change end-to-end, and re-derive conclusions from executed commands rather than restating prior claims. Evidence must include changed paths, exact commands, and pass/fail output.
30
+ 7. Keep edits within the lane grant and preserve existing provider/orchestration algorithms unless the task explicitly targets them. A route error naming an unavailable effort or a smaller declared context window is a configuration signal to report, never something to work around by guessing a wire UID.
31
+
32
+ HARD RULES: the packaged Devin harness doc is mandatory context; a local devin.md is optional; medium/high/max are the only efforts; the 1M budget never justifies unbounded dumps; maximum 2-3 active skills; never log the Devin session token, user JWT, or auth.json contents; protect-secrets applies.
33
+ ```
34
+
35
+ ## Curated skills (8)
36
+
37
+ - `packages`
38
+ - `headroom`
39
+ - `programming`
40
+ - `debugging`
41
+ - `tdd-workflow`
42
+ - `lsp`
43
+ - `ast-grep`
44
+ - `understand-anything`
45
+
46
+ ## Curated MCP servers (4)
47
+
48
+ - `fetch`
49
+ - `context7`
50
+ - `understand-anything`
51
+ - `playwright`
52
+
53
+ ## Curated hooks (6)
54
+
55
+ - `pre-shell-guard`
56
+ - `protect-secrets`
57
+ - `typecheck-after-edit`
58
+ - `stop-verify`
59
+ - `session-context`
60
+ - `precompact-checkpoint`
61
+
62
+ ## Routing triggers (7)
63
+
64
+ | kind | pattern | weight |
65
+ |---|---|---|
66
+ | keyword | `devin` | 8 |
67
+ | keyword | `swe-2` | 8 |
68
+ | keyword | `swe2` | 8 |
69
+ | keyword | `cognition` | 6 |
70
+ | keyword | `devin cli` | 8 |
71
+ | keyword | `1m context` | 5 |
72
+ | regex | `\b(devin|swe[- ]?2|cognition)\b` | 7 |
package/docs/metrics.md CHANGED
@@ -160,7 +160,7 @@ node scripts/tb-mini-suite.mjs --json # feed a runner
160
160
  node scripts/tb-mini-suite.mjs --seed 7 # a different fixed subset
161
161
  ```
162
162
 
163
- With identical task metadata, seed, size, and collation, selection is repeatable.
163
+ With identical task metadata, selection version, seed, size, and collation, selection is repeatable.
164
164
  The default 15-task subset oversamples easy tasks and prioritizes shorter expert
165
165
  time estimates; it is a regression signal, not a population-representative score
166
166
  or an agent runtime bound. Selection alone is not a capability result. Scoring
@@ -171,12 +171,35 @@ population. Missing difficulty quotas are filled from unselected tasks using the
171
171
  same ordering, so a valid request returns exactly that many distinct tasks.
172
172
  `--seed` accepts integers from `0` through `4294967295`. Invalid or missing option
173
173
  values and oversized requests exit with code `2`; absent, empty, or non-directory
174
- task paths exit with code `1`. The existing default subset remains unchanged.
174
+ task paths exit with code `1`.
175
+
176
+ The JSON output now declares `selectionVersion: 2`. Missing, empty, nonfinite, or
177
+ negative expert-time estimates are `null`, not zero. Within each difficulty band
178
+ and in quota refill, known estimates sort before unknown estimates. A genuine zero
179
+ or fractional estimate remains valid. Difficulty quotas still take precedence, so
180
+ unknown-estimate tasks can be selected to fill a band.
181
+
182
+ `knownExpertMinutes` sums known estimates; `unknownExpertEstimates` counts selected
183
+ tasks with unknown estimates. `totalExpertMinutes` is `null` when any selected
184
+ estimate is unknown. A nonfinite sum is refused with exit `1`, not serialized as
185
+ an apparently missing total. Expert estimates are not agent timeout limits.
186
+
187
+ For sizes 1–2, available slots go first to the highest-weight difficulty bands
188
+ (medium, then hard), with normal refill if those bands are unavailable. Task names
189
+ no longer decide which excess band quota is discarded. The normal-size quota rule
190
+ is preserved. Human-readable counts use a `Map`, including for labels such as
191
+ `__proto__` that overlap JavaScript object properties.
192
+
193
+ This is a selection and output-contract change: default membership can change when
194
+ metadata is incomplete, and the prior default JSON digest is historical only.
195
+ Freeze new task manifests before comparison; do not combine version 1 and 2 runs
196
+ as if their selection policy were identical. The limited flat TOML field reader
197
+ is unchanged; this does not add general TOML syntax support.
175
198
 
176
199
  Run the offline CLI regression tests without downloading tasks or calling models:
177
200
 
178
201
  ```bash
179
- node --test scripts/test/tb-mini-suite.test.mjs
202
+ node --test --test-concurrency=1 scripts/test/tb-mini-suite.test.mjs scripts/test/tb-mini-suite-ranking.test.mjs
180
203
  ```
181
204
 
182
205
  See [the harness roadmap](../../../ROADMAP.md) for the dated OMK versus Terminus-2
@@ -187,8 +210,8 @@ criteria. Planned runtime improvements are not measured benchmark gains.
187
210
 
188
211
  The checkout-only `scripts/tb21-audit.mjs` audits explicitly selected Harbor jobs
189
212
  against a caller-pinned manifest digest. It rejects duplicate tasks/trials, missing
190
- results or costs, mismatched task checksums or configured model labels, and
191
- contradictory success records. It never starts a model, picks the latest job, joins
213
+ results or costs, mismatched task checksums or configured model labels,
214
+ missing/invalid completion times, and contradictory success records. It never starts a model, picks the latest job, joins
192
215
  requests by timestamp, or rewrites evidence. See [TB 2.1 offline audit](tb21-audit.md)
193
216
  for the schema, invocation, error codes, and limitations.
194
217
 
@@ -196,14 +219,32 @@ A complete audit means recorded outcomes passed these checks, not that every
196
219
  provider request obeyed a single-model contract. Wire provenance, actual billing,
197
220
  repeated-trial analysis, and statistical superiority need separate evidence.
198
221
 
199
- ### Explicit output-limit validation
200
-
201
- For callers that supply `AgentLoopConfig.modelContract`, both the contract's
202
- `maxOutputTokens` and an explicitly supplied request `maxTokens` must be positive
203
- safe integers. Invalid explicit values are refused before `provider_request` and
204
- before calling the provider stream function; they are not treated as absent.
205
-
206
- An omitted request limit still leaves provider defaults unchecked by this
207
- predicate. This change does not activate a contract in the CLI or impose an
208
- effective cap on compaction and other provider paths. Full run-wide enforcement
209
- remains a roadmap item.
222
+ ### Output-limit validation: availability history
223
+
224
+ **2026-09-13 source snapshot (`ca75f4e5cc`):** the logical contract, CLI/SDK
225
+ wiring, and final Chat Completions model/output-limit checks are in the committed
226
+ source. [Model dispatch contracts](model-contract.md) defines their coverage;
227
+ [Verified Run](verified-run.md) describes the separate protected execution path.
228
+ Neither is a universal provider billing cap or a new controlled benchmark result.
229
+ See [the roadmap, section 16](../../../ROADMAP.md) for current local release-preparation checks.
230
+
231
+ The following paragraphs describe the earlier checkout only. Its missing modules
232
+ and failing collection are historical observations, not active release blockers.
233
+
234
+ **2026-09-08 follow-up:** the current worktree restores the logical contract and
235
+ connects it through the CLI/SDK, including SDK-stream summaries. See
236
+ [Model dispatch contracts](model-contract.md) and ROADMAP §13 for fresh evidence
237
+ and the remaining final-wire/accounting gaps. The warning below records the
238
+ preceding checkout, not the current availability of the restored module.
239
+
240
+ The previous worktree checkpoint tested positive-safe-integer validation of
241
+ `modelContract.maxOutputTokens` and explicit request `maxTokens`. During the
242
+ 2026-09-08 re-verification, the checkout changed: `run-model-contract.ts` and the
243
+ corresponding `AgentLoopConfig.modelContract` surface were absent. The remaining
244
+ `model-contract-output-limit.test.ts` fails collection against that checkout.
245
+
246
+ Do not treat the historical passing tests as proof that this guard is currently
247
+ available. Restoring or porting the runtime contract requires an explicit source
248
+ baseline decision and fresh send-boundary tests. No missing code was silently
249
+ recreated and no failing test was deleted. Full run-wide enforcement, including
250
+ omitted limits and compaction, remains unverified; see ROADMAP sections 11–12.
package/docs/providers.md CHANGED
@@ -19,6 +19,7 @@ Use `/login` in interactive mode, then select a provider:
19
19
  - Claude Pro/Max
20
20
  - GitHub Copilot
21
21
  - xAI Grok subscription OAuth
22
+ - Devin CLI subscription (SWE-2)
22
23
 
23
24
  Run `/login` and choose a configured subscription provider to open its account picker. Select an existing account by its ChatGPT, Claude, or Google email when available, or choose **Add another account** to sign in with a new one. OMK keeps and refreshes each account independently, pins the provider to the account you select, and does not silently fail over to another subscription. `/model` remains dedicated to model selection.
24
25
 
@@ -67,6 +68,80 @@ and [plan endpoint separation](https://www.alibabacloud.com/help/en/model-studio
67
68
  consulted 2026-09-08. The local tests exercise serialization, not account availability,
68
69
  provider compliance, billing, or benchmark performance.
69
70
 
71
+ ### Devin CLI
72
+
73
+ Run `/login devin`, select `devin/swe-2`, and use `/think medium`, `/think high`,
74
+ or `/think max`. For a new session after login:
75
+
76
+ ```bash
77
+ omk --provider devin --model swe-2 --thinking max
78
+ ```
79
+
80
+ The default is `medium`. Thinking-off and other levels are unsupported. `max`
81
+ is a reasoning level, not a requirement to buy the Devin Max subscription tier.
82
+ Your account's model access and quota still apply. For presets, effort guidance,
83
+ the 1M-token context budget, and the `devin-harness` loadout, see the
84
+ [Devin SWE-2 harness](devin-harness.md).
85
+
86
+ **Authentication.** OMK uses the CLI's PKCE flow at
87
+ `app.devin.ai/auth/cli/continue` and `api.devin.ai/auth/cli/token`. Its callback
88
+ binds only to `127.0.0.1:59653`, validates state, and closes after success,
89
+ rejection, cancellation, or a five-minute deadline. If the port is occupied,
90
+ paste the complete callback URL into OMK's local prompt. For remote sessions,
91
+ forward this loopback port to the machine running OMK.
92
+
93
+ Credentials use the existing protected `auth.json` storage and account picker.
94
+ Expired sessions require `/login devin` again: no refresh endpoint is verified,
95
+ and OMK does not renew the expiry locally. Use `/logout` to remove the stored
96
+ OMK credential. Alternatively, `DEVIN_API_KEY` accepts an already-owned CLI
97
+ session token, **not** a `cog_` REST API key. OMK does not install or invoke the
98
+ Devin agent, read browser cookies, or import another application's credentials.
99
+
100
+ **Transport and limits.** The Node-only `devin-agent` adapter uses Connect/protobuf
101
+ at the fixed HTTPS origin `https://server.codeium.com`. It exchanges the session
102
+ token for a user JWT and reads `GetCliModelConfigs` before each turn. The server's
103
+ SWE-2 family metadata supplies the effort's exact wire UID; OMK never invents a
104
+ `swe-2-max` ID or downgrades an unavailable route. Disabled, internal, ambiguous,
105
+ and fast-lane entries are excluded. The family's separate 1M-context entries form
106
+ a second lane that is selected only when the model's local `contextWindow` is
107
+ 1,000,000 or more (the bundled default); a smaller budget uses the standard lane.
108
+
109
+ Text, thinking, tool calls/results, and reported usage enter the normal OMK
110
+ agent loop. Image input and enterprise-origin overrides are unsupported.
111
+ Requests reject redirects. Malformed, oversized, or unterminated streams fail;
112
+ remote error bodies are not copied into diagnostics. SDK `onPayload` receives
113
+ protobuf bytes without credential metadata.
114
+
115
+ The bundled 1,000,000-token context budget and 16,384-token output cap are
116
+ local defaults, **not published SWE-2 limits**. Output is further capped against
117
+ the authenticated catalog. If the selected lane declares a smaller context window,
118
+ the request fails and names the window; lower `contextWindow` through
119
+ `modelOverrides` in [models.json](models.md#per-model-overrides) rather than
120
+ expecting a silent downgrade. Zero catalog pricing means unpriced subscription
121
+ usage, not free inference. Quota percentages are not inferred.
122
+
123
+ **Account quota.** With a Devin credential configured, the status rail's USAGE
124
+ section calls `SeatManagementService/GetUserStatus` (session-token metadata, no
125
+ user JWT) and renders the plan's daily and weekly quota meters with reset times.
126
+ Accounts whose plan reports no quota windows show the plan name and credit
127
+ balances instead. The rail mirrors the CLI's `/usage` surface; it never sends
128
+ the session token anywhere except `server.codeium.com`.
129
+
130
+ **Verification.** Local tests exercise the public stream API, real loopback
131
+ callbacks with mocked token exchange, model selection, and protocol fixtures.
132
+ Live login, catalog compatibility, SWE-2 inference, and billing remain unverified
133
+ without a Devin account. Unauthenticated catalog probes returned HTTP 400.
134
+ Live tests require `DEVIN_API_KEY` and `LIVE_E2E=1` and consume subscription quota.
135
+
136
+ Sources consulted 2026-09-12 and 2026-09-13: the [SWE-2 announcement](https://cognition.com/blog/swe-2)
137
+ (2026-09-10; CLI availability and medium/high/max; no published context window), [CLI commands](https://docs.devin.ai/cli/reference/commands),
138
+ and the [official manifest](https://static.devin.ai/cli/current/manifest.json)
139
+ (observed identity `3000.10.21`). Protocol fields follow the third-party
140
+ [oh-my-pi snapshot](https://github.com/can1357/oh-my-pi/blob/942383f768c5f2f6a620fcab57326c0f59df623a/packages/catalog/src/discovery/devin-proto.ts),
141
+ not a stable public inference contract; attribution is in `packages/ai/DEVIN-NOTICE`.
142
+ Regenerate only this logical model, preserving every other catalog entry, with
143
+ `npm --prefix packages/ai run generate-models -- --devin-only`.
144
+
70
145
  ### OpenAI Codex
71
146
 
72
147
  - Requires ChatGPT Plus or Pro subscription
@@ -111,6 +186,7 @@ omk
111
186
  | Azure OpenAI Responses | `AZURE_OPENAI_API_KEY` | `azure-openai-responses` |
112
187
  | OpenAI | `OPENAI_API_KEY` | `openai` |
113
188
  | DeepSeek | `DEEPSEEK_API_KEY` | `deepseek` |
189
+ | Devin CLI session token | `DEVIN_API_KEY` | `devin` |
114
190
  | NVIDIA NIM | `NVIDIA_API_KEY` | `nvidia` |
115
191
  | Google Gemini | `GEMINI_API_KEY` | `google` |
116
192
  | Mistral | `MISTRAL_API_KEY` | `mistral` |
@@ -76,6 +76,23 @@ An isolated HOME also hid the installed Rust toolchain: explicitly supplying its
76
76
  location restored the real cargo diagnostic check without changing the test or
77
77
  copying credentials. Neither fixture failure was treated as a product pass.
78
78
 
79
+ ## CI environment recovery
80
+
81
+ The first v0.98.5 tag run built the binaries, but its test step could not find
82
+ `/usr/bin/bwrap`. npm publication was not attempted and GitHub Release creation
83
+ was skipped. The default-branch CI workflows now install `bubblewrap` and run an
84
+ unprivileged namespace probe before the suite. Ubuntu 24.04 subsequently refused
85
+ loopback setup inside the namespace. The runtime-test jobs are pinned to Ubuntu
86
+ 22.04 LTS, with the same namespace and capability-drop checks. They do not skip
87
+ verified-run tests, disable host security controls or enable a sandbox fallback.
88
+ The older distribution `fd` lacked `--no-require-git`; CI installs the official
89
+ fd 10.4.2 static archive with SHA-256 verification before extraction and probes
90
+ that option before tests. File-search behavior and tests are not weakened.
91
+
92
+ Recovery dispatches the official workflow from `main` with both `tag` and
93
+ `source_ref` fixed to `v0.98.5`. The release tag and its source commit stay unchanged;
94
+ the existing source/tag equality checks remain mandatory.
95
+
79
96
  A source-file fingerprint is not an executed-build attestation. Linux observations
80
97
  do not establish behavior on every target platform. CI must validate the exact tag,
81
98
  build the six platform archives, run checks/tests, publish all seven npm packages
@@ -0,0 +1,68 @@
1
+ # Release audit: v0.99.0
2
+
3
+ 기준일: 2026-09-13. 사용자가 배포 준비 결과를 확인한 뒤 즉시 배포를 요청했다.
4
+ 이 기록은 후보의 검증 근거이며 GitHub/npm 게시 완료를 미리 선언하지 않는다.
5
+
6
+ ## 범위와 변경 로그
7
+
8
+ 이전 공개 릴리스 `v0.98.5`는 main의 조상이며, 준비 시점의 GitHub Release와 공개 npm
9
+ `latest` 7개도 0.98.5로 일치했다. 원격 `v0.99.0`이 없음을 확인한 뒤 준비했다.
10
+
11
+ `decee7f157`부터 `ca75f4e5cc`까지의 frontier·Devin·Codex SSE·재시도·TUI·설계 문서와,
12
+ 이번에 확정한 `3c7d3b4613`(TB 선택 v2), `b21daf1a76`(TB 감사 v2),
13
+ `003f7b081a`(문서·변경 로그·로컬 캡처 제외)을 포함한다. 마지막 세 단위는 각각
14
+ pre-commit 전체 검사를 통과했고, 선택기 40개·감사기 66개 CLI 검사를 통과했다.
15
+
16
+ TB 출력 계약 변경은 minor 증가로 처리했다. `selectionVersion: 2`의 nullable 예상
17
+ 시간·총량과 `omk-tb21-audit-report-2`의 필수 완료시각을 Breaking Changes에 기록했다.
18
+ 입력 manifest는 v1을 유지한다. 이전 버전 changelog 본문은 바꾸지 않았으며 다음
19
+ 작업용 `[Unreleased]`는 비워 두었다. 새로운 벤치마크 성능이나 SOTA 우위는 주장하지 않는다.
20
+
21
+ ## 버전·의존성
22
+
23
+ 공개 7개 패키지, root/example manifests, 내부 의존 범위, lockfiles, CLI shrinkwrap,
24
+ book compiler의 `PACKAGE_VERSION`, README 버전 링크와 릴리스 노트를 0.99.0에 맞춘다.
25
+ 외부 의존성의 버전·resolved URL·integrity 값은 변경하지 않았다. 모델 카탈로그도
26
+ 재생성하지 않았다. 운영자 설정·인증·MCP·스킬 활성 목록은 변경하지 않았다.
27
+
28
+ 첫 `version:minor` 실행은 manifest 증가 뒤 아직 이전 버전을 가리키는 내부 의존성을
29
+ npm 출시일 제한에 대조하다 중단됐다. 버전 증가를 재실행하지 않고, 기존
30
+ `sync-versions.js`와 lockfile/설치 동기화 단계만 이어갔다. 제한을 완화하거나 기존
31
+ 태그를 이동하지 않았다. 추가적인 registry 패키지 버전 변경은 없음을 대조했다.
32
+
33
+ ## 관측한 로컬 검증
34
+
35
+ - 환경: Linux, Node.js 24.19.0, npm 11.14.1.
36
+ - 새 버전의 `npm run build`가 7개 workspace 전체에서 종료 0이었다.
37
+ - `test.sh`를 인증 없는 별도 HOME·최소 환경에서 실행했다. 실제 운영자 auth 파일은
38
+ 건드리지 않았다. `taskset`으로 네 CPU에 제한했고 `LIVE_E2E=0`,
39
+ `OMK_NO_LOCAL_LLM=1`, `OMK_OFFLINE=1`을 사용했다.
40
+ - 전체 테스트: **8,279 통과, 852 환경·실계정 조건 skip, 실패 0, 종료 0**.
41
+ WPL 149, agent 870, AI 676, book compiler 22, coding-agent 5,695,
42
+ protocol 137, TUI 730개가 통과했다. 이 수치는 CI 실행 결과가 아니다.
43
+ - 준비 단계의 TB 106개와 전체 guard 378개 통과를 전체 제품 테스트 수에 다시 더하지 않는다.
44
+ - 최종 후보 확인 중 공유 트리에 Devin 요청 코드·테스트·문서와 그 변경 로그가 별도로
45
+ 바뀐 것을 발견했다. 이 4개 파일의 후속 변경은 제외하고 `003f7b081a`와 릴리스
46
+ 메타데이터만 별도 detached worktree에 옮겼다. 변경 로그의 같은 파일에 섞인 후속
47
+ 항목도 원래 커밋의 본문과 대조해 분리했다. 운영자의 변경·인증은 덮어쓰지 않는다.
48
+ - 격리 후보에서도 전체 테스트 8,279개 통과·852개 조건 skip·실패 0을 재확인했다.
49
+ 공유 트리 실행과 같은 검사를 중복 합산하지 않았다.
50
+ - 격리 후보의 `npm run check`(Node guard 378개 포함), 7개 pack dry-run·공개 entrypoint
51
+ import, 빌드 CLI의 `--version`·`--help`·`run --help`가 모두 종료 0이었다.
52
+ 모든 pack의 버전은 0.99.0이고, 필수 진입점 누락·비공개 상태 경로는 없었다.
53
+ - 후보 diff의 Gitleaks 검사는 완전 redaction과 기존 규칙으로 종료 0, 탐지 0건이었다.
54
+ - `--release`의 stale-worktree guard는 별도 최종 배포 gate다. 후보를 main에 연결하고
55
+ 이 작업의 임시 체크아웃을 제거한 뒤 기본 checkout에서 실행한다. guard나 이름을
56
+ 바꿔 검사를 피하지 않는다.
57
+
58
+ ## 배포 완료 조건
59
+
60
+ 기존 `build-binaries.yml`의 태그 기반 경로만 사용한다. 로컬 `npm publish`, 인증 변경,
61
+ 새로운 실행 권한·모델 호출은 없다. release source와 tag의 동일 SHA 검사를 유지한다.
62
+ 후보는 검토한 명시 경로만 stage하고 staged diff 전체를 확인한 뒤 commit·tag·push한다.
63
+
64
+ 공식 workflow가 6개 플랫폼 바이너리 빌드, 검사·테스트, npm 7개 패키지 게시,
65
+ GitHub Release 생성을 완료해야 한다. 최종 판정은 태그의 main 포함,
66
+ GitHub `v0.99.0` Release와 7개 npm `latest`의 일치다. 기존 token 기반 인증을 사용하며
67
+ OIDC/Sigstore provenance는 주장하지 않는다. 실패 시 원인을 확인하고 기존 배포의
68
+ 무결성 경계를 유지하며, 완료 전에는 배포 성공이라고 보고하지 않는다.
@@ -41,7 +41,8 @@ Recovery commands share command IDs and the generation cap; none resets budgets.
41
41
 
42
42
  `linux-command-dag-v1` adds `RunDagWriter` / `RunDagTask` with 1–16 nodes, explicit
43
43
  artifact dependencies, disjoint write scopes and 1–2 preapproved command attempts per
44
- node. `orderRunDag()` provides deterministic FIFO topological order and
44
+ node. Optional `maxConcurrentTasks` accepts only 1 or 2; omission preserves legacy
45
+ serialization and serial execution. `orderRunDag()` provides deterministic FIFO topological order and
45
46
  `runDagAncestors()` includes the full transitive input closure. `RunTaskRetryCommand`
46
47
  / `parseRunTaskRetryCommand()` pins a task selection to the original input and exact
47
48
  run revision/generation. These pure contracts never authorize dispatch or authenticate
@@ -53,8 +54,8 @@ then supplies authenticated checks to the existing claim-closure reducer.
53
54
  This does not replace the task/attempt/evaluation contracts or add execution to
54
55
  this package. See [Verified Run](verified-run.md) for the implemented CLI/SDK
55
56
  path, candidate recovery and checkpoint-based local writer restart, plus the
56
- serial command DAG and selective retry, plus remaining live-model, parallel frontier,
57
- plan amendment and control-surface work.
57
+ bounded command DAG, eager frontier and selective retry, plus remaining live-model,
58
+ verification-edge, plan amendment and control-surface work.
58
59
 
59
60
  ## Durable goal lifecycle
60
61
 
@@ -154,15 +154,18 @@ Evidence:
154
154
  - `packages/coding-agent/test/context-budget-selection-policy-version.test.ts`
155
155
  - `packages/coding-agent/test/context-budget-cache-disk.test.ts`
156
156
 
157
- **Working tree:** non-queued native `xai` requests started through `AgentSession.prompt()` now derive a bounded automatic skill grant from live discovered descriptions after ordinary prompt-template expansion. The selector scores task text separately from camelCase-aware path-to-skill-name signals, excludes explicit-only skills, caps automatic matches at three, and adds `headroom` only under lexical or measured context pressure. `AgentSession.prompt()` merges the result with settings/SDK/bang selections only for that request. Queued steering/follow-up messages reuse the active run's system prompt and do not trigger another selection pass.
157
+ **Working tree:** non-queued native `xai` and `devin` requests started through `AgentSession.prompt()` now derive a bounded automatic skill grant from live discovered descriptions after ordinary prompt-template expansion. The selector scores task text separately from camelCase-aware path-to-skill-name signals, excludes explicit-only skills, caps automatic matches at three, and adds `headroom` only under lexical or measured context pressure. `AgentSession.prompt()` merges the result with settings/SDK/bang selections only for that request. Queued steering/follow-up messages reuse the active run's system prompt and do not trigger another selection pass.
158
158
 
159
159
  Evidence:
160
160
 
161
161
  - `packages/coding-agent/src/core/active-skill-state.ts`
162
162
  - `packages/coding-agent/src/core/skill-selector.ts`
163
+ - `packages/coding-agent/src/core/harness-skills.ts`
163
164
  - `packages/coding-agent/src/core/grok-harness.ts`
165
+ - `packages/coding-agent/src/core/devin-harness.ts`
164
166
  - `packages/coding-agent/src/core/agent-session.ts`
165
167
  - `packages/coding-agent/test/grok-active-skills.test.ts`
168
+ - `packages/coding-agent/test/devin-active-skills.test.ts`
166
169
  - `packages/coding-agent/test/skill-selector.property.test.ts`
167
170
 
168
171
  **Working tree:** context files now treat their global/local relevance baseline