agentic-engineering-harness 0.8.1 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/README.md +8 -0
  2. package/dist/agents/config.js +7 -5
  3. package/dist/agents/config.js.map +1 -1
  4. package/dist/agents/outputContracts.d.ts +1 -1
  5. package/dist/agents/permissions.d.ts +3 -2
  6. package/dist/agents/permissions.js +4 -4
  7. package/dist/agents/permissions.js.map +1 -1
  8. package/dist/agents/routing.js +1 -1
  9. package/dist/agents/routing.js.map +1 -1
  10. package/dist/agents/types.d.ts +15 -0
  11. package/dist/audit/intent.d.ts +12 -1
  12. package/dist/audit/intent.js +44 -13
  13. package/dist/audit/intent.js.map +1 -1
  14. package/dist/audit/intentDecision.d.ts +67 -0
  15. package/dist/audit/intentDecision.js +100 -0
  16. package/dist/audit/intentDecision.js.map +1 -0
  17. package/dist/audit/run.d.ts +3 -0
  18. package/dist/audit/run.js +1 -2
  19. package/dist/audit/run.js.map +1 -1
  20. package/dist/context/gateway.js +73 -19
  21. package/dist/context/gateway.js.map +1 -1
  22. package/dist/context/preflight.d.ts +9 -2
  23. package/dist/context/preflight.js +30 -10
  24. package/dist/context/preflight.js.map +1 -1
  25. package/dist/context/transport.d.ts +33 -1
  26. package/dist/context/transport.js +127 -23
  27. package/dist/context/transport.js.map +1 -1
  28. package/dist/context/types.d.ts +4 -4
  29. package/dist/core/run.js +0 -2
  30. package/dist/core/run.js.map +1 -1
  31. package/dist/entry.js +15 -8
  32. package/dist/entry.js.map +1 -1
  33. package/dist/informational/answer.d.ts +25 -0
  34. package/dist/informational/answer.js +90 -0
  35. package/dist/informational/answer.js.map +1 -0
  36. package/dist/issues/intake.d.ts +1 -1
  37. package/dist/main.js +4 -1
  38. package/dist/main.js.map +1 -1
  39. package/dist/operations/change.js +0 -2
  40. package/dist/operations/change.js.map +1 -1
  41. package/dist/operations/completion.js +2 -0
  42. package/dist/operations/completion.js.map +1 -1
  43. package/dist/operations/controller.js +7 -3
  44. package/dist/operations/controller.js.map +1 -1
  45. package/dist/operations/evidence.d.ts +20 -0
  46. package/dist/operations/evidence.js +38 -0
  47. package/dist/operations/evidence.js.map +1 -0
  48. package/dist/operations/mcp.d.ts +7 -0
  49. package/dist/operations/mcp.js +62 -12
  50. package/dist/operations/mcp.js.map +1 -1
  51. package/dist/operations/state.d.ts +5 -0
  52. package/dist/operations/state.js.map +1 -1
  53. package/dist/paseo/deterministicSession.d.ts +10 -2
  54. package/dist/paseo/deterministicSession.js +39 -5
  55. package/dist/paseo/deterministicSession.js.map +1 -1
  56. package/dist/paseo/launchSpec.d.ts +2 -0
  57. package/dist/paseo/launchSpec.js +5 -5
  58. package/dist/paseo/launchSpec.js.map +1 -1
  59. package/dist/paseo/start.d.ts +1 -1
  60. package/dist/paseo/start.js +6 -1
  61. package/dist/paseo/start.js.map +1 -1
  62. package/dist/workers/agentPrompt.d.ts +2 -0
  63. package/dist/workers/agentPrompt.js +13 -15
  64. package/dist/workers/agentPrompt.js.map +1 -1
  65. package/dist/workers/direct.js +10 -4
  66. package/dist/workers/direct.js.map +1 -1
  67. package/dist/workers/paseo.js +13 -4
  68. package/dist/workers/paseo.js.map +1 -1
  69. package/dist/workers/podman.js +10 -4
  70. package/dist/workers/podman.js.map +1 -1
  71. package/docs/ARCHITECTURE.md +40 -0
  72. package/docs/CONTEXT_EFFICIENCY.md +18 -2
  73. package/docs/EVALS.md +12 -0
  74. package/docs/OPERATION_SUPERVISION.md +14 -0
  75. package/docs/PASEO.md +35 -1
  76. package/docs/VALIDATION.md +8 -0
  77. package/package.json +1 -1
  78. package/presets/agents/default.jsonc +3 -3
  79. package/presets/agents/orchestration.jsonc +1 -0
@@ -14,6 +14,37 @@ Paseo is the reference adapter. It owns process/session/worktree/mobile control,
14
14
 
15
15
  A lead agent (reference: Codex) owns requirement interpretation, architecture, SDD and review.
16
16
 
17
+ Conversational intent is a separate routing boundary. For a managed
18
+ conversation, the lead is the only natural-language semantic authority. It
19
+ translates each human turn, including negation, referents and follow-ups, into
20
+ a compact, versioned `IntentDecisionV1`; the controller never reclassifies the
21
+ original sentence after that decision.
22
+
23
+ ```text
24
+ human turn -> Paseo lead -> IntentDecisionV1 -> selected AEH route
25
+ | |
26
+ | +-> informational / audit / change / run / status / cancel
27
+ v
28
+ durable userTurnId, outcome, referents, constraints
29
+ ```
30
+
31
+ The decision is descriptive, not a permission grant. Its contract contains
32
+ `version`, `source` (`lead-semantic`, `explicit-cli` or
33
+ `heuristic-fallback`), optional `userTurnId`, `intent`, compact
34
+ `requestedOutcome`, effect booleans, optional continuation references,
35
+ constraints, confidence and resolution state. AEH validates only this typed
36
+ contract and its internal effect invariants. Unknown policy claims such as
37
+ `skipValidation`, `allowNetwork`, `gitWrite` or `bypassProvenance` are rejected
38
+ by strict schema validation.
39
+
40
+ Explanatory repository questions use a bounded read-only context answer and do
41
+ not create lifecycle state. Audit, change and run routes enter their existing
42
+ deterministic contracts. The controller remains the authority for TaskContract,
43
+ SDD/seal, scope, capabilities, permissions, validators, evidence, lifecycle,
44
+ provenance and delivery. The retained `classifyEngineeringIntentHeuristic`
45
+ surface is diagnostic/evaluation/fallback infrastructure only; it cannot veto a
46
+ lead-selected route.
47
+
17
48
  ### 3. Implementation workers
18
49
 
19
50
  Workers (reference: OpenCode + cost-efficient model) implement frozen, scoped tasks. They have no authority to redefine acceptance.
@@ -80,6 +111,15 @@ The deterministic harness evaluates build/type/lint/tests, scope, immutable file
80
111
 
81
112
  `ContextBudgetGateway` is the single controller-owned path for bounded agent context. It retrieves, classifies, selects, deterministically projects, budgets, selectively compresses and delivers a versioned `ContextEnvelope`. Required `VERBATIM` content is never lossy-compressed or character-truncated. Raw evidence remains in an AEH artifact and is available only through an operation/agent-authorized retrieval gateway.
82
113
 
114
+ Context capability requirements are scoped to the selected execution contract:
115
+ each capability is `REQUIRED`, `OPTIONAL` or `FORBIDDEN`. Runtime and transport
116
+ registries describe actual MCP projection surfaces independently of runtime
117
+ names. The resolved capability object is reused for pre-materialization
118
+ readiness, MCP injection, prompt policy and diagnostics. The coordinator and
119
+ operation supervisor forbid repository/semantic/raw context by default, so a
120
+ global project requirement cannot leak Serena or raw retrieval into the
121
+ supervisory control plane.
122
+
83
123
  ```text
84
124
  Graphify -> macro structural topology and advisory dependency/community signals
85
125
  Serena -> micro semantic repository retrieval (symbol, overview, references)
@@ -13,10 +13,19 @@ envelope, agent charter, frozen skills, normative contract/seal/source,
13
13
  assignment, operation projection, RepoMap, advisory memory and evidence
14
14
  references. Normative fragments remain byte-for-byte `VERBATIM`; the charter is
15
15
  not used as a single catch-all fragment. RepoMap construction is skipped when
16
- `context.repositoryMap.enabled` is false. Transport capabilities are carried
16
+ the selected execution contract forbids it. Transport capabilities are carried
17
17
  into preparation so a direct Codex or hardened Podman turn never advertises a
18
18
  retrieval tool it cannot expose.
19
19
 
20
+ Context requirements are explicit per agent: `REQUIRED`, `OPTIONAL` or
21
+ `FORBIDDEN` for repository-map, semantic retrieval, raw retrieval and
22
+ compression. Project-level `semanticRetrieval.required: true` is therefore
23
+ scoped by the routed execution contract. Coordinators/supervisors default to
24
+ `FORBIDDEN` for repository and raw semantic context; reviewers and workers may
25
+ require Serena when their runtime/transport can actually project it. The same
26
+ resolved capability result controls readiness, MCP injection, prompt policy
27
+ and degradation diagnostics.
28
+
20
29
  ## Preservation classes
21
30
 
22
31
  - `VERBATIM`: exact normative requirements, contracts, schemas, hashes, anchors and critical diagnostics. It cannot be lossy-compressed.
@@ -43,6 +52,13 @@ Envelopes carry operation, agent, phase, budget, fragment projections, allowed r
43
52
 
44
53
  ## Runtime and troubleshooting
45
54
 
46
- Run `aeh doctor` after `aeh init --setup`. Context readiness reports the gateway, estimator, retrieval gateway, Serena and Headroom independently. A missing mandatory provider is a deterministic readiness failure; AEH does not silently fall back to bulk repository reads or unoptimized compression. Fix it with `aeh setup --update-lock` and rerun doctor. `--help`, `--version` and status inspection do not install tools.
55
+ Run `aeh doctor` after `aeh init --setup`. Doctor reports the gateway,
56
+ estimator, retrieval gateway, Serena and Headroom independently. Operation
57
+ readiness is evaluated again after routing, against the concrete runtime and
58
+ transport; a project-level required provider does not block a role whose
59
+ contract forbids or does not require that capability. A required capability
60
+ for a selected worker/reviewer fails closed before materialization; an
61
+ optional capability records an explicit bounded fallback. `--help`,
62
+ `--version` and status inspection do not install tools.
47
63
 
48
64
  Context telemetry emits numeric/hash-only events including `harness.context.prepare`, `project`, `compress`, `deliver` and `operation_summary`. It intentionally does not emit prompt bodies. Evaluation should compare baseline/observe and optimized/enforce with the same task, repository, model and provider, and report cost per successful operation rather than tokens removed alone.
package/docs/EVALS.md CHANGED
@@ -26,3 +26,15 @@ Primary metrics:
26
26
  - elapsed time.
27
27
 
28
28
  Production bugs should be converted into permanent regression/eval cases when practical.
29
+
30
+ ## Semantic routing corpus
31
+
32
+ `evals/corpus/intent-routing.json` is a separate lead-semantic evaluation
33
+ corpus. It includes Spanish and English prompts, negation, mixed requests,
34
+ finding referents, constrained follow-ups, status/cancel, prepared-run
35
+ continuation and adversarial policy wording. Each case records an expected
36
+ semantic route and a compact scripted `IntentDecisionV1` fixture. The mandatory
37
+ `tests/intentRoutingCorpus.test.ts` validates the structured route/effect
38
+ contract without invoking the heuristic classifier. A model-backed eval lane
39
+ may compare a configured lead's decision against the prompt labels without
40
+ making external inference a requirement of ordinary pull-request CI.
@@ -27,6 +27,14 @@ The lead owns user intent, priorities, cross-operation dependencies, true except
27
27
 
28
28
  The operation supervisor owns semantic coordination and consolidation for one operation. It may merge semantically duplicate findings, identify conflicts and request bounded follow-up, but it cannot overrule deterministic state, validation or normative artifacts.
29
29
 
30
+ The supervisor charter is coordination-only. Its execution contract forbids
31
+ repository-map, Serena semantic retrieval and raw artifact retrieval, even when
32
+ the project globally configures semantic retrieval as required. Reviewers and
33
+ workers receive those capabilities only when their own runtime/transport
34
+ contract resolves them as available. A capability failure is rejected before
35
+ the candidate agent is materialized; optional capabilities produce an explicit
36
+ degradation rather than an unmarked policy change.
37
+
30
38
  The controller owns lifecycle, stage transitions, participant state, seals, validators, rollback, quality gates, delivery and terminalization. `OperationRecord` is the durable source of lifecycle truth.
31
39
 
32
40
  ## OperationRecord v2
@@ -94,6 +102,12 @@ A stall targets the operation supervisor first. Missing/busy/unreachable supervi
94
102
 
95
103
  Healthy non-terminal progress wakes are internal. They should not generate chat noise.
96
104
 
105
+ The conversational lead also has a bounded informational path. A request such
106
+ as “explain how validation works” is answered from a small read-only repository
107
+ context and creates no OperationRecord, TaskContract, reviewer, report or
108
+ delivery artifact. A request to find defects, assess safety/correctness or
109
+ report problems remains an AUDIT and follows the supervised lifecycle.
110
+
97
111
  ## Supervisor context generations
98
112
 
99
113
  Supervisors are proactively replaced rather than compacted as their canonical Paseo context approaches the configured handoff threshold.
package/docs/PASEO.md CHANGED
@@ -43,6 +43,28 @@ aeh paseo agents --operation <operation-id> --phase review
43
43
 
44
44
  Synchronous `aeh audit` / `aeh run` remain valid compatibility entrypoints for non-interactive automation.
45
45
 
46
+ The conversational lead is the semantic authority for translating each human
47
+ turn into a typed, versioned `IntentDecisionV1`. It resolves the requested
48
+ outcome, negation, referents, constraints and follow-ups, then selects the
49
+ corresponding route. AEH does not re-interpret the original sentence after
50
+ that decision. The deterministic controller validates only the structured
51
+ decision and remains authoritative for contracts, capabilities, permissions,
52
+ validators, evidence, lifecycle, provenance and delivery.
53
+
54
+ Each decision is bound to the originating `userTurnId` when available and is
55
+ stored with the operation's durable intent state (or the deterministic
56
+ session's compact turn record). This lets lead rotation recover operation and
57
+ finding references without replaying the old model conversation. An unresolved
58
+ mutating referent is rejected by the route contract until the lead resolves it;
59
+ the controller never guesses a target from the human text.
60
+
61
+ Explanations and orientation use the bounded `aeh_informational_context` tool
62
+ and do not create an operation. Defect discovery, correctness/safety judgments
63
+ and formal review use the AUDIT start tool; mixed requests preserve the lead's
64
+ explicit semantic decision. A heuristic classifier remains available for the
65
+ `aeh intent` diagnostic/evaluation surface and explicitly marked compatibility
66
+ fallbacks only; disagreement cannot veto a lead decision.
67
+
46
68
  ## SDK-first control plane
47
69
 
48
70
  AEH uses Paseo's published TypeScript client package, `@getpaseo/client`, as the primary control surface for agent creation, follow-up turns, status lookup and directory queries. Paseo currently documents that package as public but **not yet a stable public SDK**, so AEH deliberately resolves the copy bundled with the active `@getpaseo/cli` installation first instead of independently selecting a client version.
@@ -92,15 +114,27 @@ command: <exact Node executable>
92
114
  args: [<exact dist/main.js>, operation, mcp]
93
115
  ```
94
116
 
95
- Only four MCP tools are preapproved:
117
+ The control MCP is preapproved only for these bounded tools:
96
118
 
97
119
  ```text
120
+ aeh_informational_context
98
121
  aeh_operation_start_audit
99
122
  aeh_operation_start_run
123
+ aeh_operation_start_change
124
+ aeh_operation_digest
100
125
  aeh_operation_status
126
+ aeh_operation_ack
127
+ aeh_operation_portfolio
101
128
  aeh_operation_cancel
129
+ aeh_context_status
102
130
  ```
103
131
 
132
+ `aeh_informational_context` is read-only and accepts a lead-produced
133
+ `IntentDecisionV1` whose route is INFORMATIONAL. AEH validates the decision's
134
+ effects without scanning the human request. The lead is never given Serena through this control surface;
135
+ Serena is projected only to a routed worker/reviewer whose resolved capability
136
+ contract permits it.
137
+
104
138
  Paseo's `toolPolicy.preapproved` is scoped to those exact MCP server/tool identities; native shell/edit tools are not broadened by this configuration. If the AEH invocation cannot be parsed as a safe command vector, MCP injection is skipped rather than evaluating shell syntax, and the short `aeh operation ...` CLI surface remains the fallback.
105
139
 
106
140
  The lead bootstrap version is incremented when this managed-session contract changes, so explicit resume cannot silently reuse an older lead that lacks the current operation-control surface.
@@ -2,6 +2,14 @@
2
2
 
3
3
  LLM output is an untrusted proposal. Acceptance is based on executable evidence.
4
4
 
5
+ ## User-facing evidence discipline
6
+
7
+ Operation completion messages distinguish durable operation results, audit or
8
+ validation reports, control-plane context and inference. A failed or blocked
9
+ operation with no result artifact, AuditReport or findings is not evidence that
10
+ the repository was inspected. The lead must state the blocker and keep any
11
+ pre-existing context separate from operation-produced claims.
12
+
5
13
  ## Capability registry
6
14
 
7
15
  The normative unit of validation is a capability. TaskContracts should prefer:
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "agentic-engineering-harness",
3
- "version": "0.8.1",
3
+ "version": "0.8.2",
4
4
  "description": "OSS-first engineering harness for deterministic, spec-driven, issue-driven, audit-governed and orchestration-first multi-agent software delivery.",
5
5
  "type": "module",
6
6
  "bin": {
@@ -3,15 +3,15 @@
3
3
  "activeProfile": "balanced",
4
4
  "skillRoots": [".harness/skills", ".agents/skills", ".opencode/skills"],
5
5
  "runtimes": {
6
- "codex": { "adapter": "codex", "paseoProvider": "codex", "command": "codex", "capabilities": { "nativeAgent": false, "nativeAgentViaPaseo": false, "modelSelection": true, "variantSelection": true, "sessions": true } },
7
- "opencode": { "adapter": "opencode", "paseoProvider": "opencode", "command": "opencode", "capabilities": { "nativeAgent": true, "nativeAgentViaPaseo": false, "modelSelection": true, "variantSelection": true, "sessions": true, "structuredOutput": true } }
6
+ "codex": { "adapter": "codex", "paseoProvider": "codex", "command": "codex", "capabilities": { "nativeAgent": false, "nativeAgentViaPaseo": false, "modelSelection": true, "variantSelection": true, "sessions": true, "mcp": true, "stdioMcp": true, "localMcp": true, "nativeToolProjection": true } },
7
+ "opencode": { "adapter": "opencode", "paseoProvider": "opencode", "command": "opencode", "capabilities": { "nativeAgent": true, "nativeAgentViaPaseo": false, "modelSelection": true, "variantSelection": true, "sessions": true, "structuredOutput": true, "mcp": true, "stdioMcp": true, "localMcp": true, "runtimeConfigInjection": true, "nativeToolProjection": true } }
8
8
  },
9
9
  "models": {
10
10
  "brain": { "runtime": "codex", "provider": "openai", "model": "gpt-5.6-luna", "variant": "max" },
11
11
  "workhorse": { "runtime": "opencode", "provider": "opencode-go", "model": "deepseek-v4-flash" }
12
12
  },
13
13
  "agents": {
14
- "lead": { "role": "orchestrator", "domains": ["*"], "description": "Own intent, decomposition, dependency ordering, risk decisions and final semantic acceptance. Delegate repository work to the narrowest specialist; do not perform implementation edits. Preserve sealed requirements and use human-on-exception only when the answer cannot be derived from repository evidence or normative artifacts.", "execution": { "model": "@brain" }, "skills": ["engineering-workflow", "lead-engineer", "verification-planning", "worktree-lifecycle"], "permissions": { "read": "allow", "write": "deny", "shell": "allow", "network": "ask", "delegate": "allow", "review": "allow", "validate": "deny", "gitWrite": "deny" }, "outputContract": "orchestrator" },
14
+ "lead": { "role": "orchestrator", "domains": ["*"], "description": "Own intent, decomposition, dependency ordering, risk decisions and final semantic acceptance. Delegate repository work to the narrowest specialist; do not perform implementation edits. Preserve sealed requirements and use human-on-exception only when the answer cannot be derived from repository evidence or normative artifacts.", "execution": { "model": "@brain" }, "contextRequirements": { "repositoryMap": "FORBIDDEN", "semanticRetrieval": "FORBIDDEN", "rawRetrieval": "FORBIDDEN" }, "skills": ["engineering-workflow", "lead-engineer", "verification-planning", "worktree-lifecycle"], "permissions": { "read": "allow", "write": "deny", "shell": "allow", "network": "ask", "delegate": "allow", "review": "allow", "validate": "deny", "gitWrite": "deny" }, "outputContract": "orchestrator" },
15
15
  "planner": { "role": "planner", "domains": ["*"], "description": "Produce bounded delegation plans from requirements, changed areas, dependencies, findings and validation evidence. Remain read-only, identify parallelizable work, required reviewers and validation gates, and never silently change normative scope.", "execution": { "model": "@brain" }, "skills": ["routing-normalizer", "verification-planning", "acceptance-traceability"], "permissions": { "read": "allow", "write": "deny", "shell": "allow", "network": "deny", "delegate": "allow", "review": "allow", "validate": "deny", "gitWrite": "deny" }, "outputContract": "planner" },
16
16
  "oracle": { "role": "escalation", "domains": ["*"], "description": "Diagnose persistent, ambiguous, cycling or cross-cutting failures using repository evidence and sealed requirements. Distinguish implementation defects from spec contradictions, missing product decisions and unavailable external resources. Do not edit files.", "execution": { "model": "@brain" }, "skills": ["recovery-classifier", "simplify"], "permissions": { "read": "allow", "write": "deny", "shell": "allow", "network": "ask", "delegate": "deny", "review": "allow", "validate": "allow", "gitWrite": "deny" }, "outputContract": "recovery" },
17
17
  "explorer": { "role": "explorer", "domains": ["*"], "description": "Perform fast repository discovery: locate relevant files, symbols, tests, ownership boundaries and nearby patterns. Return evidence and paths rather than speculative redesigns. Remain read-only.", "execution": { "model": "@workhorse" }, "permissions": { "read": "allow", "write": "deny", "shell": "allow", "network": "deny", "delegate": "deny", "review": "deny", "validate": "deny", "gitWrite": "deny" } },
@@ -22,6 +22,7 @@
22
22
  "domains": ["*"],
23
23
  "description": "Own operation-local semantic coordination, delegation, consolidation, conflict detection and evidence gaps for exactly one durable AEH operation.",
24
24
  "execution": { "model": "@brain" },
25
+ "contextRequirements": { "repositoryMap": "FORBIDDEN", "semanticRetrieval": "FORBIDDEN", "rawRetrieval": "FORBIDDEN", "compression": "OPTIONAL" },
25
26
  "temperature": 0.1,
26
27
  "skills": [],
27
28
  "permissions": { "read": "allow", "write": "deny", "shell": "allow", "network": "deny", "delegate": "allow", "review": "allow", "validate": "deny", "gitWrite": "deny" },