create-mercato-app 0.6.7-develop.6814.1.0627c7e9f1 → 0.6.7-develop.6825.1.85bbf320ad
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/agentic/shared/ai/harness/README.md +2 -2
- package/agentic/shared/ai/harness/RELEASE.md +4 -4
- package/agentic/shared/ai/harness/cases.json +91 -1
- package/agentic/shared/ai/harness/cases.schema.json +3 -3
- package/agentic/shared/ai/harness/validators.json +1 -1
- package/agentic/shared/ai/skills/om-system-extension/SKILL.md +1 -1
- package/agentic/shared/scripts/evaluate-agent-harness.mjs +37 -12
- package/dist/agentic/guides/module-facts.json +110 -110
- package/dist/agentic/guides/modules/ai_assistant.md +1 -1
- package/dist/agentic/guides/modules/api_docs.md +1 -1
- package/dist/agentic/guides/modules/api_keys.md +1 -1
- package/dist/agentic/guides/modules/attachments.md +1 -1
- package/dist/agentic/guides/modules/audit_logs.md +1 -1
- package/dist/agentic/guides/modules/auth.md +1 -1
- package/dist/agentic/guides/modules/business_rules.md +1 -1
- package/dist/agentic/guides/modules/catalog.md +1 -1
- package/dist/agentic/guides/modules/channel_gmail.md +1 -1
- package/dist/agentic/guides/modules/channel_imap.md +1 -1
- package/dist/agentic/guides/modules/checkout.md +1 -1
- package/dist/agentic/guides/modules/communication_channels.md +1 -1
- package/dist/agentic/guides/modules/configs.md +1 -1
- package/dist/agentic/guides/modules/content.md +1 -1
- package/dist/agentic/guides/modules/currencies.md +1 -1
- package/dist/agentic/guides/modules/customer_accounts.md +1 -1
- package/dist/agentic/guides/modules/customers.md +1 -1
- package/dist/agentic/guides/modules/dashboards.md +1 -1
- package/dist/agentic/guides/modules/data_sync.md +1 -1
- package/dist/agentic/guides/modules/design_system.md +1 -1
- package/dist/agentic/guides/modules/dictionaries.md +1 -1
- package/dist/agentic/guides/modules/directory.md +1 -1
- package/dist/agentic/guides/modules/entities.md +1 -1
- package/dist/agentic/guides/modules/events.md +1 -1
- package/dist/agentic/guides/modules/feature_toggles.md +1 -1
- package/dist/agentic/guides/modules/gateway_stripe.md +1 -1
- package/dist/agentic/guides/modules/generators.md +1 -1
- package/dist/agentic/guides/modules/inbox_ops.md +1 -1
- package/dist/agentic/guides/modules/integrations.md +1 -1
- package/dist/agentic/guides/modules/messages.md +1 -1
- package/dist/agentic/guides/modules/notifications.md +1 -1
- package/dist/agentic/guides/modules/onboarding.md +1 -1
- package/dist/agentic/guides/modules/payment_gateways.md +1 -1
- package/dist/agentic/guides/modules/perspectives.md +1 -1
- package/dist/agentic/guides/modules/planner.md +1 -1
- package/dist/agentic/guides/modules/portal.md +1 -1
- package/dist/agentic/guides/modules/progress.md +1 -1
- package/dist/agentic/guides/modules/query_index.md +1 -1
- package/dist/agentic/guides/modules/record_locks.md +1 -1
- package/dist/agentic/guides/modules/resources.md +1 -1
- package/dist/agentic/guides/modules/sales.md +1 -1
- package/dist/agentic/guides/modules/scheduler.md +1 -1
- package/dist/agentic/guides/modules/search.md +1 -1
- package/dist/agentic/guides/modules/security.md +1 -1
- package/dist/agentic/guides/modules/shipping_carriers.md +1 -1
- package/dist/agentic/guides/modules/sso.md +1 -1
- package/dist/agentic/guides/modules/staff.md +1 -1
- package/dist/agentic/guides/modules/storage_s3.md +1 -1
- package/dist/agentic/guides/modules/sync_akeneo.md +1 -1
- package/dist/agentic/guides/modules/sync_excel.md +1 -1
- package/dist/agentic/guides/modules/system_status_overlays.md +1 -1
- package/dist/agentic/guides/modules/translations.md +1 -1
- package/dist/agentic/guides/modules/webhooks.md +1 -1
- package/dist/agentic/guides/modules/wms.md +1 -1
- package/dist/agentic/guides/modules/workflows.md +1 -1
- package/dist/agentic/guides/upstream/BACKWARD_COMPATIBILITY.md +2 -0
- package/dist/agentic/guides/upstream/manifest.json +2 -2
- package/dist/agentic/shared/ai/harness/README.md +2 -2
- package/dist/agentic/shared/ai/harness/RELEASE.md +4 -4
- package/dist/agentic/shared/ai/harness/cases.json +91 -1
- package/dist/agentic/shared/ai/harness/cases.schema.json +3 -3
- package/dist/agentic/shared/ai/harness/validators.json +1 -1
- package/dist/agentic/shared/ai/skills/om-system-extension/SKILL.md +1 -1
- package/dist/agentic/shared/scripts/evaluate-agent-harness.mjs +37 -12
- package/package.json +3 -3
package/README.md
CHANGED
|
@@ -144,7 +144,7 @@ yarn install-skills
|
|
|
144
144
|
yarn harness:release --runner codex --prepare-targets /absolute/empty-release-targets --acknowledge-writes
|
|
145
145
|
```
|
|
146
146
|
|
|
147
|
-
The target directory must be absolute, new or empty, and outside the controller app. Select one blocking primary runner with `--runner codex` or `--runner claude`; it owns all
|
|
147
|
+
The target directory must be absolute, new or empty, and outside the controller app. Select one blocking primary runner with `--runner codex` or `--runner claude`; it owns all 203 routing cases and every writable/review lane, with no per-case fallback. Optionally add the different authenticated runner through `--portability-runner` for the exact 46-case representative read-only lane. Omitting it is valid and recorded as not requested; once requested, its failures are blocking. Use a fresh, sanitized controller: automatic preparation fails before copying `.env`/`.env.*` local configuration (safe example/sample/template files remain allowed), credential files, or private-key files. The complete gate requires Linux with trusted system Bubblewrap (`bwrap`) and user namespaces because its Playwright API/browser lanes need a loopback namespace isolated from the host. Preflight rejects untrusted/no-op/pass-through executables and proves isolated loopback plus a capability-free payload before target preparation, provider invocation, or writes; native macOS and Windows therefore fail closed. The command also fails closed when a required runner, browser, or test runtime is unavailable. The 203-case catalog includes 93 framework-neutral business prompts and 46 writable implementation/regression cases (22.7%). The release command runs live routing, writable trusted oracles, per-target `generate`/`typecheck`/`lint`/`build`, any declared generated test, and isolated generated-code review for every writable result. Foundation and target validation—including `yarn build`—receive a minimal environment with network access denied, and persisted diagnostics redact sensitive environment values and URL userinfo. Test-authoring coverage executes a Jest unit test plus Linux/Bubblewrap loopback-only Playwright API and browser tests through fixed controller-owned commands against a read-only target; runtime reports must attest at least one passed test and zero skipped, todo, focused, flaky, or expected-failure tests. The suite then writes a schema-valid sanitized mode-`0600` report under `.ai/harness/results/` with the selected primary and optional portability runner policy.
|
|
148
148
|
|
|
149
149
|
Use the bundled `om-evolve-harness` skill to add a real case: reproduce failure first, select one smallest knowledge owner, run any generated unit/integration tests plus target checks, require code review, and finish with the full release suite. Open Mercato framework maintainers use the monorepo-only `$om-refresh-standalone-harness --from <ref> --to <ref>` workflow for every release range and retain its sanitized maintenance report.
|
|
150
150
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Agent harness evaluations
|
|
2
2
|
|
|
3
|
-
`cases.json` is the
|
|
3
|
+
`cases.json` is the 203-case standalone-app contract. Run `yarn harness:validate --all` for the deterministic gate. Live routing uses a fresh read-only process per case:
|
|
4
4
|
|
|
5
5
|
UMES routing is fact-first. The additive and unified-override audit evaluations, plus their targeted cases, resolve exact/pattern hosts, outgoing contributions, correlation provenance, round-trip groups, framework-owned targets, and override domain/key/mode from the generated module sheets and `.ai/guides/framework-extension-points.md` before bounded installed source. The repository UMES umbrella spec may appear as optional source-checkout provenance; it is never required in a standalone scaffold.
|
|
6
6
|
|
|
@@ -18,7 +18,7 @@ yarn harness:release --runner codex --prepare-targets /absolute/empty-release-ta
|
|
|
18
18
|
yarn harness:release --runner codex --portability-runner claude --prepare-targets /absolute/empty-release-targets --acknowledge-writes
|
|
19
19
|
```
|
|
20
20
|
|
|
21
|
-
The primary runner owns all
|
|
21
|
+
The primary runner owns all 203 routing cases, all 46 writable cases, and all generative-judge runs. No per-case fallback or mixed primary ownership is allowed. Omitting `--portability-runner` is valid and the sanitized report records `portabilityRunner: null`; explicitly requesting an unavailable or failing secondary runner fails that extended run.
|
|
22
22
|
|
|
23
23
|
Writable evaluation is intentionally opt-in. The expanded catalog has a 46-case writable release target, but only cases registered in `release-matrix.json` and backed by controller-owned fixtures and oracles are executable. Copy or create a fresh standalone app for one registered case, then seed only that case and mark the target disposable:
|
|
24
24
|
|
|
@@ -6,7 +6,7 @@ Run the complete per-release gate from a generated standalone app with one comma
|
|
|
6
6
|
yarn harness:release --runner codex --prepare-targets /absolute/empty-release-targets --acknowledge-writes
|
|
7
7
|
```
|
|
8
8
|
|
|
9
|
-
Choose exactly one blocking primary runner with `--runner codex` or `--runner claude`. That runner owns the complete
|
|
9
|
+
Choose exactly one blocking primary runner with `--runner codex` or `--runner claude`. That runner owns the complete 203-case routing gate and every writable/review lane. To add cross-model portability evidence, explicitly pass the other runner as `--portability-runner claude` or `--portability-runner codex`; it runs only the exact 46-case representative read-only set. The two runners must differ. Omitting the portability option is valid and is recorded as `portabilityRunner: null`; no secondary result is claimed. There is no per-case fallback or mixed primary ownership. Once requested, a portability failure or unavailable runner fails that extended release run.
|
|
10
10
|
|
|
11
11
|
`--prepare-targets` accepts only an absolute, new or empty regular directory outside the controller app. The controller must be a sanitized fresh scaffold: automatic preparation fails before copying when it finds `.env`, `.env.*` (except `.env.example`, `.env.sample`, and `.env.template`), credential files, or private-key files. Never use a configured development or production app as the controller. It copies the fresh scaffold once per catalog case whose `evaluationKind` is `implementation` or `regression`, while excluding `.git`, `node_modules`, build/cache/coverage output, `.ai/harness/results`, `.ai/reports`, and `.ai/framework-context`. Each target receives a guarded link to the controller's installed dependency tree. The OS sandbox resolves that link as read-only during both the writable model run and the target command gate. The release gate also hashes every dependency entry and regular-file body once before execution and once after the complete suite, and fails if any nested content or metadata changed. A generated `release-targets.json` records the local mapping.
|
|
12
12
|
|
|
@@ -34,17 +34,17 @@ For externally prepared apps, `--writable-targets /absolute/release-targets.json
|
|
|
34
34
|
}
|
|
35
35
|
```
|
|
36
36
|
|
|
37
|
-
The current catalog contains
|
|
37
|
+
The current catalog contains 203 cases, including 46 writable implementation/regression cases (22.7%). The command still derives all counts and case IDs from `cases.json`, `validators.json`, and `release-matrix.json`; those figures are documented release facts, not runner constants. The matrix keeps both supported runner model selectors, an exact all-case primary profile, an exact 46-case portability profile, and runner-neutral writable assignments. Run `yarn install-skills` first so the pinned external `om-code-review` skill and ownership evidence are present; the reusable local `om-judge-agent-session` skill ships with the scaffold. Before running a model or writing a fixture, the release command requires complete deterministic, primary live-routing, writable, trusted-oracle, target, generated-test, and generative-judge coverage. Every one of the 46 writable cases must have a judge assignment composing both skills. Missing business fixtures or release-matrix entries fail preflight and are listed by exact case ID in the report.
|
|
38
38
|
|
|
39
39
|
## PR #4529 remediation evidence
|
|
40
40
|
|
|
41
|
-
The PR's focused remediation evidence is not a release-certification substitute. Fresh emitted controllers pass deterministic 192/192, and the field-tested OMH-188–192 generative cohort passes on default Codex, Claude Sonnet, and high-effort gpt-5.4-mini. Fresh OMH-185 writable attempts fixed concrete organization-scope, command-object, module-activation, command-snapshot, schema, custom-field, UI, and Jest guidance defects at their routed owners without relaxing trusted oracles. The final attempt reached the case's fixed 600-second ceiling and is excluded from pass evidence. Issue #4670 now owns the complete selected-primary
|
|
41
|
+
The PR's focused remediation evidence is not a release-certification substitute. Fresh emitted controllers pass deterministic 192/192, and the field-tested OMH-188–192 generative cohort passes on default Codex, Claude Sonnet, and high-effort gpt-5.4-mini. Fresh OMH-185 writable attempts fixed concrete organization-scope, command-object, module-activation, command-snapshot, schema, custom-field, UI, and Jest guidance defects at their routed owners without relaxing trusted oracles. The final attempt reached the case's fixed 600-second ceiling and is excluded from pass evidence. Issue #4670 now owns the complete selected-primary 203-case routing and 46-case writable/generated-test/review certification, prioritizing the generative cohort and recording unavailable Claude lanes without fallback or mixed-runner ownership.
|
|
42
42
|
|
|
43
43
|
After preflight it runs, in order:
|
|
44
44
|
|
|
45
45
|
1. deterministic validation for the complete catalog;
|
|
46
46
|
2. the release matrix's fixed `yarn generate`, `yarn typecheck`, `yarn lint`, and `yarn build` foundation;
|
|
47
|
-
3. the selected primary runner across all
|
|
47
|
+
3. the selected primary runner across all 203 live-routing cases, followed by the optional distinct portability runner across the exact 46-case read-only sample when requested;
|
|
48
48
|
4. fixture preparation and the selected primary runner for every writable case, including the controller-owned AST/behavior oracles and target typecheck;
|
|
49
49
|
5. `yarn generate`, `yarn typecheck`, `yarn lint`, and `yarn build` in every writable target, after its trusted oracles;
|
|
50
50
|
6. real generated-code execution for OMH-163 and OMH-192 through fixed Jest, OMH-164 through API-only Playwright, and OMH-165 through real-browser Playwright; and
|
|
@@ -13586,5 +13586,95 @@
|
|
|
13586
13586
|
{"id":"OMH-199","title":"Load a supplier spreadsheet in bulk without a hand-written parser","family":"architecture","mode":"analysis","evaluationKind":"routing","risk":"medium","prompt":"In a freshly scaffolded standalone Open Mercato app, a supplier sends a monthly spreadsheet and back-office staff retype several hundred rows by hand. They want to upload that file themselves and have the rows loaded, with a record of who uploaded what and whether the load succeeded. Decide the smallest safe design against what the installed modules already provide, keep tenant and organization boundaries intact, state the access-control posture for viewing an upload versus running one, and identify the smallest relevant validation. Do not write a bespoke spreadsheet parser or upload endpoint before establishing whether an installed module already owns this capability.","tags":["architecture","module-facts","reuse-installed","bulk-import"],"owner":{"kind":"facts","path":".ai/guides/modules/sync_excel.md","ruleIds":["BC-01","BC-04"]},"expectedRouter":{"required":["architecture"],"allowedExtra":["integration","module-data","framework-context","umes"]},"requiredSkills":["om-integration-builder"],"context":{"required":["AGENTS.md",".ai/guides/architecture.md",".ai/skills/om-integration-builder/SKILL.md",".ai/guides/modules/sync_excel.md"],"allowedExtra":[".ai/guides/integrations.md",".ai/guides/contracts.md",".ai/guides/modules/data_sync.md",".ai/skills/om-help/SKILL.md"],"warn":["node_modules/@open-mercato/*/src/**"],"forbidden":[".env*",".git/**"]},"requiredDecisions":["facts-first","tenant-scope","acl-features","smallest-validation"],"forbiddenPatterns":["node_modules.{0,40}(?:write|edit|patch)","(?:tenant|organization).{0,30}(?:unscoped|scope optional)"],"validators":["catalog.schema","owner.reference","skills.reference","router.contract","context.budget","context.forbidden","patterns.forbidden"],"maxContextFiles":11,"maxInitialContextBytes":57344,"maxTotalContextBytes":147456,"relatedCases":["OMH-002","OMH-195"]},
|
|
13587
13587
|
{"id":"OMH-200","title":"Charge cards through the installed payment provider rather than a bespoke client","family":"architecture","mode":"analysis","evaluationKind":"routing","risk":"high","prompt":"In a freshly scaffolded standalone Open Mercato app, the business has a Stripe account and wants to start charging cards at checkout this week. Decide the smallest safe design against what the installed modules already provide, say where a payment provider belongs in a standalone app and how it is turned on for this one, state the access-control posture for viewing versus configuring it and how the provider credential is held, keep tenant and organization boundaries intact, and identify the smallest relevant validation. Do not write a bespoke provider client or a new payments module before establishing whether an installed module already owns this capability.","tags":["architecture","module-facts","reuse-installed","payment-provider"],"owner":{"kind":"facts","path":".ai/guides/modules/gateway_stripe.md","ruleIds":["BC-01","BC-04"]},"expectedRouter":{"required":["architecture"],"allowedExtra":["integration","module-data","framework-context","umes"]},"requiredSkills":["om-integration-builder"],"context":{"required":["AGENTS.md",".ai/guides/architecture.md",".ai/skills/om-integration-builder/SKILL.md",".ai/guides/modules/gateway_stripe.md"],"allowedExtra":[".ai/guides/integrations.md",".ai/guides/contracts.md",".ai/guides/modules/payment_gateways.md",".ai/skills/om-help/SKILL.md"],"warn":["node_modules/@open-mercato/*/src/**"],"forbidden":[".env*",".git/**"]},"requiredDecisions":["facts-first","app-module-activation","acl-features","smallest-validation"],"forbiddenPatterns":["node_modules.{0,40}(?:write|edit|patch)","(?:tenant|organization).{0,30}(?:unscoped|scope optional)"],"validators":["catalog.schema","owner.reference","skills.reference","router.contract","context.budget","context.forbidden","patterns.forbidden"],"maxContextFiles":11,"maxInitialContextBytes":57344,"maxTotalContextBytes":147456,"relatedCases":["OMH-002","OMH-195"]},
|
|
13588
13588
|
{"id":"OMH-201","title":"Pull product data from the PIM the business already runs without a new connector","family":"architecture","mode":"analysis","evaluationKind":"routing","risk":"medium","prompt":"In a freshly scaffolded standalone Open Mercato app, the merchandising team maintains product data in Akeneo and wants it to reach this application on a schedule instead of being copied over by hand. Decide the smallest safe design against what the installed modules already provide, say how that capability is turned on for this application, keep tenant and organization boundaries intact, and identify the smallest relevant validation. Do not design a bespoke connector, mapping layer, or import endpoint before establishing whether an installed module already owns this capability.","tags":["architecture","module-facts","reuse-installed","pim-sync"],"owner":{"kind":"facts","path":".ai/guides/modules/sync_akeneo.md","ruleIds":["BC-01","BC-04"]},"expectedRouter":{"required":["architecture"],"allowedExtra":["integration","module-data","framework-context","umes"]},"requiredSkills":["om-integration-builder"],"context":{"required":["AGENTS.md",".ai/guides/architecture.md",".ai/skills/om-integration-builder/SKILL.md",".ai/guides/modules/sync_akeneo.md"],"allowedExtra":[".ai/guides/integrations.md",".ai/guides/contracts.md",".ai/guides/modules/data_sync.md",".ai/skills/om-help/SKILL.md"],"warn":["node_modules/@open-mercato/*/src/**"],"forbidden":[".env*",".git/**"]},"requiredDecisions":["facts-first","app-module-activation","tenant-scope","smallest-validation"],"forbiddenPatterns":["node_modules.{0,40}(?:write|edit|patch)","(?:tenant|organization).{0,30}(?:unscoped|scope optional)"],"validators":["catalog.schema","owner.reference","skills.reference","router.contract","context.budget","context.forbidden","patterns.forbidden"],"maxContextFiles":11,"maxInitialContextBytes":57344,"maxTotalContextBytes":147456,"relatedCases":["OMH-002","OMH-195"]},
|
|
13589
|
-
{"id":"OMH-202","title":"Keep stock counts per warehouse honest and hold goods for confirmed orders without a new module","family":"architecture","mode":"analysis","evaluationKind":"routing","risk":"medium","prompt":"In a freshly scaffolded standalone Open Mercato app, the operations team keeps stock for two storage sites in a spreadsheet: how much of each product sits in which aisle, how much is already promised to confirmed orders, and what has been received, moved, or counted since the last check. They also want a heads-up when an item drops below the level they consider safe. Decide the smallest safe design against what the installed modules already provide, say how that capability is turned on for this application, keep tenant and organization boundaries intact, state the access-control posture for viewing counts versus adjusting them, and identify the smallest relevant validation. Do not model your own stock, location, or reservation tables before establishing whether an installed module already owns this capability.","tags":["architecture","module-facts","reuse-installed","stock-on-hand"],"owner":{"kind":"facts","path":".ai/guides/modules/wms.md","ruleIds":["BC-01","BC-04"]},"expectedRouter":{"required":["architecture"],"allowedExtra":["module-data","backend-ui","umes","framework-context"]},"requiredSkills":["om-help"],"context":{"required":["AGENTS.md",".ai/guides/architecture.md",".ai/skills/om-help/SKILL.md",".ai/guides/modules/wms.md"],"allowedExtra":[".ai/guides/contracts.md",".ai/guides/modules/catalog.md",".ai/guides/modules/sales.md",".ai/skills/om-data-model-design/SKILL.md"],"warn":["node_modules/@open-mercato/*/src/**"],"forbidden":[".env*",".git/**"]},"requiredDecisions":["facts-first","app-module-activation","tenant-scope","acl-features","smallest-validation"],"forbiddenPatterns":["node_modules.{0,40}(?:write|edit|patch)","(?:tenant|organization).{0,30}(?:unscoped|scope optional)"],"validators":["catalog.schema","owner.reference","skills.reference","router.contract","context.budget","context.forbidden","patterns.forbidden"],"maxContextFiles":11,"maxInitialContextBytes":57344,"maxTotalContextBytes":147456,"relatedCases":["OMH-002","OMH-194"]}
|
|
13589
|
+
{"id":"OMH-202","title":"Keep stock counts per warehouse honest and hold goods for confirmed orders without a new module","family":"architecture","mode":"analysis","evaluationKind":"routing","risk":"medium","prompt":"In a freshly scaffolded standalone Open Mercato app, the operations team keeps stock for two storage sites in a spreadsheet: how much of each product sits in which aisle, how much is already promised to confirmed orders, and what has been received, moved, or counted since the last check. They also want a heads-up when an item drops below the level they consider safe. Decide the smallest safe design against what the installed modules already provide, say how that capability is turned on for this application, keep tenant and organization boundaries intact, state the access-control posture for viewing counts versus adjusting them, and identify the smallest relevant validation. Do not model your own stock, location, or reservation tables before establishing whether an installed module already owns this capability.","tags":["architecture","module-facts","reuse-installed","stock-on-hand"],"owner":{"kind":"facts","path":".ai/guides/modules/wms.md","ruleIds":["BC-01","BC-04"]},"expectedRouter":{"required":["architecture"],"allowedExtra":["module-data","backend-ui","umes","framework-context"]},"requiredSkills":["om-help"],"context":{"required":["AGENTS.md",".ai/guides/architecture.md",".ai/skills/om-help/SKILL.md",".ai/guides/modules/wms.md"],"allowedExtra":[".ai/guides/contracts.md",".ai/guides/modules/catalog.md",".ai/guides/modules/sales.md",".ai/skills/om-data-model-design/SKILL.md"],"warn":["node_modules/@open-mercato/*/src/**"],"forbidden":[".env*",".git/**"]},"requiredDecisions":["facts-first","app-module-activation","tenant-scope","acl-features","smallest-validation"],"forbiddenPatterns":["node_modules.{0,40}(?:write|edit|patch)","(?:tenant|organization).{0,30}(?:unscoped|scope optional)"],"validators":["catalog.schema","owner.reference","skills.reference","router.contract","context.budget","context.forbidden","patterns.forbidden"],"maxContextFiles":11,"maxInitialContextBytes":57344,"maxTotalContextBytes":147456,"relatedCases":["OMH-002","OMH-194"]},
|
|
13590
|
+
{
|
|
13591
|
+
"id": "OMH-203",
|
|
13592
|
+
"title": "Evaluate CRM detail-tab UMES routing",
|
|
13593
|
+
"family": "umes",
|
|
13594
|
+
"mode": "analysis",
|
|
13595
|
+
"evaluationKind": "routing",
|
|
13596
|
+
"risk": "medium",
|
|
13597
|
+
"prompt": "How can I add a tab to the customer page?",
|
|
13598
|
+
"tags": [
|
|
13599
|
+
"umes",
|
|
13600
|
+
"backend-ui",
|
|
13601
|
+
"framework-context"
|
|
13602
|
+
],
|
|
13603
|
+
"owner": {
|
|
13604
|
+
"kind": "skill",
|
|
13605
|
+
"path": ".ai/skills/om-system-extension/SKILL.md",
|
|
13606
|
+
"ruleIds": [
|
|
13607
|
+
"BC-06"
|
|
13608
|
+
]
|
|
13609
|
+
},
|
|
13610
|
+
"expectedRouter": {
|
|
13611
|
+
"required": [
|
|
13612
|
+
"umes",
|
|
13613
|
+
"backend-ui",
|
|
13614
|
+
"framework-context"
|
|
13615
|
+
]
|
|
13616
|
+
},
|
|
13617
|
+
"requiredSkills": [
|
|
13618
|
+
"om-system-extension",
|
|
13619
|
+
"om-backend-ui-design",
|
|
13620
|
+
"om-framework-context"
|
|
13621
|
+
],
|
|
13622
|
+
"context": {
|
|
13623
|
+
"required": [
|
|
13624
|
+
"AGENTS.md",
|
|
13625
|
+
".ai/guides/extensions.md",
|
|
13626
|
+
".ai/guides/backend-ui.md",
|
|
13627
|
+
".ai/guides/modules/customers.md",
|
|
13628
|
+
".ai/skills/om-system-extension/SKILL.md",
|
|
13629
|
+
".ai/skills/om-backend-ui-design/SKILL.md",
|
|
13630
|
+
".ai/skills/om-framework-context/SKILL.md"
|
|
13631
|
+
],
|
|
13632
|
+
"allowedExtra": [
|
|
13633
|
+
".ai/skills/om-system-extension/references/extension-branches.md",
|
|
13634
|
+
".ai/skills/om-system-extension/references/mechanism-selector.md"
|
|
13635
|
+
],
|
|
13636
|
+
"forbidden": [
|
|
13637
|
+
".env*",
|
|
13638
|
+
".git/**",
|
|
13639
|
+
"node_modules/**"
|
|
13640
|
+
]
|
|
13641
|
+
},
|
|
13642
|
+
"requiredDecisions": [
|
|
13643
|
+
"extension-mechanism",
|
|
13644
|
+
"additive-before-replacement",
|
|
13645
|
+
"extension-entity",
|
|
13646
|
+
"eject-last",
|
|
13647
|
+
"widget-injection-files",
|
|
13648
|
+
"person-detail-tab-spot",
|
|
13649
|
+
"company-detail-tab-spot",
|
|
13650
|
+
"guidance-before-framework-context",
|
|
13651
|
+
"installed-packages-read-only"
|
|
13652
|
+
],
|
|
13653
|
+
"forbiddenPatterns": [
|
|
13654
|
+
"node_modules.{0,40}(?:write|edit|patch)",
|
|
13655
|
+
"(?:tenant|organization).{0,30}(?:unscoped|scope optional)"
|
|
13656
|
+
],
|
|
13657
|
+
"validators": [
|
|
13658
|
+
"catalog.schema",
|
|
13659
|
+
"owner.reference",
|
|
13660
|
+
"skills.reference",
|
|
13661
|
+
"router.contract",
|
|
13662
|
+
"context.budget",
|
|
13663
|
+
"context.forbidden",
|
|
13664
|
+
"patterns.forbidden"
|
|
13665
|
+
],
|
|
13666
|
+
"frameworkContext": [
|
|
13667
|
+
{
|
|
13668
|
+
"module": "customers",
|
|
13669
|
+
"query": "detail:customers."
|
|
13670
|
+
}
|
|
13671
|
+
],
|
|
13672
|
+
"maxContextFiles": 10,
|
|
13673
|
+
"maxInitialContextBytes": 57344,
|
|
13674
|
+
"maxTotalContextBytes": 131072,
|
|
13675
|
+
"relatedCases": [
|
|
13676
|
+
"OMH-026",
|
|
13677
|
+
"OMH-027"
|
|
13678
|
+
]
|
|
13679
|
+
}
|
|
13590
13680
|
]
|
|
@@ -3,8 +3,8 @@
|
|
|
3
3
|
"$id": "https://open-mercato.dev/schemas/standalone-harness-cases.schema.json",
|
|
4
4
|
"title": "Open Mercato standalone harness case catalog",
|
|
5
5
|
"type": "array",
|
|
6
|
-
"minItems":
|
|
7
|
-
"maxItems":
|
|
6
|
+
"minItems": 203,
|
|
7
|
+
"maxItems": 203,
|
|
8
8
|
"items": {
|
|
9
9
|
"type": "object",
|
|
10
10
|
"additionalProperties": false,
|
|
@@ -15,7 +15,7 @@
|
|
|
15
15
|
"maxTotalContextBytes", "relatedCases"
|
|
16
16
|
],
|
|
17
17
|
"properties": {
|
|
18
|
-
"id": { "type": "string", "pattern": "^OMH-(00[1-9]|0[1-9][0-9]|1[0-9][0-9]|20[0-
|
|
18
|
+
"id": { "type": "string", "pattern": "^OMH-(00[1-9]|0[1-9][0-9]|1[0-9][0-9]|20[0-3])$" },
|
|
19
19
|
"title": { "type": "string", "minLength": 12, "maxLength": 180 },
|
|
20
20
|
"family": { "enum": ["architecture", "module", "umes", "integration", "ai-workflow", "bugfix", "business", "testing"] },
|
|
21
21
|
"mode": { "enum": ["analysis", "one-shot", "spec", "bugfix", "review"] },
|
|
@@ -13,7 +13,7 @@ Route before reading: choose routes from the request and mechanism selector, the
|
|
|
13
13
|
|
|
14
14
|
1. Read `.ai/guides/extensions.md` and `references/mechanism-selector.md`; choose UMES, supported override, package, or eject.
|
|
15
15
|
2. Resolve host entity/route/spot/component/event IDs from generated facts. Invoke `om-framework-context` only when facts omit the needed contract.
|
|
16
|
-
3. Follow the
|
|
16
|
+
3. Follow the matching `references/extension-branches.md` branch. UI widgets live in `src/modules/<id>/widgets/injection/**` and register an exact facts- or context-resolved spot in `src/modules/<id>/widgets/injection-table.ts`. For `entry.overrides`, load `references/unified-overrides.md` and select the exact domain/key.
|
|
17
17
|
4. Invoke `om-data-model-design` only when the extension adds app-owned persistence; an enricher/interceptor/widget-only round trip does not need it.
|
|
18
18
|
5. For editable additions, follow `references/read-write-roundtrip.md`; implement input, authenticated write, stored data, list/detail read, UI hydration, clear-to-null, and conflict behavior. An editable addition that must survive reload is necessarily a persisted field: select `module-data`, read contracts, and invoke `om-data-model-design`.
|
|
19
19
|
6. Run `yarn generate`; verify host-present/absent, authorized/denied/wildcard, cache/search, and failure fallback using `references/verification.md`.
|
|
@@ -518,8 +518,8 @@ function validateCatalog({ root, cases, registry, releaseMatrix, fixtures, seeds
|
|
|
518
518
|
for (const related of item.relatedCases ?? []) if (!idSet.has(related)) add(id, `dangling related case ${related}`)
|
|
519
519
|
const writable = WRITABLE_KINDS.has(item.evaluationKind)
|
|
520
520
|
if (item.frameworkContext !== undefined) {
|
|
521
|
-
if (!
|
|
522
|
-
add(id, 'frameworkContext requires one to three
|
|
521
|
+
if (!Array.isArray(item.frameworkContext) || item.frameworkContext.length === 0 || item.frameworkContext.length > 3) {
|
|
522
|
+
add(id, 'frameworkContext requires one to three queries')
|
|
523
523
|
}
|
|
524
524
|
for (const request of item.frameworkContext ?? []) {
|
|
525
525
|
const selectors = [request?.module, request?.package].filter((value) => value !== undefined)
|
|
@@ -1420,7 +1420,16 @@ function prepareCaseFrameworkContext(caseRecord, controllerRoot, runRoot) {
|
|
|
1420
1420
|
throw new Error(`framework context queries resolved to the same output root for ${caseRecord.id}: ${outputRoot}`)
|
|
1421
1421
|
}
|
|
1422
1422
|
patterns.add(outputPattern)
|
|
1423
|
-
|
|
1423
|
+
const sourceFiles = [...new Set(fs.readFileSync(searchPath, 'utf8')
|
|
1424
|
+
.split(/\r?\n/)
|
|
1425
|
+
.flatMap((line) => {
|
|
1426
|
+
const match = /^(.+?):\d+:/.exec(line)
|
|
1427
|
+
return match && isSafeRelative(match[1]) && match[1].startsWith(`${sourceRoot}/`) ? [match[1]] : []
|
|
1428
|
+
}))].sort()
|
|
1429
|
+
if (sourceFiles.length === 0) {
|
|
1430
|
+
throw new Error(`framework context query returned no exact source matches for ${caseRecord.id}: ${request.query}`)
|
|
1431
|
+
}
|
|
1432
|
+
entries.push({ manifest, searchResult, sourceRoot, sourceFiles, query: request.query })
|
|
1424
1433
|
}
|
|
1425
1434
|
return { patterns: [...patterns].sort(), entries }
|
|
1426
1435
|
}
|
|
@@ -1515,6 +1524,7 @@ function observedContext(stdout, root, caseRecord, writable, reviewExpectedReads
|
|
|
1515
1524
|
for (const event of events) collectRefusedToolCallIds(event, state.refusedCallIds)
|
|
1516
1525
|
for (const event of events) recursivelyFindTraceCandidates(event, state)
|
|
1517
1526
|
const paths = new Set()
|
|
1527
|
+
const readOrder = []
|
|
1518
1528
|
const refusedReads = new Set()
|
|
1519
1529
|
const metadataPaths = new Set()
|
|
1520
1530
|
let metadataEntries = 0
|
|
@@ -1595,7 +1605,10 @@ function observedContext(stdout, root, caseRecord, writable, reviewExpectedReads
|
|
|
1595
1605
|
// Keep it out of instruction/fact budgets and selectedContext accounting while
|
|
1596
1606
|
// still failing closed above for every non-allowlisted read.
|
|
1597
1607
|
else if (isAllowedObservedPath(file, caseRecord, writable)) {
|
|
1598
|
-
if (!writable || reviewExpectedReads || permittedContextPath(file, caseRecord))
|
|
1608
|
+
if (!writable || reviewExpectedReads || permittedContextPath(file, caseRecord)) {
|
|
1609
|
+
paths.add(file)
|
|
1610
|
+
if (!readOrder.includes(file)) readOrder.push(file)
|
|
1611
|
+
}
|
|
1599
1612
|
}
|
|
1600
1613
|
else violations.add(`unsafe arbitrary app-root read ${file}`)
|
|
1601
1614
|
}
|
|
@@ -1610,6 +1623,7 @@ function observedContext(stdout, root, caseRecord, writable, reviewExpectedReads
|
|
|
1610
1623
|
return {
|
|
1611
1624
|
available: state.available,
|
|
1612
1625
|
paths: [...paths].sort(),
|
|
1626
|
+
readOrder,
|
|
1613
1627
|
refusedPaths: [...refusedReads].sort(),
|
|
1614
1628
|
metadataPaths: [...metadataPaths].sort(),
|
|
1615
1629
|
metadataEntries,
|
|
@@ -1666,7 +1680,7 @@ function contextStats(root, paths, metadata = {}) {
|
|
|
1666
1680
|
}
|
|
1667
1681
|
}
|
|
1668
1682
|
|
|
1669
|
-
function evaluateRouting(caseRecord, response, stats) {
|
|
1683
|
+
function evaluateRouting(caseRecord, response, stats, readOrder = []) {
|
|
1670
1684
|
const failures = []
|
|
1671
1685
|
const selectedRoutes = new Set(response.selectedRouter)
|
|
1672
1686
|
for (const required of caseRecord.expectedRouter.required) if (!selectedRoutes.has(required)) failures.push(`missing route ${required}`)
|
|
@@ -1709,6 +1723,19 @@ function evaluateRouting(caseRecord, response, stats) {
|
|
|
1709
1723
|
failures.push(`observed context not declared ${observed}`)
|
|
1710
1724
|
}
|
|
1711
1725
|
}
|
|
1726
|
+
if (caseRecord.materializedFrameworkContextPatterns?.length) {
|
|
1727
|
+
const frameworkReads = readOrder.filter((observed) =>
|
|
1728
|
+
caseRecord.materializedFrameworkContextPatterns.some((pattern) => globToRegExp(pattern).test(observed)))
|
|
1729
|
+
const sourceReads = frameworkReads.filter((observed) => observed.includes('/source/'))
|
|
1730
|
+
if (sourceReads.length === 0) failures.push('framework context source not observed')
|
|
1731
|
+
const firstFrameworkRead = frameworkReads.length ? readOrder.indexOf(frameworkReads[0]) : -1
|
|
1732
|
+
if (firstFrameworkRead >= 0) {
|
|
1733
|
+
for (const required of caseRecord.context.required.filter((reference) => !reference.startsWith('.ai/framework-context/'))) {
|
|
1734
|
+
const guidanceRead = readOrder.findIndex((observed) => globToRegExp(required).test(observed))
|
|
1735
|
+
if (guidanceRead > firstFrameworkRead) failures.push(`framework context read before required guidance ${required}`)
|
|
1736
|
+
}
|
|
1737
|
+
}
|
|
1738
|
+
}
|
|
1712
1739
|
const standardGuides = new Map()
|
|
1713
1740
|
for (const [route, standard] of Object.entries(ROUTE_STANDARD_CONTEXT)) {
|
|
1714
1741
|
for (const guide of standard.guides) {
|
|
@@ -1738,7 +1765,7 @@ function evaluateRouting(caseRecord, response, stats) {
|
|
|
1738
1765
|
}
|
|
1739
1766
|
|
|
1740
1767
|
function isCorrectableRoutingFailure(violation) {
|
|
1741
|
-
return /^(?:missing (?:route|skill|context|decision)|unexpected (?:route|skill|context|decision)|unmandated decision|selected skill context (?:missing|not observed)|optional skill .+ requires route|standard (?:skill|context) .+ requires route|required context not observed|selected context not observed|observed context not declared|initial context (?:file|byte) budget exceeded|context byte budget exceeded)/.test(violation)
|
|
1768
|
+
return /^(?:missing (?:route|skill|context|decision)|unexpected (?:route|skill|context|decision)|unmandated decision|selected skill context (?:missing|not observed)|optional skill .+ requires route|standard (?:skill|context) .+ requires route|required context not observed|selected context not observed|observed context not declared|framework context (?:source not observed|read before required guidance)|initial context (?:file|byte) budget exceeded|context byte budget exceeded)/.test(violation)
|
|
1742
1769
|
}
|
|
1743
1770
|
|
|
1744
1771
|
function isCorrectableTraceStartupFailure(violation) {
|
|
@@ -1800,7 +1827,7 @@ function buildPrompt(caseRecord, root, writable, frameworkContextEntries = []) {
|
|
|
1800
1827
|
? 'This is an explicitly disposable writable evaluation. You must implement the complete request with repeated use of the allowlisted harness write tool before returning the routing object, then re-read the completed implementation. A manifest of intended files, metadata-only stub, TODO, placeholder, or response that only plans/routes fails: create every requested source, API, command, UI, registration, locale, migration snapshot, and focused test surface inside the write allowlist. You may read allowlisted target source files as implementation inputs, but selectedContext records only instruction/fact paths and must never include those target source paths. Write only inside the allowlist provided after the task; do not use network access or inspect environment values.'
|
|
1801
1828
|
: 'Work read-only: do not edit files, run mutations, use network access, or inspect environment values. Do not implement the request. Accept the task premise for routing; fixture and implementation-target paths may be absent in this controller, so do not inspect them or report their absence as a blocker.'
|
|
1802
1829
|
const frameworkContextInstruction = frameworkContextEntries.length
|
|
1803
|
-
? ` The controller has already materialized bounded read-only installed-package evidence for this case. Read
|
|
1830
|
+
? ` The controller has already materialized bounded read-only installed-package evidence for this case. Read every required routed guide and skill before this evidence, then read each exact manifest and search result and every exact source path named by those search results beneath the supplied source root: ${frameworkContextEntries.map((entry) => `manifest=${entry.manifest}; search=${entry.searchResult}; source=${entry.sourceRoot}; query=${JSON.stringify(entry.query)}`).join(' | ')}. You may also follow an exact installed-source link from a generated module fact when it is useful; never enumerate node_modules or read outside @open-mercato package src trees. These successful reads belong in selectedContext.`
|
|
1804
1831
|
: ''
|
|
1805
1832
|
const toolInstruction = `The only MCP server is harness. Its exact-path tool is named read and takes JSON arguments {"path":"<exact app-relative path>"}${writable ? '; its allowlisted replacement tool is named write and takes {"path":"<exact app-relative path>","content":"<complete file>"}' : ''}. Call harness.read${writable ? ' or harness.write' : ''}; never call read_mcp_resource or any resource API.`
|
|
1806
1833
|
return `You are evaluating routing for a standalone Open Mercato application. ${modeInstruction}${frameworkContextInstruction} ${toolInstruction} No shell, process, environment, discovery, or network tool exists. Your first tool action must call harness.read with {"path":"AGENTS.md"}, even when the runner auto-injected it, then load only the smallest task-matching context. Do not execute framework-context, generation, test, package, release, installer, or skill workflow commands during routing; select the instructions that would govern that later execution. Do not emit a provisional structured response. Do not inspect .ai/harness/**; those are evaluator internals, and decision labels never name readable paths. Never enumerate, glob, recursively search, or bulk-read .ai/guides, .ai/skills, .agents/skills, or module fact directories; an all-guides/all-skills/all-facts read is an automatic failure. The emitted AGENTS.md and the context it routes are the only task-routing authority. Before the final response, open every instruction or fact path you will put in selectedContext with direct harness read calls. Never rely only on skill descriptions, filenames, discovery, metadata, or prior knowledge: an unobserved selected path automatically fails this evaluation.
|
|
@@ -2705,9 +2732,7 @@ function liveRun({ options, selected, registry, releaseMatrix, fixtures, root, h
|
|
|
2705
2732
|
const targetErrors = verifyWritableTarget(runRoot, root, caseRecord, fixtures)
|
|
2706
2733
|
if (targetErrors.length) throw new Error(`${caseRecord.id}: ${targetErrors.join('; ')}`)
|
|
2707
2734
|
}
|
|
2708
|
-
const preparedFrameworkContext =
|
|
2709
|
-
? prepareCaseFrameworkContext(caseRecord, root, runRoot)
|
|
2710
|
-
: { patterns: [], entries: [] }
|
|
2735
|
+
const preparedFrameworkContext = prepareCaseFrameworkContext(caseRecord, root, runRoot)
|
|
2711
2736
|
const evaluationCase = preparedFrameworkContext.patterns.length
|
|
2712
2737
|
? {
|
|
2713
2738
|
...caseRecord,
|
|
@@ -2715,7 +2740,7 @@ function liveRun({ options, selected, registry, releaseMatrix, fixtures, root, h
|
|
|
2715
2740
|
...caseRecord.context,
|
|
2716
2741
|
required: [
|
|
2717
2742
|
...caseRecord.context.required,
|
|
2718
|
-
...preparedFrameworkContext.entries.flatMap(({ manifest, searchResult }) => [manifest, searchResult]),
|
|
2743
|
+
...preparedFrameworkContext.entries.flatMap(({ manifest, searchResult, sourceFiles }) => [manifest, searchResult, ...sourceFiles]),
|
|
2719
2744
|
],
|
|
2720
2745
|
},
|
|
2721
2746
|
materializedFrameworkContextPatterns: preparedFrameworkContext.patterns,
|
|
@@ -2771,7 +2796,7 @@ function liveRun({ options, selected, registry, releaseMatrix, fixtures, root, h
|
|
|
2771
2796
|
const declared = response.selectedContext
|
|
2772
2797
|
.filter((entry) => isSafeRelative(entry) && isPathInside(runRoot, path.resolve(runRoot, entry)) && fs.existsSync(path.resolve(runRoot, entry)))
|
|
2773
2798
|
declaredStats = contextStats(runRoot, [...new Set(declared)].sort())
|
|
2774
|
-
violations.push(...evaluateRouting(evaluationCase, response, stats))
|
|
2799
|
+
violations.push(...evaluateRouting(evaluationCase, response, stats, trace.readOrder))
|
|
2775
2800
|
} else violations.push(`${attempt.kind}: ${sanitize(attempt.error, runRoot)}`)
|
|
2776
2801
|
return { response, trace, stats, declaredStats, violations }
|
|
2777
2802
|
}
|