create-mercato-app 0.7.1-develop.7175.1.d49ab48ee2 → 0.7.1-develop.7177.1.9827d081d4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/agentic/shared/ai/harness/README.md +2 -2
- package/agentic/shared/ai/harness/RELEASE.md +5 -5
- package/agentic/shared/ai/harness/cases.json +167 -0
- package/agentic/shared/ai/harness/cases.schema.json +4 -4
- package/agentic/shared/ai/harness/validators.json +1 -1
- package/agentic/shared/scripts/run-agent-harness-release.mjs +1 -1
- package/dist/agentic/guides/module-facts.json +1049 -132
- package/dist/agentic/guides/module-facts.v2.json +1111 -132
- package/dist/agentic/guides/modules/ai_assistant/index.md +1 -1
- package/dist/agentic/guides/modules/api_docs/index.md +1 -1
- package/dist/agentic/guides/modules/api_keys/index.md +1 -1
- package/dist/agentic/guides/modules/attachments/index.md +1 -1
- package/dist/agentic/guides/modules/audit_logs/index.md +1 -1
- package/dist/agentic/guides/modules/auth/index.md +1 -1
- package/dist/agentic/guides/modules/business_rules/index.md +1 -1
- package/dist/agentic/guides/modules/catalog/index.md +1 -1
- package/dist/agentic/guides/modules/channel_apns/index.md +1 -1
- package/dist/agentic/guides/modules/channel_discord/index.md +1 -1
- package/dist/agentic/guides/modules/channel_expo/index.md +1 -1
- package/dist/agentic/guides/modules/channel_fcm/index.md +1 -1
- package/dist/agentic/guides/modules/channel_gmail/index.md +1 -1
- package/dist/agentic/guides/modules/channel_imap/index.md +1 -1
- package/dist/agentic/guides/modules/channel_resend/index.md +1 -1
- package/dist/agentic/guides/modules/channel_ses/index.md +1 -1
- package/dist/agentic/guides/modules/checkout/index.md +1 -1
- package/dist/agentic/guides/modules/communication_channels/index.md +1 -1
- package/dist/agentic/guides/modules/configs/index.md +1 -1
- package/dist/agentic/guides/modules/content/index.md +1 -1
- package/dist/agentic/guides/modules/currencies/index.md +1 -1
- package/dist/agentic/guides/modules/customer_accounts/index.md +1 -1
- package/dist/agentic/guides/modules/customers/index.md +1 -1
- package/dist/agentic/guides/modules/dashboards/index.md +1 -1
- package/dist/agentic/guides/modules/data_sync/index.md +1 -1
- package/dist/agentic/guides/modules/design_system/index.md +1 -1
- package/dist/agentic/guides/modules/devices/index.md +1 -1
- package/dist/agentic/guides/modules/dictionaries/index.md +1 -1
- package/dist/agentic/guides/modules/directory/index.md +1 -1
- package/dist/agentic/guides/modules/documents/index.md +1 -1
- package/dist/agentic/guides/modules/entities/index.md +1 -1
- package/dist/agentic/guides/modules/eudr/index.md +1 -1
- package/dist/agentic/guides/modules/events/index.md +1 -1
- package/dist/agentic/guides/modules/feature_toggles/index.md +1 -1
- package/dist/agentic/guides/modules/gateway_stripe/index.md +1 -1
- package/dist/agentic/guides/modules/generators/index.md +1 -1
- package/dist/agentic/guides/modules/inbox_ops/index.md +1 -1
- package/dist/agentic/guides/modules/integrations/index.md +1 -1
- package/dist/agentic/guides/modules/messages/index.md +1 -1
- package/dist/agentic/guides/modules/notifications/index.md +1 -1
- package/dist/agentic/guides/modules/onboarding/index.md +1 -1
- package/dist/agentic/guides/modules/payment_gateways/index.md +1 -1
- package/dist/agentic/guides/modules/perspectives/index.md +1 -1
- package/dist/agentic/guides/modules/phone_calls/acl-features.md +12 -0
- package/dist/agentic/guides/modules/phone_calls/backend-pages.md +11 -0
- package/dist/agentic/guides/modules/phone_calls/di-registrations-rich.md +12 -0
- package/dist/agentic/guides/modules/phone_calls/domain-commands.md +11 -0
- package/dist/agentic/guides/modules/phone_calls/encryption.md +12 -0
- package/dist/agentic/guides/modules/phone_calls/entities.md +12 -0
- package/dist/agentic/guides/modules/phone_calls/events.md +13 -0
- package/dist/agentic/guides/modules/phone_calls/exact-override-targets.md +18 -0
- package/dist/agentic/guides/modules/phone_calls/index.md +22 -0
- package/dist/agentic/guides/modules/phone_calls/owned-contract-module-metadata.md +11 -0
- package/dist/agentic/guides/modules/phone_calls/setup.md +11 -0
- package/dist/agentic/guides/modules/phone_calls/umes-hosts.md +15 -0
- package/dist/agentic/guides/modules/planner/index.md +1 -1
- package/dist/agentic/guides/modules/portal/index.md +1 -1
- package/dist/agentic/guides/modules/progress/index.md +1 -1
- package/dist/agentic/guides/modules/push_notifications/index.md +1 -1
- package/dist/agentic/guides/modules/query_index/index.md +1 -1
- package/dist/agentic/guides/modules/record_locks/index.md +1 -1
- package/dist/agentic/guides/modules/resources/index.md +1 -1
- package/dist/agentic/guides/modules/sales/index.md +1 -1
- package/dist/agentic/guides/modules/scheduler/index.md +1 -1
- package/dist/agentic/guides/modules/search/index.md +1 -1
- package/dist/agentic/guides/modules/security/index.md +1 -1
- package/dist/agentic/guides/modules/shipping_carriers/index.md +1 -1
- package/dist/agentic/guides/modules/sso/index.md +1 -1
- package/dist/agentic/guides/modules/staff/index.md +1 -1
- package/dist/agentic/guides/modules/storage_s3/index.md +1 -1
- package/dist/agentic/guides/modules/sync_akeneo/index.md +1 -1
- package/dist/agentic/guides/modules/sync_excel/index.md +1 -1
- package/dist/agentic/guides/modules/system_status_overlays/index.md +1 -1
- package/dist/agentic/guides/modules/tillio/acl-features.md +11 -0
- package/dist/agentic/guides/modules/tillio/cli-commands.md +11 -0
- package/dist/agentic/guides/modules/tillio/contribution-resolutions.md +12 -0
- package/dist/agentic/guides/modules/tillio/di-registrations-rich.md +11 -0
- package/dist/agentic/guides/modules/tillio/di-service-tokens.md +11 -0
- package/dist/agentic/guides/modules/tillio/exact-override-targets.md +17 -0
- package/dist/agentic/guides/modules/tillio/index.md +21 -0
- package/dist/agentic/guides/modules/tillio/owned-contract-module-metadata.md +11 -0
- package/dist/agentic/guides/modules/tillio/setup.md +11 -0
- package/dist/agentic/guides/modules/tillio/umes-contributions.md +12 -0
- package/dist/agentic/guides/modules/tillio/workers.md +11 -0
- package/dist/agentic/guides/modules/translations/index.md +1 -1
- package/dist/agentic/guides/modules/warranty_claims/index.md +1 -1
- package/dist/agentic/guides/modules/webhooks/index.md +1 -1
- package/dist/agentic/guides/modules/wms/index.md +1 -1
- package/dist/agentic/guides/modules/workflows/index.md +1 -1
- package/dist/agentic/guides/reference-module-facts.json +1 -1
- package/dist/agentic/guides/upstream/manifest.json +1 -1
- package/dist/agentic/shared/ai/harness/README.md +2 -2
- package/dist/agentic/shared/ai/harness/RELEASE.md +5 -5
- package/dist/agentic/shared/ai/harness/cases.json +167 -0
- package/dist/agentic/shared/ai/harness/cases.schema.json +4 -4
- package/dist/agentic/shared/ai/harness/validators.json +1 -1
- package/dist/agentic/shared/scripts/run-agent-harness-release.mjs +1 -1
- package/package.json +3 -3
- package/template/.env.example +14 -0
- package/template/package.json.template +1 -0
- package/template/src/modules.ts +2 -0
package/README.md
CHANGED
|
@@ -152,7 +152,7 @@ yarn install-skills
|
|
|
152
152
|
yarn harness:release --runner codex --prepare-targets /absolute/empty-release-targets --acknowledge-writes
|
|
153
153
|
```
|
|
154
154
|
|
|
155
|
-
The target directory must be absolute, new or empty, and outside the controller app. Select one blocking primary runner with `--runner codex` or `--runner claude`; it owns all
|
|
155
|
+
The target directory must be absolute, new or empty, and outside the controller app. Select one blocking primary runner with `--runner codex` or `--runner claude`; it owns all 236 routing cases and every writable/review lane, with no per-case fallback. Optionally add the different authenticated runner through `--portability-runner` for the exact 49-case representative read-only lane. Omitting it is valid and recorded as not requested; once requested, its failures are blocking. Use a fresh, sanitized controller: automatic preparation fails before copying `.env`/`.env.*` local configuration (safe example/sample/template files remain allowed), credential files, or private-key files. The complete gate requires Linux with trusted system Bubblewrap (`bwrap`) and user namespaces because its Playwright API/browser lanes need a loopback namespace isolated from the host. Preflight rejects untrusted/no-op/pass-through executables and proves isolated loopback plus a capability-free payload before target preparation, provider invocation, or writes; native macOS and Windows therefore fail closed. The command also fails closed when a required runner, browser, or test runtime is unavailable. The 236-case catalog includes 93 framework-neutral business prompts and 49 writable implementation/regression cases (20.8%). The release command runs live routing, writable trusted oracles, per-target `generate`/`typecheck`/`lint`/`build`, any declared generated test, and isolated generated-code review for every writable result. Foundation and target validation—including `yarn build`—receive a minimal environment with network access denied, and persisted diagnostics redact sensitive environment values and URL userinfo. Test-authoring coverage executes a Jest unit test plus Linux/Bubblewrap loopback-only Playwright API and browser tests through fixed controller-owned commands against a read-only target; runtime reports must attest at least one passed test and zero skipped, todo, focused, flaky, or expected-failure tests. The suite then writes a schema-valid sanitized mode-`0600` report under `.ai/harness/results/` with the selected primary and optional portability runner policy.
|
|
156
156
|
|
|
157
157
|
Use the bundled `om-evolve-harness` skill to add a real case: reproduce failure first, select one smallest knowledge owner, run any generated unit/integration tests plus target checks, require code review, and finish with the full release suite. Open Mercato framework maintainers use the monorepo-only `$om-refresh-standalone-harness --from <ref> --to <ref>` workflow for every release range and retain its sanitized maintenance report.
|
|
158
158
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Agent harness evaluations
|
|
2
2
|
|
|
3
|
-
`cases.json` is the
|
|
3
|
+
`cases.json` is the 236-case standalone-app contract. Run `yarn harness:validate --all` for the deterministic gate. Live routing uses a fresh read-only process per case:
|
|
4
4
|
|
|
5
5
|
UMES routing is fact-first. The additive and unified-override audit evaluations, plus their targeted cases, resolve exact/pattern hosts, outgoing contributions, correlation provenance, round-trip groups, framework-owned targets, and override domain/key/mode from the generated module sheets and `.ai/guides/framework-extension-points.md` before bounded installed source. The repository UMES umbrella spec may appear as optional source-checkout provenance; it is never required in a standalone scaffold.
|
|
6
6
|
|
|
@@ -18,7 +18,7 @@ yarn harness:release --runner codex --prepare-targets /absolute/empty-release-ta
|
|
|
18
18
|
yarn harness:release --runner codex --portability-runner claude --prepare-targets /absolute/empty-release-targets --acknowledge-writes
|
|
19
19
|
```
|
|
20
20
|
|
|
21
|
-
The primary runner owns all
|
|
21
|
+
The primary runner owns all 236 routing cases, all 49 writable cases, and all generative-judge runs. No per-case fallback or mixed primary ownership is allowed. Omitting `--portability-runner` is valid and the sanitized report records `portabilityRunner: null`; explicitly requesting an unavailable or failing secondary runner fails that extended run.
|
|
22
22
|
|
|
23
23
|
Writable evaluation is intentionally opt-in. The expanded catalog has a 49-case writable release target, but only cases registered in `release-matrix.json` and backed by controller-owned fixtures and oracles are executable. Copy or create a fresh standalone app for one registered case, then seed only that case and mark the target disposable:
|
|
24
24
|
|
|
@@ -6,7 +6,7 @@ Run the complete per-release gate from a generated standalone app with one comma
|
|
|
6
6
|
yarn harness:release --runner codex --prepare-targets /absolute/empty-release-targets --acknowledge-writes
|
|
7
7
|
```
|
|
8
8
|
|
|
9
|
-
Choose exactly one blocking primary runner with `--runner codex` or `--runner claude`. That runner owns the complete
|
|
9
|
+
Choose exactly one blocking primary runner with `--runner codex` or `--runner claude`. That runner owns the complete 236-case routing gate and every writable/review lane. To add cross-model portability evidence, explicitly pass the other runner as `--portability-runner claude` or `--portability-runner codex`; it runs only the exact 49-case representative read-only set. The two runners must differ. Omitting the portability option is valid and is recorded as `portabilityRunner: null`; no secondary result is claimed. There is no per-case fallback or mixed primary ownership. Once requested, a portability failure or unavailable runner fails that extended release run.
|
|
10
10
|
|
|
11
11
|
`--prepare-targets` accepts only an absolute, new or empty regular directory outside the controller app. The controller must be a sanitized fresh scaffold: automatic preparation fails before copying when it finds `.env`, `.env.*` (except `.env.example`, `.env.sample`, and `.env.template`), credential files, or private-key files. Never use a configured development or production app as the controller. It copies the fresh scaffold once per catalog case whose `evaluationKind` is `implementation` or `regression`, while excluding `.git`, `node_modules`, build/cache/coverage output, `.ai/harness/results`, `.ai/reports`, and `.ai/framework-context`. Each target receives a guarded link to the controller's installed dependency tree. The OS sandbox resolves that link as read-only during both the writable model run and the target command gate. The release gate also hashes every dependency entry and regular-file body once before execution and once after the complete suite, and fails if any nested content or metadata changed. A generated `release-targets.json` records the local mapping.
|
|
12
12
|
|
|
@@ -34,17 +34,17 @@ For externally prepared apps, `--writable-targets /absolute/release-targets.json
|
|
|
34
34
|
}
|
|
35
35
|
```
|
|
36
36
|
|
|
37
|
-
The current catalog contains
|
|
37
|
+
The current catalog contains 236 cases, including 49 writable implementation/regression cases (20.8%). The command still derives all counts and case IDs from `cases.json`, `validators.json`, and `release-matrix.json`; those figures are documented release facts, not runner constants. The matrix keeps both supported runner model selectors, an exact all-case primary profile, an exact 49-case portability profile, and runner-neutral writable assignments. Run `yarn install-skills` first so the pinned external `om-code-review` skill and ownership evidence are present; the reusable local `om-judge-agent-session` skill ships with the scaffold. Before running a model or writing a fixture, the release command requires complete deterministic, primary live-routing, writable, trusted-oracle, target, generated-test, and generative-judge coverage. Every one of the 49 writable cases must have a judge assignment composing both skills. Missing business fixtures or release-matrix entries fail preflight and are listed by exact case ID in the report.
|
|
38
38
|
|
|
39
39
|
## PR #4529 remediation evidence
|
|
40
40
|
|
|
41
|
-
The PR's focused remediation evidence is not a release-certification substitute. Fresh emitted controllers pass deterministic 192/192, and the field-tested OMH-188–192 generative cohort passes on default Codex, Claude Sonnet, and high-effort gpt-5.4-mini. Fresh OMH-185 writable attempts fixed concrete organization-scope, command-object, module-activation, command-snapshot, schema, custom-field, UI, and Jest guidance defects at their routed owners without relaxing trusted oracles. The final attempt reached the case's fixed 600-second ceiling and is excluded from pass evidence. Issue #4670 now owns the complete selected-primary
|
|
41
|
+
The PR's focused remediation evidence is not a release-certification substitute. Fresh emitted controllers pass deterministic 192/192, and the field-tested OMH-188–192 generative cohort passes on default Codex, Claude Sonnet, and high-effort gpt-5.4-mini. Fresh OMH-185 writable attempts fixed concrete organization-scope, command-object, module-activation, command-snapshot, schema, custom-field, UI, and Jest guidance defects at their routed owners without relaxing trusted oracles. The final attempt reached the case's fixed 600-second ceiling and is excluded from pass evidence. Issue #4670 now owns the complete selected-primary 236-case routing and 49-case writable/generated-test/review certification, prioritizing the generative cohort and recording unavailable Claude lanes without fallback or mixed-runner ownership.
|
|
42
42
|
|
|
43
43
|
After preflight it runs, in order:
|
|
44
44
|
|
|
45
45
|
1. deterministic validation for the complete catalog;
|
|
46
46
|
2. the release matrix's fixed `yarn generate`, `yarn typecheck`, `yarn lint`, and `yarn build` foundation;
|
|
47
|
-
3. the selected primary runner across all
|
|
47
|
+
3. the selected primary runner across all 236 live-routing cases, followed by the optional distinct portability runner across the exact 49-case read-only sample when requested;
|
|
48
48
|
4. fixture preparation and the selected primary runner for every writable case, including the controller-owned AST/behavior oracles and target typecheck;
|
|
49
49
|
5. `yarn generate`, `yarn typecheck`, `yarn lint`, and `yarn build` in every writable target, after its trusted oracles;
|
|
50
50
|
6. real generated-code execution for OMH-163 and OMH-192 through fixed Jest, OMH-164 through API-only Playwright, and OMH-165 through real-browser Playwright; and
|
|
@@ -54,7 +54,7 @@ Each writable target is single-use because fixture preparation marks it disposab
|
|
|
54
54
|
|
|
55
55
|
Routing cases carry no case-local duration budget, and that is a decision rather than an omission. `maxContextFiles` and the byte budgets measure the agent's context discipline, which is intrinsic to the case and portable between machines; duration measures the runner and the attempt count, which are not — the audited cohort spanned 71 s to 231 s for passing routing runs, one case measured 147 s and 132 s on two runs of the same model, and OMH-139 exhausted the evaluator's 300000 ms default outright. The operator budget carries that variance instead of the catalog. On this release path the lever is `--case-timeout` (default 600000 ms), which the release command passes on to the evaluator explicitly for every routing step; because that pass-through marks the timeout explicit, the evaluator's own runner-aware floors — 600000 ms for Claude, 900000 ms for a Codex `gpt-5.4-mini` high-effort run, 300000 ms for every other runner — apply only to direct `evaluate-agent-harness.mjs` invocations and never fire under `yarn harness:release`. Keeping that pass-through is deliberate rather than inherited: the routing step derives its own process budget from the same value, as slack plus the sum of the per-case ceilings it hands out, so letting the floors raise the inner ceiling while the outer budget still followed `--case-timeout` would kill an entire routing step instead of failing one slow case. Leaving it explicit also keeps the budget runner-independent, which is what lets the primary and portability lanes in one report be compared as models rather than as budgets. The default clears the slowest audited passing routing run — 231 s against 600000 ms, about 62% headroom — and matches both the Claude floor and the `timeoutMs` ceiling `cases.schema.json` enforces, so the gate carries one upper number instead of three; lower `--case-timeout` when a hung case should fail sooner, and note that the step's process budget drops with it — though at a lowered budget the declared writable ceilings keep their own slots, so the step's budget drops less than proportionally. That value was chosen from evaluator-path measurements — the audited cohort above and the live evidence recorded on #5068 — rather than from a driven `yarn harness:release` run, because the complete gate fails closed off the Linux-with-Bubblewrap host stated above; #5078 records the decision and that deviation, and #5433 carries the unmet measurement forward so the default is confirmed or corrected against the release path's own numbers. `--case-timeout` is one budget for three lanes, not a routing-only lever: the writable and review lanes resolve their own ceilings from the same value, so raising it raises theirs too. A declared writable `timeoutMs` is combined with whichever operator value applies as a maximum, so it raises the floor and never lowers it — but because the schema caps it at the shipped default, it can only raise a budget an operator has lowered; routing cases declare none, so there the operator value always stands alone. Writable cases keep their own `timeoutMs` because a writable one-shot's cost is dominated by the slice it must produce, which the case does define.
|
|
56
56
|
|
|
57
|
-
`--case-timeout` governs the model lanes and only those: routing, writable, and the judge invocation that reviews a writable result. The two steps that invoke no model carry their own flat ceilings instead. Fixture preparation has always used 120000 ms, and the deterministic step now uses the exported `DETERMINISTIC_STEP_TIMEOUT_MS`, the same 120000 ms, rather than the per-model ceiling multiplied by catalog size it derived before. That earlier derivation handed a model-free step a budget that moved whenever an operator changed how patient the gate is with a language model, and it grew that step's ceiling by a factor of five when #5180 raised the `--case-timeout` default from 120000 ms to 600000 ms — roughly 39 hours over the shipped catalog for a pass that finishes in under a second. The replacement is measured rather than chosen freehand: timed on Linux x86_64, a complete
|
|
57
|
+
`--case-timeout` governs the model lanes and only those: routing, writable, and the judge invocation that reviews a writable result. The two steps that invoke no model carry their own flat ceilings instead. Fixture preparation has always used 120000 ms, and the deterministic step now uses the exported `DETERMINISTIC_STEP_TIMEOUT_MS`, the same 120000 ms, rather than the per-model ceiling multiplied by catalog size it derived before. That earlier derivation handed a model-free step a budget that moved whenever an operator changed how patient the gate is with a language model, and it grew that step's ceiling by a factor of five when #5180 raised the `--case-timeout` default from 120000 ms to 600000 ms — roughly 39 hours over the shipped catalog for a pass that finishes in under a second. The replacement is measured rather than chosen freehand: timed on Linux x86_64, a complete deterministic pass finished in 768 ms to 998 ms over five runs, and narrowing the selection down to a single case measured 842 ms to 890 ms — inside the same spread, so the pass is dominated by process start and catalog load rather than by how many cases it validates. A ceiling roughly 120 times the slowest observed pass therefore bounds a hang without ever bounding a healthy run, and it has no reason to scale with the catalog. `packages/create-app/src/lib/agent-harness-release.test.ts` pins the deterministic argv together with that budget, so the model ceiling and the model-free one cannot silently converge again.
|
|
58
58
|
|
|
59
59
|
UI-routed implementation reviews receive only the bounded backend UI guide and `om-backend-ui-design` design-system references. Non-UI reviews do not receive that extra context.
|
|
60
60
|
|
|
@@ -21185,5 +21185,172 @@
|
|
|
21185
21185
|
"OMH-194",
|
|
21186
21186
|
"OMH-210"
|
|
21187
21187
|
]
|
|
21188
|
+
},
|
|
21189
|
+
{
|
|
21190
|
+
"id": "OMH-235",
|
|
21191
|
+
"title": "Keep a searchable history of phone calls without a bespoke call table",
|
|
21192
|
+
"family": "architecture",
|
|
21193
|
+
"mode": "analysis",
|
|
21194
|
+
"evaluationKind": "routing",
|
|
21195
|
+
"risk": "high",
|
|
21196
|
+
"prompt": "In a freshly scaffolded standalone Open Mercato app, support wants every inbound and outbound phone call kept so an agent can search the history by number, direction and outcome and open a recording where one exists. Before proposing any schema, establish whether an installed module already owns call records and name it, or state positively that none does. Then decide the smallest safe design against what the installed modules already provide: which record holds a call and which holds the people on it, which call fields are held encrypted and what reading them back therefore requires, and how a call reaches the application given that no screen in this module creates one. Keep tenant and organization boundaries intact, state the access-control posture for reading calls versus managing ingestion, and identify the smallest relevant validation. Do not design your own calls table, participant table, or recording store before that check is done.",
|
|
21197
|
+
"tags": [
|
|
21198
|
+
"architecture",
|
|
21199
|
+
"module-facts",
|
|
21200
|
+
"reuse-installed",
|
|
21201
|
+
"call-history",
|
|
21202
|
+
"encrypted-pii"
|
|
21203
|
+
],
|
|
21204
|
+
"owner": {
|
|
21205
|
+
"kind": "facts",
|
|
21206
|
+
"path": ".ai/guides/modules/phone_calls/index.md",
|
|
21207
|
+
"ruleIds": [
|
|
21208
|
+
"BC-01",
|
|
21209
|
+
"BC-04"
|
|
21210
|
+
]
|
|
21211
|
+
},
|
|
21212
|
+
"expectedRouter": {
|
|
21213
|
+
"required": [
|
|
21214
|
+
"architecture"
|
|
21215
|
+
],
|
|
21216
|
+
"allowedExtra": [
|
|
21217
|
+
"integration",
|
|
21218
|
+
"module-data",
|
|
21219
|
+
"umes",
|
|
21220
|
+
"framework-context"
|
|
21221
|
+
]
|
|
21222
|
+
},
|
|
21223
|
+
"requiredSkills": [
|
|
21224
|
+
"om-help"
|
|
21225
|
+
],
|
|
21226
|
+
"context": {
|
|
21227
|
+
"required": [
|
|
21228
|
+
"AGENTS.md",
|
|
21229
|
+
".ai/guides/architecture.md",
|
|
21230
|
+
".ai/skills/om-help/SKILL.md",
|
|
21231
|
+
".ai/guides/modules/phone_calls/index.md"
|
|
21232
|
+
],
|
|
21233
|
+
"allowedExtra": [
|
|
21234
|
+
".ai/guides/integrations.md",
|
|
21235
|
+
".ai/guides/contracts.md",
|
|
21236
|
+
".ai/guides/modules/tillio/index.md"
|
|
21237
|
+
],
|
|
21238
|
+
"forbidden": [
|
|
21239
|
+
".env*",
|
|
21240
|
+
".git/**"
|
|
21241
|
+
]
|
|
21242
|
+
},
|
|
21243
|
+
"requiredDecisions": [
|
|
21244
|
+
"facts-first",
|
|
21245
|
+
"tenant-scope",
|
|
21246
|
+
"acl-features",
|
|
21247
|
+
"encrypted-pii",
|
|
21248
|
+
"smallest-validation"
|
|
21249
|
+
],
|
|
21250
|
+
"forbiddenPatterns": [
|
|
21251
|
+
"node_modules.{0,40}(?:write|edit|patch)",
|
|
21252
|
+
"(?:tenant|organization).{0,30}(?:unscoped|scope optional)"
|
|
21253
|
+
],
|
|
21254
|
+
"validators": [
|
|
21255
|
+
"catalog.schema",
|
|
21256
|
+
"owner.reference",
|
|
21257
|
+
"skills.reference",
|
|
21258
|
+
"router.contract",
|
|
21259
|
+
"context.budget",
|
|
21260
|
+
"context.forbidden",
|
|
21261
|
+
"patterns.forbidden"
|
|
21262
|
+
],
|
|
21263
|
+
"maxContextFiles": 9,
|
|
21264
|
+
"maxInitialContextBytes": 57344,
|
|
21265
|
+
"maxTotalContextBytes": 98304,
|
|
21266
|
+
"relatedCases": [
|
|
21267
|
+
"OMH-233",
|
|
21268
|
+
"OMH-236"
|
|
21269
|
+
]
|
|
21270
|
+
},
|
|
21271
|
+
{
|
|
21272
|
+
"id": "OMH-236",
|
|
21273
|
+
"title": "Connect the telephony provider the business already pays for without a new client",
|
|
21274
|
+
"family": "architecture",
|
|
21275
|
+
"mode": "analysis",
|
|
21276
|
+
"evaluationKind": "routing",
|
|
21277
|
+
"risk": "medium",
|
|
21278
|
+
"prompt": "In a freshly scaffolded standalone Open Mercato app, operations already run their telephony on Tillio and want the calls it records to reach this application on a schedule instead of being exported by hand. Before proposing any connector, establish whether an installed provider module already owns this connection, and name both it and the module that owns the stored calls, or state positively that none does. Then decide the smallest safe design against what the installed modules already provide: how that provider package is turned on for this application, which access-control features gate attaching an operator versus starting a pull, and how a pull that runs for minutes is carried out without holding the request that asked for it. Keep tenant and organization boundaries intact and identify the smallest relevant validation. Do not write your own Tillio client, credential store, or import endpoint before that check is done.",
|
|
21279
|
+
"tags": [
|
|
21280
|
+
"architecture",
|
|
21281
|
+
"module-facts",
|
|
21282
|
+
"reuse-installed",
|
|
21283
|
+
"telephony-provider",
|
|
21284
|
+
"provider-package"
|
|
21285
|
+
],
|
|
21286
|
+
"owner": {
|
|
21287
|
+
"kind": "facts",
|
|
21288
|
+
"path": ".ai/guides/modules/tillio/index.md",
|
|
21289
|
+
"ruleIds": [
|
|
21290
|
+
"BC-01",
|
|
21291
|
+
"BC-04"
|
|
21292
|
+
]
|
|
21293
|
+
},
|
|
21294
|
+
"expectedRouter": {
|
|
21295
|
+
"required": [
|
|
21296
|
+
"architecture"
|
|
21297
|
+
],
|
|
21298
|
+
"allowedExtra": [
|
|
21299
|
+
"integration",
|
|
21300
|
+
"module-data",
|
|
21301
|
+
"framework-context",
|
|
21302
|
+
"umes"
|
|
21303
|
+
]
|
|
21304
|
+
},
|
|
21305
|
+
"requiredSkills": [
|
|
21306
|
+
"om-integration-builder"
|
|
21307
|
+
],
|
|
21308
|
+
"context": {
|
|
21309
|
+
"required": [
|
|
21310
|
+
"AGENTS.md",
|
|
21311
|
+
".ai/guides/architecture.md",
|
|
21312
|
+
".ai/skills/om-integration-builder/SKILL.md",
|
|
21313
|
+
".ai/guides/modules/tillio/index.md",
|
|
21314
|
+
".ai/guides/modules/phone_calls/index.md"
|
|
21315
|
+
],
|
|
21316
|
+
"allowedExtra": [
|
|
21317
|
+
".ai/guides/integrations.md",
|
|
21318
|
+
".ai/guides/contracts.md",
|
|
21319
|
+
".ai/skills/om-help/SKILL.md",
|
|
21320
|
+
".ai/guides/modules/integrations/index.md"
|
|
21321
|
+
],
|
|
21322
|
+
"forbidden": [
|
|
21323
|
+
".env*",
|
|
21324
|
+
".git/**"
|
|
21325
|
+
]
|
|
21326
|
+
},
|
|
21327
|
+
"requiredDecisions": [
|
|
21328
|
+
"facts-first",
|
|
21329
|
+
"app-module-activation",
|
|
21330
|
+
"tenant-scope",
|
|
21331
|
+
"provider-package",
|
|
21332
|
+
"smallest-validation"
|
|
21333
|
+
],
|
|
21334
|
+
"forbiddenPatterns": [
|
|
21335
|
+
"node_modules.{0,40}(?:write|edit|patch)",
|
|
21336
|
+
"(?:tenant|organization).{0,30}(?:unscoped|scope optional)",
|
|
21337
|
+
"(?:token|secret|password).{0,20}(?:console|log)"
|
|
21338
|
+
],
|
|
21339
|
+
"validators": [
|
|
21340
|
+
"catalog.schema",
|
|
21341
|
+
"owner.reference",
|
|
21342
|
+
"skills.reference",
|
|
21343
|
+
"router.contract",
|
|
21344
|
+
"context.budget",
|
|
21345
|
+
"context.forbidden",
|
|
21346
|
+
"patterns.forbidden"
|
|
21347
|
+
],
|
|
21348
|
+
"maxContextFiles": 9,
|
|
21349
|
+
"maxInitialContextBytes": 57344,
|
|
21350
|
+
"maxTotalContextBytes": 98304,
|
|
21351
|
+
"relatedCases": [
|
|
21352
|
+
"OMH-201",
|
|
21353
|
+
"OMH-235"
|
|
21354
|
+
]
|
|
21188
21355
|
}
|
|
21189
21356
|
]
|
|
@@ -3,8 +3,8 @@
|
|
|
3
3
|
"$id": "https://open-mercato.dev/schemas/standalone-harness-cases.schema.json",
|
|
4
4
|
"title": "Open Mercato standalone harness case catalog",
|
|
5
5
|
"type": "array",
|
|
6
|
-
"minItems":
|
|
7
|
-
"maxItems":
|
|
6
|
+
"minItems": 236,
|
|
7
|
+
"maxItems": 236,
|
|
8
8
|
"items": {
|
|
9
9
|
"type": "object",
|
|
10
10
|
"additionalProperties": false,
|
|
@@ -15,7 +15,7 @@
|
|
|
15
15
|
"maxTotalContextBytes", "relatedCases"
|
|
16
16
|
],
|
|
17
17
|
"properties": {
|
|
18
|
-
"id": { "type": "string", "pattern": "^OMH-(00[1-9]|0[1-9][0-9]|1[0-9][0-9]|20[0-9]|21[0-9]|22[0-9]|23[0-
|
|
18
|
+
"id": { "type": "string", "pattern": "^OMH-(00[1-9]|0[1-9][0-9]|1[0-9][0-9]|20[0-9]|21[0-9]|22[0-9]|23[0-6])$" },
|
|
19
19
|
"title": { "type": "string", "minLength": 12, "maxLength": 180 },
|
|
20
20
|
"family": { "enum": ["architecture", "module", "umes", "integration", "ai-workflow", "bugfix", "business", "testing"] },
|
|
21
21
|
"mode": { "enum": ["analysis", "one-shot", "spec", "bugfix", "review"] },
|
|
@@ -168,7 +168,7 @@
|
|
|
168
168
|
"maxInitialContextBytes": { "type": "integer", "minimum": 4096, "maximum": 98304 },
|
|
169
169
|
"maxTotalContextBytes": { "type": "integer", "minimum": 8192, "maximum": 1048576 },
|
|
170
170
|
"timeoutMs": { "type": "integer", "minimum": 1000, "maximum": 600000 },
|
|
171
|
-
"relatedCases": { "type": "array", "minItems": 1, "uniqueItems": true, "items": { "type": "string", "pattern": "^OMH-(00[1-9]|0[1-9][0-9]|1[0-9][0-9]|20[0-9]|21[0-9]|22[0-9]|23[0-
|
|
171
|
+
"relatedCases": { "type": "array", "minItems": 1, "uniqueItems": true, "items": { "type": "string", "pattern": "^OMH-(00[1-9]|0[1-9][0-9]|1[0-9][0-9]|20[0-9]|21[0-9]|22[0-9]|23[0-6])$" } },
|
|
172
172
|
"source": {
|
|
173
173
|
"type": "object",
|
|
174
174
|
"additionalProperties": false,
|
|
@@ -42,7 +42,7 @@ const SENSITIVE_ENV_KEY = /(?:^|_)(?:api_?key|auth|credential|credentials|passwo
|
|
|
42
42
|
const GENERATED_YARN_CONFIG_PATH = '.yarnrc.yml'
|
|
43
43
|
const GENERATED_YARN_CONFIG_LIMIT = 16_384
|
|
44
44
|
// The deterministic step invokes no model, so its ceiling must not ride --case-timeout. Timed on
|
|
45
|
-
// Linux x86_64, the complete
|
|
45
|
+
// Linux x86_64, the complete catalog finished in 768-998 ms, and narrowing the selection
|
|
46
46
|
// down to a single case measured 842-890 ms, inside the same spread, so the run is dominated by
|
|
47
47
|
// fixed process and catalog load rather than by case count. A flat allowance is therefore the
|
|
48
48
|
// honest shape, and 120000 ms is both about 120x the slowest observed run and the value this gate
|