create-mercato-app 0.6.8-develop.7042.1.413df59570 → 0.6.8-develop.7046.1.153faed87a

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. package/agentic/shared/ai/harness/RELEASE.md +2 -2
  2. package/agentic/shared/scripts/run-agent-harness-release.mjs +21 -9
  3. package/dist/agentic/guides/module-facts.json +126 -126
  4. package/dist/agentic/guides/module-facts.v2.json +126 -126
  5. package/dist/agentic/guides/modules/ai_assistant/index.md +1 -1
  6. package/dist/agentic/guides/modules/api_docs/index.md +1 -1
  7. package/dist/agentic/guides/modules/api_keys/index.md +1 -1
  8. package/dist/agentic/guides/modules/attachments/index.md +1 -1
  9. package/dist/agentic/guides/modules/audit_logs/index.md +1 -1
  10. package/dist/agentic/guides/modules/auth/index.md +1 -1
  11. package/dist/agentic/guides/modules/business_rules/index.md +1 -1
  12. package/dist/agentic/guides/modules/catalog/index.md +1 -1
  13. package/dist/agentic/guides/modules/channel_apns/index.md +1 -1
  14. package/dist/agentic/guides/modules/channel_expo/index.md +1 -1
  15. package/dist/agentic/guides/modules/channel_fcm/index.md +1 -1
  16. package/dist/agentic/guides/modules/channel_gmail/index.md +1 -1
  17. package/dist/agentic/guides/modules/channel_imap/index.md +1 -1
  18. package/dist/agentic/guides/modules/checkout/index.md +1 -1
  19. package/dist/agentic/guides/modules/communication_channels/index.md +1 -1
  20. package/dist/agentic/guides/modules/configs/index.md +1 -1
  21. package/dist/agentic/guides/modules/content/index.md +1 -1
  22. package/dist/agentic/guides/modules/currencies/index.md +1 -1
  23. package/dist/agentic/guides/modules/customer_accounts/index.md +1 -1
  24. package/dist/agentic/guides/modules/customers/index.md +1 -1
  25. package/dist/agentic/guides/modules/dashboards/index.md +1 -1
  26. package/dist/agentic/guides/modules/data_sync/index.md +1 -1
  27. package/dist/agentic/guides/modules/design_system/index.md +1 -1
  28. package/dist/agentic/guides/modules/devices/index.md +1 -1
  29. package/dist/agentic/guides/modules/dictionaries/index.md +1 -1
  30. package/dist/agentic/guides/modules/directory/index.md +1 -1
  31. package/dist/agentic/guides/modules/documents/index.md +1 -1
  32. package/dist/agentic/guides/modules/entities/index.md +1 -1
  33. package/dist/agentic/guides/modules/eudr/index.md +1 -1
  34. package/dist/agentic/guides/modules/events/index.md +1 -1
  35. package/dist/agentic/guides/modules/feature_toggles/index.md +1 -1
  36. package/dist/agentic/guides/modules/gateway_stripe/index.md +1 -1
  37. package/dist/agentic/guides/modules/generators/index.md +1 -1
  38. package/dist/agentic/guides/modules/inbox_ops/index.md +1 -1
  39. package/dist/agentic/guides/modules/integrations/index.md +1 -1
  40. package/dist/agentic/guides/modules/messages/index.md +1 -1
  41. package/dist/agentic/guides/modules/notifications/index.md +1 -1
  42. package/dist/agentic/guides/modules/onboarding/index.md +1 -1
  43. package/dist/agentic/guides/modules/payment_gateways/index.md +1 -1
  44. package/dist/agentic/guides/modules/perspectives/index.md +1 -1
  45. package/dist/agentic/guides/modules/planner/index.md +1 -1
  46. package/dist/agentic/guides/modules/portal/index.md +1 -1
  47. package/dist/agentic/guides/modules/progress/index.md +1 -1
  48. package/dist/agentic/guides/modules/push_notifications/index.md +1 -1
  49. package/dist/agentic/guides/modules/query_index/index.md +1 -1
  50. package/dist/agentic/guides/modules/record_locks/index.md +1 -1
  51. package/dist/agentic/guides/modules/resources/index.md +1 -1
  52. package/dist/agentic/guides/modules/sales/index.md +1 -1
  53. package/dist/agentic/guides/modules/scheduler/index.md +1 -1
  54. package/dist/agentic/guides/modules/search/index.md +1 -1
  55. package/dist/agentic/guides/modules/security/index.md +1 -1
  56. package/dist/agentic/guides/modules/shipping_carriers/index.md +1 -1
  57. package/dist/agentic/guides/modules/sso/index.md +1 -1
  58. package/dist/agentic/guides/modules/staff/index.md +1 -1
  59. package/dist/agentic/guides/modules/storage_s3/index.md +1 -1
  60. package/dist/agentic/guides/modules/sync_akeneo/index.md +1 -1
  61. package/dist/agentic/guides/modules/sync_excel/index.md +1 -1
  62. package/dist/agentic/guides/modules/system_status_overlays/index.md +1 -1
  63. package/dist/agentic/guides/modules/translations/index.md +1 -1
  64. package/dist/agentic/guides/modules/warranty_claims/index.md +1 -1
  65. package/dist/agentic/guides/modules/webhooks/index.md +1 -1
  66. package/dist/agentic/guides/modules/wms/index.md +1 -1
  67. package/dist/agentic/guides/modules/workflows/index.md +1 -1
  68. package/dist/agentic/guides/reference-module-facts.json +1 -1
  69. package/dist/agentic/guides/upstream/manifest.json +1 -1
  70. package/dist/agentic/shared/ai/harness/RELEASE.md +2 -2
  71. package/dist/agentic/shared/scripts/run-agent-harness-release.mjs +21 -9
  72. package/package.json +3 -3
@@ -50,9 +50,9 @@ After preflight it runs, in order:
50
50
  6. real generated-code execution for OMH-163 and OMH-192 through fixed Jest, OMH-164 through API-only Playwright, and OMH-165 through real-browser Playwright; and
51
51
  7. explicit isolated `om-judge-agent-session` for every writable result, composing `om-code-review` and applicable design-system guidance, bound to its passing command attestation, any required generated-test result and artifact hash, and the final target fingerprint.
52
52
 
53
- Each writable target is single-use because fixture preparation marks it disposable. Externally supplied target realpaths must be pairwise disjoint and neither equal to, contain, nor be contained by the controller. A failed deterministic or foundation-validation step prevents model execution. Once fixture preparation succeeds, all four target commands run even when the writable gate itself fails, so every generated target has exact diagnostics. A writable case may declare `timeoutMs` only to raise the release `--case-timeout` floor (never lower it); OMH-185 and its business-language parity case OMH-193 use 600000 ms because the complete module slice exceeded the generic five-minute evaluator default while actively producing source. Generated tests run only after the trusted writable oracle and all four target commands pass; review then requires all applicable gates. A target command or generated-test failure is recorded with its sanitized diagnostic and review is skipped. Other matrix entries continue so the report remains useful.
53
+ Each writable target is single-use because fixture preparation marks it disposable. Externally supplied target realpaths must be pairwise disjoint and neither equal to, contain, nor be contained by the controller. A failed deterministic or foundation-validation step prevents model execution. Once fixture preparation succeeds, all four target commands run even when the writable gate itself fails, so every generated target has exact diagnostics. A writable case may declare `timeoutMs` only to raise the release `--case-timeout` floor (never lower it); OMH-185 and its business-language parity case OMH-193 use 600000 ms because the complete module slice exceeded the generic five-minute evaluator default while actively producing source. Since the schema caps `timeoutMs` at exactly the shipped `--case-timeout` default, a declared value only raises anything for an operator who lowered that default — it is the floor under a lowered budget, not an addition to the shipped one. Generated tests run only after the trusted writable oracle and all four target commands pass; review then requires all applicable gates. A target command or generated-test failure is recorded with its sanitized diagnostic and review is skipped. Other matrix entries continue so the report remains useful.
54
54
 
55
- Routing cases carry no case-local duration budget, and that is a decision rather than an omission. `maxContextFiles` and the byte budgets measure the agent's context discipline, which is intrinsic to the case and portable between machines; duration measures the runner and the attempt count, which are not — the audited cohort spanned 71 s to 231 s for passing routing runs, one case measured 147 s and 132 s on two runs of the same model, and OMH-139 exhausted the evaluator's 300000 ms default outright. The operator budget carries that variance instead of the catalog. On this release path the lever is `--case-timeout` (default 120000 ms), which the release command passes on to the evaluator explicitly for every routing step; because that pass-through marks the timeout explicit, the evaluator's own runner-aware floors — 600000 ms for Claude, 900000 ms for a Codex `gpt-5.4-mini` high-effort run, 300000 ms for every other runner — apply only to direct `evaluate-agent-harness.mjs` invocations and never fire under `yarn harness:release`. The release default sits below the slowest audited passing routing run, so raise `--case-timeout` for a slow model rather than expecting the default to absorb the tail; on the direct evaluator path that run leaves about 62% headroom against the Claude floor and about 23% against the generic 300000 ms default, and OMH-139 shows the latter is not always enough. A declared writable `timeoutMs` is combined with whichever operator value applies as a maximum, so it raises the floor and never lowers it; routing cases declare none, so there the operator value stands alone. Writable cases keep their own `timeoutMs` because a writable one-shot's cost is dominated by the slice it must produce, which the case does define.
55
+ Routing cases carry no case-local duration budget, and that is a decision rather than an omission. `maxContextFiles` and the byte budgets measure the agent's context discipline, which is intrinsic to the case and portable between machines; duration measures the runner and the attempt count, which are not — the audited cohort spanned 71 s to 231 s for passing routing runs, one case measured 147 s and 132 s on two runs of the same model, and OMH-139 exhausted the evaluator's 300000 ms default outright. The operator budget carries that variance instead of the catalog. On this release path the lever is `--case-timeout` (default 600000 ms), which the release command passes on to the evaluator explicitly for every routing step; because that pass-through marks the timeout explicit, the evaluator's own runner-aware floors — 600000 ms for Claude, 900000 ms for a Codex `gpt-5.4-mini` high-effort run, 300000 ms for every other runner — apply only to direct `evaluate-agent-harness.mjs` invocations and never fire under `yarn harness:release`. Keeping that pass-through is deliberate rather than inherited: the routing step derives its own process budget from the same value, as slack plus the sum of the per-case ceilings it hands out, so letting the floors raise the inner ceiling while the outer budget still followed `--case-timeout` would kill an entire routing step instead of failing one slow case. Leaving it explicit also keeps the budget runner-independent, which is what lets the primary and portability lanes in one report be compared as models rather than as budgets. The default clears the slowest audited passing routing run — 231 s against 600000 ms, about 62% headroom — and matches both the Claude floor and the `timeoutMs` ceiling `cases.schema.json` enforces, so the gate carries one upper number instead of three; lower `--case-timeout` when a hung case should fail sooner, and note that the step's process budget drops with it — though at a lowered budget the declared writable ceilings keep their own slots, so the step's budget drops less than proportionally. That value was chosen from evaluator-path measurements — the audited cohort above and the live evidence recorded on #5068 — rather than from a driven `yarn harness:release` run, because the complete gate fails closed off the Linux-with-Bubblewrap host stated above; #5078 records the decision and that deviation, and #5433 carries the unmet measurement forward so the default is confirmed or corrected against the release path's own numbers. `--case-timeout` is one budget for three lanes, not a routing-only lever: the writable and review lanes resolve their own ceilings from the same value, so raising it raises theirs too. A declared writable `timeoutMs` is combined with whichever operator value applies as a maximum, so it raises the floor and never lowers it — but because the schema caps it at the shipped default, it can only raise a budget an operator has lowered; routing cases declare none, so there the operator value always stands alone. Writable cases keep their own `timeoutMs` because a writable one-shot's cost is dominated by the slice it must produce, which the case does define.
56
56
 
57
57
  UI-routed implementation reviews receive only the bounded backend UI guide and `om-backend-ui-design` design-system references. Non-UI reviews do not receive that extra context.
58
58
 
@@ -25,6 +25,11 @@ const GENERATED_TEST_RUNNERS = new Set(['jest', 'playwright-api', 'playwright-br
25
25
  const RESULT_LIMIT = 262_144
26
26
  const ERROR_LIMIT = 2_000
27
27
  const VIOLATION_LIMIT = 300
28
+ // The routing step hands this to the evaluator as --timeout and derives its own process budget from
29
+ // the same value, so the two cannot be stated separately; the help text reads it rather than
30
+ // repeating it (#5078).
31
+ export const DEFAULT_CASE_TIMEOUT_MS = 600_000
32
+ export const ROUTING_STEP_SLACK_MS = 60_000
28
33
  const COPY_EXCLUDED_PREFIXES = [
29
34
  '.git', '.next', '.turbo', '.cache', 'build', 'coverage', 'dist', 'node_modules', 'out',
30
35
  '.ai/framework-context', '.ai/harness/results', '.ai/reports',
@@ -50,7 +55,7 @@ Options:
50
55
  --portability-runner <runner> Optional different runner for the 49-case read-only portability lane
51
56
  --prepare-targets <absolute> Clone this fresh scaffold once per writable case under an empty/new directory
52
57
  --writable-targets <absolute> JSON map of every writable case to a fresh disposable app
53
- --case-timeout <ms> Per-model invocation timeout floor (default: 120000; a writable case may raise it)
58
+ --case-timeout <ms> Per-model invocation timeout floor for the routing, writable, and review lanes (default: ${DEFAULT_CASE_TIMEOUT_MS})
54
59
  --validation-timeout <ms> Timeout for each yarn validation (default: 1800000)
55
60
  --acknowledge-writes Required: fixture preparation and validation commands write files
56
61
  --help Show this help
@@ -76,7 +81,7 @@ function parseArgs(argv) {
76
81
  portabilityRunner: undefined,
77
82
  prepareTargets: undefined,
78
83
  writableTargets: undefined,
79
- caseTimeout: 120_000,
84
+ caseTimeout: DEFAULT_CASE_TIMEOUT_MS,
80
85
  validationTimeout: 1_800_000,
81
86
  acknowledgeWrites: false,
82
87
  help: false,
@@ -118,6 +123,17 @@ export function effectiveCaseTimeout(cases, caseId, fallback) {
118
123
  return Math.max(fallback, Number.isInteger(declared) ? declared : 0)
119
124
  }
120
125
 
126
+ export function routingInvocation({ evaluator, root, step, cases, caseTimeout }) {
127
+ const args = [evaluator, '--root', root, '--runner', step.runner]
128
+ if (step.lane === 'primary') args.push('--all')
129
+ args.push('--model', step.modelSelector, '--timeout', String(caseTimeout))
130
+ const timeout = step.expectedCaseIds.reduce(
131
+ (total, caseId) => total + effectiveCaseTimeout(cases, caseId, caseTimeout),
132
+ ROUTING_STEP_SLACK_MS,
133
+ )
134
+ return { args, timeout }
135
+ }
136
+
121
137
  function readJson(file) {
122
138
  return JSON.parse(fs.readFileSync(file, 'utf8'))
123
139
  }
@@ -1614,13 +1630,9 @@ export function main(argv = process.argv.slice(2)) {
1614
1630
  } else {
1615
1631
  for (const step of plan.steps.filter((entry) => entry.kind === 'routing')) {
1616
1632
  const before = resultFiles(root)
1617
- const routingArgs = [evaluator, '--root', root, '--runner', step.runner]
1618
- if (step.lane === 'primary') routingArgs.push('--all')
1619
- routingArgs.push('--model', step.modelSelector, '--timeout', String(options.caseTimeout))
1620
- const routingTimeout = step.expectedCaseIds.reduce(
1621
- (total, caseId) => total + effectiveCaseTimeout(cases, caseId, options.caseTimeout),
1622
- 60_000,
1623
- )
1633
+ const { args: routingArgs, timeout: routingTimeout } = routingInvocation({
1634
+ evaluator, root, step, cases, caseTimeout: options.caseTimeout,
1635
+ })
1624
1636
  const execution = execute(process.execPath, routingArgs, root, routingTimeout)
1625
1637
  const artifacts = readNewResults(root, before)
1626
1638
  resultArtifacts.push(...artifacts)