create-mercato-app 0.6.8-develop.7042.1.413df59570 → 0.6.8-develop.7046.1.153faed87a
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/agentic/shared/ai/harness/RELEASE.md +2 -2
- package/agentic/shared/scripts/run-agent-harness-release.mjs +21 -9
- package/dist/agentic/guides/module-facts.json +126 -126
- package/dist/agentic/guides/module-facts.v2.json +126 -126
- package/dist/agentic/guides/modules/ai_assistant/index.md +1 -1
- package/dist/agentic/guides/modules/api_docs/index.md +1 -1
- package/dist/agentic/guides/modules/api_keys/index.md +1 -1
- package/dist/agentic/guides/modules/attachments/index.md +1 -1
- package/dist/agentic/guides/modules/audit_logs/index.md +1 -1
- package/dist/agentic/guides/modules/auth/index.md +1 -1
- package/dist/agentic/guides/modules/business_rules/index.md +1 -1
- package/dist/agentic/guides/modules/catalog/index.md +1 -1
- package/dist/agentic/guides/modules/channel_apns/index.md +1 -1
- package/dist/agentic/guides/modules/channel_expo/index.md +1 -1
- package/dist/agentic/guides/modules/channel_fcm/index.md +1 -1
- package/dist/agentic/guides/modules/channel_gmail/index.md +1 -1
- package/dist/agentic/guides/modules/channel_imap/index.md +1 -1
- package/dist/agentic/guides/modules/checkout/index.md +1 -1
- package/dist/agentic/guides/modules/communication_channels/index.md +1 -1
- package/dist/agentic/guides/modules/configs/index.md +1 -1
- package/dist/agentic/guides/modules/content/index.md +1 -1
- package/dist/agentic/guides/modules/currencies/index.md +1 -1
- package/dist/agentic/guides/modules/customer_accounts/index.md +1 -1
- package/dist/agentic/guides/modules/customers/index.md +1 -1
- package/dist/agentic/guides/modules/dashboards/index.md +1 -1
- package/dist/agentic/guides/modules/data_sync/index.md +1 -1
- package/dist/agentic/guides/modules/design_system/index.md +1 -1
- package/dist/agentic/guides/modules/devices/index.md +1 -1
- package/dist/agentic/guides/modules/dictionaries/index.md +1 -1
- package/dist/agentic/guides/modules/directory/index.md +1 -1
- package/dist/agentic/guides/modules/documents/index.md +1 -1
- package/dist/agentic/guides/modules/entities/index.md +1 -1
- package/dist/agentic/guides/modules/eudr/index.md +1 -1
- package/dist/agentic/guides/modules/events/index.md +1 -1
- package/dist/agentic/guides/modules/feature_toggles/index.md +1 -1
- package/dist/agentic/guides/modules/gateway_stripe/index.md +1 -1
- package/dist/agentic/guides/modules/generators/index.md +1 -1
- package/dist/agentic/guides/modules/inbox_ops/index.md +1 -1
- package/dist/agentic/guides/modules/integrations/index.md +1 -1
- package/dist/agentic/guides/modules/messages/index.md +1 -1
- package/dist/agentic/guides/modules/notifications/index.md +1 -1
- package/dist/agentic/guides/modules/onboarding/index.md +1 -1
- package/dist/agentic/guides/modules/payment_gateways/index.md +1 -1
- package/dist/agentic/guides/modules/perspectives/index.md +1 -1
- package/dist/agentic/guides/modules/planner/index.md +1 -1
- package/dist/agentic/guides/modules/portal/index.md +1 -1
- package/dist/agentic/guides/modules/progress/index.md +1 -1
- package/dist/agentic/guides/modules/push_notifications/index.md +1 -1
- package/dist/agentic/guides/modules/query_index/index.md +1 -1
- package/dist/agentic/guides/modules/record_locks/index.md +1 -1
- package/dist/agentic/guides/modules/resources/index.md +1 -1
- package/dist/agentic/guides/modules/sales/index.md +1 -1
- package/dist/agentic/guides/modules/scheduler/index.md +1 -1
- package/dist/agentic/guides/modules/search/index.md +1 -1
- package/dist/agentic/guides/modules/security/index.md +1 -1
- package/dist/agentic/guides/modules/shipping_carriers/index.md +1 -1
- package/dist/agentic/guides/modules/sso/index.md +1 -1
- package/dist/agentic/guides/modules/staff/index.md +1 -1
- package/dist/agentic/guides/modules/storage_s3/index.md +1 -1
- package/dist/agentic/guides/modules/sync_akeneo/index.md +1 -1
- package/dist/agentic/guides/modules/sync_excel/index.md +1 -1
- package/dist/agentic/guides/modules/system_status_overlays/index.md +1 -1
- package/dist/agentic/guides/modules/translations/index.md +1 -1
- package/dist/agentic/guides/modules/warranty_claims/index.md +1 -1
- package/dist/agentic/guides/modules/webhooks/index.md +1 -1
- package/dist/agentic/guides/modules/wms/index.md +1 -1
- package/dist/agentic/guides/modules/workflows/index.md +1 -1
- package/dist/agentic/guides/reference-module-facts.json +1 -1
- package/dist/agentic/guides/upstream/manifest.json +1 -1
- package/dist/agentic/shared/ai/harness/RELEASE.md +2 -2
- package/dist/agentic/shared/scripts/run-agent-harness-release.mjs +21 -9
- package/package.json +3 -3
|
@@ -50,9 +50,9 @@ After preflight it runs, in order:
|
|
|
50
50
|
6. real generated-code execution for OMH-163 and OMH-192 through fixed Jest, OMH-164 through API-only Playwright, and OMH-165 through real-browser Playwright; and
|
|
51
51
|
7. explicit isolated `om-judge-agent-session` for every writable result, composing `om-code-review` and applicable design-system guidance, bound to its passing command attestation, any required generated-test result and artifact hash, and the final target fingerprint.
|
|
52
52
|
|
|
53
|
-
Each writable target is single-use because fixture preparation marks it disposable. Externally supplied target realpaths must be pairwise disjoint and neither equal to, contain, nor be contained by the controller. A failed deterministic or foundation-validation step prevents model execution. Once fixture preparation succeeds, all four target commands run even when the writable gate itself fails, so every generated target has exact diagnostics. A writable case may declare `timeoutMs` only to raise the release `--case-timeout` floor (never lower it); OMH-185 and its business-language parity case OMH-193 use 600000 ms because the complete module slice exceeded the generic five-minute evaluator default while actively producing source. Generated tests run only after the trusted writable oracle and all four target commands pass; review then requires all applicable gates. A target command or generated-test failure is recorded with its sanitized diagnostic and review is skipped. Other matrix entries continue so the report remains useful.
|
|
53
|
+
Each writable target is single-use because fixture preparation marks it disposable. Externally supplied target realpaths must be pairwise disjoint and neither equal to, contain, nor be contained by the controller. A failed deterministic or foundation-validation step prevents model execution. Once fixture preparation succeeds, all four target commands run even when the writable gate itself fails, so every generated target has exact diagnostics. A writable case may declare `timeoutMs` only to raise the release `--case-timeout` floor (never lower it); OMH-185 and its business-language parity case OMH-193 use 600000 ms because the complete module slice exceeded the generic five-minute evaluator default while actively producing source. Since the schema caps `timeoutMs` at exactly the shipped `--case-timeout` default, a declared value only raises anything for an operator who lowered that default — it is the floor under a lowered budget, not an addition to the shipped one. Generated tests run only after the trusted writable oracle and all four target commands pass; review then requires all applicable gates. A target command or generated-test failure is recorded with its sanitized diagnostic and review is skipped. Other matrix entries continue so the report remains useful.
|
|
54
54
|
|
|
55
|
-
Routing cases carry no case-local duration budget, and that is a decision rather than an omission. `maxContextFiles` and the byte budgets measure the agent's context discipline, which is intrinsic to the case and portable between machines; duration measures the runner and the attempt count, which are not — the audited cohort spanned 71 s to 231 s for passing routing runs, one case measured 147 s and 132 s on two runs of the same model, and OMH-139 exhausted the evaluator's 300000 ms default outright. The operator budget carries that variance instead of the catalog. On this release path the lever is `--case-timeout` (default
|
|
55
|
+
Routing cases carry no case-local duration budget, and that is a decision rather than an omission. `maxContextFiles` and the byte budgets measure the agent's context discipline, which is intrinsic to the case and portable between machines; duration measures the runner and the attempt count, which are not — the audited cohort spanned 71 s to 231 s for passing routing runs, one case measured 147 s and 132 s on two runs of the same model, and OMH-139 exhausted the evaluator's 300000 ms default outright. The operator budget carries that variance instead of the catalog. On this release path the lever is `--case-timeout` (default 600000 ms), which the release command passes on to the evaluator explicitly for every routing step; because that pass-through marks the timeout explicit, the evaluator's own runner-aware floors — 600000 ms for Claude, 900000 ms for a Codex `gpt-5.4-mini` high-effort run, 300000 ms for every other runner — apply only to direct `evaluate-agent-harness.mjs` invocations and never fire under `yarn harness:release`. Keeping that pass-through is deliberate rather than inherited: the routing step derives its own process budget from the same value, as slack plus the sum of the per-case ceilings it hands out, so letting the floors raise the inner ceiling while the outer budget still followed `--case-timeout` would kill an entire routing step instead of failing one slow case. Leaving it explicit also keeps the budget runner-independent, which is what lets the primary and portability lanes in one report be compared as models rather than as budgets. The default clears the slowest audited passing routing run — 231 s against 600000 ms, about 62% headroom — and matches both the Claude floor and the `timeoutMs` ceiling `cases.schema.json` enforces, so the gate carries one upper number instead of three; lower `--case-timeout` when a hung case should fail sooner, and note that the step's process budget drops with it — though at a lowered budget the declared writable ceilings keep their own slots, so the step's budget drops less than proportionally. That value was chosen from evaluator-path measurements — the audited cohort above and the live evidence recorded on #5068 — rather than from a driven `yarn harness:release` run, because the complete gate fails closed off the Linux-with-Bubblewrap host stated above; #5078 records the decision and that deviation, and #5433 carries the unmet measurement forward so the default is confirmed or corrected against the release path's own numbers. `--case-timeout` is one budget for three lanes, not a routing-only lever: the writable and review lanes resolve their own ceilings from the same value, so raising it raises theirs too. A declared writable `timeoutMs` is combined with whichever operator value applies as a maximum, so it raises the floor and never lowers it — but because the schema caps it at the shipped default, it can only raise a budget an operator has lowered; routing cases declare none, so there the operator value always stands alone. Writable cases keep their own `timeoutMs` because a writable one-shot's cost is dominated by the slice it must produce, which the case does define.
|
|
56
56
|
|
|
57
57
|
UI-routed implementation reviews receive only the bounded backend UI guide and `om-backend-ui-design` design-system references. Non-UI reviews do not receive that extra context.
|
|
58
58
|
|
|
@@ -25,6 +25,11 @@ const GENERATED_TEST_RUNNERS = new Set(['jest', 'playwright-api', 'playwright-br
|
|
|
25
25
|
const RESULT_LIMIT = 262_144
|
|
26
26
|
const ERROR_LIMIT = 2_000
|
|
27
27
|
const VIOLATION_LIMIT = 300
|
|
28
|
+
// The routing step hands this to the evaluator as --timeout and derives its own process budget from
|
|
29
|
+
// the same value, so the two cannot be stated separately; the help text reads it rather than
|
|
30
|
+
// repeating it (#5078).
|
|
31
|
+
export const DEFAULT_CASE_TIMEOUT_MS = 600_000
|
|
32
|
+
export const ROUTING_STEP_SLACK_MS = 60_000
|
|
28
33
|
const COPY_EXCLUDED_PREFIXES = [
|
|
29
34
|
'.git', '.next', '.turbo', '.cache', 'build', 'coverage', 'dist', 'node_modules', 'out',
|
|
30
35
|
'.ai/framework-context', '.ai/harness/results', '.ai/reports',
|
|
@@ -50,7 +55,7 @@ Options:
|
|
|
50
55
|
--portability-runner <runner> Optional different runner for the 49-case read-only portability lane
|
|
51
56
|
--prepare-targets <absolute> Clone this fresh scaffold once per writable case under an empty/new directory
|
|
52
57
|
--writable-targets <absolute> JSON map of every writable case to a fresh disposable app
|
|
53
|
-
--case-timeout <ms> Per-model invocation timeout floor
|
|
58
|
+
--case-timeout <ms> Per-model invocation timeout floor for the routing, writable, and review lanes (default: ${DEFAULT_CASE_TIMEOUT_MS})
|
|
54
59
|
--validation-timeout <ms> Timeout for each yarn validation (default: 1800000)
|
|
55
60
|
--acknowledge-writes Required: fixture preparation and validation commands write files
|
|
56
61
|
--help Show this help
|
|
@@ -76,7 +81,7 @@ function parseArgs(argv) {
|
|
|
76
81
|
portabilityRunner: undefined,
|
|
77
82
|
prepareTargets: undefined,
|
|
78
83
|
writableTargets: undefined,
|
|
79
|
-
caseTimeout:
|
|
84
|
+
caseTimeout: DEFAULT_CASE_TIMEOUT_MS,
|
|
80
85
|
validationTimeout: 1_800_000,
|
|
81
86
|
acknowledgeWrites: false,
|
|
82
87
|
help: false,
|
|
@@ -118,6 +123,17 @@ export function effectiveCaseTimeout(cases, caseId, fallback) {
|
|
|
118
123
|
return Math.max(fallback, Number.isInteger(declared) ? declared : 0)
|
|
119
124
|
}
|
|
120
125
|
|
|
126
|
+
export function routingInvocation({ evaluator, root, step, cases, caseTimeout }) {
|
|
127
|
+
const args = [evaluator, '--root', root, '--runner', step.runner]
|
|
128
|
+
if (step.lane === 'primary') args.push('--all')
|
|
129
|
+
args.push('--model', step.modelSelector, '--timeout', String(caseTimeout))
|
|
130
|
+
const timeout = step.expectedCaseIds.reduce(
|
|
131
|
+
(total, caseId) => total + effectiveCaseTimeout(cases, caseId, caseTimeout),
|
|
132
|
+
ROUTING_STEP_SLACK_MS,
|
|
133
|
+
)
|
|
134
|
+
return { args, timeout }
|
|
135
|
+
}
|
|
136
|
+
|
|
121
137
|
function readJson(file) {
|
|
122
138
|
return JSON.parse(fs.readFileSync(file, 'utf8'))
|
|
123
139
|
}
|
|
@@ -1614,13 +1630,9 @@ export function main(argv = process.argv.slice(2)) {
|
|
|
1614
1630
|
} else {
|
|
1615
1631
|
for (const step of plan.steps.filter((entry) => entry.kind === 'routing')) {
|
|
1616
1632
|
const before = resultFiles(root)
|
|
1617
|
-
const
|
|
1618
|
-
|
|
1619
|
-
|
|
1620
|
-
const routingTimeout = step.expectedCaseIds.reduce(
|
|
1621
|
-
(total, caseId) => total + effectiveCaseTimeout(cases, caseId, options.caseTimeout),
|
|
1622
|
-
60_000,
|
|
1623
|
-
)
|
|
1633
|
+
const { args: routingArgs, timeout: routingTimeout } = routingInvocation({
|
|
1634
|
+
evaluator, root, step, cases, caseTimeout: options.caseTimeout,
|
|
1635
|
+
})
|
|
1624
1636
|
const execution = execute(process.execPath, routingArgs, root, routingTimeout)
|
|
1625
1637
|
const artifacts = readNewResults(root, before)
|
|
1626
1638
|
resultArtifacts.push(...artifacts)
|