tickmarkr 1.90.4 → 1.90.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -89,12 +89,20 @@ function providerModels(modelsDev, preferred) {
|
|
|
89
89
|
: [];
|
|
90
90
|
// Price is provider-specific. A provider-qualified query must never borrow an identically named
|
|
91
91
|
// model from another provider, where subscription and metered costs can differ materially.
|
|
92
|
-
|
|
92
|
+
// D-OBS-11 follow-up: that rule presumes the hinted provider EXISTS in the catalog. A hint that
|
|
93
|
+
// matches nothing (kimi's "moonshot" vs the catalog's "moonshotai"/"kimi-for-coding"; cursor has
|
|
94
|
+
// no provider at all) used to ZERO the search space and blanket-uncover the whole adapter.
|
|
95
|
+
// Fail open to the full scan instead — advisory evidence with a visible models.dev id beats none.
|
|
96
|
+
const selected = preferred && preferredEntries.length > 0 ? preferredEntries : entries;
|
|
93
97
|
return selected
|
|
94
98
|
.map(([, provider]) => record(record(provider)?.models))
|
|
95
99
|
.filter((models) => models !== undefined);
|
|
96
100
|
}
|
|
97
101
|
function findModelsDevModel(catalog, provider, modelId) {
|
|
102
|
+
// CLI namespaces prefix their catalog ids (kimi-code/k3 vs catalog key k3; omp's openai/gpt-4):
|
|
103
|
+
// after exact key/id misses, retry with the bare segment after the last "/". Deterministic
|
|
104
|
+
// provider order; first hit wins — acceptable for advisory evidence, never routing.
|
|
105
|
+
const bare = modelId.includes("/") ? modelId.slice(modelId.lastIndexOf("/") + 1) : undefined;
|
|
98
106
|
for (const models of providerModels(catalog.modelsDev, provider)) {
|
|
99
107
|
const direct = record(models[modelId]);
|
|
100
108
|
if (direct)
|
|
@@ -103,6 +111,13 @@ function findModelsDevModel(catalog, provider, modelId) {
|
|
|
103
111
|
if (byId)
|
|
104
112
|
return byId;
|
|
105
113
|
}
|
|
114
|
+
if (bare) {
|
|
115
|
+
for (const models of providerModels(catalog.modelsDev, provider)) {
|
|
116
|
+
const direct = record(models[bare]);
|
|
117
|
+
if (direct)
|
|
118
|
+
return direct;
|
|
119
|
+
}
|
|
120
|
+
}
|
|
106
121
|
return undefined;
|
|
107
122
|
}
|
|
108
123
|
function artificialAnalysisRows(value) {
|
|
@@ -11,8 +11,9 @@ export const MODEL_STALE_DAYS = 30;
|
|
|
11
11
|
const DAY_MS = 86400000;
|
|
12
12
|
// cursor-agent 2026.07.08 reports 193 mostly-parameterized ids (e.g. gpt-5.3-codex-high-fast); filter the `auto`
|
|
13
13
|
// pseudo-model + effort/speed variant suffixes from the unconfigured-lint aggregation ONLY — doctor.json keeps the
|
|
14
|
-
// raw list (verified 2026-07-10). Data stays raw; lints stay signal.
|
|
15
|
-
|
|
14
|
+
// raw list (verified 2026-07-10). Data stays raw; lints stay signal. -max/-none/-thinking joined the suffix set
|
|
15
|
+
// 2026-08-12 (D-OBS-11: cursor's residual lint list was still mostly effort variants of configured bases).
|
|
16
|
+
const LINT_VARIANT_RE = /^auto$|-(fast|minimal|none|low|medium|high|xhigh|max|thinking)$/;
|
|
16
17
|
const LINT_CAP = 5;
|
|
17
18
|
const TTY_LINT_CAP = 3;
|
|
18
19
|
const DEFAULT_STATE_DIR = ".tickmarkr";
|
package/dist/gates/baseline.js
CHANGED
|
@@ -51,9 +51,19 @@ const TRAILING_FAIL_RE = /^\s*(?:test\s+)?\S+(?:\s+\([^)]*\))?\s+(?:\.{3}|-{3,})
|
|
|
51
51
|
// runner sharing the glyph protocol is read without being enumerated; a status strip drawing ✖
|
|
52
52
|
// mid-line inside chrome matches nothing, the same position rule as the other shapes.
|
|
53
53
|
const GLYPH_FAIL_RE = /^\s*(?:(?:[\w@./-]+:\s*)*)✖\s+\S/;
|
|
54
|
+
// Q137s (verbatim capture, dossier run-2 consult dossier, turbo 2.9.14 + pnpm 10): monorepo
|
|
55
|
+
// drivers prefix child output and summarize failures in their own grammar. Four shapes, all
|
|
56
|
+
// positional — `<pkg>:<task>:` prefixes reuse GLYPH_FAIL_RE's prefix idiom; `Failed:`/`ERROR`
|
|
57
|
+
// require the driver's own `pkg#task` identifier or its literal terminus, so prose or drawn
|
|
58
|
+
// chrome containing the words matches nothing (OBS-278 discipline):
|
|
59
|
+
// `intake-frontend:lint: ELIFECYCLE Command failed with exit code 1.`
|
|
60
|
+
// `Failed: intake-frontend#lint`
|
|
61
|
+
// ` ERROR intake-frontend#lint: command (…) … exited (1)`
|
|
62
|
+
// ` ERROR run failed: command exited (1)`
|
|
63
|
+
const TURBO_FAIL_RE = /^\s*(?:[\w@./-]+:\s*)*ELIFECYCLE\s+Command failed\b|^\s*Failed:\s+\S+#\S+|^\s*ERROR\s+(?:\S+#\S+:|run failed\b)/;
|
|
54
64
|
// Lines that NAME a failing test — the ones worth headlining to the operator. One list, so recognition
|
|
55
65
|
// and reporting cannot drift apart (a shape that blocks but never gets named cost 3 attempts once).
|
|
56
|
-
const namesFailure = (l) => FAIL_ANCHOR_RE.test(l) || RUNNER_FAIL_RE.test(l) || TRAILING_FAIL_RE.test(l) || GLYPH_FAIL_RE.test(l);
|
|
66
|
+
const namesFailure = (l) => FAIL_ANCHOR_RE.test(l) || RUNNER_FAIL_RE.test(l) || TRAILING_FAIL_RE.test(l) || GLYPH_FAIL_RE.test(l) || TURBO_FAIL_RE.test(l);
|
|
57
67
|
const isFailureShaped = (l) => namesFailure(l) || SUMMARY_FAIL_RE.test(l) || ERROR_ANCHOR_RE.test(l)
|
|
58
68
|
|| TSC_ERROR_RE.test(l) || LINTER_ERROR_RE.test(l);
|
|
59
69
|
const VOCAB_RE = /\b(?:error|fail(?:ed|ure|ing)?)\b/i;
|
package/dist/run/consult.js
CHANGED
|
@@ -2,7 +2,7 @@ import { mkdirSync, writeFileSync } from "node:fs";
|
|
|
2
2
|
import { join } from "node:path";
|
|
3
3
|
import { getAdapter } from "../adapters/registry.js";
|
|
4
4
|
import { bannerShell, paneDispatchCommand } from "../brand.js";
|
|
5
|
-
import { extractVerdictJson, gateExitTrailer, gatePaneName, generateVerdictNonce, verdictNonceLine } from "../gates/llm.js";
|
|
5
|
+
import { dewrapPaneVerdict, extractVerdictJson, gateExitTrailer, gatePaneName, generateVerdictNonce, verdictNonceLine } from "../gates/llm.js";
|
|
6
6
|
import { classifyVerdictCause } from "../gates/verdict-cause.js";
|
|
7
7
|
import { disallowedBy } from "../route/preference.js";
|
|
8
8
|
import { sh } from "./git.js";
|
|
@@ -173,7 +173,22 @@ opts = {}) {
|
|
|
173
173
|
await driver.close(slot);
|
|
174
174
|
}
|
|
175
175
|
}
|
|
176
|
-
|
|
176
|
+
// D-OBS-12 (dossier runs 1-2, 3/3 valid verdicts destroyed): the pane path is a terminal
|
|
177
|
+
// scrape — a single-line ~2000-char verdict soft-wraps in rendering and brace balance dies.
|
|
178
|
+
// Judge/review got dewrapPaneVerdict for exactly this (llm.ts, OBS-209 lineage); the consult
|
|
179
|
+
// seat never did. Headless output is machine-read stdout and needs no reconstruction.
|
|
180
|
+
const effective = cfg.visibility.llm === "headless" ? out : dewrapPaneVerdict(out, nonce);
|
|
181
|
+
const parsed = parseConsultVerdict(effective, nonce);
|
|
182
|
+
if (!parsed.verdict) {
|
|
183
|
+
// Q144s / OBS-196 parity: an unparseable verdict persists its raw bytes (pre-dewrap,
|
|
184
|
+
// rendering verbatim) beside the prompt — three human parks carried zero diagnostic
|
|
185
|
+
// payload because this artifact did not exist.
|
|
186
|
+
try {
|
|
187
|
+
writeFileSync(join(dir, `${d.taskId}-${n}${seatIdx > 0 ? `-s${seatIdx}` : ""}-response.txt`), out);
|
|
188
|
+
}
|
|
189
|
+
catch { /* evidence persistence must never fail the seat walk */ }
|
|
190
|
+
}
|
|
191
|
+
return parsed;
|
|
177
192
|
};
|
|
178
193
|
// v1.54 T1: ranked seat failover. Walk consult.prefer (adapter:model entries) to the first entry
|
|
179
194
|
// whose adapter is in the live channel set; a failed seat or unparseable verdict falls to the next;
|