pi-smart-compact 9.6.1 → 9.6.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/ARCHITECTURE.md +21 -9
- package/CHANGELOG.md +26 -0
- package/README.md +27 -8
- package/dist/app/mode-policy.d.ts +3 -1
- package/dist/app/mode-policy.d.ts.map +1 -1
- package/dist/app/preflight.d.ts.map +1 -1
- package/dist/app/register-smart-compact-tool.d.ts.map +1 -1
- package/dist/app/run-smart-compact.d.ts.map +1 -1
- package/dist/app/steps/extract.d.ts.map +1 -1
- package/dist/app/steps/prepare.d.ts.map +1 -1
- package/dist/app/steps/state.d.ts.map +1 -1
- package/dist/app/steps/synthesize.d.ts.map +1 -1
- package/dist/app/steps/window.d.ts.map +1 -1
- package/dist/constants.d.ts +1 -1
- package/dist/domain/provider-evaluation.d.ts.map +1 -1
- package/dist/domain/telemetry.d.ts.map +1 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +997 -962
- package/dist/infra/ai-messages.d.ts +6 -2
- package/dist/infra/ai-messages.d.ts.map +1 -1
- package/dist/phases/verify.d.ts.map +1 -1
- package/dist/provider-eval.js +224 -1
- package/dist/provider-scenario-eval.js +234 -9
- package/dist/telemetry-report.js +1 -1
- package/dist/types.d.ts +3 -0
- package/dist/types.d.ts.map +1 -1
- package/dist/ui/error-format.d.ts +3 -1
- package/dist/ui/error-format.d.ts.map +1 -1
- package/dist/ui/metrics-report.d.ts.map +1 -1
- package/dist/utils/cache.d.ts.map +1 -1
- package/dist/utils/helpers.d.ts.map +1 -1
- package/dist/utils/session-log.d.ts.map +1 -1
- package/package.json +1 -1
|
@@ -15,9 +15,13 @@
|
|
|
15
15
|
* conversation buffer in native `Message[]` directly (see `phases/explore.ts`),
|
|
16
16
|
* which is the preferred pattern for any new code that talks to a provider.
|
|
17
17
|
*/
|
|
18
|
-
import type
|
|
19
|
-
import type { LlmMessage } from "../types.ts";
|
|
18
|
+
import { type Message } from "@earendil-works/pi-ai";
|
|
19
|
+
import type { LlmMessage, SessionMessageEntry } from "../types.ts";
|
|
20
20
|
import type { SecretScrubber } from "../domain/scrub.ts";
|
|
21
|
+
/** Preserve the host's context projection and original entry IDs, including custom and branch summaries. */
|
|
22
|
+
export declare function contextMessageEntries(entries: readonly unknown[]): SessionMessageEntry[];
|
|
23
|
+
/** Text transcript without the host summarizer's implicit 2,000-character tool-result cap. */
|
|
24
|
+
export declare function serializeConversationText(messages: LlmMessage[]): string;
|
|
21
25
|
/**
|
|
22
26
|
* Upcast a raw branch entry's `message` to a `Message` for `convertToLlm`.
|
|
23
27
|
*
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"ai-messages.d.ts","sourceRoot":"","sources":["../../src/infra/ai-messages.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;GAgBG;AAEH,OAAO,KAAK,
|
|
1
|
+
{"version":3,"file":"ai-messages.d.ts","sourceRoot":"","sources":["../../src/infra/ai-messages.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;GAgBG;AAEH,OAAO,EAAe,KAAK,OAAO,EAAE,MAAM,uBAAuB,CAAC;AAClE,OAAO,KAAK,EAAE,UAAU,EAAE,mBAAmB,EAAE,MAAM,aAAa,CAAC;AAEnE,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,oBAAoB,CAAC;AAEzD,4GAA4G;AAC5G,wBAAgB,qBAAqB,CAAC,OAAO,EAAE,SAAS,OAAO,EAAE,GAAG,mBAAmB,EAAE,CAKxF;AAED,8FAA8F;AAC9F,wBAAgB,yBAAyB,CAAC,QAAQ,EAAE,UAAU,EAAE,GAAG,MAAM,CAKxE;AAED;;;;;;;;GAQG;AACH,wBAAgB,eAAe,CAAC,OAAO,EAAE,OAAO,GAAG,OAAO,CAEzD;AAED;;;;;;;;GAQG;AACH,wBAAgB,sBAAsB,CAAC,IAAI,EAAE,UAAU,EAAE,GAAG,OAAO,EAAE,CAIpE;AAED;;;GAGG;AACH,wBAAgB,gBAAgB,CAAC,IAAI,EAAE,UAAU,EAAE,EAAE,QAAQ,EAAE,cAAc,GAAG,UAAU,EAAE,CAE3F"}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"verify.d.ts","sourceRoot":"","sources":["../../src/phases/verify.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AAEH,OAAO,KAAK,EACX,KAAK,EACL,GAAG,EACH,eAAe,EAEf,MAAM,uBAAuB,CAAC;AAC/B,OAAO,KAAK,EACX,eAAe,EACf,oBAAoB,EACpB,eAAe,EACf,qBAAqB,EACrB,kBAAkB,EAClB,UAAU,EACV,MAAM,aAAa,CAAC;AAoCrB,OAAO,KAAK,EAAE,oBAAoB,EAAE,MAAM,sBAAsB,CAAC;AACjE,MAAM,WAAW,oBAAoB;IACpC,cAAc,CAAC,EAAE,SAAS,UAAU,EAAE,CAAC;IACvC,QAAQ,CAAC,EAAE;QAAE,KAAK,CAAC,EAAE,MAAM,CAAC;QAAC,IAAI,CAAC,EAAE,MAAM,CAAA;KAAE,CAAC;IAC7C,mBAAmB,CAAC,EAAE,MAAM,CAAC;CAC7B;
|
|
1
|
+
{"version":3,"file":"verify.d.ts","sourceRoot":"","sources":["../../src/phases/verify.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AAEH,OAAO,KAAK,EACX,KAAK,EACL,GAAG,EACH,eAAe,EAEf,MAAM,uBAAuB,CAAC;AAC/B,OAAO,KAAK,EACX,eAAe,EACf,oBAAoB,EACpB,eAAe,EACf,qBAAqB,EACrB,kBAAkB,EAClB,UAAU,EACV,MAAM,aAAa,CAAC;AAoCrB,OAAO,KAAK,EAAE,oBAAoB,EAAE,MAAM,sBAAsB,CAAC;AACjE,MAAM,WAAW,oBAAoB;IACpC,cAAc,CAAC,EAAE,SAAS,UAAU,EAAE,CAAC;IACvC,QAAQ,CAAC,EAAE;QAAE,KAAK,CAAC,EAAE,MAAM,CAAC;QAAC,IAAI,CAAC,EAAE,MAAM,CAAA;KAAE,CAAC;IAC7C,mBAAmB,CAAC,EAAE,MAAM,CAAC;CAC7B;AAiTD,wBAAgB,qBAAqB,CAAC,GAAG,EAAE,eAAe,GAAG,MAAM,CAkClE;AAED,wBAAgB,0BAA0B,CACzC,MAAM,EAAE,kBAAkB,GACxB,MAAM,GAAG,IAAI,CAcf;AAED,wFAAwF;AACxF,qBAAa,qBAAsB,SAAQ,KAAK;IAC/C,QAAQ,CAAC,KAAK,EAAE,MAAM,CAAC;IACvB,QAAQ,CAAC,YAAY,EAAE,MAAM,CAAC;IAC9B,QAAQ,CAAC,QAAQ,EAAE,eAAe,CAAC,MAAM,CAAC,EAAE,CAAC;IAC7C,QAAQ,CAAC,KAAK,EAAE,qBAAqB,CAAC;IACtC,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IAE1B,YACC,MAAM,EAAE,kBAAkB,EAC1B,YAAY,EAAE,MAAM,EACpB,KAAK,EAAE,qBAAqB,EAW5B;CACD;AA4UD,wBAAgB,4BAA4B,CAAC,GAAG,EAAE,eAAe,GAAG,OAAO,CAK1E;AAED,qGAAqG;AACrG,wBAAgB,8BAA8B,CAC7C,OAAO,EAAE,MAAM,EACf,MAAM,EAAE,kBAAkB,EAC1B,UAAU,EAAE,oBAAoB,EAChC,UAAU,GAAE,eAAe,GAAG,IAAW,EACzC,QAAQ,GAAE,oBAAyB,EACnC,SAAS,SAAI,GACX;IAAE,OAAO,EAAE,MAAM,CAAC;IAAC,MAAM,EAAE,kBAAkB,CAAC;IAAC,OAAO,EAAE,eAAe,EAAE,CAAA;CAAE,CAyB7E;AAkYD,wBAAgB,aAAa,CAC5B,OAAO,EAAE,MAAM,EACf,UAAU,EAAE,oBAAoB,EAChC,UAAU,GAAE,eAAe,GAAG,IAAW,EACzC,QAAQ,GAAE,oBAAyB,GACjC,kBAAkB,CA4CpB;AAED,iHAAiH;AACjH,wBAAgB,kBAAkB,CACjC,OAAO,EAAE,MAAM,EACf,IAAI,EAAE,eAAe,EAAE,EACvB,UAAU,EAAE,oBAAoB,EAChC,UAAU,GAAE,eAAe,GAAG,IAAW,EACzC,QAAQ,GAAE,oBAAyB,GACjC,MAAM,CA4MR;AA2CD,wBAAsB,YAAY,CACjC,OAAO,EAAE,MAAM,EACf,IAAI,EAAE,eAAe,EAAE,EACvB,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,EACjB,IAAI,EAAE;IAAE,MAAM,EAAE,MAAM,CAAC;IAAC,OAAO,CAAC,EAAE,eAAe,CAAA;CAAE,EACnD,MAAM,CAAC,EAAE,WAAW,EACpB,QAAQ,CAAC,EAAE,oBAAoB,GAC7B,OAAO,CAAC,MAAM,CAAC,CA2DjB"}
|
package/dist/provider-eval.js
CHANGED
|
@@ -156,7 +156,7 @@ function formatProviderEvaluation(report) {
|
|
|
156
156
|
import fs2 from "fs";
|
|
157
157
|
|
|
158
158
|
// src/constants.ts
|
|
159
|
-
var VERSION = "9.6.
|
|
159
|
+
var VERSION = "9.6.2";
|
|
160
160
|
var SETTLED_TRIGGER_COOLDOWN_MS = 10 * 60000;
|
|
161
161
|
var COMPACT_SYSTEM_PREFIX = "You are an expert conversation summarizer for a coding agent. " + "Produce structured markdown summaries. " + "Follow output format exactly. " + "Use EXACT names \u2014 never paraphrase code identifiers. " + "Trust deterministic extraction data over intuition.";
|
|
162
162
|
var PROFILES = {
|
|
@@ -1815,6 +1815,229 @@ var processToolSupport = new ToolSupportCache;
|
|
|
1815
1815
|
var processTokenCalibration = new TokenCalibrationStore;
|
|
1816
1816
|
var _default = createServices();
|
|
1817
1817
|
|
|
1818
|
+
// src/domain/telemetry.ts
|
|
1819
|
+
function p95(values) {
|
|
1820
|
+
if (!values.length)
|
|
1821
|
+
return 0;
|
|
1822
|
+
const sorted = [...values].sort((a, b) => a - b);
|
|
1823
|
+
return sorted[Math.max(0, Math.ceil(sorted.length * 0.95) - 1)] ?? 0;
|
|
1824
|
+
}
|
|
1825
|
+
function stats(entries, damage) {
|
|
1826
|
+
const evidence = entries.filter((entry) => entry.status !== "dry-run");
|
|
1827
|
+
const successfulRuns = evidence.filter((entry) => entry.status === "success");
|
|
1828
|
+
const quality = successfulRuns.filter((entry) => typeof entry.verificationScore === "number");
|
|
1829
|
+
const appliedRunIds = new Set(successfulRuns.filter((entry) => typeof entry.runId === "string" && entry.runId.length >= 8).map((entry) => entry.runId));
|
|
1830
|
+
const observedScores = new Map;
|
|
1831
|
+
for (const observation of damage) {
|
|
1832
|
+
if (!observation.runId || !appliedRunIds.has(observation.runId) || typeof observation.damageScore !== "number" || !Number.isFinite(observation.damageScore))
|
|
1833
|
+
continue;
|
|
1834
|
+
observedScores.set(observation.runId, Math.max(observedScores.get(observation.runId) ?? 0, Math.max(0, Math.min(100, observation.damageScore))));
|
|
1835
|
+
}
|
|
1836
|
+
const damaging = [...observedScores.values()].filter((score) => score > 0).length;
|
|
1837
|
+
return {
|
|
1838
|
+
runs: entries.length,
|
|
1839
|
+
appliedRuns: evidence.length,
|
|
1840
|
+
successRate: evidence.length ? successfulRuns.length / evidence.length : 1,
|
|
1841
|
+
avgQuality: quality.length ? quality.reduce((sum, entry) => sum + (entry.verificationScore ?? 0), 0) / quality.length : null,
|
|
1842
|
+
qualityCoverage: evidence.length ? quality.length / evidence.length : 0,
|
|
1843
|
+
p95LatencyMs: p95(evidence.map((entry) => entry.durationMs ?? entry.avgLatency).filter(Number.isFinite)),
|
|
1844
|
+
avgTokens: evidence.length ? evidence.reduce((sum, entry) => sum + entry.totalInput + entry.totalCacheHit + (entry.totalCacheWrite ?? 0) + entry.totalOutput, 0) / evidence.length : 0,
|
|
1845
|
+
fallbackRate: evidence.length ? evidence.filter((entry) => entry.method === "heuristic" || Array.isArray(entry.providerRoutes) && entry.providerRoutes.some((route) => route.successes < route.calls)).length / evidence.length : 0,
|
|
1846
|
+
damageRate: observedScores.size ? damaging / observedScores.size : 0,
|
|
1847
|
+
damageCoverage: successfulRuns.length ? observedScores.size / successfulRuns.length : 0
|
|
1848
|
+
};
|
|
1849
|
+
}
|
|
1850
|
+
function roundStats(value) {
|
|
1851
|
+
return {
|
|
1852
|
+
...value,
|
|
1853
|
+
successRate: Math.round(value.successRate * 1000) / 1000,
|
|
1854
|
+
avgQuality: value.avgQuality == null ? null : Math.round(value.avgQuality * 10) / 10,
|
|
1855
|
+
qualityCoverage: Math.round(value.qualityCoverage * 1000) / 1000,
|
|
1856
|
+
p95LatencyMs: Math.round(value.p95LatencyMs),
|
|
1857
|
+
avgTokens: Math.round(value.avgTokens),
|
|
1858
|
+
fallbackRate: Math.round(value.fallbackRate * 1000) / 1000,
|
|
1859
|
+
damageRate: Math.round(value.damageRate * 1000) / 1000,
|
|
1860
|
+
damageCoverage: Math.round(value.damageCoverage * 1000) / 1000
|
|
1861
|
+
};
|
|
1862
|
+
}
|
|
1863
|
+
function assessCanary(entries, damageEntries, options) {
|
|
1864
|
+
const minCanaryRuns = Math.max(5, options.minCanaryRuns ?? 20);
|
|
1865
|
+
const canaryEntries = entries.filter((entry) => entry.metricsSchemaVersion === 2 && entry.version === options.version && entry.releaseChannel === "canary").slice(-Math.max(100, minCanaryRuns));
|
|
1866
|
+
const baselineEntries = entries.filter((entry) => entry.metricsSchemaVersion === 2 && (entry.releaseChannel ?? "stable") === "stable").slice(-(options.baselineRuns ?? Math.max(50, minCanaryRuns * 2)));
|
|
1867
|
+
const baseline = stats(baselineEntries, damageEntries);
|
|
1868
|
+
const canary = stats(canaryEntries, damageEntries);
|
|
1869
|
+
const triggers = [];
|
|
1870
|
+
const failureBaseline = 1 - baseline.successRate;
|
|
1871
|
+
const failureCanary = 1 - canary.successRate;
|
|
1872
|
+
if (canary.appliedRuns >= 3 && (failureCanary > 0.050001 || failureCanary - failureBaseline >= 0.050001)) {
|
|
1873
|
+
triggers.push({
|
|
1874
|
+
metric: "failure-rate",
|
|
1875
|
+
baseline: failureBaseline,
|
|
1876
|
+
canary: failureCanary,
|
|
1877
|
+
threshold: failureCanary > 0.050001 ? ">5% absolute" : "+5pp regression"
|
|
1878
|
+
});
|
|
1879
|
+
}
|
|
1880
|
+
if (canary.avgQuality != null && (canary.avgQuality < 85 || baseline.avgQuality != null && baseline.avgQuality - canary.avgQuality >= 5)) {
|
|
1881
|
+
triggers.push({
|
|
1882
|
+
metric: "quality",
|
|
1883
|
+
baseline: baseline.avgQuality ?? 0,
|
|
1884
|
+
canary: canary.avgQuality,
|
|
1885
|
+
threshold: canary.avgQuality < 85 ? "<85 absolute" : "-5 points"
|
|
1886
|
+
});
|
|
1887
|
+
}
|
|
1888
|
+
if (baseline.p95LatencyMs >= 1000 && canary.p95LatencyMs >= baseline.p95LatencyMs * 1.5) {
|
|
1889
|
+
triggers.push({ metric: "latency", baseline: baseline.p95LatencyMs, canary: canary.p95LatencyMs, threshold: "+50% p95" });
|
|
1890
|
+
}
|
|
1891
|
+
if (baseline.avgTokens >= 1000 && canary.avgTokens >= baseline.avgTokens * 1.5) {
|
|
1892
|
+
triggers.push({ metric: "tokens", baseline: baseline.avgTokens, canary: canary.avgTokens, threshold: "+50%" });
|
|
1893
|
+
}
|
|
1894
|
+
if (canary.fallbackRate - baseline.fallbackRate >= 0.1) {
|
|
1895
|
+
triggers.push({ metric: "fallback", baseline: baseline.fallbackRate, canary: canary.fallbackRate, threshold: "+10pp" });
|
|
1896
|
+
}
|
|
1897
|
+
if (canary.damageRate - baseline.damageRate >= 0.1) {
|
|
1898
|
+
triggers.push({ metric: "damage", baseline: baseline.damageRate, canary: canary.damageRate, threshold: "+10pp" });
|
|
1899
|
+
}
|
|
1900
|
+
const canarySampleAdequacy = Math.min(1, canary.appliedRuns / minCanaryRuns);
|
|
1901
|
+
const baselineSampleAdequacy = Math.min(1, baseline.appliedRuns / Math.max(20, minCanaryRuns));
|
|
1902
|
+
const dataConfidence = Math.round(100 * (canarySampleAdequacy * 0.25 + baselineSampleAdequacy * 0.15 + canary.qualityCoverage * canarySampleAdequacy * 0.2 + canary.damageCoverage * canarySampleAdequacy * 0.2 + baseline.damageCoverage * baselineSampleAdequacy * 0.2));
|
|
1903
|
+
const reasons = [];
|
|
1904
|
+
let decision = "hold";
|
|
1905
|
+
if (triggers.length && canary.appliedRuns >= 3) {
|
|
1906
|
+
decision = "rollback";
|
|
1907
|
+
reasons.push(...triggers.map((trigger) => trigger.metric + " crossed " + trigger.threshold));
|
|
1908
|
+
} else if (canary.appliedRuns < minCanaryRuns) {
|
|
1909
|
+
reasons.push("need " + (minCanaryRuns - canary.appliedRuns) + " more canary runs with applied outcomes");
|
|
1910
|
+
} else if (baseline.appliedRuns < Math.max(20, minCanaryRuns)) {
|
|
1911
|
+
reasons.push("stable baseline is too small");
|
|
1912
|
+
} else if (canary.qualityCoverage < 0.7) {
|
|
1913
|
+
reasons.push("schema-v2 quality coverage is below 70%");
|
|
1914
|
+
} else if (canary.damageCoverage < 0.7) {
|
|
1915
|
+
reasons.push("correlated canary damage-observation coverage is below 70%");
|
|
1916
|
+
} else if (baseline.damageCoverage < 0.7) {
|
|
1917
|
+
reasons.push("correlated stable damage-observation coverage is below 70%");
|
|
1918
|
+
} else if ((canary.avgQuality ?? 0) < 85) {
|
|
1919
|
+
reasons.push("absolute verifier quality is below 85");
|
|
1920
|
+
} else if (canary.successRate < 0.949999) {
|
|
1921
|
+
reasons.push("absolute success rate is below 95%");
|
|
1922
|
+
} else {
|
|
1923
|
+
decision = "promote";
|
|
1924
|
+
reasons.push("sample, absolute quality, reliability, latency, token, fallback, and damage gates passed");
|
|
1925
|
+
}
|
|
1926
|
+
return {
|
|
1927
|
+
version: options.version,
|
|
1928
|
+
decision,
|
|
1929
|
+
dataConfidence,
|
|
1930
|
+
baseline: roundStats(baseline),
|
|
1931
|
+
canary: roundStats(canary),
|
|
1932
|
+
triggers,
|
|
1933
|
+
reasons
|
|
1934
|
+
};
|
|
1935
|
+
}
|
|
1936
|
+
function safeMetricLabel(value, fallback) {
|
|
1937
|
+
if (typeof value !== "string" || !/^[\w./:@+-]{1,160}$/.test(value))
|
|
1938
|
+
return fallback;
|
|
1939
|
+
return value;
|
|
1940
|
+
}
|
|
1941
|
+
var TELEMETRY_FAILURE_KINDS = new Set([
|
|
1942
|
+
"cancelled",
|
|
1943
|
+
"timeout",
|
|
1944
|
+
"rate-limit",
|
|
1945
|
+
"authentication",
|
|
1946
|
+
"budget",
|
|
1947
|
+
"output-limit",
|
|
1948
|
+
"provider",
|
|
1949
|
+
"persistence",
|
|
1950
|
+
"validation",
|
|
1951
|
+
"verification",
|
|
1952
|
+
"yield",
|
|
1953
|
+
"internal"
|
|
1954
|
+
]);
|
|
1955
|
+
function isTelemetryFailureKind(value) {
|
|
1956
|
+
return typeof value === "string" && TELEMETRY_FAILURE_KINDS.has(value);
|
|
1957
|
+
}
|
|
1958
|
+
function buildPrivacySafeTelemetry(entries, damageEntries, options) {
|
|
1959
|
+
const groups = new Map;
|
|
1960
|
+
const failures = {};
|
|
1961
|
+
for (const entry of entries) {
|
|
1962
|
+
const version = safeMetricLabel(entry.version, "legacy");
|
|
1963
|
+
const channel = entry.releaseChannel === "canary" ? "canary" : "stable";
|
|
1964
|
+
const provider = safeMetricLabel(entry.provider, "unknown");
|
|
1965
|
+
const rawModel = safeMetricLabel(entry.model, "unknown");
|
|
1966
|
+
const model = rawModel.startsWith(provider + "/") ? rawModel.slice(provider.length + 1) : rawModel;
|
|
1967
|
+
const key = [version, channel, provider, model].join("\x00");
|
|
1968
|
+
const group = groups.get(key) ?? {
|
|
1969
|
+
version,
|
|
1970
|
+
channel,
|
|
1971
|
+
provider,
|
|
1972
|
+
model,
|
|
1973
|
+
runs: 0,
|
|
1974
|
+
successes: 0,
|
|
1975
|
+
quality: 0,
|
|
1976
|
+
qualityRuns: 0,
|
|
1977
|
+
latency: 0,
|
|
1978
|
+
input: 0,
|
|
1979
|
+
output: 0
|
|
1980
|
+
};
|
|
1981
|
+
group.runs++;
|
|
1982
|
+
if (entry.status === "success" || entry.status === "dry-run")
|
|
1983
|
+
group.successes++;
|
|
1984
|
+
if (entry.metricsSchemaVersion === 2 && typeof entry.verificationScore === "number") {
|
|
1985
|
+
group.quality += entry.verificationScore;
|
|
1986
|
+
group.qualityRuns++;
|
|
1987
|
+
}
|
|
1988
|
+
group.latency += entry.avgLatency;
|
|
1989
|
+
group.input += entry.totalInput + entry.totalCacheHit + (entry.totalCacheWrite ?? 0);
|
|
1990
|
+
group.output += entry.totalOutput;
|
|
1991
|
+
groups.set(key, group);
|
|
1992
|
+
if (isTelemetryFailureKind(entry.failureKind)) {
|
|
1993
|
+
failures[entry.failureKind] = (failures[entry.failureKind] ?? 0) + 1;
|
|
1994
|
+
}
|
|
1995
|
+
}
|
|
1996
|
+
const aggregates = [...groups.values()].map((group) => ({
|
|
1997
|
+
version: group.version,
|
|
1998
|
+
channel: group.channel,
|
|
1999
|
+
provider: group.provider,
|
|
2000
|
+
model: group.model,
|
|
2001
|
+
runs: group.runs,
|
|
2002
|
+
successes: group.successes,
|
|
2003
|
+
avgQuality: group.qualityRuns ? Math.round(group.quality / group.qualityRuns * 10) / 10 : null,
|
|
2004
|
+
avgLatencyMs: group.runs ? Math.round(group.latency / group.runs) : 0,
|
|
2005
|
+
inputTokens: group.input,
|
|
2006
|
+
outputTokens: group.output
|
|
2007
|
+
})).sort((a, b) => b.runs - a.runs || a.version.localeCompare(b.version));
|
|
2008
|
+
return {
|
|
2009
|
+
generatedAt: new Date().toISOString(),
|
|
2010
|
+
totalRuns: entries.length,
|
|
2011
|
+
aggregates,
|
|
2012
|
+
failures,
|
|
2013
|
+
canary: assessCanary(entries, damageEntries, options),
|
|
2014
|
+
privacy: "aggregate-only; no session ids, project ids, prompts, summaries, paths, or error text"
|
|
2015
|
+
};
|
|
2016
|
+
}
|
|
2017
|
+
function formatPrivacySafeTelemetry(report) {
|
|
2018
|
+
const lines = [
|
|
2019
|
+
"# Smart Compact Telemetry",
|
|
2020
|
+
"",
|
|
2021
|
+
"Privacy: " + report.privacy + ".",
|
|
2022
|
+
"",
|
|
2023
|
+
"| Version | Channel | Provider/model | Runs | Success | Quality | Latency | Input | Output |",
|
|
2024
|
+
"|---|---|---|---:|---:|---:|---:|---:|---:|"
|
|
2025
|
+
];
|
|
2026
|
+
for (const item of report.aggregates) {
|
|
2027
|
+
lines.push("| " + item.version + " | " + item.channel + " | " + item.provider + "/" + item.model + " | " + item.runs + " | " + item.successes + "/" + item.runs + " | " + (item.avgQuality == null ? "n/a" : item.avgQuality.toFixed(1)) + " | " + item.avgLatencyMs + "ms | " + item.inputTokens + " | " + item.outputTokens + " |");
|
|
2028
|
+
}
|
|
2029
|
+
lines.push("", "## Canary: " + report.canary.decision.toUpperCase() + " (data confidence " + report.canary.dataConfidence + "%)", "");
|
|
2030
|
+
const baseline = report.canary.baseline;
|
|
2031
|
+
const canary = report.canary.canary;
|
|
2032
|
+
lines.push("| Gate | Stable baseline | Canary |", "|---|---:|---:|", "| Runs (total/applied) | " + baseline.runs + "/" + baseline.appliedRuns + " | " + canary.runs + "/" + canary.appliedRuns + " |", "| Success | " + Math.round(baseline.successRate * 100) + "% | " + Math.round(canary.successRate * 100) + "% |", "| Verify quality | " + (baseline.avgQuality ?? "n/a") + " | " + (canary.avgQuality ?? "n/a") + " |", "| p95 duration | " + baseline.p95LatencyMs + "ms | " + canary.p95LatencyMs + "ms |", "| Avg tokens | " + baseline.avgTokens + " | " + canary.avgTokens + " |", "| Fallback | " + Math.round(baseline.fallbackRate * 100) + "% | " + Math.round(canary.fallbackRate * 100) + "% |", "| Damage | " + Math.round(baseline.damageRate * 100) + "% | " + Math.round(canary.damageRate * 100) + "% |", "| Damage observed | " + Math.round(baseline.damageCoverage * 100) + "% | " + Math.round(canary.damageCoverage * 100) + "% |", "");
|
|
2033
|
+
for (const reason of report.canary.reasons)
|
|
2034
|
+
lines.push("- " + reason);
|
|
2035
|
+
const failureText = Object.entries(report.failures).map(([kind, count]) => kind + "=" + count).join(", ");
|
|
2036
|
+
lines.push("", "Failures: " + (failureText || "none classified"));
|
|
2037
|
+
return lines.join(`
|
|
2038
|
+
`);
|
|
2039
|
+
}
|
|
2040
|
+
|
|
1818
2041
|
// src/utils/cache.ts
|
|
1819
2042
|
var INTERNAL_PHASES = new Set([
|
|
1820
2043
|
"explore-retry",
|
|
@@ -5,7 +5,7 @@ var __require = import.meta.require;
|
|
|
5
5
|
import { ModelRegistry, ModelRuntime } from "@earendil-works/pi-coding-agent";
|
|
6
6
|
|
|
7
7
|
// src/constants.ts
|
|
8
|
-
var VERSION = "9.6.
|
|
8
|
+
var VERSION = "9.6.2";
|
|
9
9
|
var SETTLED_TRIGGER_COOLDOWN_MS = 10 * 60000;
|
|
10
10
|
var COMPACT_SYSTEM_PREFIX = "You are an expert conversation summarizer for a coding agent. " + "Produce structured markdown summaries. " + "Follow output format exactly. " + "Use EXACT names \u2014 never paraphrase code identifiers. " + "Trust deterministic extraction data over intuition.";
|
|
11
11
|
var PROFILES = {
|
|
@@ -1667,6 +1667,229 @@ var processToolSupport = new ToolSupportCache;
|
|
|
1667
1667
|
var processTokenCalibration = new TokenCalibrationStore;
|
|
1668
1668
|
var _default = createServices();
|
|
1669
1669
|
|
|
1670
|
+
// src/domain/telemetry.ts
|
|
1671
|
+
function p95(values) {
|
|
1672
|
+
if (!values.length)
|
|
1673
|
+
return 0;
|
|
1674
|
+
const sorted = [...values].sort((a, b) => a - b);
|
|
1675
|
+
return sorted[Math.max(0, Math.ceil(sorted.length * 0.95) - 1)] ?? 0;
|
|
1676
|
+
}
|
|
1677
|
+
function stats(entries, damage) {
|
|
1678
|
+
const evidence = entries.filter((entry) => entry.status !== "dry-run");
|
|
1679
|
+
const successfulRuns = evidence.filter((entry) => entry.status === "success");
|
|
1680
|
+
const quality = successfulRuns.filter((entry) => typeof entry.verificationScore === "number");
|
|
1681
|
+
const appliedRunIds = new Set(successfulRuns.filter((entry) => typeof entry.runId === "string" && entry.runId.length >= 8).map((entry) => entry.runId));
|
|
1682
|
+
const observedScores = new Map;
|
|
1683
|
+
for (const observation of damage) {
|
|
1684
|
+
if (!observation.runId || !appliedRunIds.has(observation.runId) || typeof observation.damageScore !== "number" || !Number.isFinite(observation.damageScore))
|
|
1685
|
+
continue;
|
|
1686
|
+
observedScores.set(observation.runId, Math.max(observedScores.get(observation.runId) ?? 0, Math.max(0, Math.min(100, observation.damageScore))));
|
|
1687
|
+
}
|
|
1688
|
+
const damaging = [...observedScores.values()].filter((score) => score > 0).length;
|
|
1689
|
+
return {
|
|
1690
|
+
runs: entries.length,
|
|
1691
|
+
appliedRuns: evidence.length,
|
|
1692
|
+
successRate: evidence.length ? successfulRuns.length / evidence.length : 1,
|
|
1693
|
+
avgQuality: quality.length ? quality.reduce((sum, entry) => sum + (entry.verificationScore ?? 0), 0) / quality.length : null,
|
|
1694
|
+
qualityCoverage: evidence.length ? quality.length / evidence.length : 0,
|
|
1695
|
+
p95LatencyMs: p95(evidence.map((entry) => entry.durationMs ?? entry.avgLatency).filter(Number.isFinite)),
|
|
1696
|
+
avgTokens: evidence.length ? evidence.reduce((sum, entry) => sum + entry.totalInput + entry.totalCacheHit + (entry.totalCacheWrite ?? 0) + entry.totalOutput, 0) / evidence.length : 0,
|
|
1697
|
+
fallbackRate: evidence.length ? evidence.filter((entry) => entry.method === "heuristic" || Array.isArray(entry.providerRoutes) && entry.providerRoutes.some((route) => route.successes < route.calls)).length / evidence.length : 0,
|
|
1698
|
+
damageRate: observedScores.size ? damaging / observedScores.size : 0,
|
|
1699
|
+
damageCoverage: successfulRuns.length ? observedScores.size / successfulRuns.length : 0
|
|
1700
|
+
};
|
|
1701
|
+
}
|
|
1702
|
+
function roundStats(value) {
|
|
1703
|
+
return {
|
|
1704
|
+
...value,
|
|
1705
|
+
successRate: Math.round(value.successRate * 1000) / 1000,
|
|
1706
|
+
avgQuality: value.avgQuality == null ? null : Math.round(value.avgQuality * 10) / 10,
|
|
1707
|
+
qualityCoverage: Math.round(value.qualityCoverage * 1000) / 1000,
|
|
1708
|
+
p95LatencyMs: Math.round(value.p95LatencyMs),
|
|
1709
|
+
avgTokens: Math.round(value.avgTokens),
|
|
1710
|
+
fallbackRate: Math.round(value.fallbackRate * 1000) / 1000,
|
|
1711
|
+
damageRate: Math.round(value.damageRate * 1000) / 1000,
|
|
1712
|
+
damageCoverage: Math.round(value.damageCoverage * 1000) / 1000
|
|
1713
|
+
};
|
|
1714
|
+
}
|
|
1715
|
+
function assessCanary(entries, damageEntries, options) {
|
|
1716
|
+
const minCanaryRuns = Math.max(5, options.minCanaryRuns ?? 20);
|
|
1717
|
+
const canaryEntries = entries.filter((entry) => entry.metricsSchemaVersion === 2 && entry.version === options.version && entry.releaseChannel === "canary").slice(-Math.max(100, minCanaryRuns));
|
|
1718
|
+
const baselineEntries = entries.filter((entry) => entry.metricsSchemaVersion === 2 && (entry.releaseChannel ?? "stable") === "stable").slice(-(options.baselineRuns ?? Math.max(50, minCanaryRuns * 2)));
|
|
1719
|
+
const baseline = stats(baselineEntries, damageEntries);
|
|
1720
|
+
const canary = stats(canaryEntries, damageEntries);
|
|
1721
|
+
const triggers = [];
|
|
1722
|
+
const failureBaseline = 1 - baseline.successRate;
|
|
1723
|
+
const failureCanary = 1 - canary.successRate;
|
|
1724
|
+
if (canary.appliedRuns >= 3 && (failureCanary > 0.050001 || failureCanary - failureBaseline >= 0.050001)) {
|
|
1725
|
+
triggers.push({
|
|
1726
|
+
metric: "failure-rate",
|
|
1727
|
+
baseline: failureBaseline,
|
|
1728
|
+
canary: failureCanary,
|
|
1729
|
+
threshold: failureCanary > 0.050001 ? ">5% absolute" : "+5pp regression"
|
|
1730
|
+
});
|
|
1731
|
+
}
|
|
1732
|
+
if (canary.avgQuality != null && (canary.avgQuality < 85 || baseline.avgQuality != null && baseline.avgQuality - canary.avgQuality >= 5)) {
|
|
1733
|
+
triggers.push({
|
|
1734
|
+
metric: "quality",
|
|
1735
|
+
baseline: baseline.avgQuality ?? 0,
|
|
1736
|
+
canary: canary.avgQuality,
|
|
1737
|
+
threshold: canary.avgQuality < 85 ? "<85 absolute" : "-5 points"
|
|
1738
|
+
});
|
|
1739
|
+
}
|
|
1740
|
+
if (baseline.p95LatencyMs >= 1000 && canary.p95LatencyMs >= baseline.p95LatencyMs * 1.5) {
|
|
1741
|
+
triggers.push({ metric: "latency", baseline: baseline.p95LatencyMs, canary: canary.p95LatencyMs, threshold: "+50% p95" });
|
|
1742
|
+
}
|
|
1743
|
+
if (baseline.avgTokens >= 1000 && canary.avgTokens >= baseline.avgTokens * 1.5) {
|
|
1744
|
+
triggers.push({ metric: "tokens", baseline: baseline.avgTokens, canary: canary.avgTokens, threshold: "+50%" });
|
|
1745
|
+
}
|
|
1746
|
+
if (canary.fallbackRate - baseline.fallbackRate >= 0.1) {
|
|
1747
|
+
triggers.push({ metric: "fallback", baseline: baseline.fallbackRate, canary: canary.fallbackRate, threshold: "+10pp" });
|
|
1748
|
+
}
|
|
1749
|
+
if (canary.damageRate - baseline.damageRate >= 0.1) {
|
|
1750
|
+
triggers.push({ metric: "damage", baseline: baseline.damageRate, canary: canary.damageRate, threshold: "+10pp" });
|
|
1751
|
+
}
|
|
1752
|
+
const canarySampleAdequacy = Math.min(1, canary.appliedRuns / minCanaryRuns);
|
|
1753
|
+
const baselineSampleAdequacy = Math.min(1, baseline.appliedRuns / Math.max(20, minCanaryRuns));
|
|
1754
|
+
const dataConfidence = Math.round(100 * (canarySampleAdequacy * 0.25 + baselineSampleAdequacy * 0.15 + canary.qualityCoverage * canarySampleAdequacy * 0.2 + canary.damageCoverage * canarySampleAdequacy * 0.2 + baseline.damageCoverage * baselineSampleAdequacy * 0.2));
|
|
1755
|
+
const reasons = [];
|
|
1756
|
+
let decision = "hold";
|
|
1757
|
+
if (triggers.length && canary.appliedRuns >= 3) {
|
|
1758
|
+
decision = "rollback";
|
|
1759
|
+
reasons.push(...triggers.map((trigger) => trigger.metric + " crossed " + trigger.threshold));
|
|
1760
|
+
} else if (canary.appliedRuns < minCanaryRuns) {
|
|
1761
|
+
reasons.push("need " + (minCanaryRuns - canary.appliedRuns) + " more canary runs with applied outcomes");
|
|
1762
|
+
} else if (baseline.appliedRuns < Math.max(20, minCanaryRuns)) {
|
|
1763
|
+
reasons.push("stable baseline is too small");
|
|
1764
|
+
} else if (canary.qualityCoverage < 0.7) {
|
|
1765
|
+
reasons.push("schema-v2 quality coverage is below 70%");
|
|
1766
|
+
} else if (canary.damageCoverage < 0.7) {
|
|
1767
|
+
reasons.push("correlated canary damage-observation coverage is below 70%");
|
|
1768
|
+
} else if (baseline.damageCoverage < 0.7) {
|
|
1769
|
+
reasons.push("correlated stable damage-observation coverage is below 70%");
|
|
1770
|
+
} else if ((canary.avgQuality ?? 0) < 85) {
|
|
1771
|
+
reasons.push("absolute verifier quality is below 85");
|
|
1772
|
+
} else if (canary.successRate < 0.949999) {
|
|
1773
|
+
reasons.push("absolute success rate is below 95%");
|
|
1774
|
+
} else {
|
|
1775
|
+
decision = "promote";
|
|
1776
|
+
reasons.push("sample, absolute quality, reliability, latency, token, fallback, and damage gates passed");
|
|
1777
|
+
}
|
|
1778
|
+
return {
|
|
1779
|
+
version: options.version,
|
|
1780
|
+
decision,
|
|
1781
|
+
dataConfidence,
|
|
1782
|
+
baseline: roundStats(baseline),
|
|
1783
|
+
canary: roundStats(canary),
|
|
1784
|
+
triggers,
|
|
1785
|
+
reasons
|
|
1786
|
+
};
|
|
1787
|
+
}
|
|
1788
|
+
function safeMetricLabel(value, fallback) {
|
|
1789
|
+
if (typeof value !== "string" || !/^[\w./:@+-]{1,160}$/.test(value))
|
|
1790
|
+
return fallback;
|
|
1791
|
+
return value;
|
|
1792
|
+
}
|
|
1793
|
+
var TELEMETRY_FAILURE_KINDS = new Set([
|
|
1794
|
+
"cancelled",
|
|
1795
|
+
"timeout",
|
|
1796
|
+
"rate-limit",
|
|
1797
|
+
"authentication",
|
|
1798
|
+
"budget",
|
|
1799
|
+
"output-limit",
|
|
1800
|
+
"provider",
|
|
1801
|
+
"persistence",
|
|
1802
|
+
"validation",
|
|
1803
|
+
"verification",
|
|
1804
|
+
"yield",
|
|
1805
|
+
"internal"
|
|
1806
|
+
]);
|
|
1807
|
+
function isTelemetryFailureKind(value) {
|
|
1808
|
+
return typeof value === "string" && TELEMETRY_FAILURE_KINDS.has(value);
|
|
1809
|
+
}
|
|
1810
|
+
function buildPrivacySafeTelemetry(entries, damageEntries, options) {
|
|
1811
|
+
const groups = new Map;
|
|
1812
|
+
const failures = {};
|
|
1813
|
+
for (const entry of entries) {
|
|
1814
|
+
const version = safeMetricLabel(entry.version, "legacy");
|
|
1815
|
+
const channel = entry.releaseChannel === "canary" ? "canary" : "stable";
|
|
1816
|
+
const provider = safeMetricLabel(entry.provider, "unknown");
|
|
1817
|
+
const rawModel = safeMetricLabel(entry.model, "unknown");
|
|
1818
|
+
const model = rawModel.startsWith(provider + "/") ? rawModel.slice(provider.length + 1) : rawModel;
|
|
1819
|
+
const key = [version, channel, provider, model].join("\x00");
|
|
1820
|
+
const group = groups.get(key) ?? {
|
|
1821
|
+
version,
|
|
1822
|
+
channel,
|
|
1823
|
+
provider,
|
|
1824
|
+
model,
|
|
1825
|
+
runs: 0,
|
|
1826
|
+
successes: 0,
|
|
1827
|
+
quality: 0,
|
|
1828
|
+
qualityRuns: 0,
|
|
1829
|
+
latency: 0,
|
|
1830
|
+
input: 0,
|
|
1831
|
+
output: 0
|
|
1832
|
+
};
|
|
1833
|
+
group.runs++;
|
|
1834
|
+
if (entry.status === "success" || entry.status === "dry-run")
|
|
1835
|
+
group.successes++;
|
|
1836
|
+
if (entry.metricsSchemaVersion === 2 && typeof entry.verificationScore === "number") {
|
|
1837
|
+
group.quality += entry.verificationScore;
|
|
1838
|
+
group.qualityRuns++;
|
|
1839
|
+
}
|
|
1840
|
+
group.latency += entry.avgLatency;
|
|
1841
|
+
group.input += entry.totalInput + entry.totalCacheHit + (entry.totalCacheWrite ?? 0);
|
|
1842
|
+
group.output += entry.totalOutput;
|
|
1843
|
+
groups.set(key, group);
|
|
1844
|
+
if (isTelemetryFailureKind(entry.failureKind)) {
|
|
1845
|
+
failures[entry.failureKind] = (failures[entry.failureKind] ?? 0) + 1;
|
|
1846
|
+
}
|
|
1847
|
+
}
|
|
1848
|
+
const aggregates = [...groups.values()].map((group) => ({
|
|
1849
|
+
version: group.version,
|
|
1850
|
+
channel: group.channel,
|
|
1851
|
+
provider: group.provider,
|
|
1852
|
+
model: group.model,
|
|
1853
|
+
runs: group.runs,
|
|
1854
|
+
successes: group.successes,
|
|
1855
|
+
avgQuality: group.qualityRuns ? Math.round(group.quality / group.qualityRuns * 10) / 10 : null,
|
|
1856
|
+
avgLatencyMs: group.runs ? Math.round(group.latency / group.runs) : 0,
|
|
1857
|
+
inputTokens: group.input,
|
|
1858
|
+
outputTokens: group.output
|
|
1859
|
+
})).sort((a, b) => b.runs - a.runs || a.version.localeCompare(b.version));
|
|
1860
|
+
return {
|
|
1861
|
+
generatedAt: new Date().toISOString(),
|
|
1862
|
+
totalRuns: entries.length,
|
|
1863
|
+
aggregates,
|
|
1864
|
+
failures,
|
|
1865
|
+
canary: assessCanary(entries, damageEntries, options),
|
|
1866
|
+
privacy: "aggregate-only; no session ids, project ids, prompts, summaries, paths, or error text"
|
|
1867
|
+
};
|
|
1868
|
+
}
|
|
1869
|
+
function formatPrivacySafeTelemetry(report) {
|
|
1870
|
+
const lines = [
|
|
1871
|
+
"# Smart Compact Telemetry",
|
|
1872
|
+
"",
|
|
1873
|
+
"Privacy: " + report.privacy + ".",
|
|
1874
|
+
"",
|
|
1875
|
+
"| Version | Channel | Provider/model | Runs | Success | Quality | Latency | Input | Output |",
|
|
1876
|
+
"|---|---|---|---:|---:|---:|---:|---:|---:|"
|
|
1877
|
+
];
|
|
1878
|
+
for (const item of report.aggregates) {
|
|
1879
|
+
lines.push("| " + item.version + " | " + item.channel + " | " + item.provider + "/" + item.model + " | " + item.runs + " | " + item.successes + "/" + item.runs + " | " + (item.avgQuality == null ? "n/a" : item.avgQuality.toFixed(1)) + " | " + item.avgLatencyMs + "ms | " + item.inputTokens + " | " + item.outputTokens + " |");
|
|
1880
|
+
}
|
|
1881
|
+
lines.push("", "## Canary: " + report.canary.decision.toUpperCase() + " (data confidence " + report.canary.dataConfidence + "%)", "");
|
|
1882
|
+
const baseline = report.canary.baseline;
|
|
1883
|
+
const canary = report.canary.canary;
|
|
1884
|
+
lines.push("| Gate | Stable baseline | Canary |", "|---|---:|---:|", "| Runs (total/applied) | " + baseline.runs + "/" + baseline.appliedRuns + " | " + canary.runs + "/" + canary.appliedRuns + " |", "| Success | " + Math.round(baseline.successRate * 100) + "% | " + Math.round(canary.successRate * 100) + "% |", "| Verify quality | " + (baseline.avgQuality ?? "n/a") + " | " + (canary.avgQuality ?? "n/a") + " |", "| p95 duration | " + baseline.p95LatencyMs + "ms | " + canary.p95LatencyMs + "ms |", "| Avg tokens | " + baseline.avgTokens + " | " + canary.avgTokens + " |", "| Fallback | " + Math.round(baseline.fallbackRate * 100) + "% | " + Math.round(canary.fallbackRate * 100) + "% |", "| Damage | " + Math.round(baseline.damageRate * 100) + "% | " + Math.round(canary.damageRate * 100) + "% |", "| Damage observed | " + Math.round(baseline.damageCoverage * 100) + "% | " + Math.round(canary.damageCoverage * 100) + "% |", "");
|
|
1885
|
+
for (const reason of report.canary.reasons)
|
|
1886
|
+
lines.push("- " + reason);
|
|
1887
|
+
const failureText = Object.entries(report.failures).map(([kind, count]) => kind + "=" + count).join(", ");
|
|
1888
|
+
lines.push("", "Failures: " + (failureText || "none classified"));
|
|
1889
|
+
return lines.join(`
|
|
1890
|
+
`);
|
|
1891
|
+
}
|
|
1892
|
+
|
|
1670
1893
|
// src/utils/cache.ts
|
|
1671
1894
|
var INTERNAL_PHASES = new Set([
|
|
1672
1895
|
"explore-retry",
|
|
@@ -1796,8 +2019,9 @@ function hasListedPath(listed, file, display, normalizedOwners) {
|
|
|
1796
2019
|
}
|
|
1797
2020
|
return false;
|
|
1798
2021
|
}
|
|
1799
|
-
function outcomeClaims(summary) {
|
|
1800
|
-
|
|
2022
|
+
function outcomeClaims(summary, pathEvidence) {
|
|
2023
|
+
const pathLines = new Set(Array.from(pathEvidence, ([path2, display]) => [path2, display, "`" + path2 + "`"]).flat());
|
|
2024
|
+
return Array.from(new Set(summary.split(/\r?\n/).map((line) => line.replace(/^\s*(?:[-*+]|\d+[.)])\s+/, "").replace(/^\[[ x]\]\s+/i, "").trim()).filter((line) => line.length > 0 && !line.startsWith("#") && !pathLines.has(line)).filter((line) => HIGH_RISK_OUTCOME_RE.test(line)).filter((line) => /\bno\s+(?:errors?|failures?)\b/i.test(line) || !NEGATED_OUTCOME_RE.test(line)))).slice(0, 12);
|
|
1801
2025
|
}
|
|
1802
2026
|
function classifyOutcomeClaim(claim) {
|
|
1803
2027
|
const lower = claim.toLowerCase();
|
|
@@ -2031,7 +2255,7 @@ function stemToken(token) {
|
|
|
2031
2255
|
return lower;
|
|
2032
2256
|
}
|
|
2033
2257
|
function semanticTokens(text) {
|
|
2034
|
-
return (text.normalize("NFKC").match(/[\p{L}\p{N}_-]+/gu) ?? []).map(stemToken).filter((token) => token.length > 2);
|
|
2258
|
+
return (text.normalize("NFKC").match(/[\p{L}\p{N}_-]+/gu) ?? []).map(stemToken).filter((token) => token.length > 2 || NEGATION_MARKERS.has(token));
|
|
2035
2259
|
}
|
|
2036
2260
|
var semanticShapeCache = new Map;
|
|
2037
2261
|
var semanticFragmentCache = new Map;
|
|
@@ -2051,7 +2275,7 @@ function hasEffectiveTargetNegation(tokens, anchor) {
|
|
|
2051
2275
|
if (token !== anchor)
|
|
2052
2276
|
return false;
|
|
2053
2277
|
const nearbyStart = Math.max(0, anchorIndex - 2);
|
|
2054
|
-
const nearbyNegations = tokens.slice(nearbyStart, anchorIndex + 3).map((near, offset) => NEGATION_MARKERS.has(near) ? nearbyStart + offset : -1).filter((index) => index >= 0);
|
|
2278
|
+
const nearbyNegations = tokens.slice(nearbyStart, anchorIndex + 3).map((near, offset) => NEGATION_MARKERS.has(near) && !(near === "without" && nearbyStart + offset > anchorIndex) ? nearbyStart + offset : -1).filter((index) => index >= 0);
|
|
2055
2279
|
const governingStart = Math.max(0, anchorIndex - 3);
|
|
2056
2280
|
const preceding = tokens.slice(governingStart, anchorIndex);
|
|
2057
2281
|
const nearbyGuards = preceding.map((near, offset) => POLARITY_INVERTING_GUARDS.has(near) ? governingStart + offset : -1).filter((index) => index >= 0);
|
|
@@ -2100,12 +2324,13 @@ function hasSemanticEvidence(source, target) {
|
|
|
2100
2324
|
});
|
|
2101
2325
|
}
|
|
2102
2326
|
function hasSemanticContradiction(source, target) {
|
|
2327
|
+
const sourceFragments = new Set(semanticFragments(source).map((tokens) => tokens.join(" ")));
|
|
2103
2328
|
const { sourceTokens, concepts, anchor, negative, conditional } = semanticShape(source);
|
|
2104
2329
|
if (!anchor)
|
|
2105
2330
|
return false;
|
|
2106
2331
|
const required = Math.min(concepts.length, Math.max(1, Math.ceil(concepts.length * 0.6)));
|
|
2107
2332
|
return semanticFragments(target).some((tokens) => {
|
|
2108
|
-
if (!tokens.includes(anchor))
|
|
2333
|
+
if (!tokens.includes(anchor) || sourceFragments.has(tokens.join(" ")))
|
|
2109
2334
|
return false;
|
|
2110
2335
|
const overlap = concepts.filter((concept) => tokens.includes(concept)).length;
|
|
2111
2336
|
if (overlap < required)
|
|
@@ -2361,7 +2586,7 @@ function verifyProgressConsistency(parsed, extraction, collected, paths, accumul
|
|
|
2361
2586
|
}
|
|
2362
2587
|
}
|
|
2363
2588
|
}
|
|
2364
|
-
function verifyOpenLoopsAndClaims(summary, parsed, extraction, continuity, evidence, collected, accumulator) {
|
|
2589
|
+
function verifyOpenLoopsAndClaims(summary, parsed, paths, extraction, continuity, evidence, collected, accumulator) {
|
|
2365
2590
|
const unresolvedCount = collected.unresolved.length + (continuity?.openLoops.filter((loop) => loop.status !== "resolved").length ?? 0);
|
|
2366
2591
|
if (unresolvedCount >= 1 && !findSection(parsed, "open-loops") && !summary.toLowerCase().replace(/\\/g, "/").includes("unresolved")) {
|
|
2367
2592
|
addGap(accumulator, { kind: "missing-open-loops", unresolvedCount }, 5);
|
|
@@ -2369,7 +2594,7 @@ function verifyOpenLoopsAndClaims(summary, parsed, extraction, continuity, evide
|
|
|
2369
2594
|
if (!evidence.sourceMessages)
|
|
2370
2595
|
return;
|
|
2371
2596
|
const tools = successfulToolEvidence(evidence.sourceMessages);
|
|
2372
|
-
for (const claim of outcomeClaims(summary)) {
|
|
2597
|
+
for (const claim of outcomeClaims(summary, paths.rendered)) {
|
|
2373
2598
|
if (!successfulToolSupportsClaim(claim, tools, extraction)) {
|
|
2374
2599
|
addGap(accumulator, { kind: "unsupported-claim", claim }, 20);
|
|
2375
2600
|
}
|
|
@@ -2385,7 +2610,7 @@ function verifySummary(summary, extraction, continuity = null, evidence = {}) {
|
|
|
2385
2610
|
verifySemanticCoverage(parsed, collected, accumulator);
|
|
2386
2611
|
verifyFileReferences(summary, extraction, continuity, evidence, collected, paths, accumulator);
|
|
2387
2612
|
verifyProgressConsistency(parsed, extraction, collected, paths, accumulator);
|
|
2388
|
-
verifyOpenLoopsAndClaims(summary, parsed, extraction, continuity, evidence, collected, accumulator);
|
|
2613
|
+
verifyOpenLoopsAndClaims(summary, parsed, paths, extraction, continuity, evidence, collected, accumulator);
|
|
2389
2614
|
const score = Math.max(0, accumulator.score);
|
|
2390
2615
|
return {
|
|
2391
2616
|
ok: accumulator.gaps.length === 0 && score >= 85,
|
package/dist/telemetry-report.js
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
var __require = import.meta.require;
|
|
3
3
|
|
|
4
4
|
// src/constants.ts
|
|
5
|
-
var VERSION = "9.6.
|
|
5
|
+
var VERSION = "9.6.2";
|
|
6
6
|
var SETTLED_TRIGGER_COOLDOWN_MS = 10 * 60000;
|
|
7
7
|
var COMPACT_SYSTEM_PREFIX = "You are an expert conversation summarizer for a coding agent. " + "Produce structured markdown summaries. " + "Follow output format exactly. " + "Use EXACT names \u2014 never paraphrase code identifiers. " + "Trust deterministic extraction data over intuition.";
|
|
8
8
|
var PROFILES = {
|
package/dist/types.d.ts
CHANGED
|
@@ -84,6 +84,8 @@ export interface LLMCallMetric {
|
|
|
84
84
|
cacheWriteTokens: number;
|
|
85
85
|
latencyMs: number;
|
|
86
86
|
success: boolean;
|
|
87
|
+
/** Content-free provider failure category; never stores response/error text. */
|
|
88
|
+
failureKind?: TelemetryFailureKind;
|
|
87
89
|
/** True when provider usage was absent or partial and local estimates were used. */
|
|
88
90
|
usageEstimated?: boolean;
|
|
89
91
|
}
|
|
@@ -94,6 +96,7 @@ export interface ProviderRouteMetric {
|
|
|
94
96
|
model: string;
|
|
95
97
|
calls: number;
|
|
96
98
|
successes: number;
|
|
99
|
+
failures?: Partial<Record<TelemetryFailureKind, number>>;
|
|
97
100
|
avgLatencyMs: number;
|
|
98
101
|
inputTokens: number;
|
|
99
102
|
outputTokens: number;
|