@avocadostudio-ai/orchestrator-core 0.15.0 → 0.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -1338,6 +1338,15 @@ export async function runChatPipeline(ctx, body, options) {
|
|
|
1338
1338
|
* true sentence written about the turn is the one the user never sees.
|
|
1339
1339
|
*/
|
|
1340
1340
|
let strippedPropsNote;
|
|
1341
|
+
/*
|
|
1342
|
+
* The change_log coverage counts, hoisted for the same reason: they are
|
|
1343
|
+
* computed inside the resolve block below, and the result row that should
|
|
1344
|
+
* carry them is written far past it. They have been written to the log
|
|
1345
|
+
* since the validator was added but never to telemetry, so there was no
|
|
1346
|
+
* way to ask how often a planner under-describes what it is about to do —
|
|
1347
|
+
* only to read about one instance at a time.
|
|
1348
|
+
*/
|
|
1349
|
+
let changelogTelemetryFields = {};
|
|
1341
1350
|
const usageFields = planUsage ? {
|
|
1342
1351
|
inputTokens: planUsage.inputTokens,
|
|
1343
1352
|
outputTokens: planUsage.outputTokens,
|
|
@@ -1416,6 +1425,11 @@ export async function runChatPipeline(ctx, body, options) {
|
|
|
1416
1425
|
// Not an op narration, so not something to trim for over-describing.
|
|
1417
1426
|
notes: hallucinationResult.note ? [hallucinationResult.note] : undefined
|
|
1418
1427
|
});
|
|
1428
|
+
changelogTelemetryFields = {
|
|
1429
|
+
changelogMissingCount: changelogResult.missingCount || undefined,
|
|
1430
|
+
changelogExtraCount: changelogResult.extraCount || undefined,
|
|
1431
|
+
changelogFieldMislabelCount: changelogResult.fieldMislabelCount || undefined
|
|
1432
|
+
};
|
|
1419
1433
|
if (changelogResult.missingCount > 0) {
|
|
1420
1434
|
ctx.log.warn({
|
|
1421
1435
|
event: "planner_incomplete_changelog",
|
|
@@ -1775,6 +1789,7 @@ export async function runChatPipeline(ctx, body, options) {
|
|
|
1775
1789
|
id: chatRequestId,
|
|
1776
1790
|
at: new Date().toISOString(),
|
|
1777
1791
|
phase: "result",
|
|
1792
|
+
...changelogTelemetryFields,
|
|
1778
1793
|
session: body.session,
|
|
1779
1794
|
requestedSlug,
|
|
1780
1795
|
effectiveSlug,
|
|
@@ -1833,6 +1848,7 @@ export async function runChatPipeline(ctx, body, options) {
|
|
|
1833
1848
|
id: chatRequestId,
|
|
1834
1849
|
at: new Date().toISOString(),
|
|
1835
1850
|
phase: "result",
|
|
1851
|
+
...changelogTelemetryFields,
|
|
1836
1852
|
session: body.session,
|
|
1837
1853
|
requestedSlug,
|
|
1838
1854
|
effectiveSlug,
|
|
@@ -1936,6 +1952,7 @@ export async function runChatPipeline(ctx, body, options) {
|
|
|
1936
1952
|
id: chatRequestId,
|
|
1937
1953
|
at: new Date().toISOString(),
|
|
1938
1954
|
phase: "result",
|
|
1955
|
+
...changelogTelemetryFields,
|
|
1939
1956
|
session: body.session,
|
|
1940
1957
|
requestedSlug,
|
|
1941
1958
|
effectiveSlug,
|
|
@@ -1996,6 +2013,7 @@ export async function runChatPipeline(ctx, body, options) {
|
|
|
1996
2013
|
id: chatRequestId,
|
|
1997
2014
|
at: new Date().toISOString(),
|
|
1998
2015
|
phase: "result",
|
|
2016
|
+
...changelogTelemetryFields,
|
|
1999
2017
|
session: body.session,
|
|
2000
2018
|
requestedSlug,
|
|
2001
2019
|
effectiveSlug,
|
|
@@ -2084,6 +2102,7 @@ export async function runChatPipeline(ctx, body, options) {
|
|
|
2084
2102
|
id: chatRequestId,
|
|
2085
2103
|
at: new Date().toISOString(),
|
|
2086
2104
|
phase: "result",
|
|
2105
|
+
...changelogTelemetryFields,
|
|
2087
2106
|
session: body.session,
|
|
2088
2107
|
requestedSlug,
|
|
2089
2108
|
effectiveSlug,
|
|
@@ -2724,6 +2743,7 @@ export async function runChatPipeline(ctx, body, options) {
|
|
|
2724
2743
|
id: chatRequestId,
|
|
2725
2744
|
at: new Date().toISOString(),
|
|
2726
2745
|
phase: "result",
|
|
2746
|
+
...changelogTelemetryFields,
|
|
2727
2747
|
session: body.session,
|
|
2728
2748
|
requestedSlug,
|
|
2729
2749
|
effectiveSlug,
|
|
@@ -2819,6 +2839,7 @@ export async function runChatPipeline(ctx, body, options) {
|
|
|
2819
2839
|
id: chatRequestId,
|
|
2820
2840
|
at: new Date().toISOString(),
|
|
2821
2841
|
phase: "result",
|
|
2842
|
+
...changelogTelemetryFields,
|
|
2822
2843
|
session: body.session,
|
|
2823
2844
|
requestedSlug,
|
|
2824
2845
|
effectiveSlug,
|
|
@@ -3431,7 +3452,21 @@ export async function runChatPipeline(ctx, body, options) {
|
|
|
3431
3452
|
}
|
|
3432
3453
|
})
|
|
3433
3454
|
: planner.generatePlan;
|
|
3434
|
-
|
|
3455
|
+
/*
|
|
3456
|
+
* Two, not three.
|
|
3457
|
+
*
|
|
3458
|
+
* A retry re-sends the same prompt — the only thing that differs is the
|
|
3459
|
+
* output budget after a truncation — so what a retry buys is a fresh sample,
|
|
3460
|
+
* and the second sample is worth taking: across 5,632 real turns in
|
|
3461
|
+
* chat-telemetry, attempt 2 ended productively 53% of the time.
|
|
3462
|
+
*
|
|
3463
|
+
* The third did so 3.7% of the time. 437 of 454 turns that reached it paid a
|
|
3464
|
+
* full extra planning pass — a median 7.5s, three times the planner tokens,
|
|
3465
|
+
* and a "Retrying plan generation (3/3)" the user watches — to arrive at the
|
|
3466
|
+
* same error attempt 2 had already produced. Raising it back is a decision
|
|
3467
|
+
* about that ratio, so re-measure before you do.
|
|
3468
|
+
*/
|
|
3469
|
+
const maxPlanningAttempts = 2;
|
|
3435
3470
|
let initialPlan = null;
|
|
3436
3471
|
let routerDetectedInfo = false;
|
|
3437
3472
|
const planningErrors = [];
|
|
@@ -651,7 +651,9 @@ export function blockContractsSummary(manifest) {
|
|
|
651
651
|
* they are the phrasings this planner *can* execute without a key.
|
|
652
652
|
*/
|
|
653
653
|
export const KEYLESS_PLANNER_NOTICE = "No API key is configured, so this is the built-in demo planner, which handles only simple, literal edits. " +
|
|
654
|
-
"
|
|
654
|
+
"Visual editing is not affected: click any block to select it, edit its fields in the property panel, and publish as usual. " +
|
|
655
|
+
"It is AI editing \u2014 describing a change in a sentence and having it planned for you \u2014 that needs a model. " +
|
|
656
|
+
"Set ANTHROPIC_API_KEY or OPENAI_API_KEY in your project's .env.local and restart the dev server to turn it on.";
|
|
655
657
|
/**
|
|
656
658
|
* The same notice, without the *path*.
|
|
657
659
|
*
|
|
@@ -659,6 +659,27 @@ function remapListItemPatchKeys(args) {
|
|
|
659
659
|
}
|
|
660
660
|
return out;
|
|
661
661
|
}
|
|
662
|
+
/**
|
|
663
|
+
* A model asked for a string sometimes answers with a JSON-encoded array, and
|
|
664
|
+
* sometimes wraps that in an array of one: `["[\"did X\", \"did Y\"]"]`.
|
|
665
|
+
* Returns the decoded entries, or null when this is prose that merely opens
|
|
666
|
+
* with a bracket. Deliberately narrow — a change_log entry is text the user
|
|
667
|
+
* reads before approving, and rewriting it on a guess is worse than leaving it.
|
|
668
|
+
*/
|
|
669
|
+
function decodeJsonStringArray(value) {
|
|
670
|
+
if (typeof value !== "string" || !/^\s*\[/.test(value))
|
|
671
|
+
return null;
|
|
672
|
+
let parsed;
|
|
673
|
+
try {
|
|
674
|
+
parsed = JSON.parse(value);
|
|
675
|
+
}
|
|
676
|
+
catch {
|
|
677
|
+
return null;
|
|
678
|
+
}
|
|
679
|
+
if (!Array.isArray(parsed) || !parsed.every((entry) => typeof entry === "string"))
|
|
680
|
+
return null;
|
|
681
|
+
return parsed;
|
|
682
|
+
}
|
|
662
683
|
// ---------------------------------------------------------------------------
|
|
663
684
|
// Plan candidate normalisation
|
|
664
685
|
// ---------------------------------------------------------------------------
|
|
@@ -999,6 +1020,17 @@ export function normalizePlanCandidate(input, args) {
|
|
|
999
1020
|
if (!raw.patch) {
|
|
1000
1021
|
raw.patch = raw.props ?? raw.changes;
|
|
1001
1022
|
}
|
|
1023
|
+
// `update_item` takes `patch`; `add_item` takes `item`. Both names are
|
|
1024
|
+
// offered side by side in the flattened op schema the planner is handed,
|
|
1025
|
+
// so a planner editing one list entry reaches for the noun that names the
|
|
1026
|
+
// thing it is editing and puts the fields under `item`. Unrepaired, that
|
|
1027
|
+
// reaches Zod as "expected record, received undefined at ops.0.patch" and
|
|
1028
|
+
// the whole plan is thrown away. Scoped to update_item — on add_item,
|
|
1029
|
+
// `item` is the real field and must survive untouched.
|
|
1030
|
+
if (!raw.patch && raw.op === "update_item" && raw.item && typeof raw.item === "object" && !Array.isArray(raw.item)) {
|
|
1031
|
+
raw.patch = raw.item;
|
|
1032
|
+
delete raw.item;
|
|
1033
|
+
}
|
|
1002
1034
|
// LLMs sometimes put prop values directly on the op object instead of
|
|
1003
1035
|
// nesting them under "patch". Extract non-structural keys as the patch.
|
|
1004
1036
|
if (!raw.patch && raw.op === "update_props") {
|
|
@@ -1651,6 +1683,18 @@ export function normalizePlanCandidate(input, args) {
|
|
|
1651
1683
|
.map((s) => s.replace(/^\s*[-•*\d.)\]]+\s*/, "").trim())
|
|
1652
1684
|
.filter((s) => /\w/.test(s));
|
|
1653
1685
|
}
|
|
1686
|
+
// Unwrap a JSON-encoded change_log before the branches below read it. An
|
|
1687
|
+
// array holding one encoded array satisfies neither of them — it is already
|
|
1688
|
+
// an array — so it survives to the approval UI as a single line of escaped
|
|
1689
|
+
// JSON, and the coverage validator then pads it with synthesized duplicates
|
|
1690
|
+
// of the very changes it already describes.
|
|
1691
|
+
const decodedChangeLog = decodeJsonStringArray(root.change_log);
|
|
1692
|
+
if (decodedChangeLog) {
|
|
1693
|
+
root.change_log = decodedChangeLog;
|
|
1694
|
+
}
|
|
1695
|
+
else if (Array.isArray(root.change_log)) {
|
|
1696
|
+
root.change_log = root.change_log.flatMap((entry) => decodeJsonStringArray(entry) ?? [entry]);
|
|
1697
|
+
}
|
|
1654
1698
|
if (typeof root.change_log === "string") {
|
|
1655
1699
|
// Models sometimes emit the literal string "[]" (or "[ ]") for an empty
|
|
1656
1700
|
// change log; treat that as an empty array instead of a one-item "[]" list.
|
|
@@ -41,6 +41,19 @@ export type ChatTelemetryEntry = {
|
|
|
41
41
|
imagesRequested?: number;
|
|
42
42
|
imagesResolved?: number;
|
|
43
43
|
planningAttempts?: number;
|
|
44
|
+
/**
|
|
45
|
+
* change_log coverage drift, from `validateChangelogCoverage`: ops the
|
|
46
|
+
* planner left undescribed, changes it described but will not make, and
|
|
47
|
+
* entries naming a field the op does not touch.
|
|
48
|
+
*
|
|
49
|
+
* The validator has warned about all three since it was written, and a
|
|
50
|
+
* warning can be read one incident at a time but never counted — which is
|
|
51
|
+
* why nobody could say how often the approval UI shows a user a list that
|
|
52
|
+
* does not match the edit they are approving.
|
|
53
|
+
*/
|
|
54
|
+
changelogMissingCount?: number;
|
|
55
|
+
changelogExtraCount?: number;
|
|
56
|
+
changelogFieldMislabelCount?: number;
|
|
44
57
|
contextPackBytes?: number;
|
|
45
58
|
contractMode?: "minimal" | "targeted" | "full";
|
|
46
59
|
contractBytes?: number;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@avocadostudio-ai/orchestrator-core",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.17.0",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"exports": {
|
|
6
6
|
"./package.json": "./package.json",
|
|
@@ -22,8 +22,8 @@
|
|
|
22
22
|
"openai": "^4.87.1",
|
|
23
23
|
"sharp": "^0.35.4",
|
|
24
24
|
"zod": "^4.3.6",
|
|
25
|
-
"@avocadostudio-ai/migration-sdk": "^0.
|
|
26
|
-
"@avocadostudio-ai/shared": "^0.
|
|
25
|
+
"@avocadostudio-ai/migration-sdk": "^0.17.0",
|
|
26
|
+
"@avocadostudio-ai/shared": "^0.17.0"
|
|
27
27
|
},
|
|
28
28
|
"devDependencies": {
|
|
29
29
|
"@anthropic-ai/claude-agent-sdk": "^0.3.220",
|