@crewbie/cli 0.1.0-alpha.49 → 0.1.0-alpha.50
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +48 -2
- package/dist/cli.js +20 -2
- package/dist/cli.js.map +1 -1
- package/dist/config.d.ts +9 -1
- package/dist/config.js +6 -0
- package/dist/config.js.map +1 -1
- package/dist/core.d.ts +2 -0
- package/dist/core.js +8 -0
- package/dist/core.js.map +1 -1
- package/dist/execution/controls.d.ts +18 -1
- package/dist/execution/controls.js +63 -2
- package/dist/execution/controls.js.map +1 -1
- package/dist/execution/dispatch.js +5 -5
- package/dist/execution/dispatch.js.map +1 -1
- package/dist/execution/review-loop.js +2 -2
- package/dist/execution/review-loop.js.map +1 -1
- package/dist/memory/budget.d.ts +13 -0
- package/dist/memory/budget.js +29 -0
- package/dist/memory/budget.js.map +1 -0
- package/dist/memory/context.d.ts +2 -0
- package/dist/memory/context.js +14 -2
- package/dist/memory/context.js.map +1 -1
- package/dist/memory/demote.d.ts +17 -0
- package/dist/memory/demote.js +107 -0
- package/dist/memory/demote.js.map +1 -0
- package/dist/memory/improvement.d.ts +9 -0
- package/dist/memory/improvement.js +40 -3
- package/dist/memory/improvement.js.map +1 -1
- package/dist/memory/launch.d.ts +15 -1
- package/dist/memory/launch.js +15 -8
- package/dist/memory/launch.js.map +1 -1
- package/dist/memory/runner.js +4 -1
- package/dist/memory/runner.js.map +1 -1
- package/dist/reporting/ablation.d.ts +35 -0
- package/dist/reporting/ablation.js +67 -0
- package/dist/reporting/ablation.js.map +1 -0
- package/dist/reporting/dashboard.js +3 -3
- package/dist/reporting/records.d.ts +3 -1
- package/dist/reporting/records.js +12 -4
- package/dist/reporting/records.js.map +1 -1
- package/dist/setup/assessment.js +2 -2
- package/dist/setup/assessment.js.map +1 -1
- package/dist/setup/install.js +3 -3
- package/dist/setup/install.js.map +1 -1
- package/dist/setup/instruction-quality.d.ts +3 -2
- package/dist/setup/instruction-quality.js +63 -8
- package/dist/setup/instruction-quality.js.map +1 -1
- package/dist/setup/onboarding.js +11 -4
- package/dist/setup/onboarding.js.map +1 -1
- package/dist/setup/review.js +3 -1
- package/dist/setup/review.js.map +1 -1
- package/dist/setup/templates.d.ts +1 -1
- package/dist/setup/templates.js +13 -50
- package/dist/setup/templates.js.map +1 -1
- package/docs/operations.md +94 -9
- package/package.json +1 -1
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"templates.js","sourceRoot":"","sources":["../../src/setup/templates.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,gBAAgB,EAA0B,MAAM,cAAc,CAAC;AACxE,OAAO,EAAE,iBAAiB,EAAE,gBAAgB,EAAE,cAAc,EAAE,MAAM,wBAAwB,CAAC;AAC7F,OAAO,EAAE,WAAW,EAAE,MAAM,cAAc,CAAC;AAC3C,OAAO,EAAE,kBAAkB,EAAE,MAAM,kBAAkB,CAAC;AAEtD,MAAM,CAAC,MAAM,OAAO,GAAG;;;;;;;kEAO2C,CAAC;AAEnE,MAAM,CAAC,MAAM,WAAW,GAAG;;;;;;;;;;CAU1B,CAAC;AAEF,0GAA0G;AAC1G,MAAM,CAAC,MAAM,aAAa,GAAG,uGAAuG,CAAC;AACrI,MAAM,CAAC,MAAM,WAAW,GAAG,8BAA8B,CAAC;AAE1D,MAAM,eAAe,GAAG,IAAI,GAAG,CAAC,CAAC,aAAa,EAAE,UAAU,CAAC,CAAC,CAAC;AAC7D,MAAM,eAAe,GAAG,CAAC,MAAM,EAAE,QAAQ,CAAC,CAAC;AAC3C,MAAM,aAAa,GAAG,CAAC,MAAM,EAAE,QAAQ,EAAE,MAAM,EAAE,SAAS,CAAC,CAAC;AAE5D,MAAM,UAAU,YAAY,CAAC,IAAU;IACrC,oHAAoH;IACpH,OAAO,CAAC,GAAG,CAAC,eAAe,CAAC,GAAG,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC,CAAC,CAAC,eAAe,CAAC,CAAC,CAAC,aAAa,CAAC,CAAC,CAAC;AAC/E,CAAC;AAED,MAAM,UAAU,OAAO,CAAC,IAAU,EAAE,MAAc;IAChD,0EAA0E;IAC1E,MAAM,QAAQ,GAAG,CAAC,IAAI,CAAC,YAAY,IAAI,EAAE,CAAC,CAAC,MAAM,CAAC,CAAC,IAAI,EAAE,EAAE,CAAC,CAAC,kBAAkB,CAAC,IAAI,CAAC,CAAC,CAAC;IACvF,MAAM,MAAM,GAA2B;QACrC,WAAW,EAAE,ihCAAihC;QAC9hC,
|
|
1
|
+
{"version":3,"file":"templates.js","sourceRoot":"","sources":["../../src/setup/templates.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,gBAAgB,EAA0B,MAAM,cAAc,CAAC;AACxE,OAAO,EAAE,iBAAiB,EAAE,gBAAgB,EAAE,cAAc,EAAE,MAAM,wBAAwB,CAAC;AAC7F,OAAO,EAAE,WAAW,EAAE,MAAM,cAAc,CAAC;AAC3C,OAAO,EAAE,kBAAkB,EAAE,MAAM,kBAAkB,CAAC;AAEtD,MAAM,CAAC,MAAM,OAAO,GAAG;;;;;;;kEAO2C,CAAC;AAEnE,MAAM,CAAC,MAAM,WAAW,GAAG;;;;;;;;;;CAU1B,CAAC;AAEF,0GAA0G;AAC1G,MAAM,CAAC,MAAM,aAAa,GAAG,uGAAuG,CAAC;AACrI,MAAM,CAAC,MAAM,WAAW,GAAG,8BAA8B,CAAC;AAE1D,MAAM,eAAe,GAAG,IAAI,GAAG,CAAC,CAAC,aAAa,EAAE,UAAU,CAAC,CAAC,CAAC;AAC7D,MAAM,eAAe,GAAG,CAAC,MAAM,EAAE,QAAQ,CAAC,CAAC;AAC3C,MAAM,aAAa,GAAG,CAAC,MAAM,EAAE,QAAQ,EAAE,MAAM,EAAE,SAAS,CAAC,CAAC;AAE5D,MAAM,UAAU,YAAY,CAAC,IAAU;IACrC,oHAAoH;IACpH,OAAO,CAAC,GAAG,CAAC,eAAe,CAAC,GAAG,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC,CAAC,CAAC,eAAe,CAAC,CAAC,CAAC,aAAa,CAAC,CAAC,CAAC;AAC/E,CAAC;AAED,MAAM,UAAU,OAAO,CAAC,IAAU,EAAE,MAAc;IAChD,0EAA0E;IAC1E,MAAM,QAAQ,GAAG,CAAC,IAAI,CAAC,YAAY,IAAI,EAAE,CAAC,CAAC,MAAM,CAAC,CAAC,IAAI,EAAE,EAAE,CAAC,CAAC,kBAAkB,CAAC,IAAI,CAAC,CAAC,CAAC;IACvF,MAAM,MAAM,GAA2B;QACrC,WAAW,EAAE,ihCAAihC;QAC9hC,gIAAgI;QAChI,MAAM,EAAE;wSAC4R;QACpS,QAAQ,EAAE;qcACub;QACjc,QAAQ,EAAE;;;;;;;0QAO4P;KACvQ,CAAC;IACF,OAAO;gBACO,IAAI,CAAC,EAAE;eACR,IAAI,CAAC,SAAS,CAAC,IAAI,CAAC,OAAO,CAAC;;EAEzC,YAAY,CAAC,IAAI,CAAC,CAAC,GAAG,CAAC,CAAC,IAAI,EAAE,EAAE,CAAC,OAAO,IAAI,EAAE,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC;;IAExD,IAAI,CAAC,EAAE;;EAET,IAAI,CAAC,OAAO;;EAEZ,MAAM,CAAC,MAAM,CAAC,MAAM,EAAE,IAAI,CAAC,EAAE,CAAC,CAAC,CAAC,CAAC,MAAM,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC,CAAC,CAAC,EAAE;;EAErD,IAAI,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC,CAAC,yBAAyB,IAAI,CAAC,MAAM,CAAC,GAAG,CAAC,CAAC,KAAK,EAAE,EAAE,CAAC,KAAK,KAAK,EAAE,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE,GAAG,IAAI,CAAC,cAAc,EAAE,MAAM,CAAC,CAAC,CAAC,kCAAkC,IAAI,CAAC,cAAc,CAAC,GAAG,CAAC,CAAC,IAAI,EAAE,EAAE,CAAC,KAAK,IAAI,EAAE,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC,CAAC,EAAE;;EAEhP,aAAa;;EAEb,IAAI,CAAC,WAAW,CAAC,CAAC,CAAC,sEAAsE,gBAAgB,CAAC,IAAI,CAAC,WAAW,CAAC,4HAA4H,CAAC,CAAC,CAAC,EAAE;;EAE5P,MAAM,CAAC,YAAY,CAAC,CAAC,CAAC,KAAK,MAAM,CAAC,YAAY,MAAM,CAAC,CAAC,CAAC,EAAE;kBACzC,IAAI,CAAC,EAAE,kCAAkC,IAAI,CAAC,EAAE;;EAEhE,QAAQ,CAAC,MAAM,CAAC,CAAC,CAAC,4BAA4B,QAAQ,CAAC,GAAG,CAAC,CAAC,IAAI,EAAE,EAAE,CAAC,KAAK,IAAI,IAAI,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,KAAK,CAAC,CAAC,CAAC,EAAE;iCACzE,IAAI,CAAC,EAAE;EACtC,WAAW;CACZ,CAAC;AACF,CAAC;AAED,MAAM,CAAC,MAAM,mBAAmB,GAAG;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;EAsCjC,OAAO;CACR,CAAC;AAEF,MAAM,CAAC,MAAM,KAAK,GAAG;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;EAgGnB,OAAO;CACR,CAAC;AAEF,MAAM,UAAU,SAAS,CAAC,cAAc,GAAG,KAAK,EAAE,eAAe,GAAG,KAAK,EAAE,cAAc,GAAG,KAAK;IAC/F,MAAM,KAAK,GAAG;;;;;;;;;2DAS2C,WAAW;;;CAGrE,CAAC;IACA,OAAO;QACL,oCAAoC,EAAE,gBAAgB,CAAC,KAAK,EAAE,eAAe,CAAC;QAC9E,4CAA4C,EAAE,iBAAiB,CAAC,KAAK,EAAE,eAAe,IAAI,cAAc,CAAC;QACzG,wCAAwC,EAAE;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;EAiC5C,KAAK;;;;;;;CAON;QACG,sCAAsC,EAAE,cAAc,CAAC,KAAK,CAAC;QAC7D,mCAAmC,EAAE;;;;;;;;;;;;;;;EAevC,KAAK;;;;;;CAMN;QACG,wCAAwC,EAAE;;;EAG5C,cAAc,CAAC,CAAC,CAAC,yCAAyC,CAAC,CAAC,CAAC,EAAE;;;;;;;;;;;;;;;;;;EAkB/D,KAAK;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;EAyDL,KAAK;;;;;;;;;;CAUN;QACG,sCAAsC,EAAE;;;;;;;;;;;;;;;;;EAiB1C,KAAK;;;;;;;;;;;;;;EAcL,KAAK;;;;;;;;;;;;;;;;;;;;;EAqBL,KAAK;;;;;;;;;;;;;;;CAeN;KACE,CAAC;AACJ,CAAC"}
|
package/docs/operations.md
CHANGED
|
@@ -99,7 +99,7 @@ requires approved network/proxy/CA configuration, not `strict-ssl=false`.
|
|
|
99
99
|
Review the package contents with `npm pack --dry-run`, then bootstrap with
|
|
100
100
|
`npm publish --access public --tag next`. Complete any 2FA challenge directly
|
|
101
101
|
with npm. Verify the published version using
|
|
102
|
-
`npm view @crewbie/cli@0.1.0-alpha.
|
|
102
|
+
`npm view @crewbie/cli@0.1.0-alpha.50 version --registry=https://registry.npmjs.org`.
|
|
103
103
|
|
|
104
104
|
After the package exists, open its npm **Settings > Trusted publishing**, choose
|
|
105
105
|
GitHub Actions, and configure:
|
|
@@ -207,6 +207,14 @@ additions) must fit GitHub's documented 30,000-character custom agent maximum. I
|
|
|
207
207
|
it does not, init stops before adoption and asks the user to shorten the original;
|
|
208
208
|
it never truncates instructions. Available from alpha.12; earlier releases stopped at 400 words.
|
|
209
209
|
|
|
210
|
+
Generated (non-adopted) charters ship no generic domain advice such as "test
|
|
211
|
+
edge cases" or "consider accessibility"; capable models already know it and it
|
|
212
|
+
adds context to every launch. Only Crewbie workflow invariants remain (tester,
|
|
213
|
+
reviewer, coordinator, improver). Put the repository's own observable checks and
|
|
214
|
+
non-negotiables in each role's `checks` and `nonNegotiables`; the setup review
|
|
215
|
+
flags roles that have none. Shared rules such as PR headings live once in
|
|
216
|
+
`.crewbie/instructions.md`, which every launch embeds, rather than in each charter.
|
|
217
|
+
|
|
210
218
|
Archival requires the exact inspected source hash, rejects conflicting archives
|
|
211
219
|
or edited originals, and is idempotent. Archives are written before originals
|
|
212
220
|
are removed. Installation previews name both actions. Adding these adoptions is
|
|
@@ -601,13 +609,30 @@ observations, not a universal instruction ban or proof that a particular number
|
|
|
601
609
|
of words is harmful.
|
|
602
610
|
|
|
603
611
|
Crewbie's documentation/profile overlap, generic-advice (including generic
|
|
604
|
-
agent charters), unverified-link, npm-script/package-manifest
|
|
605
|
-
|
|
612
|
+
agent charters), unverified-link, npm-script/package-manifest and
|
|
613
|
+
unconditional full-suite signals are **engineering heuristics**,
|
|
606
614
|
not validated causal rules from the paper. They identify concrete material for
|
|
607
615
|
human review. Necessary standalone context and explicit merge/compliance gates
|
|
608
616
|
should be retained. Contradictions, domain relevance and actual benefit still
|
|
609
617
|
need semantic review and representative before/after task evidence.
|
|
610
|
-
|
|
618
|
+
|
|
619
|
+
The assessment also checks the six smells catalogued in
|
|
620
|
+
[*Configuration Smells in AGENTS.md Files*](https://arxiv.org/abs/2606.15828)
|
|
621
|
+
(see the README table). Static signals: `lint-leakage` (style rules a
|
|
622
|
+
linter/formatter enforces; a warning when one is configured, advisory
|
|
623
|
+
otherwise), `context-bloat`, `blind-reference` (a document path whose line
|
|
624
|
+
does not say what it holds or when to read it), and `init-fossilization` (a file
|
|
625
|
+
with exactly one commit while at least ten later commits landed; skipped when
|
|
626
|
+
history is unavailable, such as shallow clones). Skill leakage and conflicting
|
|
627
|
+
instructions need judgment and are checked by the model rubric only.
|
|
628
|
+
|
|
629
|
+
Always-loaded guidance is limited to 200 lines (`limits.guidanceLines`), after
|
|
630
|
+
Anthropic's recommendation for always-loaded instruction files, which the
|
|
631
|
+
smells study also uses. Longer existing files get a `context-bloat` warning.
|
|
632
|
+
Proposed guidance replacements over the limit are rejected, and nothing is
|
|
633
|
+
truncated. Custom-agent charters are not counted: GitHub's prompt character
|
|
634
|
+
limit applies to them. Gloaguen et al. establish no harmful length, so the limit
|
|
635
|
+
is a starting point to validate with `crewbie eval`. Scoped
|
|
611
636
|
Copilot instructions need valid YAML `applyTo` globs. When relocating domain
|
|
612
637
|
guidance, review the source reduction and destination together; preserve policy
|
|
613
638
|
coverage and host-specific instruction support. Every finding must explicitly
|
|
@@ -672,11 +697,21 @@ An LLM's suitability rationale is a proposal, not a benchmark certification.
|
|
|
672
697
|
"modelProfile": "balanced",
|
|
673
698
|
"execution": {
|
|
674
699
|
"maxLaunchesPerBatch": 20,
|
|
675
|
-
"maxAttemptsPerTask": 3
|
|
700
|
+
"maxAttemptsPerTask": 3,
|
|
701
|
+
"maxTokensPerFeature": 2000000
|
|
676
702
|
}
|
|
677
703
|
}
|
|
678
704
|
```
|
|
679
705
|
|
|
706
|
+
`maxTokensPerFeature` is optional and has no default, because no evidence-backed
|
|
707
|
+
value exists. When set, each launch first sums the **Observed tokens** Crewbie
|
|
708
|
+
recorded in the attribution block of every task PR into the plan's feature branch.
|
|
709
|
+
At or above the ceiling, further implementation and review launches stop with a
|
|
710
|
+
visible reason; running sessions are unchanged. PRs without measured usage are
|
|
711
|
+
counted separately, so the total is a lower bound and never zero-filled. Review-
|
|
712
|
+
report PRs and unreported subagent/tool usage are not included. Raise the ceiling
|
|
713
|
+
in a reviewed config change to continue. `preflight` shows the spent total.
|
|
714
|
+
|
|
680
715
|
These configuration fields are optional for legacy configurations so parsing does
|
|
681
716
|
not change approved plan hashes. Absent limits use 20/3. Changing the profile alone
|
|
682
717
|
does not reassign models. Each Crewbie-initiated implementation/review launch
|
|
@@ -690,7 +725,8 @@ publishes a new issue for it; that issue gets its own initial launch, which stil
|
|
|
690
725
|
counts toward the shared task and batch allowances. An earlier issue that is still
|
|
691
726
|
open must be reconciled (closed) first.
|
|
692
727
|
The budget covers Crewbie requests, not the number of internal backend sessions,
|
|
693
|
-
|
|
728
|
+
monetary spend, manual PR follow-ups, onboarding, planning or nightly work; only
|
|
729
|
+
`maxTokensPerFeature` looks at tokens.
|
|
694
730
|
|
|
695
731
|
**Upgrading an in-flight batch:** pre-alpha.10 launch claims have no trustworthy
|
|
696
732
|
complete attempt ledger. Further automatic launches in that batch stop visibly,
|
|
@@ -1031,6 +1067,15 @@ Operational cursors live on the orphan `crewbie/runtime` branch, separately from
|
|
|
1031
1067
|
human-facing memory. It records the latest reviewed fingerprint/outcome per work
|
|
1032
1068
|
item, so no-change analysis can advance without opening a pointless PR.
|
|
1033
1069
|
|
|
1070
|
+
Every proposed change states a `hypothesis` (the recurring failure it should
|
|
1071
|
+
prevent) and an `expectedOutcome` (the observable signal in later runs), each at
|
|
1072
|
+
most 60 words. The improvement PR adds them to `.crewbie/rationale.md`, a bounded
|
|
1073
|
+
ledger of the last 100 entries that lands only when a human merges the PR. It is
|
|
1074
|
+
loaded into the improver's context, never into implementation launches, so
|
|
1075
|
+
later analysis can check whether a rule delivered its expected outcome and
|
|
1076
|
+
propose narrowing or removing rules that did not. The ledger itself cannot be
|
|
1077
|
+
proposed as a change.
|
|
1078
|
+
|
|
1034
1079
|
The reserved proposal branch is `crewbie/improvements`. Concurrent file changes,
|
|
1035
1080
|
a branch behind its base, or a leftover closed proposal branch stop updates for
|
|
1036
1081
|
human reconciliation rather than force-pushing over work. Merge, close and
|
|
@@ -1054,6 +1099,21 @@ heading would block auto-merge after the session ended, with nobody left to fix
|
|
|
1054
1099
|
it. There is no default word limit because none is evidence-backed. Human review
|
|
1055
1100
|
still evaluates rationale and evidence.
|
|
1056
1101
|
|
|
1102
|
+
The same check reports the word count of every memory file the PR adds or
|
|
1103
|
+
changes, read at its head: `.crewbie/instructions.md`, hot files against
|
|
1104
|
+
`limits.hot`, and cold/archive topics against `limits.topic`. It never fails on
|
|
1105
|
+
memory. Instead, before every launch on a `crewbie/<feature>` branch, Crewbie
|
|
1106
|
+
checks the shared and role hot files. When one is over `limits.hot`, Crewbie
|
|
1107
|
+
moves its oldest list entries (the top of the file; agents add new entries at the
|
|
1108
|
+
end) into a new `cold/earlier-<date>.md` topic until the hot file is at 80% of
|
|
1109
|
+
its budget. It links that topic from the matching `index.md`, rebases relative
|
|
1110
|
+
links, and commits everything as one commit on the feature branch. Headings and
|
|
1111
|
+
prose stay in place. If the branch moved or the commit fails, the launch
|
|
1112
|
+
continues with memory as-is, and demotion is retried at the next launch. Agent
|
|
1113
|
+
PR branches and the default branch are never rewritten. Hot memory merged over
|
|
1114
|
+
budget after a feature's last launch is still loaded, and it is demoted at the
|
|
1115
|
+
next launch. Launch prompts show each embedded file's `words/limit`.
|
|
1116
|
+
|
|
1057
1117
|
Copilot's final session summary replaces the specialist's own PR description.
|
|
1058
1118
|
During attribution Crewbie restores the specialist's last description (its own
|
|
1059
1119
|
Copilot edit containing `Specialist: crewbie-<role>`) as the PR body and saves
|
|
@@ -1078,8 +1138,11 @@ PR metadata, not code, approvals or merge state, and avoids another paid
|
|
|
1078
1138
|
implementation run merely to repair prose.
|
|
1079
1139
|
|
|
1080
1140
|
Cloud hosts inject the active charter and may protect its file path from agent
|
|
1081
|
-
tools.
|
|
1082
|
-
|
|
1141
|
+
tools; respect those restrictions. Agents no longer attest which memory they
|
|
1142
|
+
read. Each launch ledger ref points at an annotated tag (which fires no
|
|
1143
|
+
workflows) recording the embedded files and their hashes, plus anything too large
|
|
1144
|
+
to embed or absent. Reports show that as "memory: embedded"; older launches show
|
|
1145
|
+
"unreported". Native Copilot Memory is a separate platform feature, not Crewbie's
|
|
1083
1146
|
reviewed role history or approval of proposed shared decisions.
|
|
1084
1147
|
|
|
1085
1148
|
## Reports and privacy
|
|
@@ -1108,7 +1171,29 @@ bundled SDK runtime against a loopback-only synthetic provider without paid mode
|
|
|
1108
1171
|
calls.
|
|
1109
1172
|
|
|
1110
1173
|
Before broader release, run a consenting personal/organization account matrix:
|
|
1111
|
-
select the actual profile/model, inspect
|
|
1174
|
+
select the actual profile/model, inspect the recorded launch contexts and issue/PR
|
|
1112
1175
|
linkage, exercise maintenance authentication, and confirm private publishing.
|
|
1113
1176
|
No live test is implied by a fixture passing. Registry publishing, trademark
|
|
1114
1177
|
clearance, and paid cloud runs require the project owner's separate decision.
|
|
1178
|
+
|
|
1179
|
+
### Guidance evaluation
|
|
1180
|
+
|
|
1181
|
+
Guidance and memory cost context on every launch; research on repository
|
|
1182
|
+
context files found they often raise cost without improving task success. Check
|
|
1183
|
+
this for your own repository rather than assuming it:
|
|
1184
|
+
|
|
1185
|
+
1. Create two sandbox repositories from the same commit. In both, install the
|
|
1186
|
+
**latest published** Crewbie release, never a local build:
|
|
1187
|
+
`gh release view --repo mvanderbend-msoft/crewbie --json assets` and
|
|
1188
|
+
`npm install --global --ignore-scripts <tgz asset URL>`.
|
|
1189
|
+
2. Keep guidance and memory in the guided sandbox. In the bare sandbox, reduce
|
|
1190
|
+
`AGENTS.md`, scoped instructions, charters' repository sections and
|
|
1191
|
+
`.crewbie` hot/index files to their Crewbie defaults.
|
|
1192
|
+
3. Approve and run the same plan in both. Wait for attribution on each task PR.
|
|
1193
|
+
4. Run `crewbie eval --guided owner/guided --bare owner/bare [--feature crewbie/x] [--json]`.
|
|
1194
|
+
|
|
1195
|
+
The comparison reports merged tasks, measured coverage and observed tokens per
|
|
1196
|
+
merged task per arm. Token cost is compared only when every task in both arms
|
|
1197
|
+
has measured usage. It warns on unequal task counts and fewer than five tasks per
|
|
1198
|
+
arm; one run does not prove a cause. When guidance costs more without merging
|
|
1199
|
+
more tasks, review what it adds.
|