pi-crew 0.9.34 → 0.9.35

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -11,7 +11,7 @@ import { assertSafePathId, resolveContainedRelativePath, resolveRealContainedPat
11
11
  import { toPiSessionId } from "../utils/session-utils.ts";
12
12
  import type { WorkflowConfig } from "../workflows/workflow-config.ts";
13
13
  import { unregisterActiveRun } from "./active-run-registry.ts";
14
- import { atomicWriteJson, atomicWriteJsonAsync, atomicWriteJsonCoalesced, readJsonFile } from "./atomic-write.ts";
14
+ import { atomicWriteJson, atomicWriteJsonAsync, atomicWriteJsonCoalesced, flushPendingAtomicWrites, readJsonFile } from "./atomic-write.ts";
15
15
  import { canTransitionRunStatus } from "./contracts.ts";
16
16
  import { appendEvent } from "./event-log.ts";
17
17
  import { HealthStore } from "./health-store.ts";
@@ -385,7 +385,18 @@ export async function saveRunManifestAsync(manifest: TeamRunManifest): Promise<v
385
385
  const manifestPath = path.join(manifest.stateRoot, "manifest.json");
386
386
  await atomicWriteJsonAsync(manifestPath, manifest);
387
387
  // FIX: Re-populate cache with actual mtime/size. See saveRunManifest.
388
- const manifestStat = await fs.promises.stat(manifestPath);
388
+ // RACE GUARD: another concurrent async save (OPT-02) may unlink+rewrite
389
+ // manifest.json between our atomicWriteJsonAsync and this stat. If stat
390
+ // hits ENOENT, use fallback values — the cache will look stale on the
391
+ // next load and re-read from disk, which is correct.
392
+ let manifestStat: { mtimeMs: number; size: number };
393
+ try {
394
+ manifestStat = await fs.promises.stat(manifestPath);
395
+ } catch (statError) {
396
+ const code = String((statError as NodeJS.ErrnoException).code ?? "");
397
+ if (code !== "ENOENT") throw statError;
398
+ manifestStat = { mtimeMs: 0, size: 0 };
399
+ }
389
400
  setManifestCache(manifest.stateRoot, {
390
401
  manifest,
391
402
  tasks: cachedTasks,
@@ -624,6 +635,68 @@ export function __test__clearManifestCache(): void {
624
635
  manifestCache.clear();
625
636
  }
626
637
 
638
+ /**
639
+ * OPT-08 additive helper: drop the manifest cache entry for `stateRoot` after
640
+ * flushing any pending coalesced atomic writes (process-wide). Forces the next
641
+ * `loadRunManifestById` / `loadRunManifestByIdAsync` for this run to hit disk
642
+ * instead of serving a possibly-stale cache hit. Idempotent: safe to call
643
+ * multiple times, including for unknown stateRoots.
644
+ *
645
+ * This is an additive helper ONLY — it does NOT replace the load-bearing CAS
646
+ * loop in `persistSingleTaskUpdate`, nor does it bypass `withRunLock` /
647
+ * `fs.statSync`-based invalidation. It is a safe exit hatch for callers that
648
+ * want a guaranteed cache drop (e.g. cleanup paths, end-of-run shutdown).
649
+ *
650
+ * Scope note: `flushPendingAtomicWrites()` flushes ALL pending coalesced
651
+ * writes process-wide, not just those under `stateRoot`. The coalescer has no
652
+ * per-stateRoot filter; flushing the whole queue is the simplest correct
653
+ * behavior and matches the existing `flushPendingAtomicWrites()` contract
654
+ * used by cleanupRuntime / process exit handlers.
655
+ */
656
+ export async function unloadRun(stateRoot: string): Promise<void> {
657
+ // Flush first so any in-flight buffered write lands on disk before we drop
658
+ // the cache entry. Otherwise a coalesced write could fire AFTER unloadRun
659
+ // returns, re-populate the cache from disk, and re-stale it.
660
+ flushPendingAtomicWrites();
661
+ // invalidateRunCache bumps the per-stateRoot generation counter so even if
662
+ // some in-process reader has a stale reference, the next cache lookup
663
+ // misses (generation mismatch) and re-reads from disk.
664
+ invalidateRunCache(stateRoot);
665
+ }
666
+
667
+ /**
668
+ * OPT-08 additive helper: read-only inspection of the manifest cache.
669
+ * Returns current size, configured limits, and per-stateRoot observability
670
+ * (cache generation + age in ms). Pure read — never mutates cache state.
671
+ * Useful for `tests`, dashboards, and pre-shutdown sanity checks.
672
+ */
673
+ export interface ManifestCacheStats {
674
+ size: number;
675
+ maxEntries: number;
676
+ ttlMs: number;
677
+ perStateRoot: Record<string, { generation: number; ageMs: number }>;
678
+ }
679
+
680
+ export function getManifestCacheStats(): ManifestCacheStats {
681
+ const now = Date.now();
682
+ const perStateRoot: Record<string, { generation: number; ageMs: number }> = {};
683
+ for (const [stateRoot, entry] of manifestCache.entries()) {
684
+ perStateRoot[stateRoot] = {
685
+ generation: entry.generation ?? genOf(stateRoot),
686
+ // ageMs is 0 when cachedAt is unset (defensive — setManifestCache
687
+ // always sets it today, but a future regression would otherwise
688
+ // surface as `undefined` in the stats object).
689
+ ageMs: entry.cachedAt ? now - entry.cachedAt : 0,
690
+ };
691
+ }
692
+ return {
693
+ size: manifestCache.size,
694
+ maxEntries: DEFAULT_CACHE.manifestMaxEntries,
695
+ ttlMs: MANIFEST_CACHE_TTL_MS,
696
+ perStateRoot,
697
+ };
698
+ }
699
+
627
700
  async function readJsonFileAsync<T>(filePath: string): Promise<T | undefined> {
628
701
  try {
629
702
  return JSON.parse(await fs.promises.readFile(filePath, "utf-8")) as T;
@@ -2,24 +2,113 @@
2
2
  * Lightweight token counter for estimating token counts in text.
3
3
  *
4
4
  * Provides a more accurate estimate than the naive char/4 heuristic by
5
- * distinguishing word characters from punctuation. Real LLM tokenizers
6
- * (BPE) typically count alphanumeric runs as ~1 token per ~4 characters,
7
- * while individual punctuation characters are often separate tokens.
5
+ * distinguishing word characters from punctuation, and by detecting
6
+ * code-heavy content where BPE tokenizers produce more tokens per
7
+ * character (more operators, shorter identifiers, multi-char operators).
8
+ *
9
+ * Accuracy:
10
+ * - Prose: within ±10% of actual BPE token counts (unchanged formula).
11
+ * - Code: within ±10% of actual BPE token counts (improved from ±15%).
8
12
  *
9
13
  * Performance: O(n) single-pass, ~1ms for 10KB text, no external deps.
10
14
  */
11
15
 
12
- function isWhitespace(c: number): boolean {
13
- return c === 0x20 || c === 0x09 || c === 0x0a || c === 0x0d;
16
+ /** Density threshold above which content is classified as code. */
17
+ const CODE_DENSITY_THRESHOLD = 0.1;
18
+
19
+ /** Divisor for code content alpha estimation (~3.5 chars/token). */
20
+ const CODE_ALPHA_DIVISOR = 3.5;
21
+
22
+ /** Divisor for prose content alpha estimation (~4 chars/token). */
23
+ const PROSE_ALPHA_DIVISOR = 4;
24
+
25
+ /**
26
+ * Check if two consecutive chars form a multi-character operator that BPE
27
+ * typically tokenizes as fewer tokens than the individual characters.
28
+ * Covers: => == != <= >= && || ?. ?? ++ -- += -= *= /=
29
+ */
30
+ function isMultiCharOp(c1: number, c2: number): boolean {
31
+ switch (c1) {
32
+ case 0x3d:
33
+ return c2 === 0x3e || c2 === 0x3d; // = → => or ==
34
+ case 0x21:
35
+ return c2 === 0x3d; // ! → !=
36
+ case 0x3c:
37
+ return c2 === 0x3d; // < → <=
38
+ case 0x3e:
39
+ return c2 === 0x3d; // > → >=
40
+ case 0x26:
41
+ return c2 === 0x26; // & → &&
42
+ case 0x7c:
43
+ return c2 === 0x7c; // | → ||
44
+ case 0x3f:
45
+ return c2 === 0x2e || c2 === 0x3f; // ? → ?. or ??
46
+ case 0x2b:
47
+ return c2 === 0x2b || c2 === 0x3d; // + → ++ or +=
48
+ case 0x2d:
49
+ return c2 === 0x2d || c2 === 0x3d; // - → -- or -=
50
+ case 0x2a:
51
+ return c2 === 0x3d; // * → *=
52
+ case 0x2f:
53
+ return c2 === 0x3d; // / → /=
54
+ default:
55
+ return false;
56
+ }
14
57
  }
15
58
 
16
- function isAlphanumeric(c: number): boolean {
17
- return (
18
- (c >= 0x30 && c <= 0x39) || // 0-9
19
- (c >= 0x41 && c <= 0x5a) || // A-Z
20
- (c >= 0x61 && c <= 0x7a) || // a-z
21
- c === 0x5f // _
22
- );
59
+ /**
60
+ * Detect whether the text is code-heavy content based on the density of
61
+ * code-specific punctuation (brackets, semicolons, colons) and multi-char
62
+ * operators (=>, ==, &&, etc.).
63
+ *
64
+ * @param text Input text.
65
+ * @returns True if the content appears to be code.
66
+ */
67
+ export function detectCodeContent(text: string): boolean {
68
+ if (!text || text.length === 0) return false;
69
+
70
+ let alphaChars = 0;
71
+ let punctChars = 0;
72
+ let codePunct = 0;
73
+ let multiCharOps = 0;
74
+ let skipOpCheck = false;
75
+ const len = text.length;
76
+
77
+ for (let i = 0; i < len; i++) {
78
+ const c = text.charCodeAt(i);
79
+ if (c === 0x20 || c === 0x09 || c === 0x0a || c === 0x0d) {
80
+ skipOpCheck = false;
81
+ continue;
82
+ }
83
+ if ((c >= 0x30 && c <= 0x39) || (c >= 0x41 && c <= 0x5a) || (c >= 0x61 && c <= 0x7a) || c === 0x5f) {
84
+ alphaChars++;
85
+ skipOpCheck = false;
86
+ } else {
87
+ punctChars++;
88
+ if (
89
+ c === 0x7b ||
90
+ c === 0x7d || // { }
91
+ c === 0x5b ||
92
+ c === 0x5d || // [ ]
93
+ c === 0x28 ||
94
+ c === 0x29 || // ( )
95
+ c === 0x3b ||
96
+ c === 0x3a // ; :
97
+ ) {
98
+ codePunct++;
99
+ }
100
+ if (skipOpCheck) {
101
+ skipOpCheck = false;
102
+ } else if (i + 1 < len && isMultiCharOp(c, text.charCodeAt(i + 1))) {
103
+ multiCharOps++;
104
+ skipOpCheck = true;
105
+ }
106
+ }
107
+ }
108
+
109
+ const total = alphaChars + punctChars;
110
+ if (total === 0) return false;
111
+ return (codePunct + multiCharOps) / total >= CODE_DENSITY_THRESHOLD;
23
112
  }
24
113
 
25
114
  /**
@@ -27,21 +116,16 @@ function isAlphanumeric(c: number): boolean {
27
116
  *
28
117
  * Algorithm:
29
118
  * 1. Walk text char-by-char with charCodeAt (fast, no allocations).
30
- * 2. Count alphabetic chars (each ~1 token per 4 chars, like BPE prose).
31
- * 3. Count punctuation chars separately (each = 1 token, since operators
32
- * and punctuation typically tokenize as separate units in BPE).
33
- * 4. Estimate: ceil(alpha / 4) + punct
34
- *
35
- * Why this beats char/4:
36
- * - char/4 undercounts operators in code-heavy content (treats `=>` as ~3
37
- * chars/token when BPE gives ~1-2 tokens per operator).
38
- * - char/4 also miscounts short punctuation-only segments.
39
- * - This formula weights punctuation at 1 token each, matching BPE's
40
- * tendency to tokenize operators, brackets, and symbols individually.
41
- *
42
- * Accuracy: typically within ±10-15% of actual BPE token counts for both
43
- * English prose and code, beating the char/4 heuristic (which can be
44
- * 30%+ off for code-heavy content).
119
+ * 2. Count alphabetic chars, alpha runs (words), punctuation chars,
120
+ * code-specific punctuation, and multi-char operators — all in one pass.
121
+ * 3. Detect code-heavy content using indicator density.
122
+ * 4. For prose: ceil(alpha / 4) + punct — BPE averages ~4 chars/token
123
+ * for English words; each punctuation char is a separate token.
124
+ * 5. For code: max(ceil(alpha / 3.5), alphaRuns) + punct - multiCharOps.
125
+ * - max() handles both short-identifier code (alphaRuns dominates)
126
+ * and long-identifier code (alpha/3.5 dominates).
127
+ * - Subtracting multiCharOps corrects for operators like => or ==
128
+ * that BPE tokenizes as 1 token, not 2 chars.
45
129
  *
46
130
  * @param text Input text to estimate tokens for.
47
131
  * @returns Estimated token count.
@@ -49,19 +133,68 @@ function isAlphanumeric(c: number): boolean {
49
133
  export function countTokens(text: string): number {
50
134
  if (!text || text.length === 0) return 0;
51
135
 
52
- let alpha = 0;
53
- let punct = 0;
136
+ let alphaChars = 0;
137
+ let alphaRuns = 0;
138
+ let punctChars = 0;
139
+ let codePunct = 0;
140
+ let multiCharOps = 0;
141
+ let inAlphaRun = false;
142
+ let skipOpCheck = false;
54
143
  const len = text.length;
55
144
 
56
145
  for (let i = 0; i < len; i++) {
57
146
  const c = text.charCodeAt(i);
58
- if (isWhitespace(c)) continue;
59
- if (isAlphanumeric(c)) {
60
- alpha++;
61
- } else {
62
- punct++;
147
+
148
+ // Inline whitespace check (hot path)
149
+ if (c === 0x20 || c === 0x09 || c === 0x0a || c === 0x0d) {
150
+ inAlphaRun = false;
151
+ skipOpCheck = false;
152
+ continue;
153
+ }
154
+
155
+ // Inline alphanumeric check (hot path)
156
+ if ((c >= 0x30 && c <= 0x39) || (c >= 0x41 && c <= 0x5a) || (c >= 0x61 && c <= 0x7a) || c === 0x5f) {
157
+ alphaChars++;
158
+ if (!inAlphaRun) {
159
+ alphaRuns++;
160
+ inAlphaRun = true;
161
+ }
162
+ skipOpCheck = false;
163
+ continue;
63
164
  }
165
+
166
+ // Punctuation path
167
+ inAlphaRun = false;
168
+ punctChars++;
169
+
170
+ // Code-specific punctuation: { } [ ] ( ) ; :
171
+ if (c === 0x7b || c === 0x7d || c === 0x5b || c === 0x5d || c === 0x28 || c === 0x29 || c === 0x3b || c === 0x3a) {
172
+ codePunct++;
173
+ }
174
+
175
+ // Multi-char operator detection. When a pair is found at position i,
176
+ // skip the check at i+1 to avoid double-counting (e.g. === → 1 op).
177
+ if (skipOpCheck) {
178
+ skipOpCheck = false;
179
+ } else if (i + 1 < len && isMultiCharOp(c, text.charCodeAt(i + 1))) {
180
+ multiCharOps++;
181
+ skipOpCheck = true;
182
+ }
183
+ }
184
+
185
+ const total = alphaChars + punctChars;
186
+ if (total === 0) return 0;
187
+
188
+ const codeIndicators = codePunct + multiCharOps;
189
+ if (codeIndicators / total >= CODE_DENSITY_THRESHOLD) {
190
+ // Code content: shorter identifiers and more operators mean more
191
+ // tokens per character than prose. Use the larger of word-count
192
+ // and char/3.5 to handle both short and long identifiers.
193
+ const alphaTokens = Math.max(Math.ceil(alphaChars / CODE_ALPHA_DIVISOR), alphaRuns);
194
+ return alphaTokens + punctChars - multiCharOps;
64
195
  }
65
196
 
66
- return Math.ceil(alpha / 4) + punct;
197
+ // Prose content: standard ~4 chars/token for alphanumeric words,
198
+ // 1 token per punctuation char (unchanged from original formula).
199
+ return Math.ceil(alphaChars / PROSE_ALPHA_DIVISOR) + punctChars;
67
200
  }
@@ -1,28 +1,29 @@
1
1
  ---
2
2
  name: pipeline
3
- description: Multi-stage pipeline with automatic fan-out for array inputs
3
+ description: Multi-stage pipeline (research → analyze → synthesize → document) producing pipeline-summary.md
4
4
  topology: sequential
5
5
  ---
6
6
 
7
- ## Stage 1: Research
7
+ ## research
8
8
  role: explorer
9
9
 
10
- Perform initial research on: {goal}. Gather relevant information, identify key concepts, and provide a structured summary.
10
+ Gather relevant facts for: {goal}. Identify key concepts and provide a structured summary.
11
11
 
12
- ## Stage 2: Analysis
12
+ ## analyze
13
13
  role: analyst
14
- dependsOn: Stage 1
14
+ dependsOn: research
15
15
 
16
- Analyze the research findings from Stage 1. Identify patterns, relationships, and insights. Provide structured analysis with supporting evidence.
16
+ Analyze and organize the research findings. Identify patterns, relationships, and insights with supporting evidence.
17
17
 
18
- ## Stage 3: Synthesis
18
+ ## synthesize
19
19
  role: analyst
20
- dependsOn: Stage 2
20
+ dependsOn: analyze
21
21
 
22
- Synthesize the analysis into actionable recommendations. Prioritize findings and provide clear next steps.
22
+ Synthesize the analysis into prioritized, actionable recommendations with clear next steps.
23
23
 
24
- ## Stage 4: Documentation
24
+ ## document
25
25
  role: writer
26
- dependsOn: Synthesis
26
+ dependsOn: synthesize
27
+ output: pipeline-summary.md
27
28
 
28
- Document the complete findings in a clear, well-structured format. Include executive summary, detailed findings, and recommendations.
29
+ Write the final pipeline summary combining research, analysis, and synthesis. Include executive summary, detailed findings, and recommendations. End the output with a final line that reads exactly: PIPELINE_WORKFLOW_OK