polymath-society 0.2.4 → 0.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +1024 -379
- package/dist/index.js +57 -23
- package/dist/pipeline/coding-agglomerate.js +174 -15
- package/dist/pipeline/coding-aggregate.js +306 -8
- package/dist/pipeline/coding-build.js +355 -8
- package/dist/pipeline/coding-coaching.js +1 -1
- package/dist/pipeline/coding-day-digest.js +343 -32
- package/dist/pipeline/coding-delegation.js +319 -17
- package/dist/pipeline/coding-expertise.js +324 -22
- package/dist/pipeline/coding-flow.js +302 -4
- package/dist/pipeline/coding-focus.js +16306 -0
- package/dist/pipeline/coding-frontier-detail.js +316 -14
- package/dist/pipeline/coding-frontier.js +350 -1
- package/dist/pipeline/coding-gap-dist.js +302 -4
- package/dist/pipeline/coding-gap.js +1 -1
- package/dist/pipeline/coding-grade.js +180 -20
- package/dist/pipeline/coding-nutshell.js +318 -16
- package/dist/pipeline/coding-projects.js +354 -5
- package/dist/pipeline/coding-walkthrough.js +317 -18
- package/dist/web/app.js +69 -19
- package/dist/web/styles.css +1 -1
- package/package.json +2 -2
|
@@ -2,7 +2,7 @@ import{createRequire as __cr}from'module';const require=__cr(import.meta.url);
|
|
|
2
2
|
|
|
3
3
|
// ../../scripts/coding-gap-dist.mts
|
|
4
4
|
import { promises as fs2 } from "fs";
|
|
5
|
-
import
|
|
5
|
+
import path2 from "path";
|
|
6
6
|
|
|
7
7
|
// ../../lib/agents/shared/cliAdapter.ts
|
|
8
8
|
var CLAUDE_BIN = process.env.CLAUDE_BIN || "claude";
|
|
@@ -243,13 +243,311 @@ function proseWords(text) {
|
|
|
243
243
|
return (kept.join(" ").trim().match(/\S+/g) || []).length;
|
|
244
244
|
}
|
|
245
245
|
|
|
246
|
+
// ../../lib/agents/coding/profile.ts
|
|
247
|
+
import path from "path";
|
|
248
|
+
|
|
249
|
+
// ../../lib/calibration/fingerprint.ts
|
|
250
|
+
import { createHash } from "node:crypto";
|
|
251
|
+
function rubricFingerprint(criteria) {
|
|
252
|
+
return createHash("sha256").update(JSON.stringify(criteria)).digest("hex").slice(0, 10);
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
// ../../lib/agents/coding/criteria.ts
|
|
256
|
+
var CODING_CRITERIA = [
|
|
257
|
+
{
|
|
258
|
+
key: "outcomes",
|
|
259
|
+
label: "Outcomes",
|
|
260
|
+
graded: true,
|
|
261
|
+
definition: "Whether they meaningfully move the needle \u2014 a real capability shipped and confirmed, a hard blocker killed, the boundary genuinely pushed. Per-conversation = THIS session's magnitude (cap 10); the headline Outcomes number is the CADENCE of meaningful days across the whole period (computed separately), and several stacked features show up at the DAY level, not in one session.",
|
|
262
|
+
read: "Judge the END STATE on an OBSERVABLE axis: does the thing exist and work at the end (verification) \xD7 how much it unblocks downstream (leverage) \xD7 how many load-bearing things landed (scope). NEVER guess 'how long this would take a strong engineer' \u2014 an LLM can't know that. A session is 'meaningful' at level \u22656 (a real thing shipped/confirmed). Weight load-bearing work over throwaway scripts; discount busywork, abandoned threads, edits that never converged. ANCHORS: L8 = 7\xD71080p Remotion comps rendered+downloaded, or all 7 analysis dimensions driven to completion + viewer rebuilt; L4 = ends on 'I don't have the recs or the script', the pivot goes nowhere; L2 = confirmed a repo is private / 'what should I work on next, I'm confused'. L11 is DAY-LEVEL (several verified load-bearing things across the day) \u2014 never assign 11 to one session.",
|
|
263
|
+
rungs: [
|
|
264
|
+
{ level: 2, marker: "Nothing exists at the end \u2014 a lookup or an orientation." },
|
|
265
|
+
{ level: 4, marker: "Partial \u2014 scaffolding or a small fix, no working capability yet; or a realization the work isn't ready." },
|
|
266
|
+
{ level: 6, marker: "One real working thing shipped and at least loosely confirmed. (meaningful floor)" },
|
|
267
|
+
{ level: 8, marker: "A substantial capability \u2014 or several real things \u2014 built AND verified this session." },
|
|
268
|
+
{ level: 10, marker: "A shipped, end-to-end-checked thing that unblocks a lot downstream \u2014 leverage, not just size. (per-session ceiling)" }
|
|
269
|
+
]
|
|
270
|
+
},
|
|
271
|
+
{
|
|
272
|
+
key: "taste",
|
|
273
|
+
label: "Taste",
|
|
274
|
+
graded: true,
|
|
275
|
+
definition: "How high and how RIGHT a quality bar they hold \u2014 and crucially, whether the call is one the AI could never have reached on its own. Taste is rare and particular: the 'oh shit, they're right' moment \u2014 highly opinionated, highly specific, vindicated. Includes a developed UP-FRONT vision (strongest where the AI is weakest \u2014 writing, design, brand, nuanced product) and knowing where NOT to impose a bar (letting the AI run on throwaway).",
|
|
276
|
+
read: "Count as taste ONLY: a high, specific, vindicated call the AI couldn't have reached (stated up front OR via correction); a developed up-front vision where the AI is weak; handwritten samples/specs; knowing where not to impose. Do NOT count: generic feedback/correction the AI could've gone either way on ('make it white not yellow', 'no dividers', 'square checkbox') \u2014 that is NON-EVIDENCE, not low taste. Counting corrections is not a bar. Reactive-but-correct cringe calls are NOT penalized (the AI is blind to its own slop). ANCHORS: L8 = 'no em dashes anywhere \u2014 it IMMEDIATELY looks like AI slop', or 'the growth curve is cringe\u2026 a timeline without a y-axis is better'; L10 = authored the homepage copy himself up front, or defined 'would a smart engineer be surprised?' as the bar up front + lets the AI run on rote. KEY DISTINCTION: a low taste score on ROTE/ambiguity-free work = NO penalty (preface gently); a low score on TASTE-HEAVY work where they demonstrably hold a bar but withheld it = a real miss \u2014 say which. L11 = aggregator-only (consistency across the corpus).",
|
|
277
|
+
rungs: [
|
|
278
|
+
{ level: 2, marker: "Accepts whatever the AI produces; can't tell good from mediocre." },
|
|
279
|
+
{ level: 4, marker: "Occasional pushback, but mostly rubber-stamps; standards are conventional." },
|
|
280
|
+
{ level: 6, marker: "Rejects the obviously bad version and iterates toward better \u2014 a real but ordinary bar a competent reviewer would also reach." },
|
|
281
|
+
{ level: 8, marker: "Makes the 'oh shit, they're right' call the AI couldn't have reached \u2014 specific, opinionated, vindicated \u2014 usually caught in the moment." },
|
|
282
|
+
{ level: 10, marker: "Brings the developed, resonant vision up front before the AI tries, AND knows where not to impose (lets the AI run on throwaway). (per-session ceiling)" }
|
|
283
|
+
]
|
|
284
|
+
},
|
|
285
|
+
{
|
|
286
|
+
key: "generative",
|
|
287
|
+
label: "Abstraction \u2014 seeing the load-bearing thing, and being right",
|
|
288
|
+
graded: true,
|
|
289
|
+
definition: "Under genuine ambiguity (an obvious path sitting right there, no answer key): did the USER see that the obvious path was subtly wrong \u2014 or that something decisive was invisible \u2014 and do the non-obvious thing that made it right, AND were they right? The graded unit is the whole move: the CATCH (noticing the fork) -> the QUESTION (formulating it) -> the CALL (resolving it). The move surfaces three interchangeable ways, scored identically: a REFRAME (collapse a mess to its real axis), a QUESTION (pose the hard thing nobody posed), or an INSTRUMENT (build the code/chart/structure that makes the truth legible, then read it). Tag the SUBSTANCE, not the medium \u2014 a chart built (technical act) to make a human-intensity call legible is a JUDGMENT catch.",
|
|
290
|
+
read: "FOUR LEVERS set the height: (1) INVISIBILITY \u2014 how non-obvious there was anything to catch; a sharp operator in the flow drives past it (the DOMINANT lever); (2) DIFFICULTY of forming the catch; (3) STAKES \u2014 'without this, things silently go wrong' is the 9-10 condition; (4) SOUNDNESS \u2014 did the call hold. THREE OVERRIDING GATES: (a) OBVIOUSNESS KILLS IT \u2014 a correct, even load-bearing catch a sharp operator reaches anyway scores LOW ('you're evaluating the person not the AI' is correct but obvious -> ~5; 'compute in the OAuth dead-time' is real but fairly obvious -> ~6). (b) LEVERAGE BEATS INSIGHT \u2014 a move whose payoff is a HIGHER RATE OF FUTURE INSIGHT (building the tool that lets him set his own boundaries; timestamp-by-timestamp structure to instruct the grader precisely) raises the bandwidth of the loop that produces every later catch, and can outrank a clean one-off reframe. (c) SOUNDNESS GATE \u2014 a confident WRONG call to a noticed fork is an ANTI-SIGNAL; an honest 'not sure, here's a range' is NOT penalized. HUNT THE FORK BLIND \u2014 reconstruct what continuing-without-noticing looked like, then check whether/why he diverged; do NOT hand yourself the fork pre-formed or you score everyone's problem-finding a 10. TAG every instance: kind = systems | judgment | both (+ optional register: human/product/measurement). Reasoning and instrument-building are the SAME criterion. ANCHORS: 10 = 'OCEAN won't change, measure effectiveness of actions' (reframe that redefines the measure) / 'don't hurt Natalie, not long-term compatible' (lived judgment, requires being the person) / 'plot it sorted, let me set the boundaries' (leverage \u2014 builds the bandwidth tool); 9 = 'word count IS the proxy \u2014 sitting there is unfakeable focus' / 'uncertainty collapses in that direction'; 8 = 'a better metric is the difficulty of content + original synthesis' (hard opening question); 7 = 'this candidate seems too strong for his judgement to be real' (real non-obvious catch, modest stakes); 6 = 'compute in the OAuth dead-time' / 'why overcomplicate it, single agent' (real but obvious); 5 = 'you're evaluating the person not the AI' (correct but obvious).",
|
|
291
|
+
rungs: [
|
|
292
|
+
{ level: 2, marker: "A fork existed; defers the call to the AI, or takes the obvious branch; vague." },
|
|
293
|
+
{ level: 4, marker: "Notices something, but the question is vague or obvious \u2014 anyone in the seat asks it." },
|
|
294
|
+
{ level: 5, marker: "Correct but OBVIOUS catch \u2014 a real fix a sharp operator reaches anyway; not really a judgment call. (obviousness gate)" },
|
|
295
|
+
{ level: 6, marker: "A real, somewhat-non-obvious catch with a workable call \u2014 but reachable with effort; not load-bearing or redefining." },
|
|
296
|
+
{ level: 7, marker: "A real, non-obvious catch called RIGHT \u2014 but a single move of modest stakes, no redefinition and no leverage." },
|
|
297
|
+
{ level: 8, marker: "A genuinely hard, non-obvious QUESTION that opens something real \u2014 catch + framing excellent \u2014 even if the call stays open/tentative." },
|
|
298
|
+
{ level: 9, marker: "Caught a real LOAD-BEARING fork and called it RIGHT (very good) \u2014 or a leverage/bandwidth move just below the redefinition line." },
|
|
299
|
+
{ level: 10, marker: "Redefines what to measure / how to see it from first principles AND is right; OR a leverage move that compounds the whole loop; OR lived judgment that required being the person. Near-invisible catch. (per-session ceiling)" }
|
|
300
|
+
]
|
|
301
|
+
},
|
|
302
|
+
{
|
|
303
|
+
key: "caliber",
|
|
304
|
+
label: "Caliber \u2014 what they can be trusted with",
|
|
305
|
+
graded: true,
|
|
306
|
+
definition: "A holistic TRUST read, SEPARATE from the per-move abstraction score: across the corpus, what TIER of person is typically trusted with the hard tasks/moves they actually pull off, and how SELF-SUFFICIENTLY they do them. Answers the hiring question \u2014 what can they be trusted with, and what not. TWO HALVES: (1) CALIBER \u2014 for each genuinely-hard moment, the difficulty of having FOUND AND DONE it, and the caliber of person usually trusted with a task like that. (2) AUTONOMY \u2014 how much they fix and reason on their OWN vs need to be unblocked by others (the AI included). NEVER needing to be unblocked across the corpus is itself a top-tier signal; needing constant rescue (doom-loops, forced descent) caps the tier no matter the occasional brilliance.",
|
|
307
|
+
read: "For each hard moment record { difficulty: how hard it was to FIND AND DO this from the info available, NOT the topic's stated difficulty; trustTier: the caliber usually trusted with a task like this; selfSourced: did they originate and unblock it themselves (true) or get handed it / rescued (false); vsTier: below|met|exceeded; quote }. TIER LADDER (anchored): T1 competent generalist (a smart person, no special background, could do it); T2 senior engineer (solid, with effort); T3 staff/principal (a non-obvious reframe a strong senior wouldn't reach unprompted); T4 big-tech-senior / specialist (real depth or a genuinely novel frame); T5 frontier-lab / Anthropic-tier (OUT-REASONS strong models \u2014 first-principles measurement/judgment a top researcher makes); T6 beyond-AI-ceiling (the AI, cold, cannot reproduce it \u2014 singular lived/measurement judgment; the honest top). JUDGE-CEILING CAVEAT: when a move out-reasons the AI itself (it concedes, and a fresh model cold can't reproduce it), that is T5-T6 and the honest verdict is 'above my ceiling \u2014 cannot bound from above' \u2014 treat as the STRONGEST signal, NOT missing data. AUTONOMY: weight SELF-SOURCED moves at full tier; a move the AI handed them does NOT count toward caliber. AGGREGATE = the tier they CONSISTENTLY operate at (not a lucky peak) + an explicit TRUSTED-WITH / NOT-TRUSTED-WITH list + an autonomy verdict (e.g. 'never needed to be unblocked across N sessions \u2014 fixed and unblocked his own path throughout'). Be honest about SHAPE: someone can be T5 on measurement/judgment AND T1 on systems architecture \u2014 name both. ANCHORS (this corpus): T5/T6 = 'OCEAN won't change, measure effectiveness of actions' / the IIT transformer-consciousness argument the AI conceded / 'uncertainty collapses in that direction' (out-reasoned the tool, self-sourced); T3-T4 = 'word count IS the proxy \u2014 sitting there is unfakeable focus' (sharp reframe, resisted a confident WRONG AI); T1-T2 = 'why overcomplicate it, single agent' / 'just change the index' (sensible, a competent dev does it too).",
|
|
308
|
+
rungs: [
|
|
309
|
+
{ level: 2, marker: "T1 \u2014 competent generalist: tasks a smart person with no special background handles; needs unblocking often." },
|
|
310
|
+
{ level: 4, marker: "T2 \u2014 senior engineer: solid execution with effort; self-sufficient on the familiar, rescued on the hard." },
|
|
311
|
+
{ level: 6, marker: "T3 \u2014 staff/principal: non-obvious reframes a strong senior wouldn't reach unprompted; rarely needs unblocking." },
|
|
312
|
+
{ level: 8, marker: "T4 \u2014 big-tech-senior / specialist: real depth or a genuinely novel frame; unblocks their own path." },
|
|
313
|
+
{ level: 9, marker: "T5 \u2014 frontier-lab / Anthropic-tier: out-reasons strong models on first-principles measurement/judgment; never needs unblocking." },
|
|
314
|
+
{ level: 10, marker: "T6 \u2014 beyond the AI's own ceiling: singular lived/measurement judgment the model cannot reproduce cold. (above-my-ceiling)" }
|
|
315
|
+
]
|
|
316
|
+
},
|
|
317
|
+
{
|
|
318
|
+
key: "expertise",
|
|
319
|
+
label: "Expertise (specific systems / domain knowledge)",
|
|
320
|
+
graded: true,
|
|
321
|
+
definition: "Depth of specific, non-generic knowledge of systems, infra, tooling, and their OWN stack \u2014 the kind a competent junior wouldn't have. It is how much they know THEIR systems, not a general 'look around corners'. Two shapes: hard constraints ('you can't build it that way because X, Y, Z') and specific directives ('use claude-p not the API', 'Opus only when extended thinking is on', 'run it on EC2 overnight', 'use Kubernetes for this, Terraform for that').",
|
|
322
|
+
read: "Grade OVERRIDES on non-obviousness (would a competent engineer / strong model do it unprompted? more 'no' = higher) \xD7 payoff (AI conceded, it worked, a problem was averted). A confident override that was WRONG is an anti-signal. NOT expertise: (a) cleverness about how to measure/grade \u2014 that's Abstraction; (b) generic caution any non-stupid person shows ('don't delete the local data', 'we're remoting in, no harm done') \u2014 not mission-critical, no special knowledge \u2192 non-evidence; (c) generic best-practice the AI already had ('you have the Supabase CLI, run it yourself'). ANCHORS: L8 = 'Opus only for reasoning convos, only when extended thinking is on', or 'remove Claude Code/codex convos from the chat parsers \u2014 different kind'; L9 = deep stack/infra directives a junior wouldn't know (EC2 overnight node + caffeinate/lid-close; 'when you type Claude in the terminal it's already logged in \u2014 not the issue'); L10 = counter-intuitive call the AI would've done the OPPOSITE of ('run it using claude-p, not the API'). L11 = aggregator-only.",
|
|
323
|
+
rungs: [
|
|
324
|
+
{ level: 2, marker: "Defers to the AI on system decisions; no specific knowledge surfaces." },
|
|
325
|
+
{ level: 4, marker: "Opinions are generic best-practice the AI already had." },
|
|
326
|
+
{ level: 6, marker: "Corrects the AI on a real systems fact it got wrong (but a competent engineer would too)." },
|
|
327
|
+
{ level: 8, marker: "Repeatedly catches what the AI misses with specific, correct systems/tooling calls a junior wouldn't make." },
|
|
328
|
+
{ level: 9, marker: "Deep, specific directives about their own stack / infra a junior wouldn't know." },
|
|
329
|
+
{ level: 10, marker: "Counter-intuitive call \u2014 the AI would have done the opposite \u2014 grounded in deep operating knowledge, and it pays off. (per-session ceiling)" }
|
|
330
|
+
]
|
|
331
|
+
},
|
|
332
|
+
{
|
|
333
|
+
key: "delegation",
|
|
334
|
+
label: "Delegation and calibration",
|
|
335
|
+
graded: true,
|
|
336
|
+
definition: "Whether they correctly judge when to do the thinking themselves vs. let the AI run \u2014 tight leash on subtle, judgement-heavy work; let it go ham on rote work \u2014 and whether that choice fits the task. Low-effort-until-needed is the IDEAL prompt economy, not a defect; never penalize a vague ask on routine work.",
|
|
337
|
+
read: "Two kinds of correction, OPPOSITE verdicts: (a) genuine AI error caught and redirected = GOOD (couldn't be front-loaded); (b) the AI would clearly have done better with more instruction, the user demonstrably HELD that standard but withheld it and had to redo = the MISS (self-inflicted under-briefing on taste-sensitive work). Also: letting the AI run on genuinely rote work = good; over-specifying what the AI would have nailed anyway = its own miscalibration (distrust, not delegation). COACHING IS MANDATORY: name how they should have delegated ('the AI could have handled this part itself; you only needed to front-load X'). ANCHORS: L8 = 'We should discuss the architecture first. Don't go build random stuff yourself' (gatekept implementation behind review); MISS L4 = corrected the same 'top 0.01% too generous' calibration five times (clearly held the rule, withheld it), or taste-sensitive layout choices not front-loaded \u2192 correction rounds. L11 = aggregator-only (sustained across the corpus).",
|
|
338
|
+
rungs: [
|
|
339
|
+
{ level: 2, marker: "Badly miscalibrated \u2014 under-briefs taste-sensitive work and/or micromanages trivia; constant avoidable rework." },
|
|
340
|
+
{ level: 4, marker: "Inconsistent \u2014 lets the AI run on things they clearly had opinions about, then redoes; or over-controls rote." },
|
|
341
|
+
{ level: 6, marker: "Roughly right \u2014 briefs the big stuff, but still avoidable rework from withheld standards." },
|
|
342
|
+
{ level: 8, marker: "Well-calibrated \u2014 front-loads the specifiable so the AI isn't redone on knowable things; reserves correction for genuine AI errors; lets rote run." },
|
|
343
|
+
{ level: 10, marker: "Within a session, almost everything knowable is front-loaded; avoidable rework near-zero; the corrections that remain are genuine AI surprises. (per-session ceiling)" }
|
|
344
|
+
]
|
|
345
|
+
},
|
|
346
|
+
{
|
|
347
|
+
key: "metacognition",
|
|
348
|
+
label: "Metacognition",
|
|
349
|
+
graded: true,
|
|
350
|
+
definition: "How aware they are of their own pace and process \u2014 noticing 'this is too slow, this is the bottleneck, fix it' \u2014 and how often they make HIGHER-ORDER changes to their own workflow mid-stream. Treating their own productivity as an engineering problem. (Replaces the old 'unblocking' criterion; the infra/leverage half of that now lives under Frontier tool use.)",
|
|
351
|
+
read: "Look for: real-time awareness of going too slow/fast; naming a recurring friction and restructuring the workflow so it stops; changing the WHOLE way of working with the AI mid-session to get unstuck; deciding to let a system run and patch its failures one-by-one rather than tweak arbitrarily. Re-explaining the same context repeatedly without fixing the recurrence is the anti-signal. ANCHORS: L8 = 'Is it possible to parallelize it even more heavily? this is literally not even sequential' (diagnosed the artificial concurrency cap), or 'let the system run, find the wrong grades, patch one-by-one'; ANTI L3 = re-explained the same architecture 3+ times rather than fixing the recurrence, or session-limit hit 6+ times with no harness assembled. L11 = aggregator-only (standing engineering discipline that compounds across the period).",
|
|
352
|
+
rungs: [
|
|
353
|
+
{ level: 2, marker: "No awareness of pace; grinds the same loop without noticing it's the loop." },
|
|
354
|
+
{ level: 4, marker: "Notices friction but only complains; no higher-order change." },
|
|
355
|
+
{ level: 6, marker: "Recognizes a bottleneck and adjusts how they work for this case." },
|
|
356
|
+
{ level: 8, marker: "Names a recurring friction and restructures their workflow so it stops recurring; aware in real time of pace." },
|
|
357
|
+
{ level: 10, marker: "Diagnoses the bottleneck mid-session and changes the whole way of working with the AI to fix it. (per-session ceiling)" }
|
|
358
|
+
]
|
|
359
|
+
},
|
|
360
|
+
{
|
|
361
|
+
key: "frontier",
|
|
362
|
+
label: "Frontier tool use",
|
|
363
|
+
graded: true,
|
|
364
|
+
definition: "How close to the frontier their WAY of using the tool is. Running sessions in parallel is now table-stakes; the frontier is loops, workflows, scale, remote/overnight execution, MCP, subagents, queue-ahead, custom harnesses, running from phone. Scores the BREADTH of techniques used (distinct from Expertise, which scores DEPTH of specific knowledge in the directives).",
|
|
365
|
+
read: "NOTE: this ladder still needs SOTA anchoring (how power users / the Claude Code team actually work) \u2014 until then grade CONSERVATIVELY and lean on the level-up list. Credit real leverage actually USED: overnight/remote runs (EC2, caffeinate/lid-close survival), subagents, MCP, custom scripts that drive the AI, queue-ahead, parallelization as a standing demand. The deterministic layer already exposes subagentSpawns / mcpTools / queueOps / modes \u2014 use them. Always emit a concrete 'how to level up' note: the specific frontier techniques they're NOT yet using. ANCHORS: L8 = EC2 overnight compute node + lid-close survival + rate-limit logging, real leverage used; recurring 'this is embarrassingly parallel' diagnosis driving real throughput wins. L11 = aggregator-only (the way they wield the tool is itself ahead of how its makers expect).",
|
|
366
|
+
rungs: [
|
|
367
|
+
{ level: 2, marker: "One session at a time, default settings; no leverage beyond chat." },
|
|
368
|
+
{ level: 4, marker: "Occasionally runs two things; no structural tooling." },
|
|
369
|
+
{ level: 6, marker: "Routinely parallel; queues work ahead; renames/organizes sessions." },
|
|
370
|
+
{ level: 8, marker: "Uses real leverage \u2014 overnight/remote runs, subagents, MCP, custom scripts to drive the AI." },
|
|
371
|
+
{ level: 10, marker: "Builds harnesses and workflows; loops; scales the AI across machines; runs from anywhere. (per-session ceiling)" }
|
|
372
|
+
]
|
|
373
|
+
},
|
|
374
|
+
{
|
|
375
|
+
key: "parallelism",
|
|
376
|
+
label: "Parallelism",
|
|
377
|
+
graded: false,
|
|
378
|
+
definition: "Concurrency \u2014 how many AI sessions run AT ONCE and how that's orchestrated (max concurrency, time at each level, overlap windows, queue-ahead, renames). The existing deterministic parallelism card (ParallelismView). Distinct from intensity.",
|
|
379
|
+
read: "Deterministic \u2014 see the parallelism panel (max concurrency, time at each level, overlap windows, queue-ahead, renames)."
|
|
380
|
+
},
|
|
381
|
+
{
|
|
382
|
+
key: "throughput",
|
|
383
|
+
label: "Intensity (throughput)",
|
|
384
|
+
graded: false,
|
|
385
|
+
definition: "Intensity \u2014 how hard and how long they work start to finish (active time, session spans, daily cadence, late-night cadence). A DIFFERENT axis from parallelism (you can be intense and serial, or parallel and idle). The existing deterministic throughput card (ThroughputView).",
|
|
386
|
+
read: "Deterministic \u2014 see the throughput/intensity panel (active hours, sessions/day, work spans, late-night cadence)."
|
|
387
|
+
}
|
|
388
|
+
];
|
|
389
|
+
var GRADED_CRITERIA = CODING_CRITERIA.filter((c) => c.graded);
|
|
390
|
+
var RUBRIC_FINGERPRINT = rubricFingerprint(GRADED_CRITERIA);
|
|
391
|
+
|
|
392
|
+
// ../../lib/agents/coding/walkthrough.ts
|
|
393
|
+
var IDLE_CAP_MS = 12 * 6e4;
|
|
394
|
+
|
|
395
|
+
// ../../lib/calibration/index.ts
|
|
396
|
+
var DEFAULT_CALIBRATION = {
|
|
397
|
+
version: "2026-07-12.2",
|
|
398
|
+
// = RUBRIC_FINGERPRINT of the shipped npm rubric (packages/coding-analyzer/
|
|
399
|
+
// src/grade/criteria.ts). A unit test fails loud when the rubric changes
|
|
400
|
+
// without this being updated (and re-pushed to the server).
|
|
401
|
+
rubricVersion: "f51b3a93a3",
|
|
402
|
+
// Canonical scale = the one the product owner set for the public report
|
|
403
|
+
// (11=0.01% … 8=2%), extended downward from the old report ladder. The old
|
|
404
|
+
// CodeAnalysisView / npm-viewer maps (7→10%/15%, 6→25%/30%) were drift, not
|
|
405
|
+
// intent.
|
|
406
|
+
scoreBands: [
|
|
407
|
+
{ score: 11, label: "Top 0.01%" },
|
|
408
|
+
{ score: 10, label: "Top 0.1%" },
|
|
409
|
+
{ score: 9, label: "Top 0.5%" },
|
|
410
|
+
{ score: 8, label: "Top 2%" },
|
|
411
|
+
{ score: 7, label: "Top 5%" },
|
|
412
|
+
{ score: 6, label: "Top 15%" },
|
|
413
|
+
{ score: 5, label: "Top 50%" },
|
|
414
|
+
{ score: 4, label: "Top 65%" },
|
|
415
|
+
{ score: 3, label: "Top 80%" },
|
|
416
|
+
{ score: 2, label: "Top 90%" },
|
|
417
|
+
{ score: 1, label: "Top 97%" }
|
|
418
|
+
],
|
|
419
|
+
minRankedScore: 6,
|
|
420
|
+
benchmarkBands: [
|
|
421
|
+
{
|
|
422
|
+
level: "11",
|
|
423
|
+
name: "Superhuman",
|
|
424
|
+
rank: "Top 0.01%",
|
|
425
|
+
oneIn: "1 in 10,000+",
|
|
426
|
+
meaning: "Beyond the normal human ceiling for the trait \u2014 the genuine power-law outlier. Almost no one earns this; when the evidence shows it, it is scored, not rounded down.",
|
|
427
|
+
foundIn: [
|
|
428
|
+
"Fields Medal and Nobel-track researchers",
|
|
429
|
+
"Founders of generational companies",
|
|
430
|
+
"The handful of people a frontier field is named after"
|
|
431
|
+
],
|
|
432
|
+
maxTopPct: 0.01
|
|
433
|
+
},
|
|
434
|
+
{
|
|
435
|
+
level: "10",
|
|
436
|
+
name: "World-class",
|
|
437
|
+
rank: "Top 0.1%",
|
|
438
|
+
oneIn: "1 in 1,000",
|
|
439
|
+
meaning: "Among the best alive at this trait \u2014 the level entire institutions are built to find.",
|
|
440
|
+
foundIn: [
|
|
441
|
+
"Researchers at frontier AI labs",
|
|
442
|
+
"Faculty at top-5 research universities",
|
|
443
|
+
"IMO / IOI medalists",
|
|
444
|
+
"Founders backed by the top decile of venture firms"
|
|
445
|
+
],
|
|
446
|
+
maxTopPct: 0.1
|
|
447
|
+
},
|
|
448
|
+
{
|
|
449
|
+
level: "9",
|
|
450
|
+
name: "Exceptional",
|
|
451
|
+
rank: "Top 1%",
|
|
452
|
+
oneIn: "1 in 100",
|
|
453
|
+
meaning: "Clearly exceptional \u2014 the strongest person on most strong teams.",
|
|
454
|
+
foundIn: [
|
|
455
|
+
"PhD students at top programs",
|
|
456
|
+
"Early engineers at breakout startups",
|
|
457
|
+
"YC-class founders",
|
|
458
|
+
"National olympiad finalists"
|
|
459
|
+
],
|
|
460
|
+
maxTopPct: 1
|
|
461
|
+
},
|
|
462
|
+
{
|
|
463
|
+
level: "8",
|
|
464
|
+
name: "Strong",
|
|
465
|
+
rank: "Top 10%",
|
|
466
|
+
oneIn: "1 in 10",
|
|
467
|
+
meaning: "Strong against the whole population \u2014 the best person in most rooms, not every room.",
|
|
468
|
+
foundIn: [
|
|
469
|
+
"Senior engineers at selective tech companies",
|
|
470
|
+
"Graduates of demanding technical programs",
|
|
471
|
+
"Operators who get promoted everywhere they go"
|
|
472
|
+
],
|
|
473
|
+
maxTopPct: 10
|
|
474
|
+
},
|
|
475
|
+
{
|
|
476
|
+
level: "5",
|
|
477
|
+
name: "Median",
|
|
478
|
+
rank: "50th percentile",
|
|
479
|
+
oneIn: "1 in 2",
|
|
480
|
+
meaning: 'The middle of the general population. The reference class is all ~8 billion humans \u2014 so even "strong" above already means top 10% of everyone.',
|
|
481
|
+
foundIn: ["The general population \u2014 most people, most places"],
|
|
482
|
+
maxTopPct: 100
|
|
483
|
+
}
|
|
484
|
+
],
|
|
485
|
+
codingTiers: {
|
|
486
|
+
push: { median: 300, sigma: 1.121, tag: "estimated \u2014 log-normal fit", source: "Anthropic 2026 Claude Code study (400k sessions): median engaged user ~8 prompts/day at ~35 words \u2014 ~300 directed words/day; tail shape from its actions-per-prompt distribution. AVERAGE over substantial days, cumulative not spiky." },
|
|
487
|
+
focus: { median: 0.08, sigma: 0.559, tag: "estimated \u2014 proxy-anchored", source: "median: Anthropic 20h/week engaged-user figure implies ~1h/day of true rapid exchange against an 8h working day; ceiling anchored to the ~4h/day deep-work limit (45% share = 1-in-1,000). AVERAGE share, cumulative not spiky." },
|
|
488
|
+
orchestration: { tag: "estimated \u2014 behavior bands", source: "no published concurrency distribution exists; ceiling anchored to the Claude Code lead's documented 10-15 parallel sessions" }
|
|
489
|
+
},
|
|
490
|
+
codingBenchmarks: {
|
|
491
|
+
flow: {
|
|
492
|
+
// typical knowledge worker ≈ 35% of an 8h day focused; ~4h/day is the
|
|
493
|
+
// recognized deep-work ceiling (Newport) ≈ 50% — most sit far below it.
|
|
494
|
+
typicalPctOfDay: 0.35,
|
|
495
|
+
topPctOfDay: 0.5,
|
|
496
|
+
tag: "measured",
|
|
497
|
+
source: "RescueTime 2019 (~2h48m focused/day) \xB7 Cal Newport, Deep Work (~4h/day deep-work ceiling)",
|
|
498
|
+
line: "the best spend ~half the workday in true deep flow; ~4h/day is the human ceiling"
|
|
499
|
+
},
|
|
500
|
+
longestRun: {
|
|
501
|
+
typicalHours: 0.67,
|
|
502
|
+
// ~40 min before the average worker is interrupted
|
|
503
|
+
topHours: 4,
|
|
504
|
+
tag: "measured",
|
|
505
|
+
source: "RescueTime (avg max ~40 min focus before interruption) \xB7 deep-work single-block ceiling ~3\u20134h",
|
|
506
|
+
line: "a top performer can hold a single ~3\u20134h uninterrupted block; the average worker breaks at ~40 min"
|
|
507
|
+
},
|
|
508
|
+
parallelism: {
|
|
509
|
+
topMaxConcurrent: 12,
|
|
510
|
+
// Boris Cherny ~10–15 interactive sessions
|
|
511
|
+
topAvgConcurrent: 6,
|
|
512
|
+
// time-weighted average while active (anecdotal, same source)
|
|
513
|
+
worktreesTip: "3\u20135 git worktrees at once",
|
|
514
|
+
tag: "anecdotal",
|
|
515
|
+
source: "Boris Cherny (Claude Code lead, Anthropic): ~10\u201315 concurrent sessions, 'dozens of Claudes running at all times'; 3\u20135 worktrees = 'the single biggest productivity unlock'",
|
|
516
|
+
line: "the Claude Code lead runs 10\u201315 sessions at once; his #1 tip is 3\u20135 git worktrees in parallel"
|
|
517
|
+
},
|
|
518
|
+
throughput: {
|
|
519
|
+
// Top-0.1% words/day reference lines, anchored to the throughput ladder
|
|
520
|
+
// (THROUGHPUT_LADDER in throughput.ts): score 10 "world-class sustained" =
|
|
521
|
+
// 9k/day, score 11 "superhuman ceiling" = 12k/day. Calibrated to the person's
|
|
522
|
+
// own anchors, not a wild guess — but still a target, not a measured population.
|
|
523
|
+
topAvgWordsPerDay: 9e3,
|
|
524
|
+
topPeakWordsPerDay: 12e3,
|
|
525
|
+
benchLabel: "top 0.1%",
|
|
526
|
+
// No published "words typed/day" for heavy users — the real story is OUTPUT.
|
|
527
|
+
loopsHeadline: ">1,000,000 lines of Rust at 99.8% tests passing",
|
|
528
|
+
loopsDetail: "the Bun runtime was ported Zig\u2192Rust largely autonomously \u2014 hundreds of parallel subagents, two AI reviewers per file",
|
|
529
|
+
tag: "measured",
|
|
530
|
+
source: "The Register, May 2026 (Bun Zig\u2192Rust rewrite)",
|
|
531
|
+
line: "autonomous fleets ported >1M lines of Rust at 99.8% tests passing; a while-loop agent delivered a $50k contract for $297 of compute"
|
|
532
|
+
}
|
|
533
|
+
}
|
|
534
|
+
};
|
|
535
|
+
|
|
536
|
+
// ../../lib/agents/coding/benchmarks.ts
|
|
537
|
+
var BEST = DEFAULT_CALIBRATION.codingBenchmarks;
|
|
538
|
+
|
|
539
|
+
// ../../lib/agents/coding/profile.ts
|
|
540
|
+
var ROOT = process.cwd();
|
|
541
|
+
var CODING_DIR = process.env.POLYMATH_DATA_DIR ? path.join(process.env.POLYMATH_DATA_DIR, "coding") : path.join(ROOT, ".data", "coding");
|
|
542
|
+
var GRADES = path.join(CODING_DIR, "grades");
|
|
543
|
+
|
|
246
544
|
// ../../scripts/coding-gap-dist.mts
|
|
247
|
-
var CODING =
|
|
545
|
+
var CODING = CODING_DIR;
|
|
248
546
|
var WPM = 130;
|
|
249
547
|
var BIN = 0.5;
|
|
250
548
|
var MAX = 20;
|
|
251
549
|
async function main() {
|
|
252
|
-
const sess = JSON.parse(await fs2.readFile(
|
|
550
|
+
const sess = JSON.parse(await fs2.readFile(path2.join(CODING, "sessions.json"), "utf8")).filter((s) => !s.duplicateOf && s.klass === "interactive");
|
|
253
551
|
const gaps = [];
|
|
254
552
|
for (const s of sess) {
|
|
255
553
|
let ms = [];
|
|
@@ -280,7 +578,7 @@ async function main() {
|
|
|
280
578
|
for (let t = 0.5; t <= 15; t = Math.round((t + 0.5) * 10) / 10) curve.push({ t, fracUnder: Math.round(gaps.filter((g) => g < t).length / gaps.length * 1e3) / 1e3 });
|
|
281
579
|
const pct = (p) => gaps[Math.min(gaps.length - 1, Math.floor(gaps.length * p))] ?? 0;
|
|
282
580
|
const out = { generatedAt: (/* @__PURE__ */ new Date()).toISOString(), totalGaps: gaps.length, binMin: BIN, maxMin: MAX, bins, over, currentFlowGap: 5, curve, median: Math.round(pct(0.5) * 100) / 100, p25: Math.round(pct(0.25) * 100) / 100, p75: Math.round(pct(0.75) * 100) / 100, p90: Math.round(pct(0.9) * 100) / 100 };
|
|
283
|
-
await fs2.writeFile(
|
|
581
|
+
await fs2.writeFile(path2.join(CODING, "gap-dist.json"), JSON.stringify(out, null, 2));
|
|
284
582
|
const mx = Math.max(...bins.map((b) => b.count), 1);
|
|
285
583
|
console.log(`gap-dist: ${gaps.length} gaps \xB7 median ${out.median}m \xB7 p75 ${out.p75}m \xB7 p90 ${out.p90}m \xB7 ${over} over ${MAX}m`);
|
|
286
584
|
bins.forEach((b) => {
|
|
@@ -16946,7 +16946,7 @@ async function materializeSession(file2) {
|
|
|
16946
16946
|
}
|
|
16947
16947
|
|
|
16948
16948
|
// ../../scripts/coding-gap.mts
|
|
16949
|
-
var CODING =
|
|
16949
|
+
var CODING = CODING_DIR;
|
|
16950
16950
|
var CONFIG = {
|
|
16951
16951
|
interface: "Claude Code INSIDE the Claude Desktop Mac app (NOT a raw terminal) \u2014 so terminal-only advice like `claude -p` does not fit; give desktop-app actions.",
|
|
16952
16952
|
permissions: "`--dangerously-skip-permissions` is ON. They are NOT worried about destructive actions (never happened; they just revert). Do NOT raise auto mode or any permissions change \u2014 it is not in the catalog.",
|