codecartographer-pi 0.17.0 → 0.17.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.codecarto/GUIDE.md +1 -1
- package/.codecarto/findings/architecture/SKILL.md +1 -0
- package/.codecarto/findings/contracts/SKILL.md +1 -0
- package/.codecarto/findings/defect-scan/SKILL.md +15 -1
- package/.codecarto/findings/defect-scan/passes/01-logic-and-correctness.md +1 -1
- package/.codecarto/findings/defect-scan/passes/02-error-handling.md +1 -1
- package/.codecarto/findings/defect-scan/passes/03-concurrency-and-resources.md +1 -1
- package/.codecarto/findings/defect-scan/passes/04-security-and-trust.md +1 -1
- package/.codecarto/findings/defect-scan/passes/05-api-contract-violations.md +2 -1
- package/.codecarto/findings/defect-scan/passes/06-config-and-environment.md +1 -1
- package/.codecarto/findings/defect-scan-mechanical/SKILL.md +3 -2
- package/.codecarto/findings/defect-scan-semantic/SKILL.md +2 -0
- package/.codecarto/findings/porting/SKILL.md +2 -1
- package/.codecarto/findings/protocols/SKILL.md +1 -0
- package/.codecarto/findings/reimplementation-spec/SKILL.md +1 -1
- package/.codecarto/templates/architecture-map.md +1 -1
- package/.codecarto/templates/defect-report.md +23 -0
- package/.codecarto/templates/mechanical-defects.md +22 -0
- package/.codecarto/templates/reverse-engineering-bundle.md +2 -2
- package/.codecarto/templates/semantic-defects.md +26 -0
- package/.codecarto/workflow/VALIDATE.md +1 -1
- package/.codecarto/workflow/pipeline-defect-scan.yaml +2 -0
- package/.codecarto/workflow/pipeline-full-with-audit.yaml +3 -1
- package/.codecarto/workflow/pipeline-full-with-deep-audit.yaml +5 -1
- package/.codecarto/workflow/pipeline-scout-first.yaml +5 -1
- package/.codecarto/workflow/scaffold-version.yaml +1 -1
- package/README.md +4 -4
- package/agent-skill/codecartographer/references/deep-audit-synthesis.md +4 -1
- package/agent-skill/codecartographer/references/library.md +1 -1
- package/agent-skill/codecartographer/references/orchestration.md +1 -1
- package/dist/core/amendment.js +2 -2
- package/dist/core/completion.d.ts +5 -0
- package/dist/core/completion.js +18 -2
- package/dist/core/dashboard.js +5 -3
- package/dist/core/findings.d.ts +59 -0
- package/dist/core/findings.js +145 -0
- package/dist/core/index.d.ts +1 -0
- package/dist/core/index.js +1 -0
- package/dist/core/library.d.ts +66 -1
- package/dist/core/library.js +161 -8
- package/dist/core/pipeline.js +15 -0
- package/dist/core/prompts.js +1 -1
- package/dist/core/status.js +14 -6
- package/dist/core/types.d.ts +6 -0
- package/dist/core/utils.d.ts +14 -0
- package/dist/core/utils.js +30 -0
- package/dist/core/workspace.js +11 -18
- package/dist/core/yaml.js +19 -4
- package/dist/extensions/codecarto/auto-runner.d.ts +2 -0
- package/dist/extensions/codecarto/broadside-flags.d.ts +7 -2
- package/dist/extensions/codecarto/broadside-flags.js +22 -9
- package/dist/extensions/codecarto/dashboard-narrator.js +1 -1
- package/dist/extensions/codecarto/index.js +40 -16
- package/dist/extensions/codecarto/phase-compaction.js +4 -0
- package/dist/mcp-server/server.js +65 -7
- package/package.json +2 -2
package/.codecarto/GUIDE.md
CHANGED
|
@@ -17,7 +17,7 @@ CodeCartographer distinguishes two roles. They are different jobs — but they a
|
|
|
17
17
|
**Orchestrator.** The persistent chat driving the run — normally the very session reading this guide. The role is defined by its **duties**, not by who executes phases:
|
|
18
18
|
|
|
19
19
|
- **Curate `CONVENTIONS.md` and `DECISIONS.md`**: promote patterns when they recur; append cross-cutting decisions as they land.
|
|
20
|
-
- **Re-triage open questions at every phase boundary.** A question's `kind` label is itself a claim that needs evidence: before accepting `needs-maintainer-decision` or `needs-runtime-test`, re-test whether the question has become answerable by reading. A mislabeled question suppresses verification for every later phase (the `orchestration` guide topic records a real four-phase failure).
|
|
20
|
+
- **Re-triage open questions at every phase boundary.** A question's `kind` label is itself a claim that needs evidence: before accepting `needs-maintainer-decision` or `needs-runtime-test`, re-test whether the question has become answerable by reading. A mislabeled question suppresses verification for every later phase (the `orchestration` guide topic records a real four-phase failure). The duty has a counterpart: when re-triage concludes a question genuinely still needs a runtime test, no finding in the next phase may assert one of that question's candidate answers with a settled action (`fix before porting`, `fix now`) — it inherits the question's uncertainty (`verify at runtime`) until runtime evidence closes the question.
|
|
21
21
|
- **Sweep for contradictions** between the incoming phase's required reads and earlier phases' `owner_notes`. A measured fact that contradicts a summarized claim is a gap to route, not a nuance to smooth over.
|
|
22
22
|
- **Route gaps**: confirm each completed phase's declared secondary outputs were written or explicitly routed, and that handoff `decisions` deferring work landed somewhere a later phase will actually see.
|
|
23
23
|
- **Gate strategic forks** with the user: pipeline switches, opinionated-vs-agnostic specs, scope changes.
|
|
@@ -84,6 +84,7 @@ Write the output in seven sections:
|
|
|
84
84
|
Mark every conclusion with one of these evidence levels:
|
|
85
85
|
- `observed fact`: direct statement from docs, tests, schemas, types, or code.
|
|
86
86
|
- `strong inference`: architectural conclusion drawn from multiple facts.
|
|
87
|
+
- `external-behavior claim`: a claim about what a system outside this source tree does (a server, engine, driver, third-party API, OS); unverifiable by reading this code, so it stays unsettled until a runtime probe or that system's own source confirms it.
|
|
87
88
|
- `portability hazard`: assumption tied to the source language, runtime, terminal, OS, or third-party SDKs.
|
|
88
89
|
- `open question`: missing or conflicting behavior that still needs evidence.
|
|
89
90
|
|
|
@@ -79,6 +79,7 @@ End with a black-box acceptance list:
|
|
|
79
79
|
Mark every finding with one of these evidence levels:
|
|
80
80
|
- `observed fact`: direct statement from docs, tests, schemas, types, or code.
|
|
81
81
|
- `strong inference`: behavioral conclusion drawn from multiple facts.
|
|
82
|
+
- `external-behavior claim`: a claim about what a system outside this source tree does (a server, engine, driver, third-party API, OS); unverifiable by reading this code, so it stays unsettled until a runtime probe or that system's own source confirms it.
|
|
82
83
|
- `portability hazard`: assumption tied to the source language, runtime, terminal, OS, or third-party SDKs.
|
|
83
84
|
- `open question`: missing or conflicting behavior that still needs evidence.
|
|
84
85
|
|
|
@@ -45,8 +45,11 @@ Not every pass applies to every codebase. Use the architecture map to decide emp
|
|
|
45
45
|
Mark every finding with one of these evidence levels:
|
|
46
46
|
- `observed fact`: the defect is directly visible in the code.
|
|
47
47
|
- `strong inference`: the defect is highly likely based on multiple code observations.
|
|
48
|
+
- `external-behavior claim`: the claim is about a component outside the analyzed source tree — a server, engine, driver, third-party API, or OS — including how it parses a payload, what it silently ignores, and version-dependent behavior. Not obtainable at any read depth of this source: settling it needs a runtime probe against the pinned version, or that system's own source at that version. This is not a weaker `strong inference`; it is a claim about a different artifact. "The client sends a map where the server's API documents an array" is an `external-behavior claim` about the server, even when both halves of the sentence are observed facts.
|
|
48
49
|
- `open question`: the code is suspicious but would need runtime testing to confirm.
|
|
49
50
|
|
|
51
|
+
**Cite or hedge quantities.** Any number in a finding — a file size, a default value, a call-site count, a version, a timeout — cites the file and line or the command output it was read from, or is written as an explicit estimate ("~300 MB, not measured"). A plausible specific stated in the same register as a read one is how a 32.8 MB artifact gets recorded as ~300 MB and a default of -20 as -5.
|
|
52
|
+
|
|
50
53
|
## Severity Classification
|
|
51
54
|
|
|
52
55
|
Assign one severity per finding:
|
|
@@ -59,10 +62,11 @@ Assign one severity per finding:
|
|
|
59
62
|
|
|
60
63
|
Tag each finding with a recommended action. Use the set that matches your pipeline:
|
|
61
64
|
|
|
62
|
-
**Pre-porting pipelines** (full-with-audit, full-with-deep-audit):
|
|
65
|
+
**Pre-porting pipelines** (full-with-audit, full-with-deep-audit, scout-first):
|
|
63
66
|
- `fix before porting`: the defect would carry into a new implementation if not addressed first.
|
|
64
67
|
- `port differently`: the new implementation should handle this case differently by design.
|
|
65
68
|
- `leave behind`: the defect is specific to the source implementation and won't survive porting.
|
|
69
|
+
- `verify at runtime`: the diagnosis names behavior of a system outside the analyzed source, or otherwise cannot be settled by reading; a runtime probe must confirm it before any fix is designed. Its destination is the spec's Spike List plus a `post_pipeline` entry of `kind: spike` — not a design change.
|
|
66
70
|
|
|
67
71
|
**Maintenance pipelines** (defect-scan):
|
|
68
72
|
- `fix now`: the defect is actively causing or risking problems.
|
|
@@ -72,6 +76,16 @@ Tag each finding with a recommended action. Use the set that matches your pipeli
|
|
|
72
76
|
|
|
73
77
|
Determine which action set to use by checking the `pipeline` field in `workflow/status.yaml`.
|
|
74
78
|
|
|
79
|
+
### Pairing rules — the evidence level bounds the action
|
|
80
|
+
|
|
81
|
+
A finding's action may not assert more certainty than its evidence level carries:
|
|
82
|
+
|
|
83
|
+
- Evidence `open question` or `external-behavior claim` ⇒ action is `verify at runtime` or `port differently` (pre-porting), or `investigate` (maintenance). **Never `fix before porting` or `fix now`.** A fix designed from an unsettled diagnosis can convert working behavior into the one shape the external system ignores — a real run recommended exactly that reshape, and runtime testing inverted it.
|
|
84
|
+
- Evidence `observed fact` ⇒ action is not `verify at runtime`. A settled label with an unsettled action contradicts itself; pick the one that is true.
|
|
85
|
+
- Every finding whose evidence is `open question` or `external-behavior claim` also appears in the report's **Open Questions** table, so the hedge travels with the finding into every document a later phase or a human reads — not only into the handoff.
|
|
86
|
+
|
|
87
|
+
Validation checks the first rule mechanically on the findings tables; on a current scaffold a violation fails the phase. If a routed `carry_forward` item derives from an open question of `kind: needs-runtime-test` that is still unresolved, the finding that addresses it inherits that uncertainty: it closes the carry-forward with `verify at runtime`, not with a settled action, unless the question itself is closed with runtime evidence in the same handoff.
|
|
88
|
+
|
|
75
89
|
## Output
|
|
76
90
|
|
|
77
91
|
Write findings to the primary output using the template at `templates/defect-report.md`.
|
|
@@ -41,7 +41,7 @@ For each finding, record:
|
|
|
41
41
|
- **Defect**: what is wrong, in one sentence.
|
|
42
42
|
- **Evidence**: what you observed that proves or strongly suggests the defect.
|
|
43
43
|
- **Severity**: critical (incorrect results in normal use), high (incorrect results in edge cases), medium (dead code or latent risk), low (style issue with correctness implications).
|
|
44
|
-
- **Evidence level**: observed fact / strong inference / open question.
|
|
44
|
+
- **Evidence level**: observed fact / strong inference / external-behavior claim / open question. A claim about what another system does with this code's output (a server, engine, driver, third-party API, OS) is an `external-behavior claim`, not a `strong inference` — see findings/defect-scan/SKILL.md §Evidence Classification.
|
|
45
45
|
|
|
46
46
|
## What to skip
|
|
47
47
|
|
|
@@ -47,7 +47,7 @@ For each finding, record:
|
|
|
47
47
|
- **Defect**: what error scenario is mishandled.
|
|
48
48
|
- **Evidence**: the specific code pattern or path that demonstrates the gap.
|
|
49
49
|
- **Severity**: critical (data loss or corruption on failure), high (silent failure in normal operations), medium (poor error messages or missing cleanup), low (observability gap).
|
|
50
|
-
- **Evidence level**: observed fact / strong inference / open question.
|
|
50
|
+
- **Evidence level**: observed fact / strong inference / external-behavior claim / open question. A claim about what another system does with this code's output (a server, engine, driver, third-party API, OS) is an `external-behavior claim`, not a `strong inference` — see findings/defect-scan/SKILL.md §Evidence Classification.
|
|
51
51
|
|
|
52
52
|
## What to skip
|
|
53
53
|
|
|
@@ -46,7 +46,7 @@ For each finding, record:
|
|
|
46
46
|
- **Defect**: what concurrent or resource scenario is mishandled.
|
|
47
47
|
- **Evidence**: the specific shared state, missing synchronization, or unclosed resource.
|
|
48
48
|
- **Severity**: critical (data corruption or deadlock in normal operation), high (race condition in common paths), medium (resource leak under error conditions), low (theoretical race in rarely-exercised path).
|
|
49
|
-
- **Evidence level**: observed fact / strong inference / open question.
|
|
49
|
+
- **Evidence level**: observed fact / strong inference / external-behavior claim / open question. A claim about what another system does with this code's output (a server, engine, driver, third-party API, OS) is an `external-behavior claim`, not a `strong inference` — see findings/defect-scan/SKILL.md §Evidence Classification.
|
|
50
50
|
|
|
51
51
|
## What to skip
|
|
52
52
|
|
|
@@ -54,7 +54,7 @@ For each finding, record:
|
|
|
54
54
|
- **Defect**: what security property is violated.
|
|
55
55
|
- **Evidence**: the specific code path, missing check, or exposed secret.
|
|
56
56
|
- **Severity**: critical (actively exploitable, data exposure, auth bypass), high (exploitable with some effort or preconditions), medium (defense-in-depth gap, hardcoded non-production secret), low (missing header, informational disclosure).
|
|
57
|
-
- **Evidence level**: observed fact / strong inference / open question.
|
|
57
|
+
- **Evidence level**: observed fact / strong inference / external-behavior claim / open question. A claim about what another system does with this code's output (a server, engine, driver, third-party API, OS) is an `external-behavior claim`, not a `strong inference` — see findings/defect-scan/SKILL.md §Evidence Classification.
|
|
58
58
|
|
|
59
59
|
## What to skip
|
|
60
60
|
|
|
@@ -47,9 +47,10 @@ For each finding, record:
|
|
|
47
47
|
- **Location**: file path and function/method name.
|
|
48
48
|
- **Defect**: what contract is violated and how.
|
|
49
49
|
- **Spec source**: where the expected behavior is documented (docstring, type signature, contracts phase, protocols phase, README).
|
|
50
|
+
- **Which side implements the contract**: the analyzed code is often the *caller* of a contract another system implements (an HTTP API, an inference engine, a driver). A divergence between what this code sends and what the other side documents is only an `observed fact` about this code; what the other side actually does with it — accepts, ignores, rejects — is an `external-behavior claim` until a runtime probe against the pinned version says otherwise, and its action is `verify at runtime`, never `fix before porting`.
|
|
50
51
|
- **Evidence**: the specific divergence between spec and implementation.
|
|
51
52
|
- **Severity**: critical (public API returns wrong results), high (documented behavior incorrect in edge cases), medium (internal API inconsistency), low (stale docs, minor parameter mismatch).
|
|
52
|
-
- **Evidence level**: observed fact / strong inference / open question.
|
|
53
|
+
- **Evidence level**: observed fact / strong inference / external-behavior claim / open question. A claim about what another system does with this code's output (a server, engine, driver, third-party API, OS) is an `external-behavior claim`, not a `strong inference` — see findings/defect-scan/SKILL.md §Evidence Classification.
|
|
53
54
|
|
|
54
55
|
## What to skip
|
|
55
56
|
|
|
@@ -50,7 +50,7 @@ For each finding, record:
|
|
|
50
50
|
- **Defect**: what configuration or environment hazard exists.
|
|
51
51
|
- **Evidence**: the specific hardcoded value, missing validation, or dangerous default.
|
|
52
52
|
- **Severity**: critical (security exposure in default config, data loss on misconfiguration), high (production failure from missing validation), medium (hardcoded value that will break in a different environment), low (undocumented config behavior).
|
|
53
|
-
- **Evidence level**: observed fact / strong inference / open question.
|
|
53
|
+
- **Evidence level**: observed fact / strong inference / external-behavior claim / open question. A claim about what another system does with this code's output (a server, engine, driver, third-party API, OS) is an `external-behavior claim`, not a `strong inference` — see findings/defect-scan/SKILL.md §Evidence Classification.
|
|
54
54
|
|
|
55
55
|
## What to skip
|
|
56
56
|
|
|
@@ -39,9 +39,10 @@ Use the architecture map to decide emphasis:
|
|
|
39
39
|
|
|
40
40
|
Use the same scheme as the legacy defect-scan SKILL:
|
|
41
41
|
|
|
42
|
-
- **Evidence levels:** `observed fact`, `strong inference`, `open question`.
|
|
42
|
+
- **Evidence levels:** `observed fact`, `strong inference`, `external-behavior claim`, `open question`.
|
|
43
43
|
- **Severity:** `critical`, `high`, `medium`, `low`.
|
|
44
|
-
- **Action (pre-porting pipelines):** `fix before porting`, `port differently`, `leave behind`.
|
|
44
|
+
- **Action (pre-porting pipelines):** `fix before porting`, `port differently`, `leave behind`, `verify at runtime`.
|
|
45
|
+
- **Pairing rule:** `open question` or `external-behavior claim` evidence takes `verify at runtime` or `port differently`, never `fix before porting`, and the finding also appears in the report's Open Questions table. Validation checks this on the findings tables.
|
|
45
46
|
|
|
46
47
|
See `findings/defect-scan/SKILL.md` for the full criteria; this phase intentionally does not duplicate them.
|
|
47
48
|
|
|
@@ -47,6 +47,8 @@ When citing a contract or protocol violation, include the contract ID or state-m
|
|
|
47
47
|
|
|
48
48
|
Read the `carry_forward` entries in `workflow/status.yaml` whose `target_phase` is `defect-scan-semantic`. The mechanical phase may have routed semantic-flavored sightings here for closure.
|
|
49
49
|
|
|
50
|
+
**Closing a routed item does not settle the question it came from.** Before closing a carry-forward, check whether it derives from an `open_questions` entry of `kind: needs-runtime-test` that is still unresolved — the mechanical phase typically registers the question and routes one of its candidate explanations onward in the same handoff. If the question stands, the finding that addresses the carry-forward inherits its uncertainty: evidence `external-behavior claim` or `open question`, action `verify at runtime`, and a row in this report's Open Questions table naming the question's id. Asserting one of the question's candidates as `strong inference` / `fix before porting` while the question remains open is the contradiction this rule exists to prevent — a real run did exactly that, and runtime testing inverted the finding. Only runtime evidence (or the external system's own source at the pinned version) closes such a question; when you have it, list the question in `open_question_closures` and cite the evidence in the finding.
|
|
51
|
+
|
|
50
52
|
## Output
|
|
51
53
|
|
|
52
54
|
Write findings to the primary output using `templates/semantic-defects.md`. Organize by pass, then by severity within each pass. End with a summary table covering only passes 3, 4, and 5.
|
|
@@ -19,6 +19,7 @@ Keep four classes of findings separate throughout:
|
|
|
19
19
|
- `observed fact`: direct statements from docs, tests, schemas, types, and code.
|
|
20
20
|
- `strong inference`: architectural conclusions drawn from multiple facts.
|
|
21
21
|
- `portability hazard`: assumptions tied to the source language, runtime, terminal, OS, or third-party SDKs.
|
|
22
|
+
- `external-behavior claim`: a claim about what a system outside this source tree does (a server, engine, driver, third-party API, OS) — unverifiable at any read depth here; carry it forward as unsettled, never promote it to a fact.
|
|
22
23
|
- `open question`: missing or conflicting behavior that still needs evidence.
|
|
23
24
|
|
|
24
25
|
Prefer concept names over source names:
|
|
@@ -39,7 +40,7 @@ Sort features by porting importance:
|
|
|
39
40
|
|
|
40
41
|
If the defect report is available, integrate defect findings into the porting bundle:
|
|
41
42
|
- Reference relevant defects in the feature contract table.
|
|
42
|
-
- Tag each referenced defect with a porting recommendation: `fix before porting` (the defect would carry into a new implementation), `port differently` (the new implementation should handle this case differently by design),
|
|
43
|
+
- Tag each referenced defect with a porting recommendation: `fix before porting` (the defect would carry into a new implementation), `port differently` (the new implementation should handle this case differently by design), `leave behind` (the defect is specific to the source implementation and won't survive porting), or `verify at runtime` (the diagnosis is an `external-behavior claim` or `open question` — carry it as a spike for the spec, and do not design around an unverified diagnosis). Preserve `verify at runtime` as written: flattening it into one of the settled three is how a hedge stops traveling.
|
|
43
44
|
- Consolidate defect-related portability hazards alongside hazards from other phases.
|
|
44
45
|
|
|
45
46
|
Use the output template at `templates/reverse-engineering-bundle.md`. Produce:
|
|
@@ -77,6 +77,7 @@ Produce four outputs:
|
|
|
77
77
|
Mark every finding with one of these evidence levels:
|
|
78
78
|
- `observed fact`: direct statement from docs, tests, schemas, types, or code.
|
|
79
79
|
- `strong inference`: protocol conclusion drawn from multiple facts.
|
|
80
|
+
- `external-behavior claim`: a claim about what the peer or a system outside this source tree does with a message (how it parses, what it ignores, version-dependent behavior); unverifiable by reading this code, so it stays unsettled until a runtime capture or that system's own source confirms it.
|
|
80
81
|
- `portability hazard`: assumption tied to the source language, runtime, terminal, OS, or third-party SDKs.
|
|
81
82
|
- `open question`: missing or conflicting behavior that still needs evidence.
|
|
82
83
|
|
|
@@ -69,7 +69,7 @@ End with a spike list:
|
|
|
69
69
|
- risky performance assumptions
|
|
70
70
|
- platform-sensitive areas that need targeted tests
|
|
71
71
|
|
|
72
|
-
For every defect in the bundle, preserve its disposition (`fix before porting`, `port differently`,
|
|
72
|
+
For every defect in the bundle, preserve its disposition (`fix before porting`, `port differently`, `leave behind`, or `verify at runtime`) and convert it into an explicit design consequence or acceptance check — except `verify at runtime`, which becomes a Spike List entry and a `post_pipeline` entry of `kind: spike` in your handoff, never a design consequence: the diagnosis has not been confirmed, and designing around it would build the unverified claim into the new system.
|
|
73
73
|
|
|
74
74
|
Use the output template at `templates/reimplementation-spec.md`.
|
|
75
75
|
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
<!--
|
|
4
4
|
Output template for the architecture phase.
|
|
5
5
|
Fill in each section. Remove placeholder text. Keep the section headers.
|
|
6
|
-
Mark every conclusion as: fact / strong inference / open question.
|
|
6
|
+
Mark every conclusion as: fact / strong inference / external-behavior claim / open question.
|
|
7
7
|
-->
|
|
8
8
|
|
|
9
9
|
## System Intent
|
|
@@ -16,6 +16,12 @@
|
|
|
16
16
|
<!-- For each finding: location, defect, evidence, severity, evidence level, action. -->
|
|
17
17
|
<!-- Sort by severity: critical → high → medium → low. -->
|
|
18
18
|
<!-- If no findings, write "No defects found in this category." -->
|
|
19
|
+
<!-- Evidence Level: observed fact / strong inference / external-behavior claim / open question.
|
|
20
|
+
Action — pre-porting pipelines: fix before porting / port differently / leave behind / verify at runtime;
|
|
21
|
+
maintenance pipelines: fix now / track / accept / investigate.
|
|
22
|
+
Pairing rule (validated mechanically): open question or external-behavior claim ⇒ verify at
|
|
23
|
+
runtime or port differently (pre-porting) / investigate (maintenance), never fix before porting
|
|
24
|
+
or fix now; and list the finding in ## Open Questions below. -->
|
|
19
25
|
|
|
20
26
|
| # | Location | Defect | Severity | Evidence Level | Action |
|
|
21
27
|
|---|----------|--------|----------|----------------|--------|
|
|
@@ -99,6 +105,21 @@
|
|
|
99
105
|
|
|
100
106
|
---
|
|
101
107
|
|
|
108
|
+
## Open Questions
|
|
109
|
+
|
|
110
|
+
<!-- Every finding whose Evidence Level is open question or external-behavior claim gets a row
|
|
111
|
+
here, so the hedge travels with the finding into this document — not only into the handoff.
|
|
112
|
+
Mirror each row into your phase handoff's open_questions (kind: needs-runtime-test unless a
|
|
113
|
+
maintainer decision or spec ruling is what is missing). Derived findings lists the finding
|
|
114
|
+
numbers (e.g. "5.2, 4.1") that depend on this question; none of them may carry a settled
|
|
115
|
+
action while the question stands. -->
|
|
116
|
+
|
|
117
|
+
| ID | Kind | Question | Why source cannot settle it | Derived findings |
|
|
118
|
+
|----|------|----------|-----------------------------|------------------|
|
|
119
|
+
| | | | | |
|
|
120
|
+
|
|
121
|
+
---
|
|
122
|
+
|
|
102
123
|
## Coverage and limits
|
|
103
124
|
|
|
104
125
|
- Inspected scope:
|
|
@@ -120,6 +141,8 @@
|
|
|
120
141
|
| 4 | Summary tables are complete and counts match the detailed findings. | PASS / PARTIAL / FAIL | |
|
|
121
142
|
| 5 | Findings are marked with evidence levels. | PASS / PARTIAL / FAIL | |
|
|
122
143
|
| 6 | Coverage and limits name inspected scope, skipped scope, evidence basis, and blind spots. | PASS / PARTIAL / FAIL | |
|
|
144
|
+
| 7 | Unsettled findings (evidence level open question or external-behavior claim) carry an unsettled action (verify at runtime or port differently on pre-porting pipelines; investigate on maintenance pipelines), never a settled one, and each appears in the Open Questions table. | PASS / PARTIAL / FAIL | |
|
|
145
|
+
| 8 | Every quantitative specific in a finding (size, count, default, version, timeout) cites the file and line or command output it was read from, or is marked as an estimate. | PASS / PARTIAL / FAIL | |
|
|
123
146
|
|
|
124
147
|
**Validated by:** [session identifier or date]
|
|
125
148
|
**Overall:** PASS / PASS WITH GAPS / FAIL
|
|
@@ -22,6 +22,11 @@
|
|
|
22
22
|
<!-- For each finding: location, defect, evidence, severity, evidence level, action. -->
|
|
23
23
|
<!-- Sort by severity: critical → high → medium → low. -->
|
|
24
24
|
<!-- If no findings, write "No defects found in this category." -->
|
|
25
|
+
<!-- Evidence Level: observed fact / strong inference / external-behavior claim / open question.
|
|
26
|
+
Action: fix before porting / port differently / leave behind / verify at runtime.
|
|
27
|
+
Pairing rule (validated mechanically): open question or external-behavior claim ⇒
|
|
28
|
+
verify at runtime or port differently, never fix before porting; and list the finding
|
|
29
|
+
in ## Open Questions below. -->
|
|
25
30
|
|
|
26
31
|
| # | Location | Defect | Severity | Evidence Level | Action |
|
|
27
32
|
|---|----------|--------|----------|----------------|--------|
|
|
@@ -90,6 +95,21 @@
|
|
|
90
95
|
|
|
91
96
|
---
|
|
92
97
|
|
|
98
|
+
## Open Questions
|
|
99
|
+
|
|
100
|
+
<!-- Every finding whose Evidence Level is open question or external-behavior claim gets a row
|
|
101
|
+
here, so the hedge travels with the finding into this document — not only into the handoff.
|
|
102
|
+
Mirror each row into your phase handoff's open_questions (kind: needs-runtime-test unless a
|
|
103
|
+
maintainer decision or spec ruling is what is missing). Derived findings lists the finding
|
|
104
|
+
numbers (e.g. "1.3, 6.1") that depend on this question; none of them may carry a settled
|
|
105
|
+
action while the question stands. -->
|
|
106
|
+
|
|
107
|
+
| ID | Kind | Question | Why source cannot settle it | Derived findings |
|
|
108
|
+
|----|------|----------|-----------------------------|------------------|
|
|
109
|
+
| | | | | |
|
|
110
|
+
|
|
111
|
+
---
|
|
112
|
+
|
|
93
113
|
## Coverage and limits
|
|
94
114
|
|
|
95
115
|
- Inspected scope:
|
|
@@ -110,6 +130,8 @@
|
|
|
110
130
|
| 4 | Summary tables are complete and counts match the detailed findings. | PASS / PARTIAL / FAIL | |
|
|
111
131
|
| 5 | Findings are marked with evidence levels. | PASS / PARTIAL / FAIL | |
|
|
112
132
|
| 6 | Coverage and limits name inspected scope, skipped scope, evidence basis, and blind spots. | PASS / PARTIAL / FAIL | |
|
|
133
|
+
| 7 | Unsettled findings (evidence level open question or external-behavior claim) carry an unsettled action (verify at runtime or port differently on pre-porting pipelines; investigate on maintenance pipelines), never a settled one, and each appears in the Open Questions table. | PASS / PARTIAL / FAIL | |
|
|
134
|
+
| 8 | Every quantitative specific in a finding (size, count, default, version, timeout) cites the file and line or command output it was read from, or is marked as an estimate. | PASS / PARTIAL / FAIL | |
|
|
113
135
|
|
|
114
136
|
**Validated by:** [session identifier or date]
|
|
115
137
|
**Overall:** PASS / PASS WITH GAPS / FAIL
|
|
@@ -86,7 +86,7 @@
|
|
|
86
86
|
|
|
87
87
|
| Defect ID | Source Report | One-line Description | Severity | Disposition | Required design consequence |
|
|
88
88
|
|-----------|---------------|----------------------|----------|-------------|-----------------------------|
|
|
89
|
-
| | | | | fix before porting / port differently / leave behind | |
|
|
89
|
+
| | | | | fix before porting / port differently / leave behind / verify at runtime | |
|
|
90
90
|
|
|
91
91
|
## Observed Facts vs. Inferred Structure
|
|
92
92
|
|
|
@@ -159,7 +159,7 @@
|
|
|
159
159
|
| 1 | The system summary, layer map, contract table, protocol notes, and porting findings are synthesized. | PASS / PARTIAL / FAIL | |
|
|
160
160
|
| 2 | Portability hazards and open questions are separated from facts. | PASS / PARTIAL / FAIL | |
|
|
161
161
|
| 3 | Feature importance is sorted for porting. | PASS / PARTIAL / FAIL | |
|
|
162
|
-
| 4 | Known defects are referenced in the Defect Synthesis with porting recommendations (fix before porting / port differently / leave behind), or the section explicitly notes that no defect scan ran. | PASS / PARTIAL / FAIL | |
|
|
162
|
+
| 4 | Known defects are referenced in the Defect Synthesis with porting recommendations (fix before porting / port differently / leave behind / verify at runtime), or the section explicitly notes that no defect scan ran. | PASS / PARTIAL / FAIL | |
|
|
163
163
|
| 5 | Findings are marked with evidence levels. | PASS / PARTIAL / FAIL | |
|
|
164
164
|
| 6 | Coverage and limits name inspected scope, skipped scope, evidence basis, and blind spots. | PASS / PARTIAL / FAIL | |
|
|
165
165
|
| 7 | The Source Index makes the bundle a self-contained compression boundary and identifies targeted deep-read triggers. | PASS / PARTIAL / FAIL | |
|
|
@@ -25,6 +25,12 @@
|
|
|
25
25
|
<!-- For each finding: location, defect, evidence, severity, evidence level, action. -->
|
|
26
26
|
<!-- Cite the protocol or state-machine entry that the finding violates, when relevant. -->
|
|
27
27
|
<!-- If no findings, write "No defects found in this category." -->
|
|
28
|
+
<!-- Evidence Level: observed fact / strong inference / external-behavior claim / open question.
|
|
29
|
+
Action: fix before porting / port differently / leave behind / verify at runtime.
|
|
30
|
+
Pairing rule (validated mechanically): open question or external-behavior claim ⇒
|
|
31
|
+
verify at runtime or port differently, never fix before porting; and list the finding
|
|
32
|
+
in ## Open Questions below. A finding that closes a routed carry-forward derived from a
|
|
33
|
+
still-open needs-runtime-test question inherits that uncertainty. -->
|
|
28
34
|
|
|
29
35
|
| # | Location | Defect | Severity | Evidence Level | Action |
|
|
30
36
|
|---|----------|--------|----------|----------------|--------|
|
|
@@ -45,6 +51,9 @@
|
|
|
45
51
|
## Pass 5: API Contract Violations
|
|
46
52
|
|
|
47
53
|
<!-- Each finding should pair the source contract/protocol reference with the diverging code location. -->
|
|
54
|
+
<!-- When the analyzed code is the CALLER of a contract another system implements, what that system
|
|
55
|
+
does with the payload is an external-behavior claim (action: verify at runtime) until a runtime
|
|
56
|
+
probe against the pinned version says otherwise — see passes/05 "Which side implements the contract". -->
|
|
48
57
|
|
|
49
58
|
| # | Location | Defect | Severity | Evidence Level | Action | Spec Reference |
|
|
50
59
|
|---|----------|--------|----------|----------------|--------|----------------|
|
|
@@ -95,6 +104,21 @@
|
|
|
95
104
|
|
|
96
105
|
---
|
|
97
106
|
|
|
107
|
+
## Open Questions
|
|
108
|
+
|
|
109
|
+
<!-- Every finding whose Evidence Level is open question or external-behavior claim gets a row
|
|
110
|
+
here, so the hedge travels with the finding into this document — not only into the handoff.
|
|
111
|
+
Include any still-open needs-runtime-test question a closed carry-forward derived from.
|
|
112
|
+
Mirror each row into your phase handoff's open_questions. Derived findings lists the finding
|
|
113
|
+
numbers (e.g. "5.2") that depend on this question; none of them may carry a settled action
|
|
114
|
+
while the question stands. -->
|
|
115
|
+
|
|
116
|
+
| ID | Kind | Question | Why source cannot settle it | Derived findings |
|
|
117
|
+
|----|------|----------|-----------------------------|------------------|
|
|
118
|
+
| | | | | |
|
|
119
|
+
|
|
120
|
+
---
|
|
121
|
+
|
|
98
122
|
## Coverage and limits
|
|
99
123
|
|
|
100
124
|
- Inspected scope:
|
|
@@ -115,6 +139,8 @@
|
|
|
115
139
|
| 4 | Findings are organized by pass and sorted by severity; summary tables match the detailed findings. | PASS / PARTIAL / FAIL | |
|
|
116
140
|
| 5 | Findings are marked with evidence levels. | PASS / PARTIAL / FAIL | |
|
|
117
141
|
| 6 | Coverage and limits name inspected scope, skipped scope, evidence basis, and blind spots. | PASS / PARTIAL / FAIL | |
|
|
142
|
+
| 7 | Unsettled findings (evidence level open question or external-behavior claim) carry an unsettled action (verify at runtime or port differently on pre-porting pipelines; investigate on maintenance pipelines), never a settled one, and each appears in the Open Questions table. | PASS / PARTIAL / FAIL | |
|
|
143
|
+
| 8 | Every quantitative specific in a finding (size, count, default, version, timeout) cites the file and line or command output it was read from, or is marked as an estimate. | PASS / PARTIAL / FAIL | |
|
|
118
144
|
|
|
119
145
|
**Validated by:** [session identifier or date]
|
|
120
146
|
**Overall:** PASS / PASS WITH GAPS / FAIL
|
|
@@ -50,7 +50,7 @@ Below is what a real PASS WITH GAPS block looks like — useful when a phase fin
|
|
|
50
50
|
| 2 | The layer map and dependency direction are documented. | PASS | §Layer Map; dependency direction in §Layer Map → "Dependency Direction." |
|
|
51
51
|
| 3 | Public surfaces are identified. | PARTIAL | CLI commands and HTTP routes enumerated (§Public Surfaces). MCP server endpoints and the websocket subscription channel are listed by name only — schemas not extracted. Routed to `carry_forward` as `arch-CF2` with `target_phase: protocols`. |
|
|
52
52
|
| 4 | Runtime lifecycle, concurrency model, and porting priorities are summarized. | PASS | §Runtime Lifecycle, §Concurrency Model, §Porting Priorities (table). |
|
|
53
|
-
| 5 | Findings are marked with evidence levels. | PASS | All inferences marked `observed fact` / `strong inference` / `portability hazard` / `open question`. |
|
|
53
|
+
| 5 | Findings are marked with evidence levels. | PASS | All inferences marked `observed fact` / `strong inference` / `portability hazard` / `external-behavior claim` / `open question`. |
|
|
54
54
|
|
|
55
55
|
**Validated by:** 2026-05-02 (architecture phase, session 1)
|
|
56
56
|
**Overall:** PASS WITH GAPS
|
|
@@ -56,6 +56,8 @@ phases:
|
|
|
56
56
|
- Findings are organized by pass and sorted by severity.
|
|
57
57
|
- Summary tables are complete and counts match the detailed findings.
|
|
58
58
|
- Findings are marked with evidence levels.
|
|
59
|
+
- Unsettled findings (evidence level open question or external-behavior claim) carry an unsettled action (verify at runtime or port differently on pre-porting pipelines; investigate on maintenance pipelines), never a settled one, and each appears in the Open Questions table.
|
|
60
|
+
- Every quantitative specific in a finding (size, count, default, version, timeout) cites the file and line or command output it was read from, or is marked as an estimate.
|
|
59
61
|
- Coverage and limits name inspected scope, skipped scope, evidence basis, and blind spots.
|
|
60
62
|
handoff_requirements:
|
|
61
63
|
- Run validation per workflow/VALIDATE.md. Append validation block to primary output.
|
|
@@ -60,6 +60,8 @@ phases:
|
|
|
60
60
|
- Findings are organized by pass and sorted by severity.
|
|
61
61
|
- Summary tables are complete and counts match the detailed findings.
|
|
62
62
|
- Findings are marked with evidence levels.
|
|
63
|
+
- Unsettled findings (evidence level open question or external-behavior claim) carry an unsettled action (verify at runtime or port differently on pre-porting pipelines; investigate on maintenance pipelines), never a settled one, and each appears in the Open Questions table.
|
|
64
|
+
- Every quantitative specific in a finding (size, count, default, version, timeout) cites the file and line or command output it was read from, or is marked as an estimate.
|
|
63
65
|
- Coverage and limits name inspected scope, skipped scope, evidence basis, and blind spots.
|
|
64
66
|
handoff_requirements:
|
|
65
67
|
- Run validation per workflow/VALIDATE.md. Append validation block to primary output.
|
|
@@ -159,7 +161,7 @@ phases:
|
|
|
159
161
|
- The system summary, layer map, contract table, protocol notes, and porting findings are synthesized.
|
|
160
162
|
- Portability hazards and open questions are separated from facts.
|
|
161
163
|
- Feature importance is sorted for porting.
|
|
162
|
-
- Known defects are referenced with porting recommendations (fix before porting / port differently / leave behind).
|
|
164
|
+
- Known defects are referenced with porting recommendations (fix before porting / port differently / leave behind / verify at runtime).
|
|
163
165
|
- Findings are marked with evidence levels.
|
|
164
166
|
- Coverage and limits name inspected scope, skipped scope, evidence basis, and blind spots.
|
|
165
167
|
- The Source Index makes the bundle a self-contained compression boundary and identifies targeted deep-read triggers.
|
|
@@ -62,6 +62,8 @@ phases:
|
|
|
62
62
|
- Summary tables are complete and counts match the detailed findings.
|
|
63
63
|
- Items spotted that are actually semantic in nature are routed onward via a carry_forward entry in the phase handoff targeting defect-scan-semantic.
|
|
64
64
|
- Findings are marked with evidence levels.
|
|
65
|
+
- Unsettled findings (evidence level open question or external-behavior claim) carry an unsettled action (verify at runtime or port differently on pre-porting pipelines; investigate on maintenance pipelines), never a settled one, and each appears in the Open Questions table.
|
|
66
|
+
- Every quantitative specific in a finding (size, count, default, version, timeout) cites the file and line or command output it was read from, or is marked as an estimate.
|
|
65
67
|
- Coverage and limits name inspected scope, skipped scope, evidence basis, and blind spots.
|
|
66
68
|
handoff_requirements:
|
|
67
69
|
- Run validation per workflow/VALIDATE.md. Append validation block to primary output.
|
|
@@ -157,6 +159,8 @@ phases:
|
|
|
157
159
|
- Findings are organized by pass and sorted by severity; summary tables match the detailed findings.
|
|
158
160
|
- Any carry_forward entries that targeted defect-scan-semantic have been resolved or explicitly re-routed.
|
|
159
161
|
- Findings are marked with evidence levels.
|
|
162
|
+
- Unsettled findings (evidence level open question or external-behavior claim) carry an unsettled action (verify at runtime or port differently on pre-porting pipelines; investigate on maintenance pipelines), never a settled one, and each appears in the Open Questions table.
|
|
163
|
+
- Every quantitative specific in a finding (size, count, default, version, timeout) cites the file and line or command output it was read from, or is marked as an estimate.
|
|
160
164
|
- Coverage and limits name inspected scope, skipped scope, evidence basis, and blind spots.
|
|
161
165
|
handoff_requirements:
|
|
162
166
|
- Run validation per workflow/VALIDATE.md. Append validation block to primary output.
|
|
@@ -196,7 +200,7 @@ phases:
|
|
|
196
200
|
- The system summary, layer map, contract table, protocol notes, and porting findings are synthesized.
|
|
197
201
|
- Portability hazards and open questions are separated from facts.
|
|
198
202
|
- Feature importance is sorted for porting.
|
|
199
|
-
- Defect Synthesis consolidates mechanical-defects.md and semantic-defects.md with porting recommendations (fix before porting / port differently / leave behind).
|
|
203
|
+
- Defect Synthesis consolidates mechanical-defects.md and semantic-defects.md with porting recommendations (fix before porting / port differently / leave behind / verify at runtime).
|
|
200
204
|
- Findings are marked with evidence levels.
|
|
201
205
|
- Coverage and limits name inspected scope, skipped scope, evidence basis, and blind spots.
|
|
202
206
|
- The Source Index makes the bundle a self-contained compression boundary and identifies targeted deep-read triggers.
|
|
@@ -90,6 +90,8 @@ phases:
|
|
|
90
90
|
- Summary tables are complete and counts match the detailed findings.
|
|
91
91
|
- Items spotted that are actually semantic in nature are routed onward via a carry_forward entry in the phase handoff targeting defect-scan-semantic.
|
|
92
92
|
- Findings are marked with evidence levels.
|
|
93
|
+
- Unsettled findings (evidence level open question or external-behavior claim) carry an unsettled action (verify at runtime or port differently on pre-porting pipelines; investigate on maintenance pipelines), never a settled one, and each appears in the Open Questions table.
|
|
94
|
+
- Every quantitative specific in a finding (size, count, default, version, timeout) cites the file and line or command output it was read from, or is marked as an estimate.
|
|
93
95
|
- Coverage and limits name inspected scope, skipped scope, evidence basis, and blind spots.
|
|
94
96
|
handoff_requirements:
|
|
95
97
|
- Run validation per workflow/VALIDATE.md. Append validation block to primary output.
|
|
@@ -194,6 +196,8 @@ phases:
|
|
|
194
196
|
- Findings are organized by pass and sorted by severity; summary tables match the detailed findings.
|
|
195
197
|
- Any carry_forward entries that targeted defect-scan-semantic have been resolved or explicitly re-routed.
|
|
196
198
|
- Findings are marked with evidence levels.
|
|
199
|
+
- Unsettled findings (evidence level open question or external-behavior claim) carry an unsettled action (verify at runtime or port differently on pre-porting pipelines; investigate on maintenance pipelines), never a settled one, and each appears in the Open Questions table.
|
|
200
|
+
- Every quantitative specific in a finding (size, count, default, version, timeout) cites the file and line or command output it was read from, or is marked as an estimate.
|
|
197
201
|
- Coverage and limits name inspected scope, skipped scope, evidence basis, and blind spots.
|
|
198
202
|
handoff_requirements:
|
|
199
203
|
- Run validation per workflow/VALIDATE.md. Append validation block to primary output.
|
|
@@ -236,7 +240,7 @@ phases:
|
|
|
236
240
|
- The system summary, layer map, contract table, protocol notes, and porting findings are synthesized.
|
|
237
241
|
- Portability hazards and open questions are separated from facts.
|
|
238
242
|
- Feature importance is sorted for porting.
|
|
239
|
-
- Defect Synthesis consolidates mechanical-defects.md and semantic-defects.md with porting recommendations (fix before porting / port differently / leave behind).
|
|
243
|
+
- Defect Synthesis consolidates mechanical-defects.md and semantic-defects.md with porting recommendations (fix before porting / port differently / leave behind / verify at runtime).
|
|
240
244
|
- Findings are marked with evidence levels.
|
|
241
245
|
- Coverage and limits name inspected scope, skipped scope, evidence basis, and blind spots.
|
|
242
246
|
- The Source Index makes the bundle a self-contained compression boundary and identifies targeted deep-read triggers.
|
package/README.md
CHANGED
|
@@ -35,7 +35,7 @@ Asking an LLM to "analyze this repo" loses context halfway through, hallucinates
|
|
|
35
35
|
|
|
36
36
|
3. **The output is a spec, not a chat log.** The final `reimplementation-spec.md` is language-agnostic, module-inventoried, and carries acceptance scenarios plus known unknowns. Hand it to another agent to rebuild from.
|
|
37
37
|
|
|
38
|
-
Every finding is tagged with an evidence level: `observed fact`, `strong inference`, `portability hazard`, or `open question`.
|
|
38
|
+
Every finding is tagged with an evidence level: `observed fact`, `strong inference`, `portability hazard`, `external-behavior claim`, or `open question`.
|
|
39
39
|
|
|
40
40
|
---
|
|
41
41
|
|
|
@@ -228,7 +228,7 @@ The porting bundle is the final intentional compression boundary. It carries a s
|
|
|
228
228
|
| **Porting bundle** | Everything synthesized into a porting-oriented view with priority rankings |
|
|
229
229
|
| **Reimplementation spec** | Language-agnostic build plan with modules, acceptance scenarios, and known unknowns |
|
|
230
230
|
|
|
231
|
-
Every finding is tagged with an evidence level: `observed fact`, `strong inference`, `portability hazard`, or `open question`. Every phase output is validated against explicit completion criteria before the pipeline advances.
|
|
231
|
+
Every finding is tagged with an evidence level: `observed fact`, `strong inference`, `portability hazard`, `external-behavior claim`, or `open question`. Every phase output is validated against explicit completion criteria before the pipeline advances.
|
|
232
232
|
|
|
233
233
|
---
|
|
234
234
|
|
|
@@ -373,7 +373,7 @@ Implements MCP spec revision [`2025-11-25`](https://modelcontextprotocol.io/spec
|
|
|
373
373
|
| `codecarto_refresh_scaffold` | MCP-only ([#159](https://github.com/HuginnIndustries/CodeCartographer/issues/159)) |
|
|
374
374
|
| `codecarto_broadside` | `/codecarto-broadside` |
|
|
375
375
|
|
|
376
|
-
Each workflow tool accepts an absolute `cwd` for the target repository. `codecarto_init` requires `force: true` to overwrite an existing `.codecarto/` (instead of Pi's interactive confirmation). The library tools accept an explicit absolute `library_path` or resolve `library.path` from `.codecarto/workflow/config.yaml` / `~/.codecarto/config.yaml`. The library schema is experimental and may break before v2.
|
|
376
|
+
Each workflow tool accepts an absolute `cwd` for the target repository. `codecarto_init` requires `force: true` to overwrite an existing `.codecarto/` (instead of Pi's interactive confirmation). The library tools accept an explicit absolute `library_path` or resolve `library.path` from `.codecarto/workflow/config.yaml` / `~/.codecarto/config.yaml`. `codecarto_library_reindex` and `codecarto_library_list` also report entries whose versions disagree about `source_repo` — the shape a slug collision left behind before v0.17.0's publish guard — and leave the repair manual, since splitting an entry changes paths the library format treats as ABI. The library schema is experimental and may break before v2.
|
|
377
377
|
|
|
378
378
|
---
|
|
379
379
|
|
|
@@ -584,7 +584,7 @@ The MCP server does steps 1–3 directly; the Pi extension wraps them as slash c
|
|
|
584
584
|
- **LLM-agnostic** — works with any model that can read and write files.
|
|
585
585
|
- **Phase-gated** — one phase per session, validated before advancing.
|
|
586
586
|
- **Single source of truth** — `status.yaml` tracks progress; no duplicated state.
|
|
587
|
-
- **Evidence-classified** — every finding tagged as observed fact, strong inference, portability hazard, or open question.
|
|
587
|
+
- **Evidence-classified** — every finding tagged as observed fact, strong inference, portability hazard, external-behavior claim, or open question.
|
|
588
588
|
- **Template-driven** — consistent output structure across projects and sessions.
|
|
589
589
|
- **Drop-in** — lives inside your repo as `.codecarto/`. No symlinks, no copying source code, no runtime daemon.
|
|
590
590
|
|
|
@@ -13,8 +13,11 @@ The common failure is treating defect reports as a separate document that the po
|
|
|
13
13
|
| `fix before porting` | the defect would be reproduced by a faithful port | design it out; the spec states the correct behavior |
|
|
14
14
|
| `port differently` | the behavior is needed but the mechanism is wrong | spec the intent, not the implementation |
|
|
15
15
|
| `leave behind` | dead, vestigial, or actively harmful | name it explicitly so a later reader doesn't "restore" it |
|
|
16
|
+
| `verify at runtime` | the diagnosis is an `external-behavior claim` or `open question` — about a server, engine, driver, or API this source only calls | a spike in the spec's Spike List and a `post_pipeline` `kind: spike` entry; never a design consequence, because the claim is unconfirmed |
|
|
16
17
|
|
|
17
|
-
Add the acceptance-test implication alongside each row.
|
|
18
|
+
Add the acceptance-test implication alongside each row.
|
|
19
|
+
|
|
20
|
+
The fourth disposition exists because of a real run: a defect scan asserted, as `strong inference` / `fix before porting`, that an inference server expected a different `logit_bias` payload shape, and recommended a one-line reshape. Runtime testing against the pinned engine inverted it — the shape the code already sent worked, and the recommended one was silently ignored. The evidence level bounds the action: `open question` or `external-behavior claim` evidence never pairs with `fix before porting`, and validation now fails a defect report that does so. A hazard with no test in the spec will be reintroduced by whoever implements it.
|
|
18
21
|
|
|
19
22
|
Close a carry-forward item only once its guidance is represented in an artifact a later phase actually consumes — not merely mentioned in the phase that raised it.
|
|
20
23
|
|
|
@@ -15,7 +15,7 @@ A CodeCartographer **library** is a directory of published reimplementation-spec
|
|
|
15
15
|
| `codecarto_library_init` | Create the directory, write the marker, record `library.path` in user-global config | Idempotent; pass `namespace` to create a namespaced library |
|
|
16
16
|
| `codecarto_publish` | Publish a spec as a library entry | Required: `source_repo`, `headline`, and `spec` (inline) or `spec_path` (absolute). Content-hash idempotent: identical bytes update metadata in place, no version bump. `slug` derives from `source_repo` if omitted; namespaced libraries require `namespace` (or inherit via `cwd`). Provenance (`source_commit`, `source_branch`, `source_dirty`, `analyzed_at`, `pipeline`, `model_metadata`) is recorded; omitted generation fields default to `unknown` |
|
|
17
17
|
| `codecarto_library_list` | List entries | Filter by `namespace`, `tag`, `slug`, or `source_repo` |
|
|
18
|
-
| `codecarto_library_reindex` | Regenerate `index.yaml` + `INDEX.md` from filesystem state | For manual edits and index merge conflicts |
|
|
18
|
+
| `codecarto_library_reindex` | Regenerate `index.yaml` + `INDEX.md` from filesystem state | For manual edits and index merge conflicts. Also reports entries whose versions disagree about `source_repo` (merged by a slug collision before publish refused cross-project appends; `codecarto_library_list` flags them too) — repair is manual, split the entry by hand |
|
|
19
19
|
|
|
20
20
|
## When to publish
|
|
21
21
|
|
|
@@ -8,7 +8,7 @@ All of them happen at the **phase boundary**: after one phase completes, before
|
|
|
8
8
|
|
|
9
9
|
1. **Promote conventions and append decisions.** Phase closeouts carry "Proposed Conventions" and "Decisions Beyond Prompt" sections; handoffs carry a `decisions` array. Promotion into `CONVENTIONS.md` (when a pattern recurs or clearly generalizes) and `DECISIONS.md` (every cross-cutting decision, numbered, append-only) is your call to make at the boundary. Proposals left in closeout prose are proposals lost — a real seven-phase run stranded ~12 proposed conventions and 23 decisions this way, because nobody held the duty.
|
|
10
10
|
|
|
11
|
-
2. **Re-triage open-question labels.** An `open_questions` entry's `kind` is itself a claim that needs evidence. Before accepting `needs-maintainer-decision` or `needs-runtime-test` into the next phase, re-test: *has this become answerable by reading?* Labels are sticky — the routing machinery faithfully carries a question forward, but nothing re-examines whether the label was right, so a mislabel suppresses verification for the rest of the pipeline.
|
|
11
|
+
2. **Re-triage open-question labels.** An `open_questions` entry's `kind` is itself a claim that needs evidence. Before accepting `needs-maintainer-decision` or `needs-runtime-test` into the next phase, re-test: *has this become answerable by reading?* Labels are sticky — the routing machinery faithfully carries a question forward, but nothing re-examines whether the label was right, so a mislabel suppresses verification for the rest of the pipeline. The duty cuts both ways: when the answer is genuinely *no, this still needs a runtime test*, no finding in the next phase may assert one of the question's candidate answers with a settled action (`fix before porting`, `fix now`). The finding inherits the question's uncertainty — evidence `external-behavior claim` or `open question`, action `verify at runtime` — until runtime evidence closes the question. A real `full-with-deep-audit` run broke this: the mechanical phase wrote "source alone cannot determine which" and routed one candidate onward; the semantic phase closed it as `strong inference` / `fix before porting`; runtime testing showed the recommended fix would have converted working behavior into the one shape the engine silently ignores.
|
|
12
12
|
|
|
13
13
|
3. **Sweep for contradictions.** Compare the incoming phase's required reads against earlier phases' `owner_notes`. A measured fact that contradicts a summarized claim (a line count that belies "this layer is pure configuration", a schema that admits a value a doc says is impossible) is a gap to route — into the next phase's work, an `open_questions` entry, or a correction — not a nuance to smooth over.
|
|
14
14
|
|
package/dist/core/amendment.js
CHANGED
|
@@ -9,7 +9,7 @@ import { appendFile, mkdir, readFile, writeFile } from "node:fs/promises";
|
|
|
9
9
|
import { join } from "node:path";
|
|
10
10
|
import { getNextEligiblePhase } from "./pipeline.js";
|
|
11
11
|
import { buildTerminalNextActions, ensureArray, normalizeStatus } from "./status.js";
|
|
12
|
-
import { dateOnly, pathExists } from "./utils.js";
|
|
12
|
+
import { dateOnly, newlineIfUnterminated, pathExists } from "./utils.js";
|
|
13
13
|
import { getWorkspaceState, updateStatusAtomically } from "./workspace.js";
|
|
14
14
|
import { loadYamlFile } from "./yaml.js";
|
|
15
15
|
/** Same charset rule as phase ids: the slug becomes file names, so path shapes are refused. */
|
|
@@ -134,7 +134,7 @@ export async function applyAmendment(cwd, name) {
|
|
|
134
134
|
// Created below when absent.
|
|
135
135
|
}
|
|
136
136
|
if (!current.split(/\r?\n/).some((line) => line.includes(`[closeout](closeouts/${closeoutFile})`))) {
|
|
137
|
-
await appendFile(threadLogPath, `${entry}\n`, "utf8");
|
|
137
|
+
await appendFile(threadLogPath, `${newlineIfUnterminated(current)}${entry}\n`, "utf8");
|
|
138
138
|
}
|
|
139
139
|
closeoutNotice = `Closeout: .codecarto/closeouts/${closeoutFile}`;
|
|
140
140
|
return { state: { ...lockedState, status: nextStatus } };
|
|
@@ -2,6 +2,11 @@ import type { ValidationResult, WorkspaceState } from "./types.ts";
|
|
|
2
2
|
export type CompletionResult = {
|
|
3
3
|
updatedState: WorkspaceState;
|
|
4
4
|
closeoutNotice?: string;
|
|
5
|
+
/**
|
|
6
|
+
* Non-gating closure-integrity observations (#122): closures the handoff
|
|
7
|
+
* claims that the primary output never mentions. Empty when clean.
|
|
8
|
+
*/
|
|
9
|
+
warnings: string[];
|
|
5
10
|
/**
|
|
6
11
|
* One-line phase-boundary reminder covering what completion just mechanized
|
|
7
12
|
* (decisions appended, proposals staged) and what still needs orchestrator
|