@dogfood-lab/findings 1.3.2 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,259 @@
1
+ /**
2
+ * Apply-back for accepted recommendations (F2-INTEL-003).
3
+ *
4
+ * The intelligence layer derives recommendations whose `action` describes a
5
+ * concrete operational change: `{ type, target, details }` where
6
+ * type ∈ add_check | add_scenario | set_policy | set_evidence |
7
+ * add_review_step | set_verification
8
+ * target ≤ 100 chars
9
+ * details = FREE TEXT (the human-readable intent)
10
+ *
11
+ * This module turns an ACCEPTED recommendation into an actual edit — but only
12
+ * where it is safe and unambiguous. Honest partial automation, never a fake
13
+ * auto-apply:
14
+ *
15
+ * - Only `status === 'accepted'` is applicable. A candidate / rejected
16
+ * recommendation refuses with a structured error.
17
+ * - The only structurally-safe edit is adding the recommendation's `target`
18
+ * (a scenario / check id) to a named repo policy's
19
+ * `surfaces.<surface>.required_scenarios` list. `add_scenario` and
20
+ * `add_check` map to this; every other action type is free-text-only intent
21
+ * and REFUSES on --write with a hint to apply manually.
22
+ * - The free-text `details` is NEVER injected into the policy as logic — it is
23
+ * recorded as provenance only.
24
+ * - dry-run (default) renders the proposed change and writes nothing.
25
+ *
26
+ * Structured errors all carry { code, message, hint } so the CLI and any
27
+ * programmatic caller see the same vocabulary.
28
+ *
29
+ * TEST_ROOT-safe: every path is derived from `rootDir`; no real-tree paths are
30
+ * hard-coded.
31
+ */
32
+
33
+ import { resolve } from 'node:path';
34
+ import { existsSync, readFileSync } from 'node:fs';
35
+ import yaml from 'js-yaml';
36
+
37
+ import { isUnsafeSegment } from '@dogfood-lab/ingest/lib/unsafe-segment.js';
38
+ import { atomicWriteFileSync } from '../lib/atomic-write.js';
39
+ import { findArtifactById } from '../review/review-artifacts.js';
40
+ import { createEvent, appendEvent } from '../review/event-log.js';
41
+
42
+ /**
43
+ * Action types whose `target` is a structured id we can safely add to a policy's
44
+ * required-scenarios list. Everything else is free-text-only intent.
45
+ */
46
+ const STRUCTURED_LIST_ACTIONS = new Set(['add_scenario', 'add_check']);
47
+
48
+ function structuredError(code, message, hint) {
49
+ return { success: false, error: { code, message, hint } };
50
+ }
51
+
52
+ /**
53
+ * @param {string} rootDir
54
+ * @param {object} params
55
+ * @param {string} params.id - recommendation_id
56
+ * @param {'dry-run'|'write'} [params.mode='dry-run']
57
+ * @param {string} [params.actor='operator']
58
+ * @param {string} [params.policyRepo] - org/repo naming the policy to edit (required for --write)
59
+ * @returns {{ success: boolean, applied?: boolean, preview?: object, provenance?: object, error?: { code, message, hint } }}
60
+ */
61
+ export function applyRecommendation(rootDir, params) {
62
+ const { id } = params;
63
+ const mode = params.mode || 'dry-run';
64
+ const actor = params.actor || 'operator';
65
+
66
+ if (!id) {
67
+ return structuredError('RECOMMENDATION_ID_REQUIRED', 'id is required', 'Pass the recommendation_id to apply.');
68
+ }
69
+
70
+ // Load the recommendation.
71
+ const found = findArtifactById(rootDir, 'recommendation', id);
72
+ if (!found) {
73
+ return structuredError(
74
+ 'RECOMMENDATION_NOT_FOUND',
75
+ `recommendation not found: ${id}`,
76
+ 'Run `findings recommendations list` to see available ids.'
77
+ );
78
+ }
79
+ const rec = found.data;
80
+
81
+ // Applicability gate — only accepted recommendations are applicable.
82
+ if (rec.status !== 'accepted') {
83
+ return structuredError(
84
+ 'RECOMMENDATION_NOT_ACCEPTED',
85
+ `recommendation ${id} has status "${rec.status}" — only accepted recommendations can be applied`,
86
+ 'Accept it first: `findings recommendations accept ' + id + ' --actor <name>`.'
87
+ );
88
+ }
89
+
90
+ const action = rec.action || {};
91
+ const surfaces = rec.applies_to?.product_surfaces || [];
92
+
93
+ // findings-A-001 — path-traversal guard on the operator-supplied --policy
94
+ // <org/repo>. `policyPathFor` resolves `policies/repos/<org>/<repo>.yaml`; an
95
+ // org/repo carrying `..` or a separator would escape the policies tree on
96
+ // BOTH the dry-run (path leaked in preview) and write (file touched) paths,
97
+ // so reject here before either branch resolves a path.
98
+ if (params.policyRepo) {
99
+ const [pOrg, pRepo] = String(params.policyRepo).split('/');
100
+ if (!pOrg || !pRepo || isUnsafeSegment(pOrg) || isUnsafeSegment(pRepo)) {
101
+ return structuredError(
102
+ 'RECOMMENDATION_UNSAFE_POLICY',
103
+ `policy repo "${params.policyRepo}" is not a safe org/repo path segment`,
104
+ 'Pass --policy <org/repo> with no ".." or path separators inside the org or repo name.'
105
+ );
106
+ }
107
+ }
108
+
109
+ // Build the resolution context shared by dry-run and write.
110
+ const isStructured = STRUCTURED_LIST_ACTIONS.has(action.type);
111
+ const surface = surfaces.length === 1 ? surfaces[0] : null;
112
+
113
+ // ── dry-run: render the proposed change, write nothing ──────────────
114
+ if (mode !== 'write') {
115
+ const policyPath = params.policyRepo ? policyPathFor(rootDir, params.policyRepo) : null;
116
+ return {
117
+ success: true,
118
+ applied: false,
119
+ preview: {
120
+ recommendationId: id,
121
+ actionType: action.type,
122
+ target: action.target,
123
+ details: action.details,
124
+ surface: surface,
125
+ surfaces,
126
+ autoApplicable: isStructured && !!surface,
127
+ policyRepo: params.policyRepo || null,
128
+ policyPath: policyPath,
129
+ field: isStructured ? `surfaces.${surface || '<surface>'}.required_scenarios` : null,
130
+ note: isStructured
131
+ ? (surface
132
+ ? `Would add "${action.target}" to required_scenarios for surface "${surface}". Free-text details recorded as provenance only.`
133
+ : `Action is structurally applicable but the recommendation spans ${surfaces.length} surface(s); --write needs a single surface.`)
134
+ : `Action type "${action.type}" is free-text-only intent — review the details and apply manually. --write will refuse.`
135
+ }
136
+ };
137
+ }
138
+
139
+ // ── write: apply ONLY the safe, unambiguous structured intent ───────
140
+
141
+ // Free-text-only intent (set_policy / set_evidence / set_verification /
142
+ // add_review_step) cannot be safely auto-applied.
143
+ if (!isStructured) {
144
+ return structuredError(
145
+ 'RECOMMENDATION_NOT_AUTO_APPLICABLE',
146
+ `recommendation ${id} action type "${action.type}" carries free-text-only intent`,
147
+ `The intent lives in action.details, which must not be injected as policy logic. Apply manually: "${action.details}".`
148
+ );
149
+ }
150
+
151
+ // Need a named policy to edit.
152
+ if (!params.policyRepo) {
153
+ return structuredError(
154
+ 'RECOMMENDATION_AMBIGUOUS_TARGET',
155
+ `recommendation ${id} does not name a policy to edit`,
156
+ 'Pass --policy <org/repo> to name the repo policy whose required_scenarios should gain "' + action.target + '".'
157
+ );
158
+ }
159
+
160
+ // Need an unambiguous surface.
161
+ if (!surface) {
162
+ return structuredError(
163
+ 'RECOMMENDATION_AMBIGUOUS_TARGET',
164
+ `recommendation ${id} applies to ${surfaces.length} surface(s) — cannot pick one automatically`,
165
+ 'Narrow the recommendation to a single product_surface, then re-run, or edit the policy manually.'
166
+ );
167
+ }
168
+
169
+ if (!action.target) {
170
+ return structuredError(
171
+ 'RECOMMENDATION_AMBIGUOUS_TARGET',
172
+ `recommendation ${id} has no action.target id to add`,
173
+ 'A structured add_scenario/add_check action must carry a target id. Apply manually.'
174
+ );
175
+ }
176
+
177
+ // Load the named policy.
178
+ const policyPath = policyPathFor(rootDir, params.policyRepo);
179
+ if (!existsSync(policyPath)) {
180
+ return structuredError(
181
+ 'RECOMMENDATION_TARGET_POLICY_MISSING',
182
+ `policy not found for ${params.policyRepo} at ${policyPath}`,
183
+ 'Create the repo policy first, or pass a --policy that exists.'
184
+ );
185
+ }
186
+
187
+ let policy;
188
+ try {
189
+ policy = yaml.load(readFileSync(policyPath, 'utf-8')) || {};
190
+ } catch (err) {
191
+ return structuredError(
192
+ 'RECOMMENDATION_TARGET_POLICY_UNREADABLE',
193
+ `could not parse policy ${params.policyRepo}: ${err.message}`,
194
+ 'Fix the policy YAML, then re-run.'
195
+ );
196
+ }
197
+
198
+ // Apply the structured intent: add target to surfaces.<surface>.required_scenarios.
199
+ if (!policy.surfaces) policy.surfaces = {};
200
+ if (!policy.surfaces[surface]) policy.surfaces[surface] = {};
201
+ if (!Array.isArray(policy.surfaces[surface].required_scenarios)) {
202
+ policy.surfaces[surface].required_scenarios = [];
203
+ }
204
+ const list = policy.surfaces[surface].required_scenarios;
205
+ const alreadyPresent = list.includes(action.target);
206
+ if (!alreadyPresent) list.push(action.target);
207
+
208
+ // Record provenance — recommendation_id + (free-text) details, kept OUT of any
209
+ // logic field. Stored under a dedicated `applied_recommendations` map keyed by
210
+ // the scenario id so it is auditable but never interpreted by the policy
211
+ // engine. The free-text details live here as a human note, never as a rule.
212
+ if (!policy.applied_recommendations) policy.applied_recommendations = {};
213
+ const provenance = {
214
+ recommendation_id: id,
215
+ action_type: action.type,
216
+ target: action.target,
217
+ surface,
218
+ details: action.details,
219
+ applied_by: actor,
220
+ applied_at: new Date().toISOString()
221
+ };
222
+ policy.applied_recommendations[action.target] = provenance;
223
+
224
+ // NOTE: provenance is intentionally written under a NON-schema key. The policy
225
+ // schema (`policy.schema.json`) has `additionalProperties: false`, so a real
226
+ // policy would reject `applied_recommendations`. We keep the provenance OUT of
227
+ // the validated surface by returning it on the result for the operator and the
228
+ // event log, and we strip it before persisting so the policy stays schema-valid.
229
+ const persistable = { ...policy };
230
+ delete persistable.applied_recommendations;
231
+
232
+ atomicWriteFileSync(policyPath, yaml.dump(persistable, { lineWidth: 120, noRefs: true }));
233
+
234
+ // Log an apply event (recommendation review-event shape; action 'apply').
235
+ const event = createEvent({
236
+ artifactId: id,
237
+ artifactKind: 'recommendation',
238
+ actor,
239
+ action: 'apply',
240
+ fromStatus: 'accepted',
241
+ toStatus: 'accepted',
242
+ notes: `Applied ${action.type} target=${action.target} to ${params.policyRepo} surface=${surface}`
243
+ });
244
+ appendEvent(rootDir, event);
245
+
246
+ return {
247
+ success: true,
248
+ applied: true,
249
+ alreadyPresent,
250
+ policyPath,
251
+ provenance
252
+ };
253
+ }
254
+
255
+ /** Resolve the on-disk path for a repo policy under rootDir. */
256
+ function policyPathFor(rootDir, orgRepo) {
257
+ const [org, repo] = orgRepo.split('/');
258
+ return resolve(rootDir, 'policies', 'repos', org, `${repo}.yaml`);
259
+ }
@@ -0,0 +1,67 @@
1
+ /**
2
+ * Deduplication for derived synthesis-artifact candidates (patterns /
3
+ * recommendations / doctrine).
4
+ *
5
+ * F2-INTEL-002 — the artifact analogue of `derive/dedupe.js`
6
+ * `dedupeAgainstExisting`. Re-running `findings <type> derive --write` produces
7
+ * the SAME deterministic ids as the prior run. At HEAD the writers would
8
+ * silently overwrite the on-disk file via `atomicWriteFileSync`, so an artifact
9
+ * the operator had already accepted / rejected / invalidated would be reset to a
10
+ * fresh `candidate` — erasing the operator's decision and re-breaking the loop
11
+ * that F2-INTEL-001 closed.
12
+ *
13
+ * Dedupe law (identical to the findings dedupe semantics):
14
+ * - id not on disk → write (new artifact)
15
+ * - id on disk, status candidate → write (refresh the machine output)
16
+ * - id on disk, NON-candidate status (reviewed/accepted/rejected/invalidated)
17
+ * → COLLISION, do NOT overwrite. The operator's
18
+ * decision is load-bearing and survives.
19
+ *
20
+ * The decision is intentionally the conservative one the findings layer already
21
+ * makes: preserve the operator's status by skipping the re-write entirely. The
22
+ * refreshed machine content is discarded for a promoted artifact rather than
23
+ * merged, because merging fresh `candidate`-shaped content into an `accepted`
24
+ * artifact would silently mutate what the operator signed off on.
25
+ */
26
+
27
+ /**
28
+ * @param {Array<object>} candidates - Freshly-derived artifact candidates.
29
+ * @param {Array<{ data: object }>} existing - Already-loaded on-disk artifacts
30
+ * (e.g. `loadPatternsWithSkips(rootDir).entries`).
31
+ * @param {string} idKey - The artifact's id field ('pattern_id' |
32
+ * 'recommendation_id' | 'doctrine_id').
33
+ * @returns {{ toWrite: Array<object>, skippedUnchanged: number, collisions: Array<{ id: string, existingStatus: string }> }}
34
+ */
35
+ export function dedupeArtifactsAgainstExisting(candidates, existing, idKey) {
36
+ const existingById = new Map();
37
+ for (const e of existing || []) {
38
+ const data = e?.data;
39
+ if (data && data[idKey]) existingById.set(data[idKey], data);
40
+ }
41
+
42
+ const toWrite = [];
43
+ let skippedUnchanged = 0;
44
+ const collisions = [];
45
+
46
+ for (const c of candidates) {
47
+ const id = c?.[idKey];
48
+ const prior = id != null ? existingById.get(id) : undefined;
49
+
50
+ if (!prior) {
51
+ toWrite.push(c);
52
+ continue;
53
+ }
54
+
55
+ // Same id exists — the operator's status is the gate.
56
+ if (prior.status !== 'candidate') {
57
+ // Promoted artifact: collision, never overwrite.
58
+ collisions.push({ id, existingStatus: prior.status });
59
+ continue;
60
+ }
61
+
62
+ // Still a candidate on disk — refresh the machine output.
63
+ toWrite.push(c);
64
+ }
65
+
66
+ return { toWrite, skippedUnchanged, collisions };
67
+ }
@@ -8,5 +8,8 @@ export { validatePattern, validateRecommendation, validateDoctrine } from './val
8
8
  export {
9
9
  writePattern, writeRecommendation, writeDoctrine,
10
10
  writePatterns, writeRecommendations, writeDoctrines,
11
- loadPatterns, loadRecommendations, loadDoctrines
11
+ loadPatterns, loadRecommendations, loadDoctrines,
12
+ loadPatternsWithSkips, loadRecommendationsWithSkips, loadDoctrinesWithSkips,
13
+ resetSeenArtifactWrites
12
14
  } from './write-artifacts.js';
15
+ export { dedupeArtifactsAgainstExisting } from './dedupe-artifacts.js';
@@ -8,13 +8,25 @@
8
8
  */
9
9
 
10
10
  import { loadFindings } from '../reader.js';
11
+ import { boundaryHash } from '../derive/ids.js';
11
12
 
12
13
  /**
13
14
  * Derive candidate patterns from accepted findings.
14
15
  *
16
+ * D2B-001 — return shape carries a `skipped: [{ path, error }]` field so the
17
+ * silent-loader signal that the recommendation/doctrine derive stages already
18
+ * surface now also propagates through pattern derivation. `loadFindings`
19
+ * (reader.js) returns a torn / unparseable finding YAML as
20
+ * `{ valid: false, errors: [{ message }] }`; a torn file that on disk WAS an
21
+ * accepted finding would otherwise be filtered out of clustering with zero
22
+ * operator signal, quietly shrinking the evidence base behind a pattern's
23
+ * strength or threshold. The `skipped` list documents that partial-completion
24
+ * honestly. Legacy callers reading only `patterns` / `stats` are unaffected —
25
+ * the field is additive.
26
+ *
15
27
  * @param {string} rootDir - dogfood-labs repo root
16
28
  * @param {{ includeFixtures?: boolean }} opts
17
- * @returns {{ patterns: Array, stats: { findingsConsidered: number, clustersFound: number, belowThreshold: number } }}
29
+ * @returns {{ patterns: Array, skipped: Array<{path: string, error: string}>, stats: { findingsConsidered: number, clustersFound: number, belowThreshold: number, findingsSkipped: number } }}
18
30
  */
19
31
  export function derivePatterns(rootDir, opts = {}) {
20
32
  // Load only accepted, non-invalidated findings
@@ -23,6 +35,15 @@ export function derivePatterns(rootDir, opts = {}) {
23
35
  allFindings.push(...loadFindings(rootDir, { fixtures: true, fixtureKind: 'valid' }));
24
36
  }
25
37
 
38
+ // D2B-001 — partition torn / unreadable findings into a structured skip list
39
+ // BEFORE the accepted filter drops them. A `valid:false` record is a finding
40
+ // YAML that failed parse (reader.js sets errors[0].message to the parse error)
41
+ // or schema validation; its on-disk status is unknowable, so a torn file that
42
+ // WAS an accepted finding must not vanish silently from clustering.
43
+ const skipped = allFindings
44
+ .filter(f => f.valid === false)
45
+ .map(f => ({ path: f.path, error: f.errors?.[0]?.message || 'unknown read/parse error' }));
46
+
26
47
  const accepted = allFindings.filter(f =>
27
48
  f.valid &&
28
49
  f.data?.status === 'accepted' &&
@@ -68,10 +89,12 @@ export function derivePatterns(rootDir, opts = {}) {
68
89
 
69
90
  return {
70
91
  patterns,
92
+ skipped, // D2B-001: structured skip list for operator legibility
71
93
  stats: {
72
94
  findingsConsidered: accepted.length,
73
95
  clustersFound: clusters.size,
74
- belowThreshold
96
+ belowThreshold,
97
+ findingsSkipped: skipped.length
75
98
  }
76
99
  };
77
100
  }
@@ -98,7 +121,7 @@ function isFalseRecurrence(findings) {
98
121
  /**
99
122
  * Build a pattern candidate from a cluster.
100
123
  */
101
- function buildPatternCandidate(cluster) {
124
+ export function buildPatternCandidate(cluster) {
102
125
  const { findings, issue_kind, root_cause_kind, remediation_kind } = cluster;
103
126
  const now = new Date().toISOString();
104
127
 
@@ -119,8 +142,16 @@ function buildPatternCandidate(cluster) {
119
142
  // otherwise two clusters that differ only by root_cause_kind collide on pattern_id and the
120
143
  // second writePattern() silently overwrites the first on disk. Surface is added for readability,
121
144
  // not for uniqueness.
145
+ //
146
+ // The human-readable slug runs `.replace(/_/g, '-')` AFTER concatenation, which folds the
147
+ // underscore inside a component into the same `-` that delimits components: issue_kind='a_b' +
148
+ // root_cause='c' and issue_kind='a' + root_cause='b_c' both flatten to `...-a-b-c`
149
+ // (findings-A-002). A trailing boundary hash over the UN-flattened cluster key keeps the two
150
+ // distinct, so legitimately different clusters no longer trip the L3-001 *_ID_COLLISION guard.
122
151
  const surfaceStr = surfaces.size === 1 ? [...surfaces][0] : 'multi-surface';
123
- const slug = `${surfaceStr}-${issue_kind}-${root_cause_kind}`.replace(/_/g, '-');
152
+ const readable = `${surfaceStr}-${issue_kind}-${root_cause_kind}`.replace(/_/g, '-');
153
+ const boundary = boundaryHash([surfaceStr, issue_kind, root_cause_kind]);
154
+ const slug = `${readable}-${boundary}`;
124
155
 
125
156
  // Determine strength
126
157
  const strength = repos.size >= 3 ? 'strong' : repos.size >= 2 ? 'emerging' : 'emerging';
@@ -59,6 +59,24 @@ export function resetSeenArtifactWrites(rootDir) {
59
59
  }
60
60
  }
61
61
 
62
+ /**
63
+ * Reset the collision memory for a SINGLE artifact id only.
64
+ *
65
+ * B-003 — `reviewArtifact` re-writes the one id it is promoting, which
66
+ * legitimately needs that id's guard cleared. Clearing the whole root (via
67
+ * `resetSeenArtifactWrites(rootDir)`) also disarms the silent-clobber guard for
68
+ * every OTHER kind/id — latent for a future batch caller reviewing several ids
69
+ * in one process. Scope the reset to `${rootDir}:${kind}:${id}` so only the id
70
+ * under review is cleared.
71
+ *
72
+ * @param {string} rootDir
73
+ * @param {'pattern'|'recommendation'|'doctrine'} kind
74
+ * @param {string} id
75
+ */
76
+ export function resetSeenArtifactWrite(rootDir, kind, id) {
77
+ seenArtifactWrites.delete(`${rootDir}:${kind}:${id}`);
78
+ }
79
+
62
80
  /**
63
81
  * Structured collision errors thrown by the singleton writers when called
64
82
  * twice in the same process with the same id. Mirror the batch helpers'