devmethod-ai 0.3.1 → 0.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. package/.agents/skills/devmethod-architecture/SKILL.md +14 -0
  2. package/.agents/skills/devmethod-correct-course/SKILL.md +12 -0
  3. package/.agents/skills/devmethod-design/SKILL.md +14 -0
  4. package/.agents/skills/devmethod-explore/SKILL.md +12 -0
  5. package/.agents/skills/devmethod-frame/SKILL.md +12 -0
  6. package/.agents/skills/devmethod-handoff/SKILL.md +14 -0
  7. package/.agents/skills/devmethod-implement/SKILL.md +14 -0
  8. package/.agents/skills/devmethod-integrate/SKILL.md +14 -0
  9. package/.agents/skills/devmethod-next/SKILL.md +14 -0
  10. package/.agents/skills/devmethod-plan/SKILL.md +14 -0
  11. package/.agents/skills/devmethod-ready/SKILL.md +14 -0
  12. package/.agents/skills/devmethod-review/SKILL.md +18 -0
  13. package/.agents/skills/devmethod-status/SKILL.md +12 -0
  14. package/.agents/skills/devmethod-verify/SKILL.md +14 -0
  15. package/.agents/skills/project-foundation/SKILL.md +1 -1
  16. package/.agents/skills/project-foundation/assets/START_HERE.md +3 -1
  17. package/.agents/skills/project-foundation/references/operating-commands.md +19 -15
  18. package/.agents/skills/scoped-delivery/assets/REVIEW.md +2 -2
  19. package/.agents/skills/scoped-delivery/references/review-format.md +26 -0
  20. package/.agents/skills/scoped-delivery/references/review-report.md +16 -0
  21. package/.agents/skills/scoped-delivery/references/review-workflow.md +10 -1
  22. package/.agents/skills/scoped-delivery/references/verification-and-cost.md +2 -0
  23. package/COMPATIBILITY.md +7 -3
  24. package/README.md +33 -27
  25. package/START_HERE.md +3 -1
  26. package/dist/cli.js +3 -1
  27. package/dist/commands.js +20 -0
  28. package/dist/doctor.js +7 -3
  29. package/dist/init.js +16 -2
  30. package/dist/review-agent.js +24 -0
  31. package/dist/review-runtime.js +9 -0
  32. package/docs/ADR-009-visible-workflow-commands.md +11 -0
  33. package/docs/ADR-010-installed-review-renderer.md +9 -0
  34. package/docs/COMMANDS-VALIDATION.md +13 -0
  35. package/docs/COMMANDS.md +36 -0
  36. package/docs/EVALUATION.md +4 -0
  37. package/docs/RELEASE-0.4.0.md +15 -0
  38. package/docs/RELEASE-0.4.1.md +13 -0
  39. package/docs/REVIEW-GUIDE.md +6 -0
  40. package/docs/REVIEWS.md +9 -23
  41. package/evaluation/review-detection/README.md +30 -0
  42. package/evaluation/review-detection/fixtures/compatibility.mjs +4 -0
  43. package/evaluation/review-detection/fixtures/consumer.mjs +3 -0
  44. package/evaluation/review-detection/fixtures/control.mjs +5 -0
  45. package/evaluation/review-detection/fixtures/sensitive.mjs +11 -0
  46. package/evaluation/review-detection/fixtures/submission.mjs +8 -0
  47. package/evaluation/review-detection/observed-0.4.1/README.md +16 -0
  48. package/evaluation/review-detection/observed-0.4.1/TASK.md +1 -0
  49. package/evaluation/review-detection/observed-0.4.1/adjudication.json +65 -0
  50. package/evaluation/review-detection/observed-0.4.1/case-a/sensitive.mjs +10 -0
  51. package/evaluation/review-detection/observed-0.4.1/case-b/submission.mjs +7 -0
  52. package/evaluation/review-detection/observed-0.4.1/case-c/compatibility.mjs +3 -0
  53. package/evaluation/review-detection/observed-0.4.1/case-c/consumer.mjs +2 -0
  54. package/evaluation/review-detection/observed-0.4.1/case-d/control.mjs +5 -0
  55. package/evaluation/review-detection/observed-0.4.1/input-hashes.json +7 -0
  56. package/evaluation/review-detection/observed-0.4.1/method-hashes.json +76 -0
  57. package/evaluation/review-detection/observed-0.4.1/probe-results.json +58 -0
  58. package/evaluation/review-detection/observed-0.4.1/raw-findings.md +75 -0
  59. package/evaluation/review-detection/observed-0.4.1/review-probes.mjs +41 -0
  60. package/evaluation/review-detection/observed-0.4.1/score.json +17 -0
  61. package/evaluation/review-detection/oracle.json +9 -0
  62. package/evaluation/review-detection/reproduce.test.mjs +53 -0
  63. package/evaluation/review-detection/score.mjs +26 -0
  64. package/package.json +1 -1
  65. package/scripts/package-smoke.mjs +13 -2
@@ -0,0 +1,76 @@
1
+ {
2
+ ".agents/skills/devmethod-status/SKILL.md": "ec9aa89614a0604f35b223064b3492d5a659a1726728d2f8eca5969ac984f73c",
3
+ ".agents/skills/devmethod-plan/SKILL.md": "52cc2a73da6b8bb210e7f2ee5f3e3965594036d721069a831e28793790ad2812",
4
+ ".agents/skills/react-feature-engineering/SKILL.md": "7766b7db86833cc63d16493b1d0b006a9a2c275ac513fcbbae6d3d2e4efafd95",
5
+ ".agents/skills/devmethod-architecture/SKILL.md": "db7e80b11863176455034a1ab03dba6fcf4784ea92f0bbeee5ba223fc991189b",
6
+ ".agents/skills/devmethod-verify/SKILL.md": "00549c3fb4cfd9b7d98d37128c5c6bb117605c286afdb36df5a03c038f072e30",
7
+ ".agents/skills/design-to-code/SKILL.md": "5747b5368cf6d7bdec982b6ea1d2337bf0beb96604866ff043838d6030184c1c",
8
+ ".agents/skills/project-foundation/SKILL.md": "e9cff76b8570e22dedc0b43af816d5455ddd3848c459d8190025b6248a01107c",
9
+ ".agents/skills/devmethod-integrate/SKILL.md": "f7a83a03b3e34d654341bc1638f3c66bda43732e5a8bebb6639d7df772cf770b",
10
+ ".agents/skills/reliable-ai-integration/SKILL.md": "cc43feeb4f4ba892248ce4fe9637c31637b6078eb35de7ebcc08aa918c6ff85f",
11
+ ".agents/skills/devmethod-handoff/SKILL.md": "1bb1a8874aae43d5cff3d57863d59d4d67dcf0fd320573aa746fa39bb94d5f02",
12
+ ".agents/skills/devmethod-design/SKILL.md": "a9c3fa2168e8962d838d9eba9f6659073249695645e6348c4594818cbfcec7a4",
13
+ ".agents/skills/devmethod-next/SKILL.md": "4ad95bb95e88278fd131755d019ee085215a62f592353a6076566cc81b481aeb",
14
+ ".agents/skills/scoped-delivery/SKILL.md": "557f9624a735917bfd0eb2971a32496c17fdf9d63f6c361f0d2405aaad297065",
15
+ ".agents/skills/devmethod-ready/SKILL.md": "8501aa3a54eaa3ebafa1ae68b0364350cc6cb0082fb23cb6c46b34d4499d4314",
16
+ ".agents/skills/devmethod-explore/SKILL.md": "61eefe4ea7ace88e8baa974d9aaa9129fa8ac91b60d9033b3434283bd74b46e5",
17
+ ".agents/skills/devmethod-correct-course/SKILL.md": "774b8bc371886f386489c9de398712d4e1d35045d0ad773a3913e4b67f74ac98",
18
+ ".agents/skills/devmethod-review/SKILL.md": "ef727903c1be13dbcbfc40ebbf6a862b748f10e13aaa8e8dfe9a5569169b078d",
19
+ ".agents/skills/devmethod-implement/SKILL.md": "a5f607c29959891acfb54aa4cb3a19ecf5d58bde10125dce6dbf7f3933a69bba",
20
+ ".agents/skills/decision-architecture/SKILL.md": "31c065dd526498e3a42b9182bac38196c25b7b98311808ae6416f98949f84b3d",
21
+ ".agents/skills/devmethod-frame/SKILL.md": "371b123419d01de049ff27637f73e4d85ef167847e4d16c6aa47b2d20d4a1a6b",
22
+ ".agents/skills/decision-architecture/references/backend-boundaries.md": "df6c66f6a12bb4749c0de5e346b7d8312ad8d5e05298b6e9283fdda6f401dab0",
23
+ ".agents/skills/decision-architecture/references/product-decisions.md": "5b656d98684f57adee4980eb6853b1d2a78a7b69943b5e8e14104702adc23be2",
24
+ ".agents/skills/decision-architecture/references/api-contracts.md": "085b90efc2c84718db87f4123f68e93b78c40e7fe4fe3d347e5e9840f9a6e5c3",
25
+ ".agents/skills/decision-architecture/assets/ADR.md": "4a2525af41cca1189412bd7c36b11a41cc0a5f95beccf9fd49149ab97aa3cc4c",
26
+ ".agents/skills/scoped-delivery/references/review-workflow.md": "c049418c2cb4c1e64f462a83471ae5d1a91def2a210aeb78c4d0addd70cacfae",
27
+ ".agents/skills/scoped-delivery/references/review-report.md": "d7f0260bf58b7b65e2105031aa90c66e80fd690f7e0ec826b7af87d426690630",
28
+ ".agents/skills/scoped-delivery/references/review-format.md": "19ac482c790b752e1bc3a7ca733fa34411a0b359371b7a7d1246af8c27e05f0e",
29
+ ".agents/skills/scoped-delivery/references/verification-and-cost.md": "d732e351c8408a4848fa4afa720fc2d261ecea4158e482b6d219c32ede6a3a7f",
30
+ ".agents/skills/scoped-delivery/scripts/review-cli.mjs": "b43cebaab81857bab3c538a68581d895a88cf022559999eeebe35a4b296c79b8",
31
+ ".agents/skills/scoped-delivery/scripts/review-model.mjs": "9e295ba34f0fd87f13ef40ab96ac784bd3896f74c3d02c5e79e05606089150b0",
32
+ ".agents/skills/scoped-delivery/scripts/records.mjs": "0e02f386c3e7fad694326c13a4c17f993c1d1e1f6703fe20c981cabbfce0a3de",
33
+ ".agents/skills/scoped-delivery/scripts/review-open.mjs": "aef4e126988b9a1927cd0f52426fee52cdc3878b73f149447ba51430339f19e4",
34
+ ".agents/skills/scoped-delivery/scripts/review-agent.mjs": "c8deabf384a2fb0ce30f678ed3ed9a2fc4f3584562a86c752133ff5b8fea9ff2",
35
+ ".agents/skills/scoped-delivery/scripts/filesystem.mjs": "c655f3189d63fe09291e3932a4619b314333c72748adc1551867b9112a127643",
36
+ ".agents/skills/scoped-delivery/scripts/review.mjs": "6a594e6fdd566b244a2118aa2d5125a805f69774e466f37dda8ea950dd6cc912",
37
+ ".agents/skills/scoped-delivery/scripts/review-browser.js": "0c7bb25cb58f14eee941cd0995b0d3a544c66ef55f6ef027e8deb71e76314692",
38
+ ".agents/skills/scoped-delivery/scripts/review-ui.css": "d428603fe454dbabb2fd4b8c0d5042f959e56ffc0046bce20c7413acadbf122e",
39
+ ".agents/skills/scoped-delivery/assets/MISSION.md": "3f5d16b7763cb9a50ac9c131eae038f1fd386d6dc013247e7e3d7deb3a7246e1",
40
+ ".agents/skills/scoped-delivery/assets/CHECKPOINT.md": "f74ebb2181c1944289270e389ea1df7ab7e3feb1aa975d6f89104e82ab73717b",
41
+ ".agents/skills/scoped-delivery/assets/VERIFICATION.md": "46569dc339f41be3dfb44ed89f1ec67140b1e491c8ba11c74c5ca6e6b455f6ca",
42
+ ".agents/skills/scoped-delivery/assets/SLICE.md": "8bf1c238904da20f5fb09c42f427356988b478b2230a4ad26cff5c6dd00c4ee8",
43
+ ".agents/skills/scoped-delivery/assets/REPRISE.md": "8e43cfc004bfd8a341063d02be4d656b9d3ca71abb7082ad75f00c2bf2d4ceb2",
44
+ ".agents/skills/scoped-delivery/assets/TICKET.md": "94befa7d148691d6094684715d950abf4f3ba617adeee0b80037ee9fc1cb8f6b",
45
+ ".agents/skills/scoped-delivery/assets/PLAN.md": "5f4c7f1f3e2918008da50d1865227e97f4e438633690b8bfe58bbe3e9211fe04",
46
+ ".agents/skills/scoped-delivery/assets/REVIEW.md": "414c7f578303048dc60c816190fa88463571a75f5e747d0d2948b830fb3e0df1",
47
+ ".agents/skills/reliable-ai-integration/references/jobs-and-costs.md": "32de8493813b84dc4b2540e4cd6a682494e0357dfb4bd94002e36c5e080ad699",
48
+ ".agents/skills/reliable-ai-integration/references/evidence-and-media.md": "67100fe297e40729a8acb02840ec1d6f3974e0f082b512cd203b76c4d45dcda0",
49
+ ".agents/skills/reliable-ai-integration/assets/AI_EVALUATION.md": "9fccf7fbadb38f10210d6813fde059678646f978e72bf9e79fb5f7b2366910c1",
50
+ ".agents/skills/project-foundation/references/mission-context.md": "9738b9959a08e827997a90219fe084be537bb0093acb31e5ab1688ff6e59d75c",
51
+ ".agents/skills/project-foundation/references/operating-commands.md": "03631d3adabd62e127e6ac6780e4a94e30442ee4595cb490549ca5cc77557a01",
52
+ ".agents/skills/project-foundation/references/work-sizing.md": "d2b10e269610eb6d8adf57426477e540e519cbae2d21a90a649e0abe0179509e",
53
+ ".agents/skills/project-foundation/references/delivery-planning.md": "5727abcdaca48a7db6d65d81e55a211df07db92f755f11cd714c92e67273892d",
54
+ ".agents/skills/project-foundation/references/exploration.md": "1e6dac589bd16819416d337a3307950215d36ae80155ee7966b57e3cc201b660",
55
+ ".agents/skills/project-foundation/assets/ENGINEERING_POLICY.template.md": "341d03b7ef60d6c02badebadc950a535e3db5a82c196b773a523af052db25a98",
56
+ ".agents/skills/project-foundation/assets/OPPORTUNITES.md": "ddcb9a24b70d26194795948a95566ddf68b33628ce8b5240b05b5f818d9a644c",
57
+ ".agents/skills/project-foundation/assets/CADRAGE.md": "292974445e586b97e4ecd228f4b7147b27da98620c7066af233599b56e60b1cf",
58
+ ".agents/skills/project-foundation/assets/EXISTANT.md": "7a20805fb2a0529e858e5f029441898d4c76f9344f8a5a4290c17d9f9dedf011",
59
+ ".agents/skills/project-foundation/assets/START_HERE.md": "924a91a603c26eda7126a265d1a3f55f2cca15e5952d0df2af71ed1556d4a0a4",
60
+ ".agents/skills/project-foundation/assets/REGLES.md": "cc6ee52e50380ac4c4c9a484b3fccad4f07b968bee073ad3e5f41f2c6180ab2b",
61
+ ".agents/skills/project-foundation/assets/PROJECT_PROFILE.md": "80dcc8e1d5d5282411981b0b4b38e5cb258cc39e730a7cf914de8cfd9cb703d0",
62
+ ".agents/skills/project-foundation/assets/AGENTS.foundation.md": "9564d30890ce7133e6b1fc9a6115e2217145276d7843069c447ff456d6eea0ea",
63
+ ".agents/skills/project-foundation/references/profiles/messaging.md": "56ac919ada8deef6cda92395f48cec50cdd5777b3c5f246356195251141de61b",
64
+ ".agents/skills/project-foundation/references/profiles/react-next.md": "8819abb3f7a22fb87e26f6a5465e77aeec5505f5c6dd9462edafbee1918237c9",
65
+ ".agents/skills/project-foundation/references/profiles/cloud-delivery.md": "d64452e2e9d1e8e48a9ca4c8c14974583376cf6a1def241e0ad7843eaf02dab7",
66
+ ".agents/skills/project-foundation/references/profiles/node-nest.md": "32d3f59086dbc132f1b1c01e87bdd52fa126b01c29a1137ba99ca3a0eb5232f1",
67
+ ".agents/skills/project-foundation/references/profiles/README.md": "a05eaaeef6bf25aca4bbd1e65e79995cab8f5419c275cd4d37beeff547be469c",
68
+ ".agents/skills/project-foundation/references/profiles/postgres-drizzle.md": "115e801dc009dee06c734167e3c0c3ef1b8f240009c8f2fb3e88656a8116d24a",
69
+ ".agents/skills/project-foundation/references/profiles/ci.md": "ca7ab4034c8dae85853069f37d9fc90d7d9a8c5511267df1cf51c9ec035865b9",
70
+ ".agents/skills/project-foundation/references/profiles/typescript.md": "c6452bc3f6397eb975267ccb4909e566655513d974e91c6207a7a1b858d41147",
71
+ ".agents/skills/project-foundation/references/profiles/mongodb.md": "9ec39ab8209425d714a1d17635aaa21f18995d9fb52123754e5bac87107a9a82",
72
+ ".agents/skills/design-to-code/references/visual-creation.md": "cd557e06c3020e24d3e57dc1f862f2a1d5d16c495078fe8b3d8d655e6573806e",
73
+ ".agents/skills/design-to-code/references/ux-contract.md": "7769a8830a8b83358d6274b46a0d2f44a54cce7fc3aaef38b9890f2eec0174b1",
74
+ ".agents/skills/design-to-code/assets/UI_ACCEPTANCE.md": "fd71bcb4a9f86e36b5bc91ba5a3b4310c7c109cd3bf36cda85b6b0c0e701b817",
75
+ ".agents/skills/react-feature-engineering/references/review-and-sources.md": "43a0829aa20a95bb46c09ab5f2ae7d11965abfe54dfaf48fd1e59798f8283ad3"
76
+ }
@@ -0,0 +1,58 @@
1
+ {
2
+ "a": {
3
+ "log": {
4
+ "tokenExposed": true,
5
+ "emailExposed": true,
6
+ "providerMessageSecretExposed": true
7
+ },
8
+ "telemetry": {
9
+ "tokenExposed": true,
10
+ "emailExposed": true,
11
+ "providerMessageSecretExposed": true
12
+ },
13
+ "response": {
14
+ "tokenExposed": true,
15
+ "emailExposed": true,
16
+ "providerMessageSecretExposed": true
17
+ }
18
+ },
19
+ "bConcurrent": {
20
+ "charges": 2,
21
+ "notices": 2,
22
+ "receipts": [
23
+ {
24
+ "key": "same-key",
25
+ "sequence": 1
26
+ },
27
+ {
28
+ "key": "same-key",
29
+ "sequence": 2
30
+ }
31
+ ],
32
+ "stateSize": 1
33
+ },
34
+ "bRetry": {
35
+ "charges": 2,
36
+ "sizeAfterFailure": 0
37
+ },
38
+ "c": {
39
+ "existingStoredInput": {
40
+ "id": "order-fixture",
41
+ "amountCents": 1250
42
+ },
43
+ "oldResponse": {
44
+ "id": "order-fixture"
45
+ },
46
+ "newResponse": {
47
+ "id": "new-order",
48
+ "totalCents": 1250
49
+ },
50
+ "oldConsumerNaN": true,
51
+ "newConsumerNaN": true,
52
+ "baselineTotal": 12.5
53
+ },
54
+ "d": {
55
+ "stringExamplesPassed": 3,
56
+ "nonStringsRejected": 8
57
+ }
58
+ }
@@ -0,0 +1,75 @@
1
+ # Independent review findings
2
+
3
+ Reviewed only the supplied case files under `/private/tmp/devmethod-041-detection-run`, following TASK.md and the installed devmethod-review, project-foundation routing/stage contract, scoped-delivery and review-workflow references. Node runtime: v23.10.0. No case files changed. No external calls, browser opening, fixes, or other workflow stages executed. PROJECT_PROFILE.md is an unfilled template and adds no accepted requirements. No manifest, lockfile, CONTRIBUTING.md or AGENTS.md was found in the bounded discovery.
4
+
5
+ ## Confirmed defects
6
+
7
+ ### DET-A-001 — Credential and personal data escape through the failure detail
8
+
9
+ - Case/location: case-a, `case-a/sensitive.mjs:5-8` (construction at line 5; sinks at lines 6–8).
10
+ - Classification: confirmed defect; severity P1 (credential exposure); confidence high; status open.
11
+ - Trigger: provider rejects after receiving a token, including a rejection whose message contains credential material.
12
+ - Expected/observed: failure reporting should avoid propagating raw credentials and unnecessary personal data. Instead, the raw token, email, and unrestricted provider message are copied into log, telemetry and returned response body.
13
+ - Evidence: `review-probes.mjs` invokes the real function with synthetic sentinel credentials and an independently marked provider error. `probe-results.json:a` records tokenExposed, emailExposed and providerMessageSecretExposed as true for every sink. These booleans are computed against original captured application output before any report sanitization.
14
+ - Concrete impact: any consumer of logs, telemetry or the error response receives the original credential. Provider messages can add further sensitive content, so removing only the explicit token field would leave another exposure path. Actual recipients/retention are not supplied.
15
+ - Proposed correction: construct an allowlisted public error response and separately sanitized diagnostic event; omit token and unnecessary email, and redact or replace untrusted provider messages before sinks. Preserve a safe correlation/error code for diagnosis. Trade-off: less raw diagnostic detail requires deliberate safe diagnostics.
16
+ - Resolution verification: repeat the three-sink sentinel check, covering explicit fields and provider messages; all exposure flags should be false while failure reporting remains usable.
17
+
18
+ ### DET-B-001 — Simultaneous submissions pass the same deduplication check
19
+
20
+ - Case/location: case-b, `case-b/submission.mjs:2-5`, especially the awaited charge at line 3 before reservation.
21
+ - Classification: confirmed defect; severity P1 (duplicate charge invocation); confidence high; status open.
22
+ - Trigger: two calls submit the same absent key before either awaited charge completes.
23
+ - Expected/observed: the same-key state check implies duplicate submissions should reuse one result. Both calls observe absence and independently invoke charge and notify; only their final receipts are written to one map slot.
24
+ - Evidence: a deferred Promise in `review-probes.mjs` deliberately holds the first charge, starts a second submit with the same key, then releases both. `probe-results.json:bConcurrent` shows 2 charges, 2 notices, distinct sequence receipts, and stateSize 1.
25
+ - Concrete impact: the function dispatches duplicate charge requests and notifications while its state retains only one receipt. Actual duplicate billing depends on whether an unseen provider independently deduplicates the supplied key.
26
+ - Proposed correction: atomically reserve/share the pending same-key operation before charge. For multiple processes use a durable uniqueness/reservation mechanism, plus provider idempotency with the same operation identity. Define failure recovery so an ambiguous effect is not blindly reissued. Trade-off: reservation lifetime and recovery require explicit handling.
27
+ - Resolution verification: repeat the controlled interleaving and assert exactly one charge and consistent returned receipt for both callers.
28
+
29
+ ### DET-B-002 — Notification failure discards knowledge of an already successful charge
30
+
31
+ - Case/location: case-b, `case-b/submission.mjs:4-5`.
32
+ - Classification: confirmed defect; severity P1 (unsafe retry after partial success); confidence high; status open.
33
+ - Trigger: charge succeeds but notify rejects, followed by retry using the same key and state.
34
+ - Expected/observed: the successful financial effect should remain recorded so retry can recover notification without recharging. Because state is updated only after notify, the first call rejects with no saved receipt and the retry invokes charge again.
35
+ - Evidence: `probe-results.json:bRetry` reports sizeAfterFailure 0 and charges 2 after an injected notification exception and one retry.
36
+ - Concrete impact: an ordinary notification outage causes duplicate charge requests and loses the recovery reference to the first successful receipt. Actual duplicate billing remains conditional on provider behavior, but the missing local record and repeated invocation are reproduced.
37
+ - Proposed correction: durably record charged status/receipt at the successful charge boundary, track notification status independently and retry that step. Combine with DET-B-001's reservation and provider reconciliation for charge timeouts. Simply caching before notify must not silently suppress the outstanding notification forever. Trade-off: additional state/status handling.
38
+ - Resolution verification: inject notify failure after a successful charge; retry should preserve the first receipt, charge exactly once, and finish notification recovery.
39
+
40
+ ### DET-C-001 — Response rename breaks the existing consumer and persisted order shape
41
+
42
+ - Case/location: case-c, changed `case-c/compatibility.mjs:2`; existing context `case-c/consumer.mjs:1-2`.
43
+ - Classification: confirmed compatibility defect; severity P1 (existing monetary calculation unavailable); confidence high; status open.
44
+ - Trigger: pass existingOrder through orderResponse then invoiceTotal; also reproduced with a new totalCents-shaped order.
45
+ - Expected/observed: existing consumer expects amountCents and the supplied existing record stores amountCents. The changed response reads and exposes only totalCents. Existing records lose the amount entirely (undefined; omitted by JSON serialization), and the consumer returns NaN for both old and new response shapes.
46
+ - Evidence: `probe-results.json:c` shows baselineTotal 12.5 for the unchanged consumer and fixture; oldResponse contains only id when serialized; newResponse contains totalCents 1250; oldConsumerNaN and newConsumerNaN are both true.
47
+ - Concrete impact: existing invoice totals become NaN rather than 12.50, and serialized responses for stored records no longer carry the monetary value. Only compatibility.mjs is proposed change, so a presumed coordinated consumer/data migration cannot justify the break.
48
+ - Proposed correction: preserve amountCents for existing consumers and read existing stored amountCents (optionally add totalCents and a carefully defined fallback if new records must be supported). Alternatively perform an explicitly coordinated/versioned migration of data and consumers, outside the supplied change. Trade-off: a compatibility alias temporarily maintains two names.
49
+ - Resolution verification: test unchanged invoiceTotal(orderResponse(existingOrder)) equals 12.5; include new-format records only if that additional format is an accepted requirement.
50
+
51
+ ## Case-d: no confirmed defect
52
+
53
+ `case-d/control.mjs:1` expressly requires trimming strings and rejecting non-strings, with no layering mandate. Lines 2–4 satisfy that contract. Executed checks cover surrounding whitespace, whitespace-only input, already trimmed text and eight non-string values (including a boxed String); all passed. An extra architectural layer or rejecting an empty trimmed result would introduce a preference/new requirement, not correct an observed defect.
54
+
55
+ ## Risks and limits, separate from confirmed defects
56
+
57
+ - Case-a: nested SDK error fields other than message are not serialized by this supplied function and are not a separate demonstrated leak. Real provider errors, actual sink redaction, access controls and retention were not available. Verify integration sinks and representative safe synthetic SDK errors to assess the full deployed exposure; no claim of actual credential compromise is made.
58
+ - Case-b: provider-side idempotency, cross-process state consistency, restart durability, and charge-success-then-timeout ambiguity are unknown. The supplied key is passed to charge and might enable downstream deduplication, but no such implementation is supplied. Verify with an idempotency-aware provider stub/integration and fault injection before asserting actual duplicate billing. Notification retries might also duplicate delivered notifications if failure occurs after delivery; not exercised here.
59
+ - Case-c: no migration policy or alternative consumer exists in the supplied scope. Findings rely on the explicitly supplied existing consumer/record, not inferred external clients.
60
+ - Case-d: no length, emptiness, Unicode normalization or layering requirements are given; none are imposed.
61
+ - No external documentation consulted because the assignment forbids external calls. These conclusions rely on small local ECMAScript modules and executed behavior, not version-dependent third-party APIs. No UI exists in scope. No repository suite or CI gate is provided. This is bounded evidence, not production verification.
62
+
63
+ ## Executed checks
64
+
65
+ 1. Read TASK.md, all five source modules with numbered lines, relevant installed review references, and local profile; bounded file discovery for rules/manifests/lockfiles. Manual source inspection covers every supplied case source line and the case-c interaction.
66
+ 2. Read runtime with `node --version`: v23.10.0.
67
+ 3. Ran `node review-probes.mjs > probe-results.json`: exit 0. Assertions confirm reproduced defects, so exit 0 means the reproductions behaved as documented, not that case-a/b/c passed correctness checks.
68
+ 4. Case-a: three captured failure sinks, direct credential/email and provider-message sentinel checks; confirmed exposure in all sinks.
69
+ 5. Case-b: deterministic simultaneous same-key submission and a separate successful-charge/failed-notification/retry sequence; both demonstrate repeated charge calls.
70
+ 6. Case-c: unchanged consumer baseline and old/new stored-shape response paths; baseline 12.5, both transformed paths NaN.
71
+ 7. Case-d: 3 string examples and 8 non-string rejection checks passed.
72
+
73
+ Artifacts: `review-probes.mjs` (synthetic executable reproduction), `probe-results.json` (captured results), and this requested `raw-findings.md`. No report UI was generated/opened. All four cases were evaluated independently; no external answer material was inspected.
74
+
75
+ Recommended next command (not executed): `devmethod-implement` for the confirmed corrections, after authorizing that scope.
@@ -0,0 +1,41 @@
1
+ import assert from 'node:assert/strict';
2
+ import { connectAccount } from './case-a/sensitive.mjs';
3
+ import { submit } from './case-b/submission.mjs';
4
+ import { orderResponse } from './case-c/compatibility.mjs';
5
+ import { invoiceTotal, existingOrder } from './case-c/consumer.mjs';
6
+ import { normalizeLabel } from './case-d/control.mjs';
7
+ const evidence = {};
8
+ const token = 'SYNTHETIC_TOKEN_REVIEW_041';
9
+ const email = 'synthetic@example.invalid';
10
+ const nested = 'SYNTHETIC_ERROR_SECRET_041';
11
+ let log, telemetry;
12
+ const result = await connectAccount({ token, email }, async () => { throw new Error(`Provider rejected credential ${nested}`); }, { log: x => { log = x; }, telemetry: x => { telemetry = x; } });
13
+ evidence.a = Object.fromEntries(Object.entries({log,telemetry,response:result}).map(([sink,value]) => [sink, { tokenExposed: JSON.stringify(value).includes(token), emailExposed: JSON.stringify(value).includes(email), providerMessageSecretExposed: JSON.stringify(value).includes(nested) }]));
14
+ for (const sink of Object.values(evidence.a)) assert.deepEqual(sink,{tokenExposed:true,emailExposed:true,providerMessageSecretExposed:true});
15
+ let release;
16
+ const barrier = new Promise(resolve => { release = resolve; });
17
+ let charges = 0, notices = 0;
18
+ const state = new Map();
19
+ const charge = async key => { const sequence = ++charges; await barrier; return { key, sequence }; };
20
+ const first = submit('same-key',state,charge,async()=>{notices++;});
21
+ const second = submit('same-key',state,charge,async()=>{notices++;});
22
+ release();
23
+ const receipts = await Promise.all([first,second]);
24
+ evidence.bConcurrent = {charges,notices,receipts,stateSize:state.size};
25
+ assert.equal(charges,2);
26
+ let retriesCharged = 0;
27
+ const retryState = new Map();
28
+ try { await submit('retry-key',retryState,async()=>({sequence:++retriesCharged}),async()=>{throw new Error('notification unavailable');}); } catch {}
29
+ const sizeAfterFailure = retryState.size;
30
+ await submit('retry-key',retryState,async()=>({sequence:++retriesCharged}),async()=>{});
31
+ evidence.bRetry = {charges:retriesCharged,sizeAfterFailure};
32
+ assert.equal(retriesCharged,2); assert.equal(sizeAfterFailure,0);
33
+ const oldResponse = orderResponse(existingOrder);
34
+ const newResponse = orderResponse({id:'new-order',totalCents:1250});
35
+ evidence.c = {existingStoredInput:existingOrder,oldResponse,newResponse,oldConsumerNaN:Number.isNaN(invoiceTotal(oldResponse)),newConsumerNaN:Number.isNaN(invoiceTotal(newResponse)),baselineTotal:invoiceTotal(existingOrder)};
36
+ assert.equal(invoiceTotal(existingOrder),12.5);
37
+ assert.ok(Number.isNaN(invoiceTotal(oldResponse)));assert.ok(Number.isNaN(invoiceTotal(newResponse)));
38
+ assert.equal(normalizeLabel(' \t hello \n '),'hello');assert.equal(normalizeLabel(' '),'');assert.equal(normalizeLabel('already trimmed'),'already trimmed');
39
+ for (const value of [null,undefined,1,true,{},[],new String('text'),Symbol('label')]) assert.throws(()=>normalizeLabel(value),TypeError);
40
+ evidence.d = {stringExamplesPassed:3,nonStringsRejected:8};
41
+ console.log(JSON.stringify(evidence,null,2));
@@ -0,0 +1,17 @@
1
+ {
2
+ "expected": 5,
3
+ "detected": [
4
+ "LEAK",
5
+ "RACE",
6
+ "PARTIAL",
7
+ "CONTRACT"
8
+ ],
9
+ "missed": [
10
+ "TIMEOUT"
11
+ ],
12
+ "falsePositives": [],
13
+ "unresolved": [],
14
+ "unadjudicated": [],
15
+ "recall": 0.8,
16
+ "limitation": "Separate evidence-based adjudication required; fixture/score tests alone do not measure model detection."
17
+ }
@@ -0,0 +1,9 @@
1
+ {
2
+ "format": 1,
3
+ "cases": [
4
+ {"id":"SENSITIVE","files":["sensitive.mjs"],"defects":[{"id":"LEAK","expected":"Synthetic token and email reach log, telemetry and API response on provider failure; error text is forwarded too."}]},
5
+ {"id":"SUBMISSION","files":["submission.mjs"],"defects":[{"id":"RACE","expected":"Two concurrent calls with the same key can both charge before state is recorded."},{"id":"PARTIAL","expected":"Notification failure after a successful charge leaves no receipt in state; retry charges again."},{"id":"TIMEOUT","expected":"A charge that applies its effect then times out leaves no receipt; retry can duplicate the external effect."}]},
6
+ {"id":"COMPATIBILITY","files":["compatibility.mjs","consumer.mjs"],"defects":[{"id":"CONTRACT","expected":"Renamed field loses existing amountCents data and breaks the unchanged invoice consumer."}]},
7
+ {"id":"CONTROL","files":["control.mjs"],"defects":[]}
8
+ ]
9
+ }
@@ -0,0 +1,53 @@
1
+ import test from 'node:test';
2
+ import assert from 'node:assert/strict';
3
+ import { connectAccount } from './fixtures/sensitive.mjs';
4
+ import { submit } from './fixtures/submission.mjs';
5
+ import { orderResponse } from './fixtures/compatibility.mjs';
6
+ import { invoiceTotal, existingOrder } from './fixtures/consumer.mjs';
7
+ import { normalizeLabel } from './fixtures/control.mjs';
8
+
9
+ test('LEAK: application outputs expose synthetic values before any report masking', async () => {
10
+ const token = 'SYNTHETIC-ONLY-TOKEN', email = 'fixture@example.invalid';
11
+ const logs = [], telemetry = [];
12
+ const result = await connectAccount({ token, email }, async () => { throw new Error(`Provider rejected ${token}`); }, {
13
+ log: x => logs.push(x), telemetry: x => telemetry.push(x),
14
+ });
15
+ for (const sink of [logs, telemetry, result.body]) {
16
+ assert.ok(JSON.stringify(sink).includes(token));
17
+ assert.ok(JSON.stringify(sink).includes(email));
18
+ }
19
+ assert.ok(result.body.message.includes(token));
20
+ });
21
+ test('RACE: controlled double submission duplicates the external effect', async () => {
22
+ let release; const gate = new Promise(resolve => { release = resolve; });
23
+ let charges = 0; const state = new Map();
24
+ const charge = async () => { charges++; await gate; return 'receipt'; };
25
+ const first = submit('same', state, charge, async () => {});
26
+ const second = submit('same', state, charge, async () => {});
27
+ release(); await Promise.all([first, second]);
28
+ assert.equal(charges, 2);
29
+ });
30
+ test('PARTIAL: retry after notification failure duplicates a successful charge', async () => {
31
+ const state = new Map(); let charges = 0;
32
+ const charge = async () => `receipt-${++charges}`;
33
+ await assert.rejects(submit('same', state, charge, async () => { throw new Error('interrupted'); }));
34
+ assert.equal(state.has('same'), false);
35
+ await submit('same', state, charge, async () => {});
36
+ assert.equal(charges, 2);
37
+ });
38
+ test('TIMEOUT: applied external effect can survive a timeout and be retried', async () => {
39
+ const state = new Map(); let effects = 0;
40
+ const charge = async () => { effects++; if (effects === 1) throw new Error('timeout after commit'); return 'receipt'; };
41
+ await assert.rejects(submit('same', state, charge, async () => {}));
42
+ await submit('same', state, charge, async () => {});
43
+ assert.equal(effects, 2);
44
+ });
45
+ test('CONTRACT: existing data and unchanged consumer break outside the diff', () => {
46
+ const response = orderResponse(existingOrder);
47
+ assert.equal(response.totalCents, undefined);
48
+ assert.ok(Number.isNaN(invoiceTotal(response)));
49
+ });
50
+ test('CONTROL: accepted simple implementation meets its contract without extra layers', () => {
51
+ assert.equal(normalizeLabel(' ready '), 'ready');
52
+ assert.throws(() => normalizeLabel(null), TypeError);
53
+ });
@@ -0,0 +1,26 @@
1
+ /** Score separately adjudicated findings, never infer a match from keywords or JSON validity. */
2
+ export function scoreDetection(expectedIds, findings, adjudications) {
3
+ const expected = new Set(expectedIds);
4
+ const ids = new Set(findings.map(f => f.id));
5
+ if (expected.size !== expectedIds.length || ids.size !== findings.length) throw new Error('Duplicate IDs');
6
+ const judged = new Set(), detected = new Set(), falsePositives = [], unresolved = [];
7
+ for (const item of adjudications) {
8
+ if (!ids.has(item.findingId) || judged.has(item.findingId)) throw new Error('Unknown or duplicate finding adjudication');
9
+ judged.add(item.findingId);
10
+ if (item.verdict === 'matched') {
11
+ if (!Array.isArray(item.defectIds) || !item.defectIds.length || item.defectIds.some(id => !expected.has(id))) throw new Error('Unknown or empty defect match');
12
+ const finding = findings.find(f => f.id === item.findingId);
13
+ if (finding.confidence !== 'confirmed') throw new Error('An unconfirmed risk is not a confirmed detection');
14
+ if (!item.evidenceReviewed || !item.impactJustified) throw new Error('Match requires reviewed evidence and justified impact');
15
+ item.defectIds.forEach(id => detected.add(id));
16
+ } else if (item.verdict === 'false-positive') falsePositives.push(item.findingId);
17
+ else if (item.verdict === 'unresolved') unresolved.push(item.findingId);
18
+ else throw new Error('Unknown adjudication verdict');
19
+ }
20
+ return {
21
+ expected: expected.size, detected: [...detected], missed: expectedIds.filter(id => !detected.has(id)),
22
+ falsePositives, unresolved, unadjudicated: findings.filter(f => !judged.has(f.id)).map(f => f.id),
23
+ recall: expected.size ? detected.size / expected.size : null,
24
+ limitation: 'Separate evidence-based adjudication required; fixture/score tests alone do not measure model detection.',
25
+ };
26
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "devmethod-ai",
3
- "version": "0.3.1",
3
+ "version": "0.4.1",
4
4
  "description": "From idea to delivery with your AI coding agents",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -15,7 +15,7 @@ try {
15
15
  run('tar', ['-xzf', '-'], root, 0, fs.readFileSync(path.resolve(archive)));
16
16
  const pkg = path.join(root, 'package'); const cli = path.join(pkg, 'dist/cli.js');
17
17
  const call = (args, expected = 0) => run(process.execPath, [cli, ...args], root, expected);
18
- assert.equal(JSON.parse(fs.readFileSync(path.join(pkg, 'package.json'))).version, '0.3.1');
18
+ assert.equal(JSON.parse(fs.readFileSync(path.join(pkg, 'package.json'))).version, '0.4.1');
19
19
  run(process.execPath, ['scripts/check-docs.mjs'], pkg);
20
20
  assert.match(call(['--help']), /Markdown PLAN\/tickets and legacy missions/);
21
21
  for (const resource of ['project-foundation/references/exploration.md', 'project-foundation/references/delivery-planning.md', 'project-foundation/assets/EXISTANT.md', 'project-foundation/assets/OPPORTUNITES.md', 'project-foundation/assets/CADRAGE.md', 'project-foundation/assets/REGLES.md', 'scoped-delivery/assets/PLAN.md', 'scoped-delivery/assets/TICKET.md', 'scoped-delivery/assets/REPRISE.md', 'scoped-delivery/assets/MISSION.md', 'scoped-delivery/assets/REVIEW.md', 'scoped-delivery/references/review-workflow.md']) {
@@ -37,6 +37,17 @@ try {
37
37
  const project = path.join(root, host);
38
38
  call(['init', '--tool', host, '--dest', project]);
39
39
  assert.equal(JSON.parse(call(['doctor', '--dest', project, '--json'])).status, 'ok');
40
+ const skillRoot = { codex: '.agents/skills', claude: '.claude/skills', cursor: '.cursor/skills' }[host];
41
+ const entries = fs.readdirSync(path.join(project, skillRoot)).filter(name => name.startsWith('devmethod-'));
42
+ assert.equal(entries.length, 14);
43
+ assert.ok(entries.includes('devmethod-review'));
44
+ for (const name of entries) assert.ok(fs.statSync(path.join(project, skillRoot, name, 'SKILL.md')).size > 0);
45
+ assert.ok(fs.statSync(path.join(project, skillRoot, 'scoped-delivery/references/review-format.md')).size > 0);
46
+ fs.copyFileSync(path.join(pkg, 'examples/review/review.json'), path.join(project, 'review-fixture.json'));
47
+ const reportRun = JSON.parse(run(process.execPath, [path.join(project, skillRoot, 'scoped-delivery/scripts/review-agent.mjs'), '--dest', project, '--review', 'review-fixture.json', '--output', 'actual-report.html', '--markdown', 'actual-report.md']));
48
+ assert.equal(reportRun.status, 'corrections');
49
+ assert.ok(fs.readFileSync(path.join(project, 'actual-report.html'), 'utf8').includes('Content-Security-Policy'));
50
+ assert.ok(fs.statSync(path.join(project, 'actual-report.md')).size > 0);
40
51
  const profile = path.join(project, 'PROJECT_PROFILE.md'); fs.appendFileSync(profile, '\nFictional local customization.\n');
41
52
  const before = fs.readFileSync(profile);
42
53
  const preview = JSON.parse(call(['update-preview', '--dest', project, '--json']));
@@ -75,5 +86,5 @@ try {
75
86
  assert.deepEqual([fs.readFileSync(customized), fs.readFileSync(profile)], before);
76
87
  console.log(`Actual legacy tarball: ${upstreamChanged ? 'local/upstream conflict' : 'unchanged upstream with local customization'} detected; filled profile/custom skill preserved.`);
77
88
  }
78
- console.log('Packed 0.3.1: three host installs, subset, customization preservation, mission/context/staleness/planning and documentation links passed. No native host execution.');
89
+ console.log('Packed 0.4.1: three host installs, subset, customization preservation, mission/context/staleness/planning and documentation links passed. No native host execution.');
79
90
  } finally { fs.rmSync(root, { recursive: true, force: true }); }