@dzhechkov/skills-idea2prd 0.1.0 → 0.1.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3,7 +3,7 @@
3
3
  GOAP Research Planner with Ed25519 Verification
4
4
 
5
5
  Enhanced Goal-Oriented Action Planning for research tasks with
6
- cryptographic verification support for anti-hallucination protection.
6
+ cryptographic provenance and tamper-evidence support.
7
7
 
8
8
  Features:
9
9
  - A* search for optimal research paths
@@ -383,8 +383,8 @@ def find_research_plan(
383
383
  if goal_state.issubset(current_state):
384
384
  # Check if verification requirements are met
385
385
  if require_verification and current.unsigned_claims > 0:
386
- # In strict/paranoid mode, penalize but don't reject
387
- pass
386
+ # In strict/paranoid mode, unsigned or invalid claims are rejected.
387
+ continue
388
388
 
389
389
  # Reconstruct state progression
390
390
  states = [initial_state]
@@ -396,7 +396,7 @@ def find_research_plan(
396
396
 
397
397
  # Calculate confidence based on verification
398
398
  base_confidence = 1.0 - (current.unsigned_claims * 0.1)
399
- estimated_confidence = max(0.5, min(1.0, base_confidence))
399
+ estimated_confidence = min(1.0, base_confidence)
400
400
 
401
401
  return ResearchPlan(
402
402
  actions=current.actions,
@@ -0,0 +1,112 @@
1
+ #!/usr/bin/env python3
2
+ """Load-bearing security tests for ed25519_verifier.py."""
3
+
4
+ import copy
5
+ import unittest
6
+
7
+ import ed25519_verifier as ev
8
+
9
+
10
+ @unittest.skipIf(ev.CRYPTO_BACKEND is None, "No Ed25519 backend installed")
11
+ class Ed25519VerifierSecurityTests(unittest.TestCase):
12
+ def make_pinned_pair(self):
13
+ signer = ev.Ed25519Verifier(auto_generate_keypair=True)
14
+ verifier = ev.Ed25519Verifier(
15
+ trusted_issuers={
16
+ "nature.com": {
17
+ "pubkey_b64": signer.get_public_key_b64(),
18
+ "status": "active",
19
+ }
20
+ }
21
+ )
22
+ return signer, verifier
23
+
24
+ def test_attacker_self_signed_trusted_string_rejected(self):
25
+ _, verifier = self.make_pinned_pair()
26
+ attacker = ev.Ed25519Verifier(auto_generate_keypair=True)
27
+
28
+ fact = attacker.create_issuer_signed_fact(
29
+ claim="Fabricated result was published",
30
+ source_url="https://nature.com/articles/example",
31
+ source_content="attacker-controlled content",
32
+ issuer="nature.com",
33
+ )
34
+
35
+ result = verifier.verify_fact(fact)
36
+
37
+ self.assertFalse(result.verified)
38
+ self.assertEqual(result.confidence, 0.0)
39
+ self.assertNotEqual(result.confidence, 0.95)
40
+
41
+ def test_fact_signed_by_pinned_trusted_key_verifies(self):
42
+ signer, verifier = self.make_pinned_pair()
43
+
44
+ fact = signer.create_issuer_signed_fact(
45
+ claim="Pinned-key fact",
46
+ source_url="https://nature.com/articles/example",
47
+ source_content="source content",
48
+ issuer="nature.com",
49
+ )
50
+
51
+ result = verifier.verify_fact(fact)
52
+
53
+ self.assertTrue(result.verified)
54
+ self.assertEqual(result.trust_class, ev.TRUST_CLASS_ISSUER_SIGNED)
55
+ self.assertEqual(result.confidence, 0.95)
56
+
57
+ def test_relabelled_or_moved_fact_fails(self):
58
+ signer, verifier = self.make_pinned_pair()
59
+ fact = signer.create_issuer_signed_fact(
60
+ claim="Pinned-key fact",
61
+ source_url="https://nature.com/articles/example",
62
+ source_content="source content",
63
+ issuer="nature.com",
64
+ )
65
+
66
+ relabelled = copy.deepcopy(fact)
67
+ relabelled.issuer = "science.org"
68
+ relabelled_result = verifier.verify_fact(relabelled)
69
+
70
+ moved = copy.deepcopy(fact)
71
+ moved.source_url = "https://nature.com/articles/other"
72
+ moved_result = verifier.verify_fact(moved)
73
+
74
+ self.assertFalse(relabelled_result.verified)
75
+ self.assertEqual(relabelled_result.confidence, 0.0)
76
+ self.assertFalse(moved_result.verified)
77
+ self.assertEqual(moved_result.confidence, 0.0)
78
+
79
+ def test_reordered_citation_chain_fails(self):
80
+ signer, verifier = self.make_pinned_pair()
81
+ chain = ev.CitationChain(chain_id="test-chain")
82
+
83
+ for index in range(3):
84
+ chain.add_fact(
85
+ signer.create_issuer_signed_fact(
86
+ claim=f"Claim {index}",
87
+ source_url=f"https://nature.com/articles/{index}",
88
+ source_content=f"source content {index}",
89
+ issuer="nature.com",
90
+ )
91
+ )
92
+
93
+ signer.sign_chain(chain)
94
+ ok, _, error = verifier.verify_citation_chain(chain, signer.get_public_key_b64())
95
+ self.assertTrue(ok, error)
96
+
97
+ reordered = ev.CitationChain(
98
+ chain_id=chain.chain_id,
99
+ facts=[chain.facts[1], chain.facts[0], chain.facts[2]],
100
+ chain_signature=chain.chain_signature,
101
+ )
102
+ reordered_ok, _, reordered_error = verifier.verify_citation_chain(
103
+ reordered,
104
+ signer.get_public_key_b64(),
105
+ )
106
+
107
+ self.assertFalse(reordered_ok)
108
+ self.assertIn("Invalid", reordered_error)
109
+
110
+
111
+ if __name__ == "__main__":
112
+ unittest.main()
@@ -18,11 +18,83 @@ trust_tier: 1
18
18
 
19
19
  | Feature | Benefit |
20
20
  |---------|---------|
21
- | **Phase 4.5: Pseudocode** | +99% code quality (research-backed) |
21
+ | **Phase 4.5: Pseudocode** | Algorithm logic fixed before code, so codegen implements a decided design rather than inventing one (see `references/pseudocode-style.md`) |
22
22
  | **Phase 5: Test Scenarios** | TDD-ready Gherkin specs |
23
23
  | **Phase 6: Completion** | CI/CD, deployment, monitoring |
24
24
  | **9 Checkpoints** | Full control over each phase |
25
25
 
26
+ ## What's New in v3 (honesty + memory layer)
27
+
28
+ | Feature | Benefit |
29
+ |---------|---------|
30
+ | **Claim Discipline** | Every accuracy/count/percentage claim in an emitted doc carries an honest tag (MEASURED / CLAIMED / ESTIMATED) — no unsourced numbers |
31
+ | **ADR Confirmation stanza** | Every generated ADR names its load-bearing property + the fitness function / Gherkin test that would falsify it (Phase 3 → Phase 5 wired) |
32
+ | **Brain memory (Step 0 recall → closing teach)** | Recalls prior PRD/ADR lessons into the brief and teaches new ones at the end — idea2prd stops being write-once |
33
+
34
+ ---
35
+
36
+ ## Authoring Discipline (applies to EVERY emitted doc)
37
+
38
+ ### Claim Discipline (MEASURED / CLAIMED / ESTIMATED) — mandatory
39
+
40
+ The docs this skill emits (PRD, ADRs, fitness targets, executive summary) **must not carry untagged
41
+ accuracy, count, percentage, or performance claims.** Whenever you write a metric term next to a number
42
+ (`%`, `x faster`, coverage, latency, "N of M", "reduces … by …"), the same paragraph MUST carry one of
43
+ these honest tags, else do not write the number:
44
+
45
+ | Tag | Use when |
46
+ |-----|----------|
47
+ | `(MEASURED — <reproducer / source>)` | You actually ran/observed it — cite the command, file, or source |
48
+ | `(CLAIMED — <who/where>)` | Repeating an external/vendor claim you did not verify — attribute it |
49
+ | `(ESTIMATED — <basis>)` | A projection or target, not an observation — name the basis |
50
+
51
+ **Rule:** an untagged number that looks like a result is a defect. Prefer deleting a fabricated number
52
+ over dressing it up. This applies to **fitness-function targets and count tables too** — a target cell or
53
+ a count cell is a claim: template it as a `[placeholder]` and tag the real value MEASURED (an observed
54
+ baseline, naming its reproducer) or ESTIMATED (a chosen target, naming its basis) when it is filled in.
55
+
56
+ **Perfect-score prohibition:** a bare `100%` / `0 defects` / `never fails` / `always` cell is always a
57
+ HIGH finding — `dz claim-check` flags such a literal even when tagged, because perfection is almost never
58
+ observed. State the real value against a baseline (tag it MEASURED, name the reproducer), or express a
59
+ target as ESTIMATED with its basis (for example `target >= 95 (ESTIMATED from NFR-P01)`).
60
+
61
+ (This mirrors the harness Integrity Rule; a `dz claim-check` scan over the emitted `docs/` should pass —
62
+ run it if `dz` is available: `dz claim-check docs/ --fail-on medium`. Use `medium`, not `high`: `high`
63
+ lets ordinary untagged medium claims through — the false-green this discipline exists to stop.)
64
+
65
+ ### Brain memory — recall at Step 0, teach at the end (pin ONE canonical store)
66
+
67
+ If the `dz` CLI is available, idea2prd **learns across runs**. The learned-pattern store is pinned to
68
+ **ONE canonical brain store** — the project root as an ABSOLUTE path (the directory that will hold
69
+ `docs/`) — via `--project`, so lessons land in one store instead of fragmenting across sub-directories
70
+ (a recall from one store and a teach to another cannot reinforce each other):
71
+
72
+ - **Step 0 (before Gate 0):** recall prior PRD/ADR lessons and fold the top ones into the brief.
73
+ - **Closing (after Phase 6):** teach the genuinely new lessons from this run.
74
+
75
+ Both are guarded by "if `dz` present" — absent `dz`, skip silently and run the pipeline unchanged.
76
+
77
+ ---
78
+
79
+ ## Step 0: Brain Recall (run BEFORE Gate 0)
80
+
81
+ **Action (only if `dz` is on PATH — else skip silently):** resolve `<BRAIN>` to the **ABSOLUTE** path of
82
+ the project root (the directory that will hold `docs/`, e.g. via `pwd`) and use that SAME absolute path in
83
+ Step 0 recall and the closing teach. Via Bash run VERBATIM (`<BRAIN>` shell-quoted so a path with spaces
84
+ does not break the command):
85
+
86
+ ```bash
87
+ cd "<BRAIN>" && dz recall "<key domain terms of this idea/problem> PRD ADR bounded-context" --project "<BRAIN>"
88
+ ```
89
+
90
+ > `<BRAIN>` MUST be absolute. A relative `<BRAIN>` makes `cd "<BRAIN>"` then `--project "<BRAIN>"` resolve
91
+ > against the new working directory (a nested `<BRAIN>/<BRAIN>`), pinning the store to the wrong place.
92
+
93
+ - Log the number of patterns loaded.
94
+ - Fold the top 3 applicable patterns into the Task Brief / Requirements as a `{LEARNED_PATTERNS}` note,
95
+ and **preserve their text** so the closing teach step can compare candidate lessons against them.
96
+ - No patterns (first run) → proceed normally.
97
+
26
98
  ## Bundled Components
27
99
 
28
100
  ```
@@ -328,8 +400,14 @@ idea2prd-manual/
328
400
 
329
401
  ### Phase 3: ADR + C4
330
402
 
403
+ **Reference:** `references/adr-catalog.md` (ADR template — the `## Confirmation` stanza is REQUIRED).
404
+
331
405
  **Генерирует:**
332
- - 10+ ADRs
406
+ - 10+ ADRs — **every ADR MUST carry a `## Confirmation` stanza** (Method / Monitoring / Success metric /
407
+ Owner / **Load-bearing property** / **Required automated check: `<fitness function or Gherkin test>`**).
408
+ The named check MUST be a real Phase-5 artifact (a `FF-NNN` fitness function or a `.feature` Gherkin
409
+ scenario) — this is what wires Phase 3 → Phase 5 and turns each ADR from prose into a *checkable*
410
+ decision. An ADR whose Confirmation names no falsifying test is incomplete.
333
411
  - C4 Level 1: System Context
334
412
  - C4 Level 2: Container
335
413
  - C4 Level 3: Component
@@ -342,12 +420,15 @@ idea2prd-manual/
342
420
 
343
421
  ## ADRs Summary
344
422
 
345
- | ADR | Decision | Status |
346
- |-----|----------|--------|
347
- | ADR-001 | [Architecture]: [choice] | Accepted |
348
- | ADR-002 | [Database]: [choice] | Accepted |
423
+ | ADR | Decision | Status | Load-bearing property | Falsifying test |
424
+ |-----|----------|--------|-----------------------|-----------------|
425
+ | ADR-001 | [Architecture]: [choice] | Accepted | [property] | FF-0NN / [feature].feature |
426
+ | ADR-002 | [Database]: [choice] | Accepted | [property] | FF-0NN / [feature].feature |
349
427
  ...
350
428
 
429
+ > **Confirmation check:** every row above MUST name a load-bearing property and a real falsifying test
430
+ > (a `FF-NNN` fitness function or a Gherkin `.feature` scenario). A blank = an incomplete ADR — go back.
431
+
351
432
  ## C4 Diagrams
352
433
 
353
434
  - Level 1: System Context ✅
@@ -403,7 +484,7 @@ idea2prd-manual/
403
484
 
404
485
  **Reference:** `references/pseudocode-style.md`
405
486
 
406
- **КРИТИЧЕСКИ ВАЖНО для качества генерации кода (+99% по исследованиям).**
487
+ **Псевдокод фиксирует алгоритмическую логику ДО кодогенерации — меньше неоднозначности при реализации.** (Никаких неподтверждённых процентов: любые метрики качества эмитятся только с честным тегом — см. раздел «Claim Discipline».)
407
488
 
408
489
  **Действие Claude:**
409
490
  ```
@@ -552,7 +633,31 @@ Feature: Order Placement
552
633
 
553
634
  **Output file:** `docs/completion/COMPLETION_CHECKLIST.md`
554
635
 
555
- **После Phase 6 — финальный Executive Summary (без checkpoint).**
636
+ **После Phase 6 — Brain Teach, затем финальный Executive Summary (без checkpoint).**
637
+
638
+ ---
639
+
640
+ ## Closing: Brain Teach (run AFTER Phase 6, only if `dz` present — else skip silently)
641
+
642
+ Close the learning loop: teach the genuinely NEW lessons from this run to the SAME canonical brain store
643
+ that Step 0 recalled from — the SAME ABSOLUTE `<BRAIN>` path (the project root), pinned with `--project`
644
+ so nothing fragments.
645
+
646
+ **Action:**
647
+ 1. Compare the run's candidate lessons (a reusable PRD/ADR/DDD decision, a domain constraint that bit, a
648
+ fitness-function pattern) against the `{LEARNED_PATTERNS}` text recalled in Step 0.
649
+ 2. For a lesson ALREADY covered by a recalled pattern → do NOT re-teach it (it is already in the store).
650
+ 3. For each genuinely NEW lesson, via Bash run VERBATIM (one call per lesson):
651
+
652
+ ```bash
653
+ cd <BRAIN> && dz teach "<one-line reusable lesson>" --reward 0.7 --domain idea2prd --project <BRAIN>
654
+ ```
655
+
656
+ 4. Log how many lessons were taught (0 is a valid, honest result on a run that surfaced nothing new).
657
+
658
+ > **Pin discipline:** Step 0 recall and this teach MUST hit the SAME `<BRAIN>` (project root) via
659
+ > `--project` — if recall reads one store and teach writes another, the two stores stay separate and
660
+ > the loop does not accumulate in a single place. This mirrors the feature-adr `args.brain` rule.
556
661
 
557
662
  ---
558
663
 
@@ -43,10 +43,32 @@
43
43
  ### Risks
44
44
  - [Risk]: Mitigation: [approach]
45
45
 
46
+ ## Confirmation
47
+ <!-- REQUIRED. An ADR without this stanza is incomplete. -->
48
+ - **Load-bearing property:** [the single property that, if it silently regressed, makes this decision wrong]
49
+ - **Required automated check:** [`FF-NNN` fitness function OR `docs/tests/<feature>.feature` Gherkin scenario that FAILS if the property is violated]
50
+ - **Verification method:** [how the check is run — e.g. CI stage, fitness_validator.py, test runner]
51
+ - **Monitoring:** [what to watch in production to detect drift — metric/log/alert]
52
+ - **Success metric:** [the threshold that says the property still holds — tag it MEASURED / ESTIMATED]
53
+ - **Owner:** [role/team accountable for the check]
54
+
46
55
  ## Related ADRs
47
56
  - ADR-XXX: [relationship]
48
57
  ```
49
58
 
59
+ > **Confirmation stanza is load-bearing (idea2prd v3).** Every generated ADR MUST carry `## Confirmation`.
60
+ > Its `Required automated check` MUST name a REAL Phase-5 artifact — a `FF-NNN` fitness function
61
+ > (`references/fitness-functions-catalog.md` / `scripts/fitness_validator.py`) or a Gherkin `.feature`
62
+ > scenario — so the decision's load-bearing property is *falsifiable*, not prose. This wires Phase 3 (ADR)
63
+ > to Phase 5 (fitness/tests): a property an ADR names is often the one that ships untested (a recurring
64
+ > lesson in this repo), so the Confirmation stanza forces it to name its falsifying test.
65
+ > An ADR whose Confirmation names no falsifying check is **incomplete** — do not accept it at Checkpoint 3.
66
+
67
+ ### Supersession discipline
68
+
69
+ Never edit an Accepted/Rejected ADR's decision in place. A changed decision **mints a new ADR** that
70
+ supersedes the old one (`Status: Superseded by ADR-XXX`), and the new ADR carries its own `## Confirmation`.
71
+
50
72
  ---
51
73
 
52
74
  ## Standard ADRs (Required)
@@ -4,7 +4,7 @@
4
4
 
5
5
  Pseudocode в idea2prd используется для:
6
6
  1. Точного описания алгоритмов до написания кода
7
- 2. Улучшения качества генерации кода в Claude Code (+99% по исследованиям)
7
+ 2. Снижения неоднозначности при кодогенерации в Claude Code (алгоритм зафиксирован до кода)
8
8
  3. Документирования business logic
9
9
 
10
10
  ## Syntax Conventions