pwn 0.5.669 → 0.5.673
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.rubocop.yml +1 -1
- data/Gemfile +1 -1
- data/README.md +13 -9
- data/documentation/AI-Integration.md +1 -1
- data/documentation/Agent-Tool-Registry.md +17 -5
- data/documentation/Configuration.md +19 -8
- data/documentation/Diagrams.md +2 -2
- data/documentation/Home.md +2 -2
- data/documentation/How-PWN-Works.md +12 -10
- data/documentation/Installation.md +3 -1
- data/documentation/Mistakes.md +3 -0
- data/documentation/Persistence.md +4 -1
- data/documentation/Reinforcement-Learning.md +90 -73
- data/documentation/Skills-Memory-Learning.md +28 -4
- data/documentation/What-is-PWN.md +8 -7
- data/documentation/Why-PWN.md +3 -2
- data/documentation/diagrams/agent-tool-registry.svg +188 -160
- data/documentation/diagrams/dot/agent-tool-registry.dot +7 -4
- data/documentation/diagrams/dot/memory-skills-detailed.dot +12 -5
- data/documentation/diagrams/dot/overall-pwn-architecture.dot +4 -3
- data/documentation/diagrams/dot/persistence-filesystem.dot +2 -1
- data/documentation/diagrams/dot/pwn-ai-feedback-learning-loop.dot +11 -4
- data/documentation/diagrams/dot/reinforcement-learning.dot +8 -3
- data/documentation/diagrams/dot/task-summarizer.dot +24 -12
- data/documentation/diagrams/memory-skills-detailed.svg +252 -210
- data/documentation/diagrams/overall-pwn-architecture.svg +21 -12
- data/documentation/diagrams/persistence-filesystem.svg +127 -113
- data/documentation/diagrams/pwn-ai-feedback-learning-loop.svg +449 -397
- data/documentation/diagrams/reinforcement-learning.svg +276 -239
- data/documentation/diagrams/task-summarizer.svg +178 -126
- data/documentation/pwn-ai-Agent.md +66 -32
- data/lib/pwn/ai/agent/curriculum.rb +13 -4
- data/lib/pwn/ai/agent/dispatch.rb +3 -0
- data/lib/pwn/ai/agent/learning.rb +100 -14
- data/lib/pwn/ai/agent/loop.rb +692 -25
- data/lib/pwn/ai/agent/metrics.rb +52 -4
- data/lib/pwn/ai/agent/mistakes.rb +158 -1
- data/lib/pwn/ai/agent/policy.rb +935 -0
- data/lib/pwn/ai/agent/prompt_builder.rb +82 -6
- data/lib/pwn/ai/agent/reflect.rb +11 -3
- data/lib/pwn/ai/agent/registry.rb +18 -3
- data/lib/pwn/ai/agent/reward.rb +360 -48
- data/lib/pwn/ai/agent/task_summarizer.rb +415 -33
- data/lib/pwn/ai/agent/tool_guard.rb +157 -0
- data/lib/pwn/ai/agent/tools/policy.rb +76 -0
- data/lib/pwn/ai/agent/tools/ruby_eval.rb +27 -1
- data/lib/pwn/ai/agent/tools/shell.rb +28 -32
- data/lib/pwn/ai/agent.rb +2 -0
- data/lib/pwn/config.rb +14 -2
- data/lib/pwn/memory.rb +188 -0
- data/lib/pwn/sessions.rb +9 -4
- data/lib/pwn/version.rb +1 -1
- data/spec/integration/reinforced_feedback_loop_spec.rb +35 -9
- data/spec/lib/pwn/ai/agent/learning_spec.rb +119 -3
- data/spec/lib/pwn/ai/agent/loop_spec.rb +215 -3
- data/spec/lib/pwn/ai/agent/metrics_spec.rb +10 -0
- data/spec/lib/pwn/ai/agent/mistakes_spec.rb +33 -0
- data/spec/lib/pwn/ai/agent/policy_spec.rb +165 -0
- data/spec/lib/pwn/ai/agent/prompt_builder_spec.rb +51 -0
- data/spec/lib/pwn/ai/agent/reward_spec.rb +75 -0
- data/spec/lib/pwn/ai/agent/signal_hygiene_spec.rb +118 -0
- data/spec/lib/pwn/ai/agent/task_summarizer_spec.rb +111 -6
- data/spec/lib/pwn/ai/agent/tool_guard_spec.rb +61 -0
- data/spec/lib/pwn/ai/agent/tools/policy_spec.rb +18 -0
- data/spec/lib/pwn/memory_spec.rb +62 -0
- data/spec/support/sandbox.rb +2 -0
- data/third_party/pwn_rdoc.jsonl +108 -4
- metadata +10 -3
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
digraph "PWN_Agent_Tool_Registry" {
|
|
2
2
|
graph [
|
|
3
|
-
label=<<B>PWN::AI::Agent::Registry - Toolsets exposed to the LLM</B><BR/><FONT POINT-SIZE="11" COLOR="#94a3b8">
|
|
3
|
+
label=<<B>PWN::AI::Agent::Registry - Toolsets exposed to the LLM</B><BR/><FONT POINT-SIZE="11" COLOR="#94a3b8">13 toolsets · 85 callable tools · lib/pwn/ai/agent/tools/*</FONT>>,
|
|
4
4
|
labelloc=t, fontsize=20, fontname="Helvetica",
|
|
5
5
|
rankdir=LR, splines=spline, nodesep=0.35, ranksep=1.4,
|
|
6
6
|
bgcolor="#0f172a", fontcolor="#e2e8f0", pad=0.6, newrank=true
|
|
@@ -26,14 +26,15 @@ digraph "PWN_Agent_Tool_Registry" {
|
|
|
26
26
|
swarm [label="swarm", fillcolor="#6ee7b7"];
|
|
27
27
|
reward [label="reward", fillcolor="#6ee7b7"];
|
|
28
28
|
curric [label="curriculum", fillcolor="#6ee7b7"];
|
|
29
|
+
policy [label="policy", fillcolor="#6ee7b7"];
|
|
29
30
|
}
|
|
30
31
|
|
|
31
32
|
subgraph cluster_tools {
|
|
32
33
|
label="Tools (LLM-callable)"; fontcolor="#fde68a"; style=rounded;
|
|
33
34
|
color="#a16207"; bgcolor="#422006"; penwidth=2;
|
|
34
35
|
node [fillcolor="#fcd34d", fontsize=9];
|
|
35
|
-
t_shell [label="shell"];
|
|
36
|
-
t_eval [label="pwn_eval"];
|
|
36
|
+
t_shell [label="shell\n(ToolGuard)"];
|
|
37
|
+
t_eval [label="pwn_eval\n(ToolGuard)"];
|
|
37
38
|
t_mem [label="memory_remember\nrecall · forget · clear\nlean"];
|
|
38
39
|
t_skill [label="skill_list · view · create\nadd_reference · delete\nmigrate_legacy"];
|
|
39
40
|
t_sess [label="sessions_list · view\ncurrent · delete · stats\nlean"];
|
|
@@ -44,12 +45,13 @@ digraph "PWN_Agent_Tool_Registry" {
|
|
|
44
45
|
t_swarm [label="agent_list · spawn · ask\ndebate · broadcast\nswarm_bus · swarm_list"];
|
|
45
46
|
t_reward [label="reward_generator_mix"];
|
|
46
47
|
t_curric [label="curriculum_practice_kpi"];
|
|
48
|
+
t_policy [label="policy_stats\npolicy_evaluate\npolicy_recommend"];
|
|
47
49
|
}
|
|
48
50
|
|
|
49
51
|
Registry -> terminal; Registry -> pwn; Registry -> memory;
|
|
50
52
|
Registry -> skills; Registry -> sessions; Registry -> learning;
|
|
51
53
|
Registry -> metrics; Registry -> extro; Registry -> cron; Registry -> swarm;
|
|
52
|
-
Registry -> reward; Registry -> curric;
|
|
54
|
+
Registry -> reward; Registry -> curric; Registry -> policy;
|
|
53
55
|
|
|
54
56
|
terminal -> t_shell [color="#f59e0b"];
|
|
55
57
|
pwn -> t_eval [color="#f59e0b"];
|
|
@@ -62,6 +64,7 @@ digraph "PWN_Agent_Tool_Registry" {
|
|
|
62
64
|
cron -> t_cron [color="#f59e0b"];
|
|
63
65
|
swarm -> t_swarm [color="#f59e0b"];
|
|
64
66
|
reward -> t_reward [color="#f59e0b"];
|
|
67
|
+
policy -> t_policy [color="#f59e0b"];
|
|
65
68
|
curric -> t_curric [color="#f59e0b"];
|
|
66
69
|
|
|
67
70
|
t_metric -> Registry [label="success_rate\n→ router tie-break", style=dashed, color="#fbbf24", constraint=false];
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
digraph "PWN_MemorySkills" {
|
|
2
2
|
graph [
|
|
3
|
-
label=<<B>Persistent Knowledge - Memory · Skills · Learning · Mistakes · Sessions</B><BR/><FONT POINT-SIZE="11" COLOR="#94a3b8">everything under ~/.pwn/ that shapes future prompts</FONT>>,
|
|
3
|
+
label=<<B>Persistent Knowledge - Memory · Skills · Learning · Mistakes · Policy · Sessions</B><BR/><FONT POINT-SIZE="11" COLOR="#94a3b8">everything under ~/.pwn/ that shapes future prompts</FONT>>,
|
|
4
4
|
labelloc=t, fontsize=20, fontname="Helvetica",
|
|
5
5
|
rankdir=TB, splines=spline, nodesep=0.55, ranksep=0.95,
|
|
6
6
|
bgcolor="#0f172a", fontcolor="#e2e8f0", pad=0.6, newrank=true
|
|
@@ -20,8 +20,9 @@ digraph "PWN_MemorySkills" {
|
|
|
20
20
|
Mrec [label="mistakes_record\nmistakes_resolve", fillcolor="#6ee7b7"];
|
|
21
21
|
Xft [label="Learning.export_finetune\n(sessions → dataset)", fillcolor="#6ee7b7"];
|
|
22
22
|
Rdpo [label="Reward.record_preference\nexport_dpo", fillcolor="#6ee7b7"];
|
|
23
|
+
PolW [label="Policy.begin / observe / finish\n(Q + REINFORCE)", fillcolor="#6ee7b7"];
|
|
23
24
|
}
|
|
24
|
-
{rank=same; Rem; Skc; Note; Dist; Mrec; Xft; Rdpo}
|
|
25
|
+
{rank=same; Rem; Skc; Note; Dist; Mrec; Xft; Rdpo; PolW}
|
|
25
26
|
|
|
26
27
|
subgraph cluster_files {
|
|
27
28
|
label="~/.pwn/"; fontcolor="#fde68a"; style=rounded;
|
|
@@ -35,8 +36,9 @@ digraph "PWN_MemorySkills" {
|
|
|
35
36
|
Fext [label="extrospection.json\nsnapshot · rf · web · obs", shape=cylinder, fillcolor="#fcd34d"];
|
|
36
37
|
Fftn [label="finetune/*.jsonl\nShareGPT / OpenAI / DPO", shape=cylinder, fillcolor="#fcd34d"];
|
|
37
38
|
Fprf [label="preferences.jsonl\nDPO (prompt,rejected,chosen)", shape=cylinder, fillcolor="#fcd34d"];
|
|
39
|
+
Fpol [label="policy.json + policy_traj.jsonl\nQ table · REINFORCE · MDP log", shape=cylinder, fillcolor="#fcd34d"];
|
|
38
40
|
}
|
|
39
|
-
{rank=same; Fmem; Fidx; Fskl; Flrn; Fses; Fmis; Fext; Fftn; Fprf}
|
|
41
|
+
{rank=same; Fmem; Fidx; Fskl; Flrn; Fses; Fmis; Fext; Fftn; Fprf; Fpol}
|
|
40
42
|
|
|
41
43
|
subgraph cluster_read {
|
|
42
44
|
label="Read-Side / Injection"; fontcolor="#ddd6fe"; style=rounded;
|
|
@@ -48,10 +50,11 @@ digraph "PWN_MemorySkills" {
|
|
|
48
50
|
Sesv [label="sessions_view\nsessions_current", fillcolor="#c4b5fd"];
|
|
49
51
|
Mlst [label="mistakes_list\ncorrection_hint", fillcolor="#c4b5fd"];
|
|
50
52
|
Exem [label="Learning.exemplars_for\n(few-shot trace)", fillcolor="#c4b5fd"];
|
|
53
|
+
PolR [label="policy_stats · evaluate\npolicy_recommend", fillcolor="#c4b5fd"];
|
|
51
54
|
}
|
|
52
|
-
{rank=same; Rec; Midx; Skv; Out; Sesv; Mlst; Exem}
|
|
55
|
+
{rank=same; Rec; Midx; Skv; Out; Sesv; Mlst; Exem; PolR}
|
|
53
56
|
|
|
54
|
-
Prompt [label="PromptBuilder.build(request:)\nengine-aware .budget\nMEMORY (relevance-ranked) + SKILLS\n+ LEARNING + KNOWN MISTAKES/FIXES\n+ METRICS (per-engine) + EXTRO", fillcolor="#7dd3fc", penwidth=2];
|
|
57
|
+
Prompt [label="PromptBuilder.build(request:)\nengine-aware .budget\nMEMORY (relevance-ranked) + SKILLS\n+ LEARNING + KNOWN MISTAKES/FIXES\n+ METRICS (per-engine) + POLICY + EXTRO", fillcolor="#7dd3fc", penwidth=2];
|
|
55
58
|
|
|
56
59
|
Rem -> Fmem [color="#f59e0b"];
|
|
57
60
|
Skc -> Fskl [color="#f59e0b"];
|
|
@@ -94,6 +97,10 @@ digraph "PWN_MemorySkills" {
|
|
|
94
97
|
Teach [label="Reflect.on\n(engine: reflect_engine)\nteacher-student", fillcolor="#fda4af", penwidth=2];
|
|
95
98
|
Fses -> Teach [label="transcript", style=dashed, color="#94a3b8", constraint=false];
|
|
96
99
|
Teach -> Fmem [label="frontier-authored\n:lesson", color="#fb7185", penwidth=2];
|
|
100
|
+
|
|
101
|
+
PolW -> Fpol [color="#f59e0b"];
|
|
102
|
+
Fpol -> PolR [color="#a78bfa"];
|
|
103
|
+
PolR -> Prompt [label="POLICY block\nadvisory only", color="#38bdf8"];
|
|
97
104
|
Rdpo -> Fprf [color="#f59e0b"];
|
|
98
105
|
Mrec -> Fprf [label="resolve → pair", style=dashed, color="#fbbf24"];
|
|
99
106
|
Fprf -> Fftn [label="export_dpo", style=dashed, color="#fbbf24", constraint=false];
|
|
@@ -37,9 +37,9 @@ digraph "PWN_Overall_Architecture" {
|
|
|
37
37
|
label="PWN::AI::Agent"; fontcolor="#ddd6fe";
|
|
38
38
|
style=rounded; color="#6d28d9"; bgcolor="#2e1065"; penwidth=2;
|
|
39
39
|
Loop [label="Loop\nTaskSummarizer briefs\nplan_first → dispatch\n→ observe → escalate\nbudget-pressure caps", fillcolor="#c4b5fd"];
|
|
40
|
-
Registry [label="Registry\
|
|
40
|
+
Registry [label="Registry\n13 toolsets · 85 tools\ntool_router (CORE+topK)", fillcolor="#c4b5fd"];
|
|
41
41
|
Swarm [label="Swarm\npersonas · debate · bus", fillcolor="#c4b5fd"];
|
|
42
|
-
Prompt [label="PromptBuilder\nengine-budgeted blocks\nMemoryIndex → relevance-ranked",fillcolor="#c4b5fd"];
|
|
42
|
+
Prompt [label="PromptBuilder\nengine-budgeted blocks\nMEMORY · SKILLS · POLICY\nMemoryIndex → relevance-ranked",fillcolor="#c4b5fd"];
|
|
43
43
|
}
|
|
44
44
|
{rank=same; Loop; Registry; Swarm; Prompt}
|
|
45
45
|
|
|
@@ -74,8 +74,9 @@ digraph "PWN_Overall_Architecture" {
|
|
|
74
74
|
CronF [label="cron/jobs.yml", shape=cylinder, fillcolor="#fcd34d"];
|
|
75
75
|
Finetune [label="finetune/*.jsonl", shape=cylinder, fillcolor="#fcd34d"];
|
|
76
76
|
Prefs [label="preferences.jsonl", shape=cylinder, fillcolor="#fcd34d"];
|
|
77
|
+
PolicyF [label="policy.json\npolicy_traj.jsonl", shape=cylinder, fillcolor="#fcd34d"];
|
|
77
78
|
}
|
|
78
|
-
{rank=same; Memory; MemIdx; Skills; Learn; Metrics; MistF; Extro; Sessions; SwarmB; CronF; Finetune; Prefs}
|
|
79
|
+
{rank=same; Memory; MemIdx; Skills; Learn; Metrics; MistF; Extro; Sessions; SwarmB; CronF; Finetune; Prefs; PolicyF}
|
|
79
80
|
|
|
80
81
|
/* ── L0 → L1 ── */
|
|
81
82
|
User -> REPL [color="#38bdf8"];
|
|
@@ -21,6 +21,7 @@ digraph "PWN_Persistence" {
|
|
|
21
21
|
prf [label="preferences.jsonl\nDPO/KTO/ORPO pairs\nuser_correction · resolve · counterfactual"];
|
|
22
22
|
mis [label="mistakes.json\nfailure fingerprints · fixes\n[REPEATING] · [REGRESSED]"];
|
|
23
23
|
met [label="metrics.json\nper-tool · per-engine telemetry\ncalibration"];
|
|
24
|
+
pol [label="policy.json + policy_traj.jsonl\nlive Q / REINFORCE · MDP log"];
|
|
24
25
|
rws [label="reward_sentinel.json\nproxy vs judge vs correction gap"];
|
|
25
26
|
ext [label="extrospection.json\nsnapshot (host/net/tc/repo/env/\nrf/web/osint/serial/telecomm/\npacket/vision/voice) + prev + obs[]"];
|
|
26
27
|
eart [label="extrospection/{web,packet,voice}/*\nscreenshots · pcaps · wav/txt"];
|
|
@@ -34,7 +35,7 @@ digraph "PWN_Persistence" {
|
|
|
34
35
|
hist [label="~/.pwn_history\nREPL history"];
|
|
35
36
|
|
|
36
37
|
Root->cfg; Root->schm; Root->mem; Root->midx; Root->skl; Root->lrn;
|
|
37
|
-
Root->prf; Root->mis; Root->met; Root->rws; Root->ext; Root->eart;
|
|
38
|
+
Root->prf; Root->mis; Root->met; Root->pol; Root->rws; Root->ext; Root->eart;
|
|
38
39
|
Root->ses; Root->crn; Root->agt; Root->swm; Root->cur; Root->ftn;
|
|
39
40
|
Root->bkp; Root->hist;
|
|
40
41
|
}
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
digraph "PWN_AI_Feedback_Learning_Loop" {
|
|
2
2
|
graph [
|
|
3
|
-
label=<<B>pwn-ai - Closed Self-Improvement Loop</B><BR/><FONT POINT-SIZE="11" COLOR="#94a3b8">Introspection (self) ⟷ Extrospection (world) · Mistakes (negative) · Reward + Curriculum (RL) · Local-model scaffolding</FONT>>,
|
|
3
|
+
label=<<B>pwn-ai - Closed Self-Improvement Loop</B><BR/><FONT POINT-SIZE="11" COLOR="#94a3b8">Introspection (self) ⟷ Extrospection (world) · Mistakes (negative) · Reward + Curriculum + Policy (RL) · Local-model scaffolding</FONT>>,
|
|
4
4
|
labelloc=t, fontsize=20, fontname="Helvetica",
|
|
5
5
|
rankdir=TB, splines=spline, nodesep=0.6, ranksep=1.0,
|
|
6
6
|
bgcolor="#0f172a", fontcolor="#e2e8f0", pad=0.6, newrank=true, compound=true
|
|
@@ -21,7 +21,7 @@ digraph "PWN_AI_Feedback_Learning_Loop" {
|
|
|
21
21
|
label="PWN::AI::Agent::Loop · budget pressure · TaskSummarizer"; fontcolor="#ddd6fe";
|
|
22
22
|
style=rounded; color="#6d28d9"; bgcolor="#2e1065"; penwidth=2;
|
|
23
23
|
Prompt [label="PromptBuilder\nbudget · MemoryIndex\nrelevance-ranked ctx", fillcolor="#c4b5fd"];
|
|
24
|
-
Dispatch [label="Dispatch\ntolerant parse\
|
|
24
|
+
Dispatch [label="Dispatch\ntolerant parse\nToolGuard · repair_name", fillcolor="#c4b5fd"];
|
|
25
25
|
Router [label="Registry\ntool_router\nCORE + top-K", fillcolor="#c4b5fd"];
|
|
26
26
|
Plan [label="plan_first\n(pre-pass, local)", fillcolor="#c4b5fd"];
|
|
27
27
|
TaskSum [label="TaskSummarizer\nemit_plan! · about_to\ndedup last_brief_fp", fillcolor="#c4b5fd", penwidth=2];
|
|
@@ -40,6 +40,7 @@ Escalate [label="escalate\n>=N fails → Swarm hint", fillcolor="#c4b5fd"
|
|
|
40
40
|
ReflectT [label="Reflect\nteacher-student\n(reflect_engine)", fillcolor="#6ee7b7", penwidth=2];
|
|
41
41
|
Reward [label="Reward\njudge · prm · semantic_ok\nsentinel warm\nDPO scrub <=40%/src", fillcolor="#6ee7b7", penwidth=2];
|
|
42
42
|
Curric [label="Curriculum\npractice · offline_judge\ncritic/CF geometry · gate + diet", fillcolor="#6ee7b7", penwidth=2];
|
|
43
|
+
Policy [label="Policy\nlive Q / REINFORCE\nbegin · observe · finish\nadvisory rank only", fillcolor="#6ee7b7", penwidth=2];
|
|
43
44
|
}
|
|
44
45
|
subgraph cluster_extro {
|
|
45
46
|
label="EXTROSPECTION (world)"; fontcolor="#fde68a";
|
|
@@ -49,7 +50,7 @@ Escalate [label="escalate\n>=N fails → Swarm hint", fillcolor="#c4b5fd"
|
|
|
49
50
|
Verify [label="verify (browser)\nfact-check own claims\nrevalidate_memory", fillcolor="#fcd34d", penwidth=2];
|
|
50
51
|
RFTune [label="rf_tune 📡\ntune GQRX · demod\nRDS → now_playing", fillcolor="#fcd34d", penwidth=2];
|
|
51
52
|
}
|
|
52
|
-
{rank=same; Metrics; Learning; Mistakes; ReflectT; Reward; Curric; Snapshot; Observe; Verify; RFTune}
|
|
53
|
+
{rank=same; Metrics; Learning; Mistakes; ReflectT; Reward; Curric; Policy; Snapshot; Observe; Verify; RFTune}
|
|
53
54
|
|
|
54
55
|
/* L3 ─ correlate */
|
|
55
56
|
Correlate [label="extro_correlate()\n\"I did it wrong\" vs \"the world changed\"",
|
|
@@ -68,8 +69,9 @@ Escalate [label="escalate\n>=N fails → Swarm hint", fillcolor="#c4b5fd"
|
|
|
68
69
|
Fext [label="extrospection.json", shape=cylinder, fillcolor="#fcd34d"];
|
|
69
70
|
Fftn [label="finetune/*.jsonl", shape=cylinder, fillcolor="#fcd34d"];
|
|
70
71
|
Fprf [label="preferences.jsonl", shape=cylinder, fillcolor="#fcd34d"];
|
|
72
|
+
Fpol [label="policy.json\npolicy_traj.jsonl", shape=cylinder, fillcolor="#fcd34d"];
|
|
71
73
|
}
|
|
72
|
-
{rank=same; Fmet; Flrn; Fmis; Fmem; Fidx; Fskl; Fext; Fftn; Fprf}
|
|
74
|
+
{rank=same; Fmet; Flrn; Fmis; Fmem; Fidx; Fskl; Fext; Fftn; Fprf; Fpol}
|
|
73
75
|
|
|
74
76
|
/* L0 → L1 */
|
|
75
77
|
User -> Prompt [label="task", color="#38bdf8", penwidth=2];
|
|
@@ -87,7 +89,9 @@ Escalate [label="escalate\n>=N fails → Swarm hint", fillcolor="#c4b5fd"
|
|
|
87
89
|
|
|
88
90
|
/* L1 → L2 */
|
|
89
91
|
Dispatch -> Metrics [label="record(engine:)", color="#34d399"];
|
|
92
|
+
Dispatch -> Policy [label="observe_step", color="#34d399"];
|
|
90
93
|
Result -> Learning [label="auto_introspect", color="#34d399"];
|
|
94
|
+
Result -> Policy [label="finish (judge)", color="#34d399"];
|
|
91
95
|
Dispatch -> Mistakes [label="on failure\nrecord()", color="#fb7185", penwidth=2];
|
|
92
96
|
User -> Mistakes [label="\"that's wrong\"\ncheck_user_correction", color="#fb7185",
|
|
93
97
|
style=dashed, constraint=false];
|
|
@@ -111,6 +115,7 @@ Escalate [label="escalate\n>=N fails → Swarm hint", fillcolor="#c4b5fd"
|
|
|
111
115
|
/* L3 → L4 */
|
|
112
116
|
Correlate -> Fmet [style=invis];
|
|
113
117
|
Metrics -> Fmet [color="#94a3b8"];
|
|
118
|
+
Policy -> Fpol [color="#94a3b8"];
|
|
114
119
|
Learning -> Flrn [color="#94a3b8"];
|
|
115
120
|
Mistakes -> Fmis [color="#94a3b8"];
|
|
116
121
|
Mistakes -> Fmem [label="resolve→lesson", color="#94a3b8"];
|
|
@@ -141,6 +146,8 @@ Escalate [label="escalate\n>=N fails → Swarm hint", fillcolor="#c4b5fd"
|
|
|
141
146
|
Fidx -> Prompt [label="MemoryIndex\ncosine top-K", style=dashed, color="#fbbf24", constraint=false, penwidth=2];
|
|
142
147
|
Flrn -> Plan [label="exemplars_for\n(few-shot, local)", style=dashed, color="#fbbf24", constraint=false];
|
|
143
148
|
Fmet -> Router [label="success_rate\n→ tie-break", style=dashed, color="#fbbf24", constraint=false];
|
|
149
|
+
Fpol -> Prompt [label="POLICY block", style=dashed, color="#fbbf24", constraint=false];
|
|
150
|
+
Fpol -> Router [label="Q-advantage\nvisits≥2", style=dashed, color="#fbbf24", constraint=false];
|
|
144
151
|
Escalate -> LLM [label="Swarm.ask(persona)\nfrontier hint", color="#a78bfa", constraint=false, penwidth=2, style=dashed];
|
|
145
152
|
Escalate -> Mistakes [label="record\n(tool:'escalation')", color="#fb7185", constraint=false, style=dashed];
|
|
146
153
|
Fftn -> LLM [label="LoRA (weights)\ncron weekly", style=dashed, color="#fb7185", constraint=false, penwidth=2];
|
|
@@ -19,13 +19,14 @@ digraph "PWN_Reinforcement_Learning" {
|
|
|
19
19
|
subgraph cluster_reward {
|
|
20
20
|
label="Tier 1 · Reward (PWN::AI::Agent::Reward)"; fontcolor="#ddd6fe";
|
|
21
21
|
style=rounded; color="#6d28d9"; bgcolor="#2e1065"; penwidth=2;
|
|
22
|
-
R1 [label="judge (
|
|
22
|
+
R1 [label="judge (cheap LLM ORM)\nscore · verdict · source\nevidence-prior fallback", fillcolor="#c4b5fd", penwidth=2];
|
|
23
23
|
R2 [label="prm (process)\nper-step +1/0/-1\n→ Registry.rank", fillcolor="#c4b5fd"];
|
|
24
24
|
R3 [label="sentinel (ring N=40)\nproxy != judge != (1-corr)\nwarm + proxy distrust", fillcolor="#c4b5fd"];
|
|
25
25
|
R4 [label="semantic_ok\ngrep exit 1 != failure", fillcolor="#c4b5fd"];
|
|
26
|
+
R5 [label="Policy (live MDP)\nplan · complete · usable\nwarm Q + episode budget", fillcolor="#c4b5fd", penwidth=2];
|
|
26
27
|
E3 [label="verify_as_reward\nextro_verify caps/floors judge", fillcolor="#c4b5fd"];
|
|
27
28
|
}
|
|
28
|
-
{rank=same; R1; R2; R3; R4; E3}
|
|
29
|
+
{rank=same; R1; R2; R3; R4; R5; E3}
|
|
29
30
|
|
|
30
31
|
/* Tier 4 — Curriculum / self-play */
|
|
31
32
|
subgraph cluster_curr {
|
|
@@ -55,22 +56,26 @@ digraph "PWN_Reinforcement_Learning" {
|
|
|
55
56
|
color="#a16207"; bgcolor="#422006";
|
|
56
57
|
Fmis [label="mistakes.json", shape=cylinder, fillcolor="#fcd34d"];
|
|
57
58
|
Fses [label="sessions/*.jsonl\n(step_reward)", shape=cylinder, fillcolor="#fcd34d"];
|
|
59
|
+
Fpol [label="policy.json +\npolicy_traj.jsonl", shape=cylinder, fillcolor="#fcd34d"];
|
|
58
60
|
Fprf [label="preferences.jsonl", shape=cylinder, fillcolor="#fcd34d"];
|
|
59
61
|
Fcur [label="curriculum/", shape=cylinder, fillcolor="#fcd34d"];
|
|
60
62
|
Fftn [label="finetune/*.jsonl", shape=cylinder, fillcolor="#fcd34d"];
|
|
61
63
|
Frws [label="reward_sentinel.json", shape=cylinder, fillcolor="#fcd34d"];
|
|
62
64
|
}
|
|
63
|
-
{rank=same; Fmis; Fses; Fprf; Fcur; Fftn; Frws}
|
|
65
|
+
{rank=same; Fmis; Fses; Fpol; Fprf; Fcur; Fftn; Frws}
|
|
64
66
|
|
|
65
67
|
Ollama [label="ollama create pwn-vN+1\n(local LoRA adapter)", fillcolor="#7dd3fc"];
|
|
66
68
|
|
|
67
69
|
/* edges */
|
|
68
70
|
Req -> Loop [color="#38bdf8", penwidth=2];
|
|
69
71
|
Loop -> R4 [label="Dispatch", color="#a78bfa"];
|
|
72
|
+
Loop -> R5 [label="(s,a,r,s')", color="#38bdf8", penwidth=2];
|
|
70
73
|
Loop -> R1 [label="final", color="#a78bfa", penwidth=2];
|
|
71
74
|
Loop -> R2 [label="session", color="#a78bfa"];
|
|
72
75
|
Loop -> R3 [color="#a78bfa"];
|
|
73
76
|
R1 -> E3 [label="ground", style=dashed, color="#fbbf24"];
|
|
77
|
+
R1 -> R5 [label="terminal", color="#a78bfa"];
|
|
78
|
+
R5 -> Fpol [color="#f59e0b"];
|
|
74
79
|
Loop -> S4 [label="plan_first", color="#34d399"];
|
|
75
80
|
Loop -> S3 [label="final", color="#34d399"];
|
|
76
81
|
Loop -> S2 [label="[REPEATING]", color="#fb7185"];
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
digraph "PWN_TaskSummarizer" {
|
|
2
2
|
graph [
|
|
3
|
-
label=<<B>PWN::AI::Agent::TaskSummarizer -
|
|
3
|
+
label=<<B>PWN::AI::Agent::TaskSummarizer - Request kind + executive briefs</B><BR/><FONT POINT-SIZE="11" COLOR="#94a3b8">request_kind → (statement|question: no plan) · (autonomous_goal: emit_plan! → about_to → tools)</FONT>>,
|
|
4
4
|
labelloc=t, fontsize=20, fontname="Helvetica",
|
|
5
5
|
rankdir=TB, splines=spline, nodesep=0.55, ranksep=0.9,
|
|
6
6
|
bgcolor="#0f172a", fontcolor="#e2e8f0", pad=0.6, newrank=true, compound=true
|
|
@@ -10,25 +10,34 @@ digraph "PWN_TaskSummarizer" {
|
|
|
10
10
|
edge [fontname="Helvetica", fontsize=9, color="#94a3b8",
|
|
11
11
|
fontcolor="#cbd5e1", penwidth=1.3, arrowsize=0.8];
|
|
12
12
|
|
|
13
|
-
User [label="
|
|
14
|
-
Loop [label="Loop.run\nts_state = TaskSummarizer.fresh", fillcolor="#c4b5fd"];
|
|
15
|
-
|
|
13
|
+
User [label="User request", fillcolor="#7dd3fc"];
|
|
14
|
+
Loop [label="Loop.run\nts_state = TaskSummarizer.fresh\nrequest_kind + request_intent", fillcolor="#c4b5fd"];
|
|
15
|
+
Kind [label="request_kind (LLM + heuristic)\nstatement | question | autonomous_goal\nhost-evidence Qs → goal", fillcolor="#fcd34d", penwidth=2];
|
|
16
|
+
{rank=same; User; Loop; Kind}
|
|
17
|
+
|
|
18
|
+
subgraph cluster_nogoal {
|
|
19
|
+
label="No multi-step breakdown"; fontcolor="#fda4af";
|
|
20
|
+
style=rounded; color="#be123c"; bgcolor="#4c0519"; penwidth=2;
|
|
21
|
+
Stmt [label="statement\nanswer_statement / greeting ack", fillcolor="#fda4af"];
|
|
22
|
+
Ques [label="question (knowledge/howto/recall)\nanswer_question / howto / recall\nno tools / no plan", fillcolor="#fda4af"];
|
|
23
|
+
}
|
|
24
|
+
{rank=same; Stmt; Ques}
|
|
16
25
|
|
|
17
26
|
subgraph cluster_plan {
|
|
18
|
-
label="
|
|
27
|
+
label="Autonomous goal plan (once per turn)"; fontcolor="#ddd6fe";
|
|
19
28
|
style=rounded; color="#6d28d9"; bgcolor="#2e1065"; penwidth=2;
|
|
20
|
-
Plan [label="plan(request)\nenumerated
|
|
21
|
-
EmitP [label="emit_plan!\nGoal:
|
|
29
|
+
Plan [label="plan(request)\nenumerated · LLM · fallback\nordered work units", fillcolor="#c4b5fd"];
|
|
30
|
+
EmitP [label="emit_plan!\nKind: autonomous_goal\nGoal: ...\nTangible tasks (N):\n each may use 1+ tools", fillcolor="#c4b5fd", penwidth=2];
|
|
22
31
|
}
|
|
23
32
|
{rank=same; Plan; EmitP}
|
|
24
33
|
|
|
25
34
|
subgraph cluster_batch {
|
|
26
35
|
label="Batch layer (each tool collection)"; fontcolor="#a7f3d0";
|
|
27
36
|
style=rounded; color="#047857"; bgcolor="#022c22"; penwidth=2;
|
|
28
|
-
About [label="about_to(tools:)\
|
|
37
|
+
About [label="about_to(tools:)\ntask k/n English primary\nvia tools (intent)", fillcolor="#6ee7b7"];
|
|
29
38
|
FP [label="brief_fingerprint\nlast_brief_fp", fillcolor="#6ee7b7"];
|
|
30
39
|
Dup [label="duplicate_brief?", shape=diamond, fillcolor="#fcd34d"];
|
|
31
|
-
EmitB [label="emit task line\
|
|
40
|
+
EmitB [label="emit task line\ntask k/N: ...\nvia shell×2 (search)", fillcolor="#6ee7b7", penwidth=2];
|
|
32
41
|
Skip [label="return nil\n(suppress duplicate)", fillcolor="#fda4af"];
|
|
33
42
|
}
|
|
34
43
|
{rank=same; About; FP; Dup; EmitB; Skip}
|
|
@@ -36,7 +45,7 @@ digraph "PWN_TaskSummarizer" {
|
|
|
36
45
|
subgraph cluster_exec {
|
|
37
46
|
label="Execution (not on task row)"; fontcolor="#fde68a";
|
|
38
47
|
style=rounded; color="#a16207"; bgcolor="#422006"; penwidth=2;
|
|
39
|
-
Tools [label="Dispatch tools\nshell · pwn_eval ·
|
|
48
|
+
Tools [label="Dispatch tools\nshell · pwn_eval · ...\nown REPL lines", fillcolor="#fcd34d"];
|
|
40
49
|
Rec [label="record!\nadvance plan_idx\n(silent unless verbose)", fillcolor="#fcd34d"];
|
|
41
50
|
Flush [label="flush! / emit!\noptional closing brief", fillcolor="#fcd34d"];
|
|
42
51
|
}
|
|
@@ -45,10 +54,13 @@ digraph "PWN_TaskSummarizer" {
|
|
|
45
54
|
UI [label="REPL on_tool\nname='task', full text, result=''\n(no truncation)", fillcolor="#7dd3fc"];
|
|
46
55
|
|
|
47
56
|
User -> Loop [color="#38bdf8", penwidth=2];
|
|
48
|
-
Loop ->
|
|
57
|
+
Loop -> Kind [color="#fbbf24", penwidth=2];
|
|
58
|
+
Kind -> Stmt [label="statement", color="#fb7185"];
|
|
59
|
+
Kind -> Ques [label="question", color="#fb7185"];
|
|
60
|
+
Kind -> Plan [label="autonomous_goal", color="#a78bfa", penwidth=2];
|
|
49
61
|
Plan -> EmitP [color="#a78bfa"];
|
|
50
62
|
EmitP -> UI [label="full goal once", color="#38bdf8", penwidth=2];
|
|
51
|
-
Loop -> About [label="before each\ntool batch", color="#a78bfa"];
|
|
63
|
+
Loop -> About [label="before each\ntool batch\n(goals only)", color="#a78bfa"];
|
|
52
64
|
About -> FP [color="#34d399"];
|
|
53
65
|
FP -> Dup [color="#34d399"];
|
|
54
66
|
Dup -> EmitB [label="new", color="#34d399"];
|