packet-tracer-skill 0.3.0 → 0.3.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +116 -0
- package/README.md +55 -9
- package/SKILL.md +214 -2
- package/package.json +2 -1
- package/scripts/build_sample_catalog.py +42 -0
- package/scripts/generate_pkt.py +4408 -187
- package/scripts/intent_parser.py +31 -4
- package/scripts/lab_coherence.py +455 -0
- package/scripts/pkt_editor.py +117 -35
- package/scripts/pkt_transformer.py +99 -1
- package/scripts/sample_catalog.py +65 -12
- package/scripts/session_log.py +325 -0
- package/scripts/usage_ledger.py +230 -218
package/scripts/usage_ledger.py
CHANGED
|
@@ -1,218 +1,230 @@
|
|
|
1
|
-
#!/usr/bin/env python3
|
|
2
|
-
"""A local, append-only record of what actually worked, so the skill improves with use.
|
|
3
|
-
|
|
4
|
-
Donor selection is expensive and mostly repetitive: the same scenario families
|
|
5
|
-
come back, and the same handful of donors keep winning or keep failing for the
|
|
6
|
-
same reasons. Rediscovering that on every run wastes several seconds per
|
|
7
|
-
candidate and, worse, throws away the only real evidence the skill ever gets
|
|
8
|
-
about which donors survive a Packet Tracer open.
|
|
9
|
-
|
|
10
|
-
The ledger records the outcome of each generation and feeds it back into donor
|
|
11
|
-
ranking on the next run. Nothing is inferred or guessed: an entry is written
|
|
12
|
-
only after a real attempt produced a real result.
|
|
13
|
-
|
|
14
|
-
Privacy and safety rules, deliberately strict:
|
|
15
|
-
|
|
16
|
-
- the ledger is local only. It is written under `output/`, which is gitignored,
|
|
17
|
-
and it is never committed, packaged, or transmitted anywhere.
|
|
18
|
-
- prompts are stored as a normalised fingerprint, not verbatim text, so a lab
|
|
19
|
-
description containing names or addresses does not end up on disk.
|
|
20
|
-
- the file is bounded. Old entries are dropped once the cap is reached.
|
|
21
|
-
- a corrupt or unreadable ledger is ignored, never fatal. Learning is an
|
|
22
|
-
optimisation; the skill must work identically with the ledger deleted.
|
|
23
|
-
"""
|
|
24
|
-
|
|
25
|
-
from __future__ import annotations
|
|
26
|
-
|
|
27
|
-
import hashlib
|
|
28
|
-
import json
|
|
29
|
-
import os
|
|
30
|
-
import re
|
|
31
|
-
from collections import defaultdict
|
|
32
|
-
from dataclasses import dataclass, field
|
|
33
|
-
from datetime import datetime, timezone
|
|
34
|
-
from pathlib import Path
|
|
35
|
-
|
|
36
|
-
SKILL_ROOT = Path(__file__).resolve().parents[1]
|
|
37
|
-
DEFAULT_LEDGER_PATH = SKILL_ROOT / "output" / "usage-ledger.jsonl"
|
|
38
|
-
MAX_ENTRIES = 2000
|
|
39
|
-
LEDGER_VERSION = 1
|
|
40
|
-
|
|
41
|
-
OUTCOME_GENERATED_VERIFIED = "generated_verified"
|
|
42
|
-
OUTCOME_GENERATED_UNVERIFIED = "generated_unverified"
|
|
43
|
-
OUTCOME_REFUSED = "refused"
|
|
44
|
-
OUTCOMES = (OUTCOME_GENERATED_VERIFIED, OUTCOME_GENERATED_UNVERIFIED, OUTCOME_REFUSED)
|
|
45
|
-
|
|
46
|
-
# Outcomes that count as evidence a donor works, best first.
|
|
47
|
-
_SUCCESS_WEIGHT = {
|
|
48
|
-
OUTCOME_GENERATED_VERIFIED: 3,
|
|
49
|
-
OUTCOME_GENERATED_UNVERIFIED: 1,
|
|
50
|
-
}
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
def
|
|
64
|
-
""
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
try:
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
return
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
try:
|
|
142
|
-
|
|
143
|
-
except OSError:
|
|
144
|
-
return
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
)
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""A local, append-only record of what actually worked, so the skill improves with use.
|
|
3
|
+
|
|
4
|
+
Donor selection is expensive and mostly repetitive: the same scenario families
|
|
5
|
+
come back, and the same handful of donors keep winning or keep failing for the
|
|
6
|
+
same reasons. Rediscovering that on every run wastes several seconds per
|
|
7
|
+
candidate and, worse, throws away the only real evidence the skill ever gets
|
|
8
|
+
about which donors survive a Packet Tracer open.
|
|
9
|
+
|
|
10
|
+
The ledger records the outcome of each generation and feeds it back into donor
|
|
11
|
+
ranking on the next run. Nothing is inferred or guessed: an entry is written
|
|
12
|
+
only after a real attempt produced a real result.
|
|
13
|
+
|
|
14
|
+
Privacy and safety rules, deliberately strict:
|
|
15
|
+
|
|
16
|
+
- the ledger is local only. It is written under `output/`, which is gitignored,
|
|
17
|
+
and it is never committed, packaged, or transmitted anywhere.
|
|
18
|
+
- prompts are stored as a normalised fingerprint, not verbatim text, so a lab
|
|
19
|
+
description containing names or addresses does not end up on disk.
|
|
20
|
+
- the file is bounded. Old entries are dropped once the cap is reached.
|
|
21
|
+
- a corrupt or unreadable ledger is ignored, never fatal. Learning is an
|
|
22
|
+
optimisation; the skill must work identically with the ledger deleted.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import hashlib
|
|
28
|
+
import json
|
|
29
|
+
import os
|
|
30
|
+
import re
|
|
31
|
+
from collections import defaultdict
|
|
32
|
+
from dataclasses import dataclass, field
|
|
33
|
+
from datetime import datetime, timezone
|
|
34
|
+
from pathlib import Path
|
|
35
|
+
|
|
36
|
+
SKILL_ROOT = Path(__file__).resolve().parents[1]
|
|
37
|
+
DEFAULT_LEDGER_PATH = SKILL_ROOT / "output" / "usage-ledger.jsonl"
|
|
38
|
+
MAX_ENTRIES = 2000
|
|
39
|
+
LEDGER_VERSION = 1
|
|
40
|
+
|
|
41
|
+
OUTCOME_GENERATED_VERIFIED = "generated_verified"
|
|
42
|
+
OUTCOME_GENERATED_UNVERIFIED = "generated_unverified"
|
|
43
|
+
OUTCOME_REFUSED = "refused"
|
|
44
|
+
OUTCOMES = (OUTCOME_GENERATED_VERIFIED, OUTCOME_GENERATED_UNVERIFIED, OUTCOME_REFUSED)
|
|
45
|
+
|
|
46
|
+
# Outcomes that count as evidence a donor works, best first.
|
|
47
|
+
_SUCCESS_WEIGHT = {
|
|
48
|
+
OUTCOME_GENERATED_VERIFIED: 3,
|
|
49
|
+
OUTCOME_GENERATED_UNVERIFIED: 1,
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
# One variable carries two meanings -- a switch and a path -- and only one of
|
|
54
|
+
# them was reading the switch words. `PKT_USAGE_LEDGER=on`, the obvious way to
|
|
55
|
+
# turn learning on, made `ledger_path` treat "on" as a filename and write the
|
|
56
|
+
# ledger to a file called `on` in the working directory. Found by doing exactly
|
|
57
|
+
# that while testing something else, which left an untracked `on` in the
|
|
58
|
+
# repository root. Both readers share the vocabulary now.
|
|
59
|
+
_OFF_WORDS = {"off", "0", "false", "none"}
|
|
60
|
+
_ON_WORDS = {"on", "1", "true"}
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def ledger_path() -> Path:
|
|
64
|
+
override = (os.getenv("PKT_USAGE_LEDGER") or "").strip()
|
|
65
|
+
if override and override.lower() not in (_OFF_WORDS | _ON_WORDS):
|
|
66
|
+
return Path(override).expanduser()
|
|
67
|
+
return DEFAULT_LEDGER_PATH
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def ledger_enabled() -> bool:
|
|
71
|
+
"""Learning is on by default; `PKT_USAGE_LEDGER=off` disables it entirely."""
|
|
72
|
+
return (os.getenv("PKT_USAGE_LEDGER") or "").strip().lower() not in _OFF_WORDS
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def prompt_fingerprint(prompt: str) -> str:
|
|
76
|
+
"""A stable, non-reversible fingerprint of a prompt's *shape*.
|
|
77
|
+
|
|
78
|
+
Digits are collapsed to `#` and case and spacing are normalised, so
|
|
79
|
+
"3 switch 6 pc" and "5 switch 2 pc" share a fingerprint: they are the same
|
|
80
|
+
kind of request, which is exactly the granularity donor reuse needs. The
|
|
81
|
+
result is hashed so no prompt text is ever written to disk.
|
|
82
|
+
"""
|
|
83
|
+
normalised = re.sub(r"\d+", "#", (prompt or "").strip().lower())
|
|
84
|
+
normalised = re.sub(r"[^\w#]+", " ", normalised).strip()
|
|
85
|
+
return hashlib.sha256(normalised.encode("utf-8")).hexdigest()[:16]
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
@dataclass
|
|
89
|
+
class LedgerEntry:
|
|
90
|
+
scenario_family: str
|
|
91
|
+
donor: str
|
|
92
|
+
outcome: str
|
|
93
|
+
prompt_shape: str = ""
|
|
94
|
+
target_version: str = ""
|
|
95
|
+
donor_version: str = ""
|
|
96
|
+
rejected_donors: list[str] = field(default_factory=list)
|
|
97
|
+
rejection_codes: list[str] = field(default_factory=list)
|
|
98
|
+
recorded_at: str = ""
|
|
99
|
+
|
|
100
|
+
def to_json(self) -> dict[str, object]:
|
|
101
|
+
return {
|
|
102
|
+
"v": LEDGER_VERSION,
|
|
103
|
+
"recorded_at": self.recorded_at or datetime.now(timezone.utc).isoformat(timespec="seconds"),
|
|
104
|
+
"scenario_family": self.scenario_family,
|
|
105
|
+
"prompt_shape": self.prompt_shape,
|
|
106
|
+
"donor": self.donor,
|
|
107
|
+
"donor_version": self.donor_version,
|
|
108
|
+
"target_version": self.target_version,
|
|
109
|
+
"outcome": self.outcome,
|
|
110
|
+
"rejected_donors": self.rejected_donors[:20],
|
|
111
|
+
"rejection_codes": self.rejection_codes[:20],
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def record(entry: LedgerEntry, path: Path | None = None) -> bool:
|
|
116
|
+
"""Append one outcome. Returns False if the ledger is off or unwritable."""
|
|
117
|
+
if not ledger_enabled():
|
|
118
|
+
return False
|
|
119
|
+
if entry.outcome not in OUTCOMES:
|
|
120
|
+
raise ValueError(f"unknown outcome: {entry.outcome}")
|
|
121
|
+
|
|
122
|
+
target = path or ledger_path()
|
|
123
|
+
try:
|
|
124
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
125
|
+
with target.open("a", encoding="utf-8") as handle:
|
|
126
|
+
handle.write(json.dumps(entry.to_json(), ensure_ascii=False) + "\n")
|
|
127
|
+
except OSError:
|
|
128
|
+
return False
|
|
129
|
+
|
|
130
|
+
_trim(target)
|
|
131
|
+
return True
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _trim(path: Path) -> None:
|
|
135
|
+
try:
|
|
136
|
+
lines = path.read_text(encoding="utf-8").splitlines()
|
|
137
|
+
except OSError:
|
|
138
|
+
return
|
|
139
|
+
if len(lines) <= MAX_ENTRIES:
|
|
140
|
+
return
|
|
141
|
+
try:
|
|
142
|
+
path.write_text("\n".join(lines[-MAX_ENTRIES:]) + "\n", encoding="utf-8")
|
|
143
|
+
except OSError:
|
|
144
|
+
return
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def load_entries(path: Path | None = None) -> list[dict[str, object]]:
|
|
148
|
+
"""Read the ledger. A damaged file yields whatever lines still parse."""
|
|
149
|
+
target = path or ledger_path()
|
|
150
|
+
if not ledger_enabled() or not target.exists():
|
|
151
|
+
return []
|
|
152
|
+
entries: list[dict[str, object]] = []
|
|
153
|
+
try:
|
|
154
|
+
raw_lines = target.read_text(encoding="utf-8").splitlines()
|
|
155
|
+
except OSError:
|
|
156
|
+
return []
|
|
157
|
+
for line in raw_lines:
|
|
158
|
+
line = line.strip()
|
|
159
|
+
if not line:
|
|
160
|
+
continue
|
|
161
|
+
try:
|
|
162
|
+
parsed = json.loads(line)
|
|
163
|
+
except json.JSONDecodeError:
|
|
164
|
+
continue
|
|
165
|
+
if isinstance(parsed, dict) and parsed.get("donor"):
|
|
166
|
+
entries.append(parsed)
|
|
167
|
+
return entries
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def donor_scores(
|
|
171
|
+
scenario_family: str,
|
|
172
|
+
prompt_shape: str = "",
|
|
173
|
+
path: Path | None = None,
|
|
174
|
+
) -> dict[str, int]:
|
|
175
|
+
"""Learned preference per donor, as `relative_path -> score`.
|
|
176
|
+
|
|
177
|
+
Positive means the donor has produced output for this kind of request
|
|
178
|
+
before; negative means it has been rejected. An exact prompt-shape match
|
|
179
|
+
counts double, because it is stronger evidence than family alone.
|
|
180
|
+
"""
|
|
181
|
+
scores: dict[str, int] = defaultdict(int)
|
|
182
|
+
for entry in load_entries(path):
|
|
183
|
+
if str(entry.get("scenario_family") or "") != scenario_family:
|
|
184
|
+
continue
|
|
185
|
+
multiplier = 2 if prompt_shape and entry.get("prompt_shape") == prompt_shape else 1
|
|
186
|
+
|
|
187
|
+
donor = str(entry.get("donor") or "")
|
|
188
|
+
weight = _SUCCESS_WEIGHT.get(str(entry.get("outcome") or ""), 0)
|
|
189
|
+
if donor and weight:
|
|
190
|
+
scores[donor] += weight * multiplier
|
|
191
|
+
|
|
192
|
+
for rejected in entry.get("rejected_donors") or []:
|
|
193
|
+
name = str(rejected)
|
|
194
|
+
if name:
|
|
195
|
+
scores[name] -= multiplier
|
|
196
|
+
return dict(scores)
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def summary(path: Path | None = None) -> dict[str, object]:
|
|
200
|
+
"""Human-facing view of what the skill has learned so far."""
|
|
201
|
+
entries = load_entries(path)
|
|
202
|
+
by_outcome: dict[str, int] = defaultdict(int)
|
|
203
|
+
by_family: dict[str, int] = defaultdict(int)
|
|
204
|
+
proven: dict[str, int] = defaultdict(int)
|
|
205
|
+
for entry in entries:
|
|
206
|
+
outcome = str(entry.get("outcome") or "")
|
|
207
|
+
by_outcome[outcome] += 1
|
|
208
|
+
by_family[str(entry.get("scenario_family") or "unknown")] += 1
|
|
209
|
+
if outcome in _SUCCESS_WEIGHT:
|
|
210
|
+
proven[str(entry.get("donor") or "")] += _SUCCESS_WEIGHT[outcome]
|
|
211
|
+
return {
|
|
212
|
+
"ledger_path": str(path or ledger_path()),
|
|
213
|
+
"enabled": ledger_enabled(),
|
|
214
|
+
"entry_count": len(entries),
|
|
215
|
+
"outcomes": dict(by_outcome),
|
|
216
|
+
"scenario_families": dict(by_family),
|
|
217
|
+
"proven_donors": sorted(
|
|
218
|
+
({"donor": donor, "score": score} for donor, score in proven.items() if score > 0),
|
|
219
|
+
key=lambda item: (-int(item["score"]), str(item["donor"])),
|
|
220
|
+
)[:10],
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def main() -> int:
|
|
225
|
+
print(json.dumps(summary(), ensure_ascii=False, indent=2))
|
|
226
|
+
return 0
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
if __name__ == "__main__":
|
|
230
|
+
raise SystemExit(main())
|