firefly-judgment 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- firefly_judgment-0.1.0/LICENSE +21 -0
- firefly_judgment-0.1.0/PKG-INFO +276 -0
- firefly_judgment-0.1.0/README.md +250 -0
- firefly_judgment-0.1.0/firefly/__init__.py +6 -0
- firefly_judgment-0.1.0/firefly/bootstrap.py +38 -0
- firefly_judgment-0.1.0/firefly/bus/__init__.py +17 -0
- firefly_judgment-0.1.0/firefly/bus/audit.py +90 -0
- firefly_judgment-0.1.0/firefly/bus/bus.py +249 -0
- firefly_judgment-0.1.0/firefly/bus/cache.py +55 -0
- firefly_judgment-0.1.0/firefly/bus/fallback.py +106 -0
- firefly_judgment-0.1.0/firefly/bus/metrics.py +67 -0
- firefly_judgment-0.1.0/firefly/core/__init__.py +13 -0
- firefly_judgment-0.1.0/firefly/core/calibrator.py +77 -0
- firefly_judgment-0.1.0/firefly/core/encoder.py +126 -0
- firefly_judgment-0.1.0/firefly/core/firefly_core.py +151 -0
- firefly_judgment-0.1.0/firefly/core/heads.py +96 -0
- firefly_judgment-0.1.0/firefly/core/scorer.py +192 -0
- firefly_judgment-0.1.0/firefly/evolution/__init__.py +54 -0
- firefly_judgment-0.1.0/firefly/evolution/ab.py +130 -0
- firefly_judgment-0.1.0/firefly/evolution/change_detector.py +227 -0
- firefly_judgment-0.1.0/firefly/evolution/data_synthesizer.py +202 -0
- firefly_judgment-0.1.0/firefly/evolution/discovery.py +247 -0
- firefly_judgment-0.1.0/firefly/evolution/engine.py +496 -0
- firefly_judgment-0.1.0/firefly/evolution/forgetter.py +51 -0
- firefly_judgment-0.1.0/firefly/evolution/publisher.py +273 -0
- firefly_judgment-0.1.0/firefly/evolution/replay.py +87 -0
- firefly_judgment-0.1.0/firefly/evolution/samples.py +58 -0
- firefly_judgment-0.1.0/firefly/evolution/shadow.py +111 -0
- firefly_judgment-0.1.0/firefly/evolution/trainer.py +112 -0
- firefly_judgment-0.1.0/firefly/graph/__init__.py +4 -0
- firefly_judgment-0.1.0/firefly/graph/tool_graph.py +187 -0
- firefly_judgment-0.1.0/firefly/nodes/__init__.py +16 -0
- firefly_judgment-0.1.0/firefly/nodes/base.py +143 -0
- firefly_judgment-0.1.0/firefly/nodes/gate.py +207 -0
- firefly_judgment-0.1.0/firefly/nodes/loop.py +115 -0
- firefly_judgment-0.1.0/firefly/nodes/memory.py +136 -0
- firefly_judgment-0.1.0/firefly/nodes/meta.py +119 -0
- firefly_judgment-0.1.0/firefly/nodes/router.py +64 -0
- firefly_judgment-0.1.0/firefly/nodes/schedule.py +76 -0
- firefly_judgment-0.1.0/firefly/nodes/verify.py +99 -0
- firefly_judgment-0.1.0/firefly/protocol/__init__.py +38 -0
- firefly_judgment-0.1.0/firefly/protocol/models.py +469 -0
- firefly_judgment-0.1.0/firefly/protocol/schemas/decision_log.schema.json +24 -0
- firefly_judgment-0.1.0/firefly/protocol/schemas/decision_request.schema.json +100 -0
- firefly_judgment-0.1.0/firefly/protocol/schemas/decision_response.schema.json +74 -0
- firefly_judgment-0.1.0/firefly/protocol/schemas/tool_spec.schema.json +42 -0
- firefly_judgment-0.1.0/firefly/protocol/schemas.py +27 -0
- firefly_judgment-0.1.0/firefly/protocol/validation.py +89 -0
- firefly_judgment-0.1.0/firefly/protocol/version.py +5 -0
- firefly_judgment-0.1.0/firefly/sdk/__init__.py +4 -0
- firefly_judgment-0.1.0/firefly/sdk/client.py +160 -0
- firefly_judgment-0.1.0/firefly/server/__init__.py +4 -0
- firefly_judgment-0.1.0/firefly/server/http.py +124 -0
- firefly_judgment-0.1.0/firefly_judgment.egg-info/PKG-INFO +276 -0
- firefly_judgment-0.1.0/firefly_judgment.egg-info/SOURCES.txt +68 -0
- firefly_judgment-0.1.0/firefly_judgment.egg-info/dependency_links.txt +1 -0
- firefly_judgment-0.1.0/firefly_judgment.egg-info/entry_points.txt +2 -0
- firefly_judgment-0.1.0/firefly_judgment.egg-info/requires.txt +4 -0
- firefly_judgment-0.1.0/firefly_judgment.egg-info/top_level.txt +1 -0
- firefly_judgment-0.1.0/pyproject.toml +50 -0
- firefly_judgment-0.1.0/setup.cfg +4 -0
- firefly_judgment-0.1.0/tests/test_bus.py +207 -0
- firefly_judgment-0.1.0/tests/test_core.py +115 -0
- firefly_judgment-0.1.0/tests/test_evolution.py +705 -0
- firefly_judgment-0.1.0/tests/test_graph.py +71 -0
- firefly_judgment-0.1.0/tests/test_nodes.py +129 -0
- firefly_judgment-0.1.0/tests/test_protocol.py +88 -0
- firefly_judgment-0.1.0/tests/test_sdk_server.py +68 -0
- firefly_judgment-0.1.0/tests/test_swarm.py +267 -0
- firefly_judgment-0.1.0//350/220/244/357/274/210Firefly/357/274/211/346/236/266/346/236/204/350/256/276/350/256/241/346/226/271/346/241/210.md +842 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Firefly
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,276 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: firefly-judgment
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: 萤 (Firefly): 可嵌入 Agent Harness 的自成长毫秒级判断层
|
|
5
|
+
Author: Firefly
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Keywords: llm,agent,decision-engine,tool-routing,safety-gate,online-learning,sidecar
|
|
8
|
+
Classifier: Development Status :: 4 - Beta
|
|
9
|
+
Classifier: Intended Audience :: Developers
|
|
10
|
+
Classifier: Operating System :: OS Independent
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
18
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
19
|
+
Requires-Python: >=3.9
|
|
20
|
+
Description-Content-Type: text/markdown; charset=utf-8
|
|
21
|
+
License-File: LICENSE
|
|
22
|
+
Provides-Extra: dev
|
|
23
|
+
Requires-Dist: build>=1.2; extra == "dev"
|
|
24
|
+
Requires-Dist: twine>=5.0; extra == "dev"
|
|
25
|
+
Dynamic: license-file
|
|
26
|
+
|
|
27
|
+
# 萤(Firefly)—— 毫秒级判断层
|
|
28
|
+
|
|
29
|
+
萤是一个给 LLM Agent 用的**判断层**:把 Agent 里大量"重复、高频、可学习"的决策(选哪个工具、要不要拦截、用哪个模型、这步谁来判)从昂贵的 LLM 推理下沉为毫秒级的小模型判断;萤不确定时自动回退给 LLM/用户,LLM 的判断结果自动回流为训练数据——**越用越准,越用越省**。
|
|
30
|
+
|
|
31
|
+
```
|
|
32
|
+
LLM Agent 决策点
|
|
33
|
+
│ 低置信度 / 复杂语义 → System 2:LLM / 用户(结果回流训练)
|
|
34
|
+
▼
|
|
35
|
+
萤(System 1):毫秒级 · 可学习 · 可解释 · 可审计 · 可回滚
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
## 能力总览
|
|
39
|
+
|
|
40
|
+
| 阶段 | 能力 |
|
|
41
|
+
|---|---|
|
|
42
|
+
| P0 判断内核 | 工具路由萤、安全守门萤(硬规则不可学习覆盖)、在线自学习、置信度校准、反事实解释、审计哈希链、HTTP 边车 |
|
|
43
|
+
| P1 进化引擎 | 新工具冷启动(注册即学会)、回放防遗忘、影子评估 → A/B 灰度 → 全量发布 → 一键回滚、行为漂移自愈、淘汰遗忘、重启自动恢复、历史数据批量导入 |
|
|
44
|
+
| P2 萤群 | 校验萤、循环萤(防死循环硬规则)、调度萤(模型/推理力度)、记忆萤、元决策萤(决定"这步谁来做")、**从日志自动发现决策点** |
|
|
45
|
+
|
|
46
|
+
零第三方依赖,纯 Python 标准库,`pip install` 即用。
|
|
47
|
+
|
|
48
|
+
## 安装
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
# 方式一:从 PyPI 安装(推荐,进程内调用,零网络开销)
|
|
52
|
+
pip install firefly-judgment
|
|
53
|
+
# 之后其他项目里直接 from firefly import ...
|
|
54
|
+
|
|
55
|
+
# 本地开发安装(在本仓库根目录执行):pip install -e .
|
|
56
|
+
|
|
57
|
+
# 方式二:HTTP 边车(跨语言/跨进程共享同一个萤)
|
|
58
|
+
python -m firefly.server.http # 默认 0.0.0.0:8000
|
|
59
|
+
# 任何语言通过 /v1/decide、/v1/feedback、/v1/tools、/v1/metrics 接入
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
## 快速开始(3 分钟)
|
|
63
|
+
|
|
64
|
+
```python
|
|
65
|
+
from firefly.bootstrap import build_bus
|
|
66
|
+
from firefly.sdk import FireflyClient
|
|
67
|
+
from firefly.protocol import ToolSpec
|
|
68
|
+
|
|
69
|
+
# llm_decider:萤置信度不足时的 System 2(签名 (request, response) -> 选项id)
|
|
70
|
+
def my_llm(request, response):
|
|
71
|
+
return call_your_llm(request.state.task) # 返回它认为对的选项 id
|
|
72
|
+
|
|
73
|
+
bus = build_bus(llm_decider=my_llm)
|
|
74
|
+
firefly = FireflyClient.in_process(bus)
|
|
75
|
+
|
|
76
|
+
# 注册你的工具(动态选项空间:注册即可被路由,无需重启)
|
|
77
|
+
firefly.register_tool(ToolSpec(
|
|
78
|
+
id="search", name="search",
|
|
79
|
+
description="search the web for external information retrieval",
|
|
80
|
+
tags=["retrieval"], cost=0.001, latency_ms=500, risk="low",
|
|
81
|
+
success_rate=0.92,
|
|
82
|
+
))
|
|
83
|
+
firefly.register_tool(ToolSpec(
|
|
84
|
+
id="read_file", name="read_file",
|
|
85
|
+
description="read a local file from disk",
|
|
86
|
+
tags=["file"], cost=0.0, latency_ms=2, risk="low",
|
|
87
|
+
success_rate=0.98,
|
|
88
|
+
))
|
|
89
|
+
|
|
90
|
+
# 决策 → 执行 → 反馈(闭环的关键:把结果喂回来)
|
|
91
|
+
decision = firefly.decide(
|
|
92
|
+
node="tool_router",
|
|
93
|
+
state={"task": "查一下今天的科技新闻"}, # 中英文都能理解
|
|
94
|
+
options=[t.to_option() for t in firefly.list_tools()],
|
|
95
|
+
)
|
|
96
|
+
if not decision.fallback_used:
|
|
97
|
+
result = run_tool(decision.choice) # 你的工具执行逻辑
|
|
98
|
+
firefly.feedback(decision.trace_id,
|
|
99
|
+
"success" if result.ok else "failure")
|
|
100
|
+
# 若 fallback_used=True:decision.choice 是 LLM 给的答案,
|
|
101
|
+
# 照常执行;feedback 时该答案会作为教师信号自动训练萤
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
注意前几次决策置信度不足属正常——**萤核是随机初始化的在线学习器,需要预热**。喂 10~20 次反馈后即可形成稳定判断力;有历史数据可用批量导入(见下),几秒完成预热。
|
|
105
|
+
|
|
106
|
+
## 怎么结合到 LLM 项目
|
|
107
|
+
|
|
108
|
+
一个典型 Agent 的每一步都在问 LLM。下面是萤接管各决策点的接入姿势——每个决策点都是独立节点,按需取用:
|
|
109
|
+
|
|
110
|
+
### 1. 工具路由(最常用):agent 选工具前问萤
|
|
111
|
+
|
|
112
|
+
```python
|
|
113
|
+
decision = firefly.decide(
|
|
114
|
+
node="tool_router",
|
|
115
|
+
state={"task": user_input},
|
|
116
|
+
options=[t.to_option() for t in my_tools],
|
|
117
|
+
)
|
|
118
|
+
tool = decision.choice # 毫秒级;不确定时自动回退 LLM(fallback_used=True)
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
### 2. 安全守门:执行工具前最后一道闸(硬规则不可被学习绕过)
|
|
122
|
+
|
|
123
|
+
```python
|
|
124
|
+
gate = firefly.raw_decide({
|
|
125
|
+
"node": "safety_gate",
|
|
126
|
+
"options": [{"id": tool_call_name, "features": {"description": tool_args_text}}],
|
|
127
|
+
"state": {"task": user_input, "risk": "high"},
|
|
128
|
+
})
|
|
129
|
+
if gate.choice == "block": return refuse()
|
|
130
|
+
if gate.choice == "ask": return ask_human()
|
|
131
|
+
# gate.certificate 非空 = 硬规则裁决(如拦截 rm -rf /),带审计证书
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
### 3. 调度萤:给任务选模型档位(省钱的关键)
|
|
135
|
+
|
|
136
|
+
```python
|
|
137
|
+
r = firefly.decide(node="scheduler",
|
|
138
|
+
state={"task": user_input},
|
|
139
|
+
fallback="none") # 档位选择禁用外部回退
|
|
140
|
+
model = MY_MODELS[r.choice] # 默认三档 fast / balanced / deep
|
|
141
|
+
# 也可用 raw_decide 传入你自己的候选模型列表 options=[{"id": "gpt-4o", ...}, ...]
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
### 4. 元决策萤:这一步到底让谁来做
|
|
145
|
+
|
|
146
|
+
```python
|
|
147
|
+
r = firefly.decide(node="meta_controller",
|
|
148
|
+
state={"task": user_input, "risk": "high"},
|
|
149
|
+
fallback="none") # 元层自身不再回退,避免循环
|
|
150
|
+
# r.choice ∈ firefly / llm / user / explore
|
|
151
|
+
# 硬规则:未知未知 → 强制 user;高风险 → 禁止 explore(不可被学习绕过)
|
|
152
|
+
```
|
|
153
|
+
|
|
154
|
+
### 5. 校验萤 / 循环萤 / 记忆萤
|
|
155
|
+
|
|
156
|
+
```python
|
|
157
|
+
# 校验工具产出(置信度不足时其本身就输出 uncertain,无需外部回退)
|
|
158
|
+
v = firefly.raw_decide({"node": "verifier",
|
|
159
|
+
"state": {"task": task, "extra": {"output": tool_output}},
|
|
160
|
+
"constraints": {"fallback": {"mode": "none"}}})
|
|
161
|
+
# v.choice ∈ pass / fail / uncertain
|
|
162
|
+
|
|
163
|
+
# 控制重试循环(attempt 达上限时硬规则强制 stop,防死循环)
|
|
164
|
+
loop = firefly.raw_decide({"node": "loop_controller",
|
|
165
|
+
"state": {"task": task, "extra": {
|
|
166
|
+
"attempt": 3, "last_error": "connection timeout", "max_attempts": 5}},
|
|
167
|
+
"constraints": {"fallback": {"mode": "none"}}})
|
|
168
|
+
# loop.choice ∈ continue / retry / switch / stop
|
|
169
|
+
|
|
170
|
+
# 整理长期记忆(对每条记忆 保留/压缩/丢弃)
|
|
171
|
+
m = firefly.raw_decide({"node": "memory_curator",
|
|
172
|
+
"state": {"task": "整理记忆", "extra": {"memory_items": memories}},
|
|
173
|
+
"constraints": {"fallback": {"mode": "none"}}})
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
> P2 五节点(校验/循环/调度/记忆/元决策)是固定小动作空间的元判断,建议统一 `fallback none`:它们的输出空间内已含"不确定/询问"等保守选项,被外部 LLM 回退改写反而会破坏语义。工具路由(`tool_router`)则相反——保留回退,让 LLM 兜底正是设计意图。
|
|
177
|
+
|
|
178
|
+
### 6. 进化引擎:新工具注册即学会,历史数据秒级预热
|
|
179
|
+
|
|
180
|
+
```python
|
|
181
|
+
from firefly.evolution import EvolutionEngine
|
|
182
|
+
|
|
183
|
+
engine = EvolutionEngine(
|
|
184
|
+
bus,
|
|
185
|
+
model_dir=".firefly_models", # 版本/回放缓冲落盘,重启自动恢复
|
|
186
|
+
)
|
|
187
|
+
engine.manage_defaults() # 纳管全部节点(影子/A/B/灰度/回滚)
|
|
188
|
+
|
|
189
|
+
# 批量导入历史决策记录:直接训练线上核,几秒完成预热
|
|
190
|
+
engine.import_history([
|
|
191
|
+
{"task": "查一下今天的科技新闻", "target_id": "search"},
|
|
192
|
+
{"task": "读取本地配置文件", "target_id": "read_file"},
|
|
193
|
+
# ... 几百条历史 (任务, 正确工具) 对
|
|
194
|
+
])
|
|
195
|
+
|
|
196
|
+
# 之后:新工具注册 → 后台自动 合成数据→训练→影子→灰度→全量,全程不中断
|
|
197
|
+
bus.register_tool(new_tool_spec)
|
|
198
|
+
engine.start(interval=30) # 后台 tick;也可手动 engine.tick()
|
|
199
|
+
engine.promote("tool_router") # 影子达标后推进:灰度 10%→50%→全量
|
|
200
|
+
engine.rollback("tool_router") # 出问题一键回滚 stable 基线
|
|
201
|
+
```
|
|
202
|
+
|
|
203
|
+
### 7. 从 LLM 调用日志自动发现决策点(把重复的 LLM 判断固化成萤)
|
|
204
|
+
|
|
205
|
+
如果你的 LLM 项目里有一类判断反复出现(如意图分类、请求分派),把日志喂给挖掘器:
|
|
206
|
+
|
|
207
|
+
```python
|
|
208
|
+
from firefly.evolution import DecisionPointMiner, register_decision_point
|
|
209
|
+
|
|
210
|
+
logs = [{"point": "intent", # 同一决策点打同一个标签
|
|
211
|
+
"input": user_text,
|
|
212
|
+
"output": llm_answer_label} # 输出需为短标签(可枚举)
|
|
213
|
+
for user_text, llm_answer_label in your_history]
|
|
214
|
+
|
|
215
|
+
for s in DecisionPointMiner().mine(logs):
|
|
216
|
+
print(s.name, s.action_space, f"{s.frequency} 条日志,熵 {s.entropy:.2f}")
|
|
217
|
+
# 人工审核通过后:
|
|
218
|
+
register_decision_point(bus, s) # 注册新萤节点
|
|
219
|
+
engine.manage(s.name)
|
|
220
|
+
engine.import_history([{"task": x.task, "target_id": x.target_id}
|
|
221
|
+
for x in s.samples]) # 灌入初始样本
|
|
222
|
+
```
|
|
223
|
+
|
|
224
|
+
## 反馈闭环:outcome 从哪来
|
|
225
|
+
|
|
226
|
+
`firefly.feedback(trace_id, outcome, corrected_choice=None, extra=None)`:
|
|
227
|
+
|
|
228
|
+
| outcome | 含义 | 学习效果 |
|
|
229
|
+
|---|---|---|
|
|
230
|
+
| `success` | 工具执行成功 / 用户采纳 | 强化本次选择 |
|
|
231
|
+
| `failure` / `timeout` / `user_rejected` | 执行失败 | 惩罚本次选择 |
|
|
232
|
+
| `user_corrected` | 用户/LLM 给出正确答案 | 以 1.2 倍权重学习 `corrected_choice` |
|
|
233
|
+
|
|
234
|
+
任何节点的决策都走同一个闭环;守门硬规则、循环上限、元决策安全规则带证书,反馈无法改变其结论。
|
|
235
|
+
|
|
236
|
+
## HTTP 边车(跨语言接入)
|
|
237
|
+
|
|
238
|
+
```bash
|
|
239
|
+
python -m firefly.server.http --host 0.0.0.0 --port 8000
|
|
240
|
+
```
|
|
241
|
+
|
|
242
|
+
```bash
|
|
243
|
+
curl -X POST localhost:8000/v1/decide -H "Content-Type: application/json" -d '{
|
|
244
|
+
"node": "tool_router",
|
|
245
|
+
"state": {"task": "查一下今天的科技新闻"},
|
|
246
|
+
"options": [{"id": "search", "features": {"description": "search the web"}}]
|
|
247
|
+
}'
|
|
248
|
+
```
|
|
249
|
+
|
|
250
|
+
端点:`POST /v1/decide`、`POST /v1/feedback`、`POST /v1/tools`、`GET /v1/tools`、`GET /v1/metrics`、`GET /health`。
|
|
251
|
+
|
|
252
|
+
## API 速查
|
|
253
|
+
|
|
254
|
+
| 对象 | 关键方法 |
|
|
255
|
+
|---|---|
|
|
256
|
+
| `FireflyClient` | `in_process(bus)` / `http(url)` / `decide()` / `raw_decide()` / `feedback()` / `register_tool()` / `metrics()` |
|
|
257
|
+
| `build_bus()` | `llm_decider`、`user_decider`(回退执行者)、`swarm_nodes=True`(挂 P2 全部节点)、`gate_policy`(自定义守门规则) |
|
|
258
|
+
| `EvolutionEngine` | `manage_defaults()` / `manage(name)` / `start()` / `tick()` / `promote()` / `rollback()` / `import_history()` |
|
|
259
|
+
| `DecisionPointMiner` | `mine(logs) -> [DecisionPointSuggestion]` |
|
|
260
|
+
| `register_decision_point` | `(bus, suggestion)` 注册新节点 |
|
|
261
|
+
|
|
262
|
+
## 使用注意事项
|
|
263
|
+
|
|
264
|
+
1. **必须预热**:核随机初始化,冷启动决策接近随机。用 `import_history` 批量导入历史数据是最快的预热方式;没有历史数据就上线初期让 LLM 当老师(回退闭环会自动积累训练数据)。
|
|
265
|
+
2. **反馈必须接**:不接 `feedback` 萤就不会进化。工具执行结果、用户采纳/纠正都是现成的信号源。
|
|
266
|
+
3. **学习期隔离回退**:批量训练/演示时传 `constraints: {"fallback": {"mode": "none"}}`,避免低置信度期间决策被 explore 劫持干扰训练信号。
|
|
267
|
+
4. **语义边界**:萤的特征是哈希级语义(同义词汇能对上,深层推理不行)。需要真正理解的判断本来就该回退 LLM——这正是系统设计,调好 `fallback.threshold`(默认 0.6)即可。
|
|
268
|
+
5. **重启恢复**:给 `EvolutionEngine` 传 `model_dir`,版本历史、回放缓冲、active 萤核重启后自动恢复。
|
|
269
|
+
6. **进程常驻**:审计日志与缓存目前为内存态,适合常驻服务;HTTP 边车模式天然满足。
|
|
270
|
+
|
|
271
|
+
## 更多示例
|
|
272
|
+
|
|
273
|
+
- [examples/quickstart.py](examples/quickstart.py) — P0 全流程(路由/学习/守门/指标)
|
|
274
|
+
- [examples/evolution_demo.py](examples/evolution_demo.py) — P1 进化引擎全链路(导入历史→冷启动→影子→灰度→发布→漂移自愈→回滚)
|
|
275
|
+
- [examples/swarm_demo.py](examples/swarm_demo.py) — P2 萤群 + 决策点自动发现
|
|
276
|
+
- [tests/](tests/) — 94 个测试用例,是最好的用法文档
|
|
@@ -0,0 +1,250 @@
|
|
|
1
|
+
# 萤(Firefly)—— 毫秒级判断层
|
|
2
|
+
|
|
3
|
+
萤是一个给 LLM Agent 用的**判断层**:把 Agent 里大量"重复、高频、可学习"的决策(选哪个工具、要不要拦截、用哪个模型、这步谁来判)从昂贵的 LLM 推理下沉为毫秒级的小模型判断;萤不确定时自动回退给 LLM/用户,LLM 的判断结果自动回流为训练数据——**越用越准,越用越省**。
|
|
4
|
+
|
|
5
|
+
```
|
|
6
|
+
LLM Agent 决策点
|
|
7
|
+
│ 低置信度 / 复杂语义 → System 2:LLM / 用户(结果回流训练)
|
|
8
|
+
▼
|
|
9
|
+
萤(System 1):毫秒级 · 可学习 · 可解释 · 可审计 · 可回滚
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
## 能力总览
|
|
13
|
+
|
|
14
|
+
| 阶段 | 能力 |
|
|
15
|
+
|---|---|
|
|
16
|
+
| P0 判断内核 | 工具路由萤、安全守门萤(硬规则不可学习覆盖)、在线自学习、置信度校准、反事实解释、审计哈希链、HTTP 边车 |
|
|
17
|
+
| P1 进化引擎 | 新工具冷启动(注册即学会)、回放防遗忘、影子评估 → A/B 灰度 → 全量发布 → 一键回滚、行为漂移自愈、淘汰遗忘、重启自动恢复、历史数据批量导入 |
|
|
18
|
+
| P2 萤群 | 校验萤、循环萤(防死循环硬规则)、调度萤(模型/推理力度)、记忆萤、元决策萤(决定"这步谁来做")、**从日志自动发现决策点** |
|
|
19
|
+
|
|
20
|
+
零第三方依赖,纯 Python 标准库,`pip install` 即用。
|
|
21
|
+
|
|
22
|
+
## 安装
|
|
23
|
+
|
|
24
|
+
```bash
|
|
25
|
+
# 方式一:从 PyPI 安装(推荐,进程内调用,零网络开销)
|
|
26
|
+
pip install firefly-judgment
|
|
27
|
+
# 之后其他项目里直接 from firefly import ...
|
|
28
|
+
|
|
29
|
+
# 本地开发安装(在本仓库根目录执行):pip install -e .
|
|
30
|
+
|
|
31
|
+
# 方式二:HTTP 边车(跨语言/跨进程共享同一个萤)
|
|
32
|
+
python -m firefly.server.http # 默认 0.0.0.0:8000
|
|
33
|
+
# 任何语言通过 /v1/decide、/v1/feedback、/v1/tools、/v1/metrics 接入
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
## 快速开始(3 分钟)
|
|
37
|
+
|
|
38
|
+
```python
|
|
39
|
+
from firefly.bootstrap import build_bus
|
|
40
|
+
from firefly.sdk import FireflyClient
|
|
41
|
+
from firefly.protocol import ToolSpec
|
|
42
|
+
|
|
43
|
+
# llm_decider:萤置信度不足时的 System 2(签名 (request, response) -> 选项id)
|
|
44
|
+
def my_llm(request, response):
|
|
45
|
+
return call_your_llm(request.state.task) # 返回它认为对的选项 id
|
|
46
|
+
|
|
47
|
+
bus = build_bus(llm_decider=my_llm)
|
|
48
|
+
firefly = FireflyClient.in_process(bus)
|
|
49
|
+
|
|
50
|
+
# 注册你的工具(动态选项空间:注册即可被路由,无需重启)
|
|
51
|
+
firefly.register_tool(ToolSpec(
|
|
52
|
+
id="search", name="search",
|
|
53
|
+
description="search the web for external information retrieval",
|
|
54
|
+
tags=["retrieval"], cost=0.001, latency_ms=500, risk="low",
|
|
55
|
+
success_rate=0.92,
|
|
56
|
+
))
|
|
57
|
+
firefly.register_tool(ToolSpec(
|
|
58
|
+
id="read_file", name="read_file",
|
|
59
|
+
description="read a local file from disk",
|
|
60
|
+
tags=["file"], cost=0.0, latency_ms=2, risk="low",
|
|
61
|
+
success_rate=0.98,
|
|
62
|
+
))
|
|
63
|
+
|
|
64
|
+
# 决策 → 执行 → 反馈(闭环的关键:把结果喂回来)
|
|
65
|
+
decision = firefly.decide(
|
|
66
|
+
node="tool_router",
|
|
67
|
+
state={"task": "查一下今天的科技新闻"}, # 中英文都能理解
|
|
68
|
+
options=[t.to_option() for t in firefly.list_tools()],
|
|
69
|
+
)
|
|
70
|
+
if not decision.fallback_used:
|
|
71
|
+
result = run_tool(decision.choice) # 你的工具执行逻辑
|
|
72
|
+
firefly.feedback(decision.trace_id,
|
|
73
|
+
"success" if result.ok else "failure")
|
|
74
|
+
# 若 fallback_used=True:decision.choice 是 LLM 给的答案,
|
|
75
|
+
# 照常执行;feedback 时该答案会作为教师信号自动训练萤
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
注意前几次决策置信度不足属正常——**萤核是随机初始化的在线学习器,需要预热**。喂 10~20 次反馈后即可形成稳定判断力;有历史数据可用批量导入(见下),几秒完成预热。
|
|
79
|
+
|
|
80
|
+
## 怎么结合到 LLM 项目
|
|
81
|
+
|
|
82
|
+
一个典型 Agent 的每一步都在问 LLM。下面是萤接管各决策点的接入姿势——每个决策点都是独立节点,按需取用:
|
|
83
|
+
|
|
84
|
+
### 1. 工具路由(最常用):agent 选工具前问萤
|
|
85
|
+
|
|
86
|
+
```python
|
|
87
|
+
decision = firefly.decide(
|
|
88
|
+
node="tool_router",
|
|
89
|
+
state={"task": user_input},
|
|
90
|
+
options=[t.to_option() for t in my_tools],
|
|
91
|
+
)
|
|
92
|
+
tool = decision.choice # 毫秒级;不确定时自动回退 LLM(fallback_used=True)
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
### 2. 安全守门:执行工具前最后一道闸(硬规则不可被学习绕过)
|
|
96
|
+
|
|
97
|
+
```python
|
|
98
|
+
gate = firefly.raw_decide({
|
|
99
|
+
"node": "safety_gate",
|
|
100
|
+
"options": [{"id": tool_call_name, "features": {"description": tool_args_text}}],
|
|
101
|
+
"state": {"task": user_input, "risk": "high"},
|
|
102
|
+
})
|
|
103
|
+
if gate.choice == "block": return refuse()
|
|
104
|
+
if gate.choice == "ask": return ask_human()
|
|
105
|
+
# gate.certificate 非空 = 硬规则裁决(如拦截 rm -rf /),带审计证书
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
### 3. 调度萤:给任务选模型档位(省钱的关键)
|
|
109
|
+
|
|
110
|
+
```python
|
|
111
|
+
r = firefly.decide(node="scheduler",
|
|
112
|
+
state={"task": user_input},
|
|
113
|
+
fallback="none") # 档位选择禁用外部回退
|
|
114
|
+
model = MY_MODELS[r.choice] # 默认三档 fast / balanced / deep
|
|
115
|
+
# 也可用 raw_decide 传入你自己的候选模型列表 options=[{"id": "gpt-4o", ...}, ...]
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
### 4. 元决策萤:这一步到底让谁来做
|
|
119
|
+
|
|
120
|
+
```python
|
|
121
|
+
r = firefly.decide(node="meta_controller",
|
|
122
|
+
state={"task": user_input, "risk": "high"},
|
|
123
|
+
fallback="none") # 元层自身不再回退,避免循环
|
|
124
|
+
# r.choice ∈ firefly / llm / user / explore
|
|
125
|
+
# 硬规则:未知未知 → 强制 user;高风险 → 禁止 explore(不可被学习绕过)
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
### 5. 校验萤 / 循环萤 / 记忆萤
|
|
129
|
+
|
|
130
|
+
```python
|
|
131
|
+
# 校验工具产出(置信度不足时其本身就输出 uncertain,无需外部回退)
|
|
132
|
+
v = firefly.raw_decide({"node": "verifier",
|
|
133
|
+
"state": {"task": task, "extra": {"output": tool_output}},
|
|
134
|
+
"constraints": {"fallback": {"mode": "none"}}})
|
|
135
|
+
# v.choice ∈ pass / fail / uncertain
|
|
136
|
+
|
|
137
|
+
# 控制重试循环(attempt 达上限时硬规则强制 stop,防死循环)
|
|
138
|
+
loop = firefly.raw_decide({"node": "loop_controller",
|
|
139
|
+
"state": {"task": task, "extra": {
|
|
140
|
+
"attempt": 3, "last_error": "connection timeout", "max_attempts": 5}},
|
|
141
|
+
"constraints": {"fallback": {"mode": "none"}}})
|
|
142
|
+
# loop.choice ∈ continue / retry / switch / stop
|
|
143
|
+
|
|
144
|
+
# 整理长期记忆(对每条记忆 保留/压缩/丢弃)
|
|
145
|
+
m = firefly.raw_decide({"node": "memory_curator",
|
|
146
|
+
"state": {"task": "整理记忆", "extra": {"memory_items": memories}},
|
|
147
|
+
"constraints": {"fallback": {"mode": "none"}}})
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
> P2 五节点(校验/循环/调度/记忆/元决策)是固定小动作空间的元判断,建议统一 `fallback none`:它们的输出空间内已含"不确定/询问"等保守选项,被外部 LLM 回退改写反而会破坏语义。工具路由(`tool_router`)则相反——保留回退,让 LLM 兜底正是设计意图。
|
|
151
|
+
|
|
152
|
+
### 6. 进化引擎:新工具注册即学会,历史数据秒级预热
|
|
153
|
+
|
|
154
|
+
```python
|
|
155
|
+
from firefly.evolution import EvolutionEngine
|
|
156
|
+
|
|
157
|
+
engine = EvolutionEngine(
|
|
158
|
+
bus,
|
|
159
|
+
model_dir=".firefly_models", # 版本/回放缓冲落盘,重启自动恢复
|
|
160
|
+
)
|
|
161
|
+
engine.manage_defaults() # 纳管全部节点(影子/A/B/灰度/回滚)
|
|
162
|
+
|
|
163
|
+
# 批量导入历史决策记录:直接训练线上核,几秒完成预热
|
|
164
|
+
engine.import_history([
|
|
165
|
+
{"task": "查一下今天的科技新闻", "target_id": "search"},
|
|
166
|
+
{"task": "读取本地配置文件", "target_id": "read_file"},
|
|
167
|
+
# ... 几百条历史 (任务, 正确工具) 对
|
|
168
|
+
])
|
|
169
|
+
|
|
170
|
+
# 之后:新工具注册 → 后台自动 合成数据→训练→影子→灰度→全量,全程不中断
|
|
171
|
+
bus.register_tool(new_tool_spec)
|
|
172
|
+
engine.start(interval=30) # 后台 tick;也可手动 engine.tick()
|
|
173
|
+
engine.promote("tool_router") # 影子达标后推进:灰度 10%→50%→全量
|
|
174
|
+
engine.rollback("tool_router") # 出问题一键回滚 stable 基线
|
|
175
|
+
```
|
|
176
|
+
|
|
177
|
+
### 7. 从 LLM 调用日志自动发现决策点(把重复的 LLM 判断固化成萤)
|
|
178
|
+
|
|
179
|
+
如果你的 LLM 项目里有一类判断反复出现(如意图分类、请求分派),把日志喂给挖掘器:
|
|
180
|
+
|
|
181
|
+
```python
|
|
182
|
+
from firefly.evolution import DecisionPointMiner, register_decision_point
|
|
183
|
+
|
|
184
|
+
logs = [{"point": "intent", # 同一决策点打同一个标签
|
|
185
|
+
"input": user_text,
|
|
186
|
+
"output": llm_answer_label} # 输出需为短标签(可枚举)
|
|
187
|
+
for user_text, llm_answer_label in your_history]
|
|
188
|
+
|
|
189
|
+
for s in DecisionPointMiner().mine(logs):
|
|
190
|
+
print(s.name, s.action_space, f"{s.frequency} 条日志,熵 {s.entropy:.2f}")
|
|
191
|
+
# 人工审核通过后:
|
|
192
|
+
register_decision_point(bus, s) # 注册新萤节点
|
|
193
|
+
engine.manage(s.name)
|
|
194
|
+
engine.import_history([{"task": x.task, "target_id": x.target_id}
|
|
195
|
+
for x in s.samples]) # 灌入初始样本
|
|
196
|
+
```
|
|
197
|
+
|
|
198
|
+
## 反馈闭环:outcome 从哪来
|
|
199
|
+
|
|
200
|
+
`firefly.feedback(trace_id, outcome, corrected_choice=None, extra=None)`:
|
|
201
|
+
|
|
202
|
+
| outcome | 含义 | 学习效果 |
|
|
203
|
+
|---|---|---|
|
|
204
|
+
| `success` | 工具执行成功 / 用户采纳 | 强化本次选择 |
|
|
205
|
+
| `failure` / `timeout` / `user_rejected` | 执行失败 | 惩罚本次选择 |
|
|
206
|
+
| `user_corrected` | 用户/LLM 给出正确答案 | 以 1.2 倍权重学习 `corrected_choice` |
|
|
207
|
+
|
|
208
|
+
任何节点的决策都走同一个闭环;守门硬规则、循环上限、元决策安全规则带证书,反馈无法改变其结论。
|
|
209
|
+
|
|
210
|
+
## HTTP 边车(跨语言接入)
|
|
211
|
+
|
|
212
|
+
```bash
|
|
213
|
+
python -m firefly.server.http --host 0.0.0.0 --port 8000
|
|
214
|
+
```
|
|
215
|
+
|
|
216
|
+
```bash
|
|
217
|
+
curl -X POST localhost:8000/v1/decide -H "Content-Type: application/json" -d '{
|
|
218
|
+
"node": "tool_router",
|
|
219
|
+
"state": {"task": "查一下今天的科技新闻"},
|
|
220
|
+
"options": [{"id": "search", "features": {"description": "search the web"}}]
|
|
221
|
+
}'
|
|
222
|
+
```
|
|
223
|
+
|
|
224
|
+
端点:`POST /v1/decide`、`POST /v1/feedback`、`POST /v1/tools`、`GET /v1/tools`、`GET /v1/metrics`、`GET /health`。
|
|
225
|
+
|
|
226
|
+
## API 速查
|
|
227
|
+
|
|
228
|
+
| 对象 | 关键方法 |
|
|
229
|
+
|---|---|
|
|
230
|
+
| `FireflyClient` | `in_process(bus)` / `http(url)` / `decide()` / `raw_decide()` / `feedback()` / `register_tool()` / `metrics()` |
|
|
231
|
+
| `build_bus()` | `llm_decider`、`user_decider`(回退执行者)、`swarm_nodes=True`(挂 P2 全部节点)、`gate_policy`(自定义守门规则) |
|
|
232
|
+
| `EvolutionEngine` | `manage_defaults()` / `manage(name)` / `start()` / `tick()` / `promote()` / `rollback()` / `import_history()` |
|
|
233
|
+
| `DecisionPointMiner` | `mine(logs) -> [DecisionPointSuggestion]` |
|
|
234
|
+
| `register_decision_point` | `(bus, suggestion)` 注册新节点 |
|
|
235
|
+
|
|
236
|
+
## 使用注意事项
|
|
237
|
+
|
|
238
|
+
1. **必须预热**:核随机初始化,冷启动决策接近随机。用 `import_history` 批量导入历史数据是最快的预热方式;没有历史数据就上线初期让 LLM 当老师(回退闭环会自动积累训练数据)。
|
|
239
|
+
2. **反馈必须接**:不接 `feedback` 萤就不会进化。工具执行结果、用户采纳/纠正都是现成的信号源。
|
|
240
|
+
3. **学习期隔离回退**:批量训练/演示时传 `constraints: {"fallback": {"mode": "none"}}`,避免低置信度期间决策被 explore 劫持干扰训练信号。
|
|
241
|
+
4. **语义边界**:萤的特征是哈希级语义(同义词汇能对上,深层推理不行)。需要真正理解的判断本来就该回退 LLM——这正是系统设计,调好 `fallback.threshold`(默认 0.6)即可。
|
|
242
|
+
5. **重启恢复**:给 `EvolutionEngine` 传 `model_dir`,版本历史、回放缓冲、active 萤核重启后自动恢复。
|
|
243
|
+
6. **进程常驻**:审计日志与缓存目前为内存态,适合常驻服务;HTTP 边车模式天然满足。
|
|
244
|
+
|
|
245
|
+
## 更多示例
|
|
246
|
+
|
|
247
|
+
- [examples/quickstart.py](examples/quickstart.py) — P0 全流程(路由/学习/守门/指标)
|
|
248
|
+
- [examples/evolution_demo.py](examples/evolution_demo.py) — P1 进化引擎全链路(导入历史→冷启动→影子→灰度→发布→漂移自愈→回滚)
|
|
249
|
+
- [examples/swarm_demo.py](examples/swarm_demo.py) — P2 萤群 + 决策点自动发现
|
|
250
|
+
- [tests/](tests/) — 94 个测试用例,是最好的用法文档
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""一键装配萤总线:P0 路由/守门双节点;swarm_nodes=True 时加 P2 萤群五节点。"""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from typing import Callable, Optional
|
|
5
|
+
|
|
6
|
+
from .bus import DecisionCache, Fallback, FireflyBus
|
|
7
|
+
from .nodes import (
|
|
8
|
+
GateNode,
|
|
9
|
+
GuardPolicy,
|
|
10
|
+
LoopNode,
|
|
11
|
+
MemoryNode,
|
|
12
|
+
MetaNode,
|
|
13
|
+
RouterNode,
|
|
14
|
+
ScheduleNode,
|
|
15
|
+
VerifyNode,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def build_bus(
|
|
20
|
+
llm_decider: Optional[Callable] = None,
|
|
21
|
+
user_decider: Optional[Callable] = None,
|
|
22
|
+
cache_ttl_seconds: float = 5.0,
|
|
23
|
+
gate_policy: Optional[GuardPolicy] = None,
|
|
24
|
+
swarm_nodes: bool = False,
|
|
25
|
+
) -> FireflyBus:
|
|
26
|
+
bus = FireflyBus(
|
|
27
|
+
cache=DecisionCache(ttl_seconds=cache_ttl_seconds),
|
|
28
|
+
fallback=Fallback(llm_decider=llm_decider, user_decider=user_decider),
|
|
29
|
+
)
|
|
30
|
+
bus.register_node(RouterNode())
|
|
31
|
+
bus.register_node(GateNode(policy=gate_policy))
|
|
32
|
+
if swarm_nodes:
|
|
33
|
+
bus.register_node(VerifyNode())
|
|
34
|
+
bus.register_node(LoopNode())
|
|
35
|
+
bus.register_node(ScheduleNode())
|
|
36
|
+
bus.register_node(MemoryNode())
|
|
37
|
+
bus.register_node(MetaNode())
|
|
38
|
+
return bus
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""萤总线层:路由 / 缓存 / 回退 / 审计 / 指标。"""
|
|
2
|
+
from .audit import AuditLog
|
|
3
|
+
from .bus import (
|
|
4
|
+
FireflyBus,
|
|
5
|
+
FireflyError,
|
|
6
|
+
NodeNotFound,
|
|
7
|
+
ProtocolViolation,
|
|
8
|
+
)
|
|
9
|
+
from .cache import DecisionCache
|
|
10
|
+
from .fallback import Fallback, FallbackDecision
|
|
11
|
+
from .metrics import Metrics
|
|
12
|
+
|
|
13
|
+
__all__ = [
|
|
14
|
+
"FireflyBus", "FireflyError", "NodeNotFound", "ProtocolViolation",
|
|
15
|
+
"DecisionCache", "Fallback", "FallbackDecision",
|
|
16
|
+
"AuditLog", "Metrics",
|
|
17
|
+
]
|