dsh-industry-research 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +19 -0
- package/LICENSE +201 -0
- package/README.es.md +175 -0
- package/README.hi.md +175 -0
- package/README.md +175 -0
- package/README.pt.md +175 -0
- package/README.zh.md +175 -0
- package/THIRD_PARTY_NOTICES.md +21 -0
- package/cordis.patch.yml +44 -0
- package/lib/index.js +2080 -0
- package/lib/types/chain.d.ts +68 -0
- package/lib/types/chain.d.ts.map +1 -0
- package/lib/types/chain.js +88 -0
- package/lib/types/chain.js.map +1 -0
- package/lib/types/company.d.ts +121 -0
- package/lib/types/company.d.ts.map +1 -0
- package/lib/types/company.js +170 -0
- package/lib/types/company.js.map +1 -0
- package/lib/types/config.d.ts +72 -0
- package/lib/types/config.d.ts.map +1 -0
- package/lib/types/config.js +86 -0
- package/lib/types/config.js.map +1 -0
- package/lib/types/engine-bridge.d.ts +68 -0
- package/lib/types/engine-bridge.d.ts.map +1 -0
- package/lib/types/engine-bridge.js +18 -0
- package/lib/types/engine-bridge.js.map +1 -0
- package/lib/types/events.d.ts +79 -0
- package/lib/types/events.d.ts.map +1 -0
- package/lib/types/events.js +13 -0
- package/lib/types/events.js.map +1 -0
- package/lib/types/index.d.ts +58 -0
- package/lib/types/index.d.ts.map +1 -0
- package/lib/types/index.js +111 -0
- package/lib/types/index.js.map +1 -0
- package/lib/types/paths.d.ts +45 -0
- package/lib/types/paths.d.ts.map +1 -0
- package/lib/types/paths.js +93 -0
- package/lib/types/paths.js.map +1 -0
- package/lib/types/report.d.ts +134 -0
- package/lib/types/report.d.ts.map +1 -0
- package/lib/types/report.js +278 -0
- package/lib/types/report.js.map +1 -0
- package/lib/types/sources.d.ts +55 -0
- package/lib/types/sources.d.ts.map +1 -0
- package/lib/types/sources.js +73 -0
- package/lib/types/sources.js.map +1 -0
- package/lib/types/timeline.d.ts +78 -0
- package/lib/types/timeline.d.ts.map +1 -0
- package/lib/types/timeline.js +129 -0
- package/lib/types/timeline.js.map +1 -0
- package/lib/types/toolkit.d.ts +61 -0
- package/lib/types/toolkit.d.ts.map +1 -0
- package/lib/types/toolkit.js +77 -0
- package/lib/types/toolkit.js.map +1 -0
- package/lib/types/tools/company.d.ts +36 -0
- package/lib/types/tools/company.d.ts.map +1 -0
- package/lib/types/tools/company.js +139 -0
- package/lib/types/tools/company.js.map +1 -0
- package/lib/types/tools/map.d.ts +44 -0
- package/lib/types/tools/map.d.ts.map +1 -0
- package/lib/types/tools/map.js +211 -0
- package/lib/types/tools/map.js.map +1 -0
- package/lib/types/tools/report.d.ts +61 -0
- package/lib/types/tools/report.d.ts.map +1 -0
- package/lib/types/tools/report.js +277 -0
- package/lib/types/tools/report.js.map +1 -0
- package/lib/types/tools/track.d.ts +46 -0
- package/lib/types/tools/track.d.ts.map +1 -0
- package/lib/types/tools/track.js +188 -0
- package/lib/types/tools/track.js.map +1 -0
- package/lib/types/version.d.ts +8 -0
- package/lib/types/version.d.ts.map +1 -0
- package/lib/types/version.js +8 -0
- package/lib/types/version.js.map +1 -0
- package/lib/types/web.d.ts +74 -0
- package/lib/types/web.d.ts.map +1 -0
- package/lib/types/web.js +60 -0
- package/lib/types/web.js.map +1 -0
- package/package.json +143 -0
- package/skills/company-research-method/SKILL.md +50 -0
- package/skills/industry-research-method/SKILL.md +57 -0
- package/skills/industry-research-method/references/frameworks.md +42 -0
- package/src/chain.ts +132 -0
- package/src/company.ts +238 -0
- package/src/config.ts +154 -0
- package/src/engine-bridge.ts +62 -0
- package/src/events.ts +82 -0
- package/src/index.ts +139 -0
- package/src/paths.ts +96 -0
- package/src/report.ts +326 -0
- package/src/sources.ts +95 -0
- package/src/timeline.ts +155 -0
- package/src/toolkit.ts +87 -0
- package/src/tools/company.ts +157 -0
- package/src/tools/map.ts +241 -0
- package/src/tools/report.ts +304 -0
- package/src/tools/track.ts +223 -0
- package/src/version.ts +8 -0
- package/src/web.ts +94 -0
package/lib/index.js
ADDED
|
@@ -0,0 +1,2080 @@
|
|
|
1
|
+
import { existsSync, readdirSync } from "node:fs";
|
|
2
|
+
import { dirname, extname, isAbsolute, join, resolve, sep } from "node:path";
|
|
3
|
+
import { fileURLToPath } from "node:url";
|
|
4
|
+
import { FileSystemSkillProvider } from "@deepseek-ai/dsh-skill-filesystem";
|
|
5
|
+
import z from "@deepseek-ai/schemastery";
|
|
6
|
+
import { mkdir, readFile, readdir, stat, writeFile } from "node:fs/promises";
|
|
7
|
+
import { defineTool } from "@deepseek-ai/dsh-tools";
|
|
8
|
+
import { createHash } from "node:crypto";
|
|
9
|
+
//#region src/config.ts
|
|
10
|
+
/**
|
|
11
|
+
* Config schema and resolution for `dsh-industry-research`. Every tunable is a
|
|
12
|
+
* validated {@link Config} field changeable from cordis.yml; the resolution
|
|
13
|
+
* step validates bounds so misconfiguration fails loud at mount. With
|
|
14
|
+
* `enabled: false` the plugin registers nothing and stays inert.
|
|
15
|
+
* @module dsh-industry-research/config
|
|
16
|
+
*/
|
|
17
|
+
/** Schemastery schema: the loader validates and fills defaults before `apply`. */
|
|
18
|
+
const Config = z.object({
|
|
19
|
+
enabled: z.boolean().default(true),
|
|
20
|
+
industryRoot: z.string().default("industry-research"),
|
|
21
|
+
fetchTimeoutMs: z.number().default(2e4),
|
|
22
|
+
timelineMaxEntries: z.number().default(500),
|
|
23
|
+
sourceAllowlist: z.array(z.string()).default([]),
|
|
24
|
+
sourceBlocklist: z.array(z.string()).default([]),
|
|
25
|
+
offline: z.boolean().default(false),
|
|
26
|
+
skillsDir: z.string(),
|
|
27
|
+
track: z.object({
|
|
28
|
+
maxResultsPerTopic: z.number().default(10),
|
|
29
|
+
maxFetchesPerCall: z.number().default(10)
|
|
30
|
+
}).default({
|
|
31
|
+
maxResultsPerTopic: 10,
|
|
32
|
+
maxFetchesPerCall: 10
|
|
33
|
+
}),
|
|
34
|
+
scan: z.object({
|
|
35
|
+
maxFileBytes: z.number().default(1048576),
|
|
36
|
+
maxFigureCandidates: z.number().default(100)
|
|
37
|
+
}).default({
|
|
38
|
+
maxFileBytes: 1048576,
|
|
39
|
+
maxFigureCandidates: 100
|
|
40
|
+
})
|
|
41
|
+
});
|
|
42
|
+
/** Throw unless `value` is a positive safe integer. */
|
|
43
|
+
function assertPositiveInt(name, value) {
|
|
44
|
+
if (!Number.isSafeInteger(value) || value <= 0) throw new TypeError(`${name} must be a positive safe integer, got ${String(value)}`);
|
|
45
|
+
}
|
|
46
|
+
/** Throw unless every entry is a non-empty string. */
|
|
47
|
+
function assertStringList(name, value) {
|
|
48
|
+
for (const entry of value) if (typeof entry !== "string" || entry.trim().length === 0) throw new TypeError(`${name} entries must be non-empty strings, got ${JSON.stringify(entry)}`);
|
|
49
|
+
}
|
|
50
|
+
/**
|
|
51
|
+
* Validate raw values and fill explicit defaults. Invalid bounds throw here —
|
|
52
|
+
* misconfiguration fails loud at mount even without the Schemastery loader.
|
|
53
|
+
* @param config - raw (possibly partial) plugin config.
|
|
54
|
+
* @returns the fully resolved config.
|
|
55
|
+
*/
|
|
56
|
+
function resolveConfig(config = {}) {
|
|
57
|
+
const industryRoot = config.industryRoot ?? "industry-research";
|
|
58
|
+
if (typeof industryRoot !== "string" || industryRoot.trim().length === 0) throw new TypeError("industryRoot must be a non-empty path");
|
|
59
|
+
const fetchTimeoutMs = config.fetchTimeoutMs ?? 2e4;
|
|
60
|
+
assertPositiveInt("fetchTimeoutMs", fetchTimeoutMs);
|
|
61
|
+
const timelineMaxEntries = config.timelineMaxEntries ?? 500;
|
|
62
|
+
assertPositiveInt("timelineMaxEntries", timelineMaxEntries);
|
|
63
|
+
const sourceAllowlist = config.sourceAllowlist ?? [];
|
|
64
|
+
assertStringList("sourceAllowlist", sourceAllowlist);
|
|
65
|
+
const sourceBlocklist = config.sourceBlocklist ?? [];
|
|
66
|
+
assertStringList("sourceBlocklist", sourceBlocklist);
|
|
67
|
+
const skillsDir = config.skillsDir;
|
|
68
|
+
if (skillsDir !== void 0 && (typeof skillsDir !== "string" || skillsDir.trim().length === 0)) throw new TypeError("skillsDir must be a non-empty path when set");
|
|
69
|
+
const maxResultsPerTopic = config.track?.maxResultsPerTopic ?? 10;
|
|
70
|
+
assertPositiveInt("track.maxResultsPerTopic", maxResultsPerTopic);
|
|
71
|
+
const maxFetchesPerCall = config.track?.maxFetchesPerCall ?? 10;
|
|
72
|
+
assertPositiveInt("track.maxFetchesPerCall", maxFetchesPerCall);
|
|
73
|
+
const maxFileBytes = config.scan?.maxFileBytes ?? 1048576;
|
|
74
|
+
assertPositiveInt("scan.maxFileBytes", maxFileBytes);
|
|
75
|
+
const maxFigureCandidates = config.scan?.maxFigureCandidates ?? 100;
|
|
76
|
+
assertPositiveInt("scan.maxFigureCandidates", maxFigureCandidates);
|
|
77
|
+
return {
|
|
78
|
+
enabled: config.enabled ?? true,
|
|
79
|
+
industryRoot,
|
|
80
|
+
fetchTimeoutMs,
|
|
81
|
+
timelineMaxEntries,
|
|
82
|
+
sourceAllowlist,
|
|
83
|
+
sourceBlocklist,
|
|
84
|
+
offline: config.offline ?? false,
|
|
85
|
+
skillsDir,
|
|
86
|
+
track: {
|
|
87
|
+
maxResultsPerTopic,
|
|
88
|
+
maxFetchesPerCall
|
|
89
|
+
},
|
|
90
|
+
scan: {
|
|
91
|
+
maxFileBytes,
|
|
92
|
+
maxFigureCandidates
|
|
93
|
+
}
|
|
94
|
+
};
|
|
95
|
+
}
|
|
96
|
+
//#endregion
|
|
97
|
+
//#region src/chain.ts
|
|
98
|
+
/**
|
|
99
|
+
* The industry-chain structure model (`ChainMap`) and its validation. This is
|
|
100
|
+
* a pure data module: no I/O, no clock. A metric is either a sourced value
|
|
101
|
+
* (`value` + `sourceRef`, optionally `unit`/`asOf`) or an explicit gap slot
|
|
102
|
+
* (no `value`), so an unsourced number is a validation error by construction
|
|
103
|
+
* and a missing number is an honest, listable gap.
|
|
104
|
+
* @module dsh-industry-research/chain
|
|
105
|
+
*/
|
|
106
|
+
/** Chain tiers, upstream → downstream. */
|
|
107
|
+
const CHAIN_TIERS = [
|
|
108
|
+
"upstream",
|
|
109
|
+
"midstream",
|
|
110
|
+
"downstream"
|
|
111
|
+
];
|
|
112
|
+
/**
|
|
113
|
+
* Validate a chain map. Pure: returns the list of problems (empty when the
|
|
114
|
+
* map is well-formed). Rules: unique node ids, legal tiers, edges reference
|
|
115
|
+
* existing nodes, and every metric carrying a `value` also carries a
|
|
116
|
+
* `sourceRef` (gap slots without a value are always legal).
|
|
117
|
+
* @param map - the candidate chain map.
|
|
118
|
+
* @returns human-readable validation problems, in encounter order.
|
|
119
|
+
*/
|
|
120
|
+
function validateChainMap(map) {
|
|
121
|
+
const problems = [];
|
|
122
|
+
if (typeof map.industry !== "string" || map.industry.trim().length === 0) problems.push("chain.industry must be a non-empty name");
|
|
123
|
+
const ids = /* @__PURE__ */ new Set();
|
|
124
|
+
for (const node of map.nodes) {
|
|
125
|
+
if (typeof node.id !== "string" || node.id.trim().length === 0) {
|
|
126
|
+
problems.push(`node ${JSON.stringify(node.id)} has an empty id`);
|
|
127
|
+
continue;
|
|
128
|
+
}
|
|
129
|
+
if (ids.has(node.id)) problems.push(`duplicate node id "${node.id}"`);
|
|
130
|
+
ids.add(node.id);
|
|
131
|
+
if (typeof node.name !== "string" || node.name.trim().length === 0) problems.push(`node "${node.id}" has an empty name`);
|
|
132
|
+
if (!CHAIN_TIERS.includes(node.tier)) problems.push(`node "${node.id}" has an illegal tier ${JSON.stringify(node.tier)} (expected upstream|midstream|downstream)`);
|
|
133
|
+
for (const metric of node.metrics) {
|
|
134
|
+
if (typeof metric.key !== "string" || metric.key.trim().length === 0) {
|
|
135
|
+
problems.push(`node "${node.id}" has a metric with an empty key`);
|
|
136
|
+
continue;
|
|
137
|
+
}
|
|
138
|
+
if (metric.value !== void 0) {
|
|
139
|
+
if (typeof metric.value !== "number" || !Number.isFinite(metric.value)) problems.push(`node "${node.id}" metric "${metric.key}" carries a non-finite value`);
|
|
140
|
+
if (typeof metric.sourceRef !== "string" || metric.sourceRef.trim().length === 0) problems.push(`node "${node.id}" metric "${metric.key}" carries a value without a sourceRef — register the source or mark the slot 待补 (omit value)`);
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
for (const edge of map.edges) {
|
|
145
|
+
if (!ids.has(edge.from)) problems.push(`edge references unknown node "${edge.from}" (from)`);
|
|
146
|
+
if (!ids.has(edge.to)) problems.push(`edge references unknown node "${edge.to}" (to)`);
|
|
147
|
+
}
|
|
148
|
+
return problems;
|
|
149
|
+
}
|
|
150
|
+
/**
|
|
151
|
+
* List the explicit gaps of a well-formed chain map: metric slots without a
|
|
152
|
+
* value, nodes without any metric slot, and tiers with no node at all.
|
|
153
|
+
* @param map - the chain map (already validated).
|
|
154
|
+
* @returns human-readable gap lines, in encounter order.
|
|
155
|
+
*/
|
|
156
|
+
function chainGaps(map) {
|
|
157
|
+
const gaps = [];
|
|
158
|
+
const tiers = /* @__PURE__ */ new Set();
|
|
159
|
+
for (const node of map.nodes) {
|
|
160
|
+
tiers.add(node.tier);
|
|
161
|
+
if (node.metrics.length === 0) gaps.push(`节点「${node.name}」(${node.id}) 没有任何指标槽位`);
|
|
162
|
+
for (const metric of node.metrics) if (metric.value === void 0) gaps.push(`节点「${node.name}」(${node.id}) 的指标「${metric.key}」待补(无来源数值)`);
|
|
163
|
+
}
|
|
164
|
+
for (const tier of CHAIN_TIERS) if (!tiers.has(tier)) gaps.push(`产业链缺少 ${tier} 层节点`);
|
|
165
|
+
return gaps;
|
|
166
|
+
}
|
|
167
|
+
//#endregion
|
|
168
|
+
//#region src/sources.ts
|
|
169
|
+
/**
|
|
170
|
+
* The per-industry source registry (`sources.json`). Every artifact a metric
|
|
171
|
+
* or claim may cite — a user note, a supplied data file, a fetched page — is
|
|
172
|
+
* registered here with a stable ref (`S1`, `S2`, …), its origin (workspace
|
|
173
|
+
* path or URL), and a SHA-256 content hash, so reports can render a
|
|
174
|
+
* source-traceability appendix and byte-level checks can replay.
|
|
175
|
+
* @module dsh-industry-research/sources
|
|
176
|
+
*/
|
|
177
|
+
/** SHA-256 hex digest of a UTF-8 string. */
|
|
178
|
+
function sha256Of(content) {
|
|
179
|
+
return createHash("sha256").update(content, "utf8").digest("hex");
|
|
180
|
+
}
|
|
181
|
+
/**
|
|
182
|
+
* Load a registry from disk, tolerating a missing file (empty registry).
|
|
183
|
+
* A corrupt file fails loud — durable data must not be silently dropped.
|
|
184
|
+
* @param path - absolute path of the registry JSON.
|
|
185
|
+
* @returns the parsed registry, or a fresh one when the file does not exist.
|
|
186
|
+
*/
|
|
187
|
+
async function loadSources(path) {
|
|
188
|
+
let text;
|
|
189
|
+
try {
|
|
190
|
+
text = await readFile(path, "utf8");
|
|
191
|
+
} catch (error) {
|
|
192
|
+
if (error.code === "ENOENT") return {
|
|
193
|
+
next: 1,
|
|
194
|
+
items: []
|
|
195
|
+
};
|
|
196
|
+
throw error;
|
|
197
|
+
}
|
|
198
|
+
const parsed = JSON.parse(text);
|
|
199
|
+
if (typeof parsed.next !== "number" || !Array.isArray(parsed.items)) throw new Error(`sources registry at ${path} is malformed (expected { next, items[] })`);
|
|
200
|
+
return parsed;
|
|
201
|
+
}
|
|
202
|
+
/**
|
|
203
|
+
* Persist a registry, creating the parent directory.
|
|
204
|
+
* @param path - absolute path of the registry JSON.
|
|
205
|
+
* @param registry - the registry to write.
|
|
206
|
+
*/
|
|
207
|
+
async function saveSources(path, registry) {
|
|
208
|
+
await mkdir(dirname(path), { recursive: true });
|
|
209
|
+
await writeFile(path, `${JSON.stringify(registry, null, 2)}\n`, "utf8");
|
|
210
|
+
}
|
|
211
|
+
/**
|
|
212
|
+
* Register one source and return its ref. Re-registering the same origin
|
|
213
|
+
* refreshes the hash/timestamp of the existing entry instead of duplicating
|
|
214
|
+
* it, so refs stay stable across re-runs.
|
|
215
|
+
* @param registry - the registry to mutate.
|
|
216
|
+
* @param origin - workspace path or URL the content came from.
|
|
217
|
+
* @param content - the verbatim content snapshot (hashed).
|
|
218
|
+
* @param capturedAt - ISO-8601 registration time.
|
|
219
|
+
* @param note - optional human note (e.g. page title).
|
|
220
|
+
* @returns the stable ref of the entry.
|
|
221
|
+
*/
|
|
222
|
+
function registerSource(registry, origin, content, capturedAt, note) {
|
|
223
|
+
const sha256 = sha256Of(content);
|
|
224
|
+
const existing = registry.items.find((item) => item.origin === origin);
|
|
225
|
+
if (existing !== void 0) {
|
|
226
|
+
existing.sha256 = sha256;
|
|
227
|
+
existing.capturedAt = capturedAt;
|
|
228
|
+
if (note !== void 0) existing.note = note;
|
|
229
|
+
return existing.ref;
|
|
230
|
+
}
|
|
231
|
+
const ref = `S${registry.next}`;
|
|
232
|
+
registry.next += 1;
|
|
233
|
+
registry.items.push({
|
|
234
|
+
ref,
|
|
235
|
+
origin,
|
|
236
|
+
sha256,
|
|
237
|
+
capturedAt,
|
|
238
|
+
...note !== void 0 ? { note } : {}
|
|
239
|
+
});
|
|
240
|
+
return ref;
|
|
241
|
+
}
|
|
242
|
+
//#endregion
|
|
243
|
+
//#region src/web.ts
|
|
244
|
+
/**
|
|
245
|
+
* Look up the optional web capability.
|
|
246
|
+
* @param ctx - the plugin context.
|
|
247
|
+
* @returns the web service surface, or undefined when no web seam is mounted.
|
|
248
|
+
*/
|
|
249
|
+
function lookupWeb(ctx) {
|
|
250
|
+
return ctx.get("web");
|
|
251
|
+
}
|
|
252
|
+
/**
|
|
253
|
+
* Resolve the web capability or throw the actionable reason it cannot run:
|
|
254
|
+
* `offline: true` (deployment choice) or no mounted web seam (mount guidance
|
|
255
|
+
* naming the missing pieces). Used by tools whose work is impossible offline.
|
|
256
|
+
* @param ctx - the plugin context.
|
|
257
|
+
* @param config - the resolved plugin config.
|
|
258
|
+
* @param tool - the calling tool name (for the error message).
|
|
259
|
+
* @returns the web service surface.
|
|
260
|
+
*/
|
|
261
|
+
function requireWeb(ctx, config, tool) {
|
|
262
|
+
if (config.offline) throw new Error(`${tool} cannot run while config.offline is true — public-source tracking requires the web; set offline: false or work from local artifacts only`);
|
|
263
|
+
const web = lookupWeb(ctx);
|
|
264
|
+
if (web === void 0) throw new Error(`${tool} requires the ctx.web capability, which is not mounted in this profile — mount @deepseek-ai/dsh-web plus a search provider (e.g. @deepseek-ai/dsh-web-search-deepseek) and a fetch provider (e.g. @deepseek-ai/dsh-web-fetch-http); the official dsh-base bundle already composes them`);
|
|
265
|
+
return web;
|
|
266
|
+
}
|
|
267
|
+
/**
|
|
268
|
+
* Combine the tool's caller signal with the configured per-request timeout.
|
|
269
|
+
* @param signal - the tool execution signal (`exec.signal`).
|
|
270
|
+
* @param timeoutMs - per-request timeout in milliseconds.
|
|
271
|
+
* @returns a signal that fires on either cancellation or timeout.
|
|
272
|
+
*/
|
|
273
|
+
function requestSignal(signal, timeoutMs) {
|
|
274
|
+
return AbortSignal.any([signal, AbortSignal.timeout(timeoutMs)]);
|
|
275
|
+
}
|
|
276
|
+
/**
|
|
277
|
+
* Render a thrown web failure with its machine-routable code when one exists
|
|
278
|
+
* (`WebError` carries a stable `code`; plain errors degrade to the message).
|
|
279
|
+
* @param error - the thrown value.
|
|
280
|
+
* @returns a single-line diagnostic.
|
|
281
|
+
*/
|
|
282
|
+
function webErrorMessage(error) {
|
|
283
|
+
if (error instanceof Error) {
|
|
284
|
+
const code = error.code;
|
|
285
|
+
if (typeof code === "string" && code.length > 0) return `${code}: ${error.message}`;
|
|
286
|
+
return error.message;
|
|
287
|
+
}
|
|
288
|
+
return String(error);
|
|
289
|
+
}
|
|
290
|
+
//#endregion
|
|
291
|
+
//#region src/paths.ts
|
|
292
|
+
/**
|
|
293
|
+
* Workspace path resolution and containment for `dsh-industry-research`. All
|
|
294
|
+
* artifacts live under `<cwd>/<industryRoot>/`; every name crossing a tool
|
|
295
|
+
* argument boundary (industry, company, data file) is validated so a crafted
|
|
296
|
+
* value cannot escape the workspace. Both sides of every containment check
|
|
297
|
+
* are resolved before comparison (`path.resolve` returns backslashes on
|
|
298
|
+
* Windows, and comparing against a forward-slash input would always fail).
|
|
299
|
+
* @module dsh-industry-research/paths
|
|
300
|
+
*/
|
|
301
|
+
/** Windows device names that cannot serve as directory segments. */
|
|
302
|
+
const WINDOWS_RESERVED = /^(?:con|prn|aux|nul|com[1-9]|lpt[1-9])$/iu;
|
|
303
|
+
/** Windows-forbidden filename characters (control characters are checked by code point below). */
|
|
304
|
+
const FORBIDDEN_CHARS = /[<>:"|?*]/u;
|
|
305
|
+
/**
|
|
306
|
+
* Validate one path segment (an industry or company directory name). CJK and
|
|
307
|
+
* other Unicode letters are fine; separators, traversal, control characters,
|
|
308
|
+
* leading/trailing dots/spaces, and Windows device names are rejected.
|
|
309
|
+
* @param label - what the segment names (for the error message).
|
|
310
|
+
* @param value - the raw user/model-supplied name.
|
|
311
|
+
* @returns the trimmed, validated segment.
|
|
312
|
+
*/
|
|
313
|
+
function safeSegment(label, value) {
|
|
314
|
+
const segment = value.trim();
|
|
315
|
+
if (segment.length === 0) throw new Error(`${label} must be a non-empty name`);
|
|
316
|
+
if (segment.length > 80) throw new Error(`${label} must be at most 80 characters, got ${segment.length}`);
|
|
317
|
+
if (segment.includes("/") || segment.includes("\\")) throw new Error(`${label} must be a single path segment, got ${JSON.stringify(value)}`);
|
|
318
|
+
if (segment === "." || segment === ".." || segment.includes("..")) throw new Error(`${label} must not contain traversal, got ${JSON.stringify(value)}`);
|
|
319
|
+
if (FORBIDDEN_CHARS.test(segment)) throw new Error(`${label} must not contain Windows-reserved characters, got ${JSON.stringify(value)}`);
|
|
320
|
+
for (const char of segment) if ((char.codePointAt(0) ?? 0) < 32) throw new Error(`${label} must not contain control characters, got ${JSON.stringify(value)}`);
|
|
321
|
+
if (segment.startsWith(".") || segment.endsWith(".") || segment.endsWith(" ")) throw new Error(`${label} must not start with '.' or end with '.'/' ', got ${JSON.stringify(value)}`);
|
|
322
|
+
if (WINDOWS_RESERVED.test(segment)) throw new Error(`${label} must not be a Windows device name, got ${JSON.stringify(value)}`);
|
|
323
|
+
return segment;
|
|
324
|
+
}
|
|
325
|
+
/**
|
|
326
|
+
* Resolve `root` against the workspace cwd and return the absolute root.
|
|
327
|
+
* A relative `industryRoot` is workspace-relative; an absolute one is used
|
|
328
|
+
* as-is (deployment choice).
|
|
329
|
+
* @param cwd - absolute session workspace root.
|
|
330
|
+
* @param industryRoot - configured root (relative or absolute).
|
|
331
|
+
* @returns the absolute industry root.
|
|
332
|
+
*/
|
|
333
|
+
function resolveIndustryRoot(cwd, industryRoot) {
|
|
334
|
+
return isAbsolute(industryRoot) ? resolve(industryRoot) : resolve(cwd, industryRoot);
|
|
335
|
+
}
|
|
336
|
+
/**
|
|
337
|
+
* Resolve a user-supplied workspace-relative file path and verify containment
|
|
338
|
+
* inside the workspace. Absolute paths are accepted only when already inside
|
|
339
|
+
* the workspace; everything escaping the cwd is rejected.
|
|
340
|
+
* @param cwd - absolute session workspace root.
|
|
341
|
+
* @param file - the raw path from a tool argument.
|
|
342
|
+
* @returns the absolute, containment-verified path.
|
|
343
|
+
*/
|
|
344
|
+
function resolveWorkspaceFile(cwd, file) {
|
|
345
|
+
const base = resolve(cwd);
|
|
346
|
+
const target = isAbsolute(file) ? resolve(file) : resolve(base, file);
|
|
347
|
+
if (target !== base && !target.startsWith(base + sep)) throw new Error(`path escapes the session workspace: ${JSON.stringify(file)}`);
|
|
348
|
+
return target;
|
|
349
|
+
}
|
|
350
|
+
/**
|
|
351
|
+
* Resolve one named directory below a root and verify containment (defense in
|
|
352
|
+
* depth after {@link safeSegment}).
|
|
353
|
+
* @param root - absolute parent root.
|
|
354
|
+
* @param segment - an already validated segment.
|
|
355
|
+
* @returns the absolute, containment-verified directory path.
|
|
356
|
+
*/
|
|
357
|
+
function resolveContained(root, segment) {
|
|
358
|
+
const base = resolve(root);
|
|
359
|
+
const target = resolve(base, segment);
|
|
360
|
+
if (!target.startsWith(base + sep)) throw new Error(`segment escapes its root: ${JSON.stringify(segment)}`);
|
|
361
|
+
return target;
|
|
362
|
+
}
|
|
363
|
+
//#endregion
|
|
364
|
+
//#region src/toolkit.ts
|
|
365
|
+
/**
|
|
366
|
+
* Shared helpers for the four research tools: workspace resolution from the
|
|
367
|
+
* calling agent's session and per-industry directory layout. The layout is:
|
|
368
|
+
*
|
|
369
|
+
* ```
|
|
370
|
+
* <cwd>/<industryRoot>/<industry>/chain.json industry_map
|
|
371
|
+
* <cwd>/<industryRoot>/<industry>/timeline.jsonl industry_track
|
|
372
|
+
* <cwd>/<industryRoot>/<industry>/sources.json source registry
|
|
373
|
+
* <cwd>/<industryRoot>/<industry>/notes/ seed notes
|
|
374
|
+
* <cwd>/<industryRoot>/<industry>/reports/<ts>/ industry_report
|
|
375
|
+
* <cwd>/<industryRoot>/companies/<slug>/card.* company_scan
|
|
376
|
+
* ```
|
|
377
|
+
* @module dsh-industry-research/toolkit
|
|
378
|
+
*/
|
|
379
|
+
/**
|
|
380
|
+
* Resolve the session workspace the calling agent operates in. The tools are
|
|
381
|
+
* workspace-bound by design; without an agent-owned session cwd there is no
|
|
382
|
+
* honest place to persist artifacts, so this fails loud.
|
|
383
|
+
* @param exec - the tool execution context.
|
|
384
|
+
* @returns the absolute workspace root.
|
|
385
|
+
*/
|
|
386
|
+
function workspaceOf(exec) {
|
|
387
|
+
const cwd = exec.agent?.session.header.cwd;
|
|
388
|
+
if (cwd === void 0 || cwd.trim().length === 0) throw new Error("this tool requires an agent-owned session workspace (session.header.cwd is unset)");
|
|
389
|
+
return cwd;
|
|
390
|
+
}
|
|
391
|
+
/**
|
|
392
|
+
* Resolve and validate the directory of one industry under the configured
|
|
393
|
+
* root. The industry name is validated as a single safe path segment.
|
|
394
|
+
* @param config - resolved plugin config.
|
|
395
|
+
* @param cwd - absolute workspace root.
|
|
396
|
+
* @param industry - the raw industry argument.
|
|
397
|
+
* @returns `{ root, dir, name }` — absolute industry root, absolute industry directory, and the validated segment.
|
|
398
|
+
*/
|
|
399
|
+
function industryDirOf(config, cwd, industry) {
|
|
400
|
+
const name = safeSegment("industry", industry);
|
|
401
|
+
const root = resolveIndustryRoot(cwd, config.industryRoot);
|
|
402
|
+
return {
|
|
403
|
+
root,
|
|
404
|
+
dir: resolveContained(root, name),
|
|
405
|
+
name
|
|
406
|
+
};
|
|
407
|
+
}
|
|
408
|
+
/**
|
|
409
|
+
* Resolve and validate the directory of one company card under the shared
|
|
410
|
+
* companies root.
|
|
411
|
+
* @param config - resolved plugin config.
|
|
412
|
+
* @param cwd - absolute workspace root.
|
|
413
|
+
* @param company - the raw company name argument.
|
|
414
|
+
* @returns `{ root, dir, slug }` — absolute industry root, absolute company directory, and the validated slug.
|
|
415
|
+
*/
|
|
416
|
+
function companyDirOf(config, cwd, company) {
|
|
417
|
+
const slug = safeSegment("company", company);
|
|
418
|
+
const root = resolveIndustryRoot(cwd, config.industryRoot);
|
|
419
|
+
return {
|
|
420
|
+
root,
|
|
421
|
+
dir: join(root, "companies", slug),
|
|
422
|
+
slug
|
|
423
|
+
};
|
|
424
|
+
}
|
|
425
|
+
/** The `chain.json` path of an industry directory. */
|
|
426
|
+
function chainPathOf(dir) {
|
|
427
|
+
return join(dir, "chain.json");
|
|
428
|
+
}
|
|
429
|
+
/** The `timeline.jsonl` path of an industry directory. */
|
|
430
|
+
function timelinePathOf(dir) {
|
|
431
|
+
return join(dir, "timeline.jsonl");
|
|
432
|
+
}
|
|
433
|
+
/** The `sources.json` path of an industry directory. */
|
|
434
|
+
function sourcesPathOf(dir) {
|
|
435
|
+
return join(dir, "sources.json");
|
|
436
|
+
}
|
|
437
|
+
/** The `notes/` directory of an industry directory. */
|
|
438
|
+
function notesDirOf(dir) {
|
|
439
|
+
return join(dir, "notes");
|
|
440
|
+
}
|
|
441
|
+
/** The `reports/` directory of an industry directory. */
|
|
442
|
+
function reportsDirOf(dir) {
|
|
443
|
+
return join(dir, "reports");
|
|
444
|
+
}
|
|
445
|
+
//#endregion
|
|
446
|
+
//#region src/tools/map.ts
|
|
447
|
+
/**
|
|
448
|
+
* The `industry_map` tool: build or update one industry's chain map. The
|
|
449
|
+
* model authors the ChainMap (guided by the industry-research-method skill);
|
|
450
|
+
* the tool validates it (dangling edges, unsourced numbers, illegal tiers),
|
|
451
|
+
* persists `chain.json`, registers seed/web material as citable sources, and
|
|
452
|
+
* lists the explicit gap slots. Called without a `chain` argument it returns
|
|
453
|
+
* the current map plus the registered sources, so the model can iterate.
|
|
454
|
+
* @module dsh-industry-research/tools/map
|
|
455
|
+
*/
|
|
456
|
+
/** The ChainMap parameter schema (semantic checks live in {@link validateChainMap}). */
|
|
457
|
+
const CHAIN_PARAMETER = {
|
|
458
|
+
type: "object",
|
|
459
|
+
properties: {
|
|
460
|
+
industry: {
|
|
461
|
+
type: "string",
|
|
462
|
+
required: true,
|
|
463
|
+
description: "行业显示名"
|
|
464
|
+
},
|
|
465
|
+
nodes: {
|
|
466
|
+
type: "array",
|
|
467
|
+
required: true,
|
|
468
|
+
items: {
|
|
469
|
+
type: "object",
|
|
470
|
+
properties: {
|
|
471
|
+
id: {
|
|
472
|
+
type: "string",
|
|
473
|
+
required: true,
|
|
474
|
+
description: "节点 id(边引用用,如 upstream-1)"
|
|
475
|
+
},
|
|
476
|
+
name: {
|
|
477
|
+
type: "string",
|
|
478
|
+
required: true,
|
|
479
|
+
description: "节点显示名(如 高粱种植)"
|
|
480
|
+
},
|
|
481
|
+
tier: {
|
|
482
|
+
type: "string",
|
|
483
|
+
enum: [
|
|
484
|
+
"upstream",
|
|
485
|
+
"midstream",
|
|
486
|
+
"downstream"
|
|
487
|
+
],
|
|
488
|
+
required: true
|
|
489
|
+
},
|
|
490
|
+
metrics: {
|
|
491
|
+
type: "array",
|
|
492
|
+
required: true,
|
|
493
|
+
items: {
|
|
494
|
+
type: "object",
|
|
495
|
+
properties: {
|
|
496
|
+
key: {
|
|
497
|
+
type: "string",
|
|
498
|
+
required: true,
|
|
499
|
+
description: "指标名(如 市场规模)"
|
|
500
|
+
},
|
|
501
|
+
value: {
|
|
502
|
+
type: "number",
|
|
503
|
+
description: "数值;缺省即待补槽位"
|
|
504
|
+
},
|
|
505
|
+
unit: {
|
|
506
|
+
type: "string",
|
|
507
|
+
description: "单位(如 亿元、%)"
|
|
508
|
+
},
|
|
509
|
+
asOf: {
|
|
510
|
+
type: "string",
|
|
511
|
+
description: "数值截止日期(ISO-8601)"
|
|
512
|
+
},
|
|
513
|
+
sourceRef: {
|
|
514
|
+
type: "string",
|
|
515
|
+
description: "来源引用:sources.json 的 ref(如 S1)、URL 或工作区路径;有 value 时必填"
|
|
516
|
+
}
|
|
517
|
+
},
|
|
518
|
+
additionalProperties: false
|
|
519
|
+
},
|
|
520
|
+
description: "指标槽位:有 value 必须带 sourceRef;无 value 即显式待补"
|
|
521
|
+
}
|
|
522
|
+
},
|
|
523
|
+
additionalProperties: false
|
|
524
|
+
}
|
|
525
|
+
},
|
|
526
|
+
edges: {
|
|
527
|
+
type: "array",
|
|
528
|
+
required: true,
|
|
529
|
+
items: {
|
|
530
|
+
type: "object",
|
|
531
|
+
properties: {
|
|
532
|
+
from: {
|
|
533
|
+
type: "string",
|
|
534
|
+
required: true,
|
|
535
|
+
description: "源节点 id"
|
|
536
|
+
},
|
|
537
|
+
to: {
|
|
538
|
+
type: "string",
|
|
539
|
+
required: true,
|
|
540
|
+
description: "目标节点 id"
|
|
541
|
+
},
|
|
542
|
+
note: {
|
|
543
|
+
type: "string",
|
|
544
|
+
description: "关系说明(如 原料供应)"
|
|
545
|
+
}
|
|
546
|
+
},
|
|
547
|
+
additionalProperties: false
|
|
548
|
+
}
|
|
549
|
+
}
|
|
550
|
+
},
|
|
551
|
+
additionalProperties: false,
|
|
552
|
+
description: "模型撰写的产业链结构图;语义校验(悬空边、无来源数值)由工具执行"
|
|
553
|
+
};
|
|
554
|
+
/**
|
|
555
|
+
* Build the `industry_map` tool definition.
|
|
556
|
+
* @param ctx - the plugin context (event emission + optional web lookup).
|
|
557
|
+
* @param config - the resolved plugin config.
|
|
558
|
+
* @returns the tool definition to register.
|
|
559
|
+
*/
|
|
560
|
+
function buildIndustryMapTool(ctx, config) {
|
|
561
|
+
return defineTool({
|
|
562
|
+
name: "industry_map",
|
|
563
|
+
description: "行业研究员的产业链建图工具:校验并落盘一份产业链结构图(上/中/下游节点 + 边 + 指标槽位),或在不带 chain 时返回当前图与已登记来源。每个有数值的指标必须带来源引用(sourceRef),缺数值的槽位即显式待补;禁止编造数据。仅供研究,不构成投资建议。",
|
|
564
|
+
parameters: {
|
|
565
|
+
industry: {
|
|
566
|
+
type: "string",
|
|
567
|
+
required: true,
|
|
568
|
+
description: "行业名(作为目录段,如「白酒」)"
|
|
569
|
+
},
|
|
570
|
+
seed: {
|
|
571
|
+
type: "string",
|
|
572
|
+
description: "自由文本笔记,写入该行业的 notes/ 并登记为来源"
|
|
573
|
+
},
|
|
574
|
+
seedFiles: {
|
|
575
|
+
type: "array",
|
|
576
|
+
items: { type: "string" },
|
|
577
|
+
description: "工作区内已有笔记/材料文件的相对路径列表,登记为来源"
|
|
578
|
+
},
|
|
579
|
+
web: {
|
|
580
|
+
type: "boolean",
|
|
581
|
+
description: "是否做 web 辅助检索(默认 true;config.offline 时自动跳过)"
|
|
582
|
+
},
|
|
583
|
+
chain: CHAIN_PARAMETER
|
|
584
|
+
},
|
|
585
|
+
output: {
|
|
586
|
+
schema: {
|
|
587
|
+
type: "object",
|
|
588
|
+
properties: {
|
|
589
|
+
industry: {
|
|
590
|
+
type: "string",
|
|
591
|
+
required: true
|
|
592
|
+
},
|
|
593
|
+
dir: {
|
|
594
|
+
type: "string",
|
|
595
|
+
required: true
|
|
596
|
+
},
|
|
597
|
+
chainPath: {
|
|
598
|
+
type: "string",
|
|
599
|
+
required: true
|
|
600
|
+
},
|
|
601
|
+
updated: {
|
|
602
|
+
type: "boolean",
|
|
603
|
+
required: true
|
|
604
|
+
},
|
|
605
|
+
chain: {
|
|
606
|
+
type: "json",
|
|
607
|
+
required: true
|
|
608
|
+
},
|
|
609
|
+
gaps: {
|
|
610
|
+
type: "array",
|
|
611
|
+
items: { type: "string" },
|
|
612
|
+
required: true
|
|
613
|
+
},
|
|
614
|
+
sources: {
|
|
615
|
+
type: "array",
|
|
616
|
+
items: { type: "json" },
|
|
617
|
+
required: true
|
|
618
|
+
},
|
|
619
|
+
seedRefs: {
|
|
620
|
+
type: "array",
|
|
621
|
+
items: { type: "string" },
|
|
622
|
+
required: true
|
|
623
|
+
},
|
|
624
|
+
webDigest: {
|
|
625
|
+
oneOf: [{
|
|
626
|
+
type: "array",
|
|
627
|
+
items: { type: "json" }
|
|
628
|
+
}, { type: "null" }],
|
|
629
|
+
required: true
|
|
630
|
+
},
|
|
631
|
+
webNote: {
|
|
632
|
+
oneOf: [{ type: "string" }, { type: "null" }],
|
|
633
|
+
required: true
|
|
634
|
+
}
|
|
635
|
+
},
|
|
636
|
+
additionalProperties: false
|
|
637
|
+
},
|
|
638
|
+
render: (_args, value) => {
|
|
639
|
+
const current = value;
|
|
640
|
+
const lines = [`行业「${current.industry}」产业链图:${current.updated ? "已校验并写入" : "未更新(仅读取)"} ${current.chainPath}`, current.chain === null ? "当前无图:请撰写 chain 后再次调用。" : `节点 ${current.chain.nodes.length} / 边 ${current.chain.edges.length};缺口 ${current.gaps.length} 项。`];
|
|
641
|
+
if (current.gaps.length > 0) lines.push(`缺口清单:${current.gaps.join(";")}`);
|
|
642
|
+
if (current.seedRefs.length > 0) lines.push(`本次登记来源:${current.seedRefs.join(", ")}(共 ${current.sources.length} 条,可用于 sourceRef)`);
|
|
643
|
+
if (current.webDigest !== null && current.webDigest.length > 0) lines.push(`web 辅助检索 ${current.webDigest.length} 条:${current.webDigest.map((source) => source.title ?? source.url).join(";")}`);
|
|
644
|
+
if (current.webNote !== null) lines.push(`web 辅助未执行:${current.webNote}`);
|
|
645
|
+
return [{
|
|
646
|
+
type: "text",
|
|
647
|
+
text: lines.join("\n")
|
|
648
|
+
}];
|
|
649
|
+
}
|
|
650
|
+
},
|
|
651
|
+
timeoutMs: Math.max(3e4, config.fetchTimeoutMs * 3),
|
|
652
|
+
async execute(args, exec) {
|
|
653
|
+
const cwd = workspaceOf(exec);
|
|
654
|
+
const { dir, name } = industryDirOf(config, cwd, args.industry);
|
|
655
|
+
const sourcesPath = sourcesPathOf(dir);
|
|
656
|
+
const registry = await loadSources(sourcesPath);
|
|
657
|
+
const now = (/* @__PURE__ */ new Date()).toISOString();
|
|
658
|
+
const seedRefs = [];
|
|
659
|
+
if (args.seed !== void 0 && args.seed.trim().length > 0) {
|
|
660
|
+
const notesDir = notesDirOf(dir);
|
|
661
|
+
await mkdir(notesDir, { recursive: true });
|
|
662
|
+
const notePath = join(notesDir, `seed-${now.replaceAll(":", "-")}.md`);
|
|
663
|
+
const noteContent = `# 种子笔记(${now})\n\n${args.seed.trim()}\n`;
|
|
664
|
+
await writeFile(notePath, noteContent, "utf8");
|
|
665
|
+
seedRefs.push(registerSource(registry, notePath, noteContent, now, "seed note"));
|
|
666
|
+
}
|
|
667
|
+
for (const file of args.seedFiles ?? []) {
|
|
668
|
+
const absolute = resolveWorkspaceFile(cwd, file);
|
|
669
|
+
const content = await readFile(absolute, "utf8");
|
|
670
|
+
seedRefs.push(registerSource(registry, absolute, content, now, "seed file"));
|
|
671
|
+
}
|
|
672
|
+
let webDigest = null;
|
|
673
|
+
let webNote = null;
|
|
674
|
+
if (args.web === false) webNote = "调用方指定 web: false";
|
|
675
|
+
else if (config.offline) webNote = "config.offline 为 true";
|
|
676
|
+
else {
|
|
677
|
+
const web = lookupWeb(ctx);
|
|
678
|
+
if (web === void 0) webNote = "ctx.web 未挂载(可选能力;不影响建图)";
|
|
679
|
+
else try {
|
|
680
|
+
const outcome = await web.search({
|
|
681
|
+
query: `${name} 产业链 上游 中游 下游 结构`,
|
|
682
|
+
maxResults: 5
|
|
683
|
+
}, requestSignal(exec.signal, config.fetchTimeoutMs));
|
|
684
|
+
webDigest = [...outcome.sources];
|
|
685
|
+
for (const source of outcome.sources) {
|
|
686
|
+
const digest = `${source.title ?? ""}\n${source.snippet ?? ""}`;
|
|
687
|
+
seedRefs.push(registerSource(registry, source.url, digest, now, source.title));
|
|
688
|
+
}
|
|
689
|
+
} catch (error) {
|
|
690
|
+
webNote = `web 检索失败:${webErrorMessage(error)}`;
|
|
691
|
+
}
|
|
692
|
+
}
|
|
693
|
+
if (seedRefs.length > 0) await saveSources(sourcesPath, registry);
|
|
694
|
+
const chainPath = chainPathOf(dir);
|
|
695
|
+
let updated = false;
|
|
696
|
+
if (args.chain !== void 0) {
|
|
697
|
+
const candidate = args.chain;
|
|
698
|
+
const problems = validateChainMap(candidate);
|
|
699
|
+
if (problems.length > 0) throw new Error(`chain 校验失败(${problems.length} 项):${problems.join(";")}`);
|
|
700
|
+
await mkdir(dir, { recursive: true });
|
|
701
|
+
await writeFile(chainPath, `${JSON.stringify(candidate, null, 2)}\n`, "utf8");
|
|
702
|
+
updated = true;
|
|
703
|
+
}
|
|
704
|
+
let current = null;
|
|
705
|
+
try {
|
|
706
|
+
current = JSON.parse(await readFile(chainPath, "utf8"));
|
|
707
|
+
} catch (error) {
|
|
708
|
+
if (error.code !== "ENOENT") throw error;
|
|
709
|
+
}
|
|
710
|
+
const gaps = current === null ? ["尚无产业链结构图:请基于 seed 与来源撰写 chain 后再次调用"] : chainGaps(current);
|
|
711
|
+
if (updated && current !== null) {
|
|
712
|
+
const payload = {
|
|
713
|
+
industry: name,
|
|
714
|
+
path: chainPath,
|
|
715
|
+
nodes: current.nodes.length,
|
|
716
|
+
edges: current.edges.length,
|
|
717
|
+
gaps: gaps.length
|
|
718
|
+
};
|
|
719
|
+
ctx.emit("industry-research/map", payload);
|
|
720
|
+
}
|
|
721
|
+
return {
|
|
722
|
+
industry: name,
|
|
723
|
+
dir,
|
|
724
|
+
chainPath,
|
|
725
|
+
updated,
|
|
726
|
+
chain: current,
|
|
727
|
+
gaps,
|
|
728
|
+
sources: registry.items,
|
|
729
|
+
seedRefs,
|
|
730
|
+
webDigest,
|
|
731
|
+
webNote
|
|
732
|
+
};
|
|
733
|
+
}
|
|
734
|
+
});
|
|
735
|
+
}
|
|
736
|
+
//#endregion
|
|
737
|
+
//#region src/timeline.ts
|
|
738
|
+
/**
|
|
739
|
+
* The per-industry timeline store (`timeline.jsonl`): one JSON entry per line,
|
|
740
|
+
* appended by `industry_track`, deduplicated by normalized URL, and capped by
|
|
741
|
+
* `timelineMaxEntries` (oldest entries dropped first). Pure file store — the
|
|
742
|
+
* web retrieval policy lives in `tools/track.ts`.
|
|
743
|
+
* @module dsh-industry-research/timeline
|
|
744
|
+
*/
|
|
745
|
+
/**
|
|
746
|
+
* Normalize a URL for dedupe: lowercase scheme and host, strip default ports,
|
|
747
|
+
* strip a lone trailing slash on the path. Unparseable URLs are returned
|
|
748
|
+
* verbatim (dedupe then falls back to exact string equality).
|
|
749
|
+
* @param url - the raw URL.
|
|
750
|
+
* @returns the normalized dedupe key.
|
|
751
|
+
*/
|
|
752
|
+
function normalizeUrl(url) {
|
|
753
|
+
try {
|
|
754
|
+
const parsed = new URL(url);
|
|
755
|
+
parsed.protocol = parsed.protocol.toLowerCase();
|
|
756
|
+
parsed.hostname = parsed.hostname.toLowerCase();
|
|
757
|
+
if (parsed.protocol === "https:" && parsed.port === "443" || parsed.protocol === "http:" && parsed.port === "80") parsed.port = "";
|
|
758
|
+
if (parsed.pathname.length > 1 && parsed.pathname.endsWith("/")) parsed.pathname = parsed.pathname.slice(0, -1);
|
|
759
|
+
return parsed.toString();
|
|
760
|
+
} catch {
|
|
761
|
+
return url;
|
|
762
|
+
}
|
|
763
|
+
}
|
|
764
|
+
/**
|
|
765
|
+
* Decide whether a source URL passes the host allow/block lists. Entries are
|
|
766
|
+
* host suffixes (`gov.cn` matches `www.gov.cn`) or URL prefixes
|
|
767
|
+
* (`https://www.gov.cn/zhengce/`). The blocklist wins. An empty allowlist
|
|
768
|
+
* allows every host.
|
|
769
|
+
* @param url - the candidate URL.
|
|
770
|
+
* @param allowlist - configured allow entries.
|
|
771
|
+
* @param blocklist - configured block entries.
|
|
772
|
+
* @returns whether the URL may be tracked.
|
|
773
|
+
*/
|
|
774
|
+
function sourceAllowed(url, allowlist, blocklist) {
|
|
775
|
+
let host = "";
|
|
776
|
+
try {
|
|
777
|
+
host = new URL(url).hostname.toLowerCase();
|
|
778
|
+
} catch {
|
|
779
|
+
host = "";
|
|
780
|
+
}
|
|
781
|
+
const matches = (entry) => {
|
|
782
|
+
const normalized = entry.toLowerCase();
|
|
783
|
+
if (normalized.includes("://")) return url.toLowerCase().startsWith(normalized);
|
|
784
|
+
return host === normalized || host.endsWith(`.${normalized}`);
|
|
785
|
+
};
|
|
786
|
+
if (blocklist.some(matches)) return false;
|
|
787
|
+
if (allowlist.length === 0) return true;
|
|
788
|
+
return allowlist.some(matches);
|
|
789
|
+
}
|
|
790
|
+
/**
|
|
791
|
+
* Read a timeline store, tolerating a missing file. Corrupt lines are skipped
|
|
792
|
+
* and counted (never silently: the caller surfaces the count), because a
|
|
793
|
+
* single torn line must not drop the rest of a durable log.
|
|
794
|
+
* @param path - absolute path of the JSONL file.
|
|
795
|
+
* @returns the parsed entries plus the number of skipped corrupt lines.
|
|
796
|
+
*/
|
|
797
|
+
async function readTimeline(path) {
|
|
798
|
+
let text;
|
|
799
|
+
try {
|
|
800
|
+
text = await readFile(path, "utf8");
|
|
801
|
+
} catch (error) {
|
|
802
|
+
if (error.code === "ENOENT") return {
|
|
803
|
+
entries: [],
|
|
804
|
+
corrupt: 0
|
|
805
|
+
};
|
|
806
|
+
throw error;
|
|
807
|
+
}
|
|
808
|
+
const entries = [];
|
|
809
|
+
let corrupt = 0;
|
|
810
|
+
for (const line of text.split("\n")) {
|
|
811
|
+
const trimmed = line.trim();
|
|
812
|
+
if (trimmed.length === 0) continue;
|
|
813
|
+
try {
|
|
814
|
+
const parsed = JSON.parse(trimmed);
|
|
815
|
+
if (typeof parsed.url !== "string" || typeof parsed.title !== "string") throw new Error("missing fields");
|
|
816
|
+
entries.push(parsed);
|
|
817
|
+
} catch {
|
|
818
|
+
corrupt += 1;
|
|
819
|
+
}
|
|
820
|
+
}
|
|
821
|
+
return {
|
|
822
|
+
entries,
|
|
823
|
+
corrupt
|
|
824
|
+
};
|
|
825
|
+
}
|
|
826
|
+
/**
|
|
827
|
+
* Merge a batch into the store and persist: dedupe by normalized URL (both
|
|
828
|
+
* against the store and within the batch), append the survivors, and rewrite
|
|
829
|
+
* the file keeping only the newest `maxEntries` when the cap is exceeded.
|
|
830
|
+
* @param path - absolute path of the JSONL file.
|
|
831
|
+
* @param batch - candidate entries, in arrival order.
|
|
832
|
+
* @param maxEntries - retention cap (oldest dropped first).
|
|
833
|
+
* @returns the merge outcome.
|
|
834
|
+
*/
|
|
835
|
+
async function mergeTimeline(path, batch, maxEntries) {
|
|
836
|
+
const { entries } = await readTimeline(path);
|
|
837
|
+
const seen = new Set(entries.map((entry) => normalizeUrl(entry.url)));
|
|
838
|
+
const added = [];
|
|
839
|
+
let duplicates = 0;
|
|
840
|
+
for (const entry of batch) {
|
|
841
|
+
const key = normalizeUrl(entry.url);
|
|
842
|
+
if (seen.has(key)) {
|
|
843
|
+
duplicates += 1;
|
|
844
|
+
continue;
|
|
845
|
+
}
|
|
846
|
+
seen.add(key);
|
|
847
|
+
added.push(entry);
|
|
848
|
+
}
|
|
849
|
+
const merged = [...entries, ...added];
|
|
850
|
+
const kept = merged.length > maxEntries ? merged.slice(merged.length - maxEntries) : merged;
|
|
851
|
+
const truncated = kept.length !== merged.length;
|
|
852
|
+
await mkdir(dirname(path), { recursive: true });
|
|
853
|
+
await writeFile(path, kept.map((entry) => JSON.stringify(entry)).join("\n") + (kept.length > 0 ? "\n" : ""), "utf8");
|
|
854
|
+
return {
|
|
855
|
+
added,
|
|
856
|
+
duplicates,
|
|
857
|
+
total: kept.length,
|
|
858
|
+
truncated
|
|
859
|
+
};
|
|
860
|
+
}
|
|
861
|
+
//#endregion
|
|
862
|
+
//#region src/tools/track.ts
|
|
863
|
+
/** How many snapshot fetches may run concurrently. */
|
|
864
|
+
const FETCH_CONCURRENCY = 4;
|
|
865
|
+
/**
|
|
866
|
+
* Run `limit`-bounded concurrent workers over `items`.
|
|
867
|
+
* @param items - the work items.
|
|
868
|
+
* @param limit - concurrency bound.
|
|
869
|
+
* @param worker - the per-item async worker.
|
|
870
|
+
*/
|
|
871
|
+
async function pool(items, limit, worker) {
|
|
872
|
+
let next = 0;
|
|
873
|
+
const runners = Array.from({ length: Math.min(limit, items.length) }, async () => {
|
|
874
|
+
while (next < items.length) {
|
|
875
|
+
const item = items[next];
|
|
876
|
+
next += 1;
|
|
877
|
+
if (item !== void 0) await worker(item);
|
|
878
|
+
}
|
|
879
|
+
});
|
|
880
|
+
await Promise.all(runners);
|
|
881
|
+
}
|
|
882
|
+
/**
|
|
883
|
+
* Build the `industry_track` tool definition.
|
|
884
|
+
* @param ctx - the plugin context (event emission + optional web lookup).
|
|
885
|
+
* @param config - the resolved plugin config.
|
|
886
|
+
* @returns the tool definition to register.
|
|
887
|
+
*/
|
|
888
|
+
function buildIndustryTrackTool(ctx, config) {
|
|
889
|
+
return defineTool({
|
|
890
|
+
name: "industry_track",
|
|
891
|
+
description: "行业研究员的政策与动态跟踪工具:经官方 ctx.web 检索行业政策/要闻,产出带日期、标题、来源 URL、摘要与抓取快照哈希的结构化时间线条目,追加去重写入 timeline.jsonl。只使用公开源;来源不可达或数据缺口会显式列出,禁止编造。仅供研究,不构成投资建议。",
|
|
892
|
+
parameters: {
|
|
893
|
+
industry: {
|
|
894
|
+
type: "string",
|
|
895
|
+
required: true,
|
|
896
|
+
description: "行业名(作为目录段,如「白酒」)"
|
|
897
|
+
},
|
|
898
|
+
topics: {
|
|
899
|
+
type: "array",
|
|
900
|
+
items: { type: "string" },
|
|
901
|
+
description: "检索主题列表;缺省用「<行业> 行业 政策」与「<行业> 行业 动态 要闻」"
|
|
902
|
+
},
|
|
903
|
+
since: {
|
|
904
|
+
type: "string",
|
|
905
|
+
description: "只保留该日期(ISO-8601)及之后的条目(按来源发布日期过滤;无日期的来源保留)"
|
|
906
|
+
}
|
|
907
|
+
},
|
|
908
|
+
output: {
|
|
909
|
+
schema: {
|
|
910
|
+
type: "object",
|
|
911
|
+
properties: {
|
|
912
|
+
industry: {
|
|
913
|
+
type: "string",
|
|
914
|
+
required: true
|
|
915
|
+
},
|
|
916
|
+
path: {
|
|
917
|
+
type: "string",
|
|
918
|
+
required: true
|
|
919
|
+
},
|
|
920
|
+
added: {
|
|
921
|
+
type: "array",
|
|
922
|
+
items: { type: "json" },
|
|
923
|
+
required: true
|
|
924
|
+
},
|
|
925
|
+
duplicates: {
|
|
926
|
+
type: "number",
|
|
927
|
+
required: true
|
|
928
|
+
},
|
|
929
|
+
blocked: {
|
|
930
|
+
type: "number",
|
|
931
|
+
required: true
|
|
932
|
+
},
|
|
933
|
+
tooOld: {
|
|
934
|
+
type: "number",
|
|
935
|
+
required: true
|
|
936
|
+
},
|
|
937
|
+
fetchFailed: {
|
|
938
|
+
type: "array",
|
|
939
|
+
items: { type: "json" },
|
|
940
|
+
required: true
|
|
941
|
+
},
|
|
942
|
+
total: {
|
|
943
|
+
type: "number",
|
|
944
|
+
required: true
|
|
945
|
+
},
|
|
946
|
+
truncated: {
|
|
947
|
+
type: "boolean",
|
|
948
|
+
required: true
|
|
949
|
+
},
|
|
950
|
+
corruptSkipped: {
|
|
951
|
+
type: "number",
|
|
952
|
+
required: true
|
|
953
|
+
}
|
|
954
|
+
},
|
|
955
|
+
additionalProperties: false
|
|
956
|
+
},
|
|
957
|
+
render: (_args, value) => {
|
|
958
|
+
const current = value;
|
|
959
|
+
const lines = [`行业「${current.industry}」政策与动态:新增 ${current.added.length} 条(去重 ${current.duplicates},拦截 ${current.blocked},过早 ${current.tooOld}),时间线共 ${current.total} 条 → ${current.path}`];
|
|
960
|
+
for (const entry of current.added.slice(0, 10)) lines.push(`- ${entry.date ?? "日期未知"} — ${entry.title}(${entry.url})`);
|
|
961
|
+
if (current.added.length > 10) lines.push(`- ……其余 ${current.added.length - 10} 条见 timeline.jsonl`);
|
|
962
|
+
if (current.fetchFailed.length > 0) lines.push(`快照抓取失败 ${current.fetchFailed.length} 条(仍以纯引用条目记录):${current.fetchFailed.map((failure) => failure.url).join(";")}`);
|
|
963
|
+
if (current.truncated) lines.push("已达 retention 上限,最旧条目已被裁剪。");
|
|
964
|
+
if (current.corruptSkipped > 0) lines.push(`警告:跳过 ${current.corruptSkipped} 行损坏的既有时间线记录。`);
|
|
965
|
+
return [{
|
|
966
|
+
type: "text",
|
|
967
|
+
text: lines.join("\n")
|
|
968
|
+
}];
|
|
969
|
+
}
|
|
970
|
+
},
|
|
971
|
+
timeoutMs: Math.max(6e4, config.fetchTimeoutMs * config.track.maxFetchesPerCall),
|
|
972
|
+
async execute(args, exec) {
|
|
973
|
+
const web = requireWeb(ctx, config, "industry_track");
|
|
974
|
+
const { dir, name } = industryDirOf(config, workspaceOf(exec), args.industry);
|
|
975
|
+
const path = timelinePathOf(dir);
|
|
976
|
+
const topics = (args.topics ?? [`${name} 行业 政策`, `${name} 行业 动态 要闻`]).map((topic) => topic.trim()).filter((topic) => topic.length > 0);
|
|
977
|
+
if (topics.length === 0) throw new Error("topics must contain at least one non-empty topic");
|
|
978
|
+
const since = args.since?.trim();
|
|
979
|
+
if (since !== void 0 && Number.isNaN(Date.parse(since))) throw new Error(`since must be an ISO-8601 date, got ${JSON.stringify(args.since)}`);
|
|
980
|
+
const byUrl = /* @__PURE__ */ new Map();
|
|
981
|
+
for (const topic of topics) {
|
|
982
|
+
const outcome = await web.search({
|
|
983
|
+
query: topic,
|
|
984
|
+
maxResults: config.track.maxResultsPerTopic
|
|
985
|
+
}, requestSignal(exec.signal, config.fetchTimeoutMs));
|
|
986
|
+
for (const source of outcome.sources) {
|
|
987
|
+
const existing = byUrl.get(source.url);
|
|
988
|
+
if (existing !== void 0) {
|
|
989
|
+
if (!existing.topics.includes(topic)) existing.topics.push(topic);
|
|
990
|
+
continue;
|
|
991
|
+
}
|
|
992
|
+
byUrl.set(source.url, {
|
|
993
|
+
url: source.url,
|
|
994
|
+
title: source.title ?? null,
|
|
995
|
+
snippet: source.snippet ?? null,
|
|
996
|
+
publishedAt: source.publishedAt ?? null,
|
|
997
|
+
topics: [topic]
|
|
998
|
+
});
|
|
999
|
+
}
|
|
1000
|
+
}
|
|
1001
|
+
let blocked = 0;
|
|
1002
|
+
let tooOld = 0;
|
|
1003
|
+
const candidates = [];
|
|
1004
|
+
for (const candidate of byUrl.values()) {
|
|
1005
|
+
if (!sourceAllowed(candidate.url, config.sourceAllowlist, config.sourceBlocklist)) {
|
|
1006
|
+
blocked += 1;
|
|
1007
|
+
continue;
|
|
1008
|
+
}
|
|
1009
|
+
if (since !== void 0 && candidate.publishedAt !== null) {
|
|
1010
|
+
const published = Date.parse(candidate.publishedAt);
|
|
1011
|
+
if (!Number.isNaN(published) && published < Date.parse(since)) {
|
|
1012
|
+
tooOld += 1;
|
|
1013
|
+
continue;
|
|
1014
|
+
}
|
|
1015
|
+
}
|
|
1016
|
+
candidates.push(candidate);
|
|
1017
|
+
}
|
|
1018
|
+
const now = (/* @__PURE__ */ new Date()).toISOString();
|
|
1019
|
+
const fetchFailed = [];
|
|
1020
|
+
const snapshots = /* @__PURE__ */ new Map();
|
|
1021
|
+
const fetchable = candidates.slice(0, config.track.maxFetchesPerCall);
|
|
1022
|
+
await pool(fetchable, FETCH_CONCURRENCY, async (candidate) => {
|
|
1023
|
+
try {
|
|
1024
|
+
const outcome = await web.fetch({ url: candidate.url }, requestSignal(exec.signal, config.fetchTimeoutMs));
|
|
1025
|
+
snapshots.set(candidate.url, sha256Of(outcome.body.content));
|
|
1026
|
+
} catch (error) {
|
|
1027
|
+
fetchFailed.push({
|
|
1028
|
+
url: candidate.url,
|
|
1029
|
+
note: webErrorMessage(error)
|
|
1030
|
+
});
|
|
1031
|
+
}
|
|
1032
|
+
});
|
|
1033
|
+
const batch = candidates.map((candidate) => {
|
|
1034
|
+
const snapshotHash = snapshots.get(candidate.url) ?? null;
|
|
1035
|
+
const failure = fetchFailed.find((entry) => entry.url === candidate.url);
|
|
1036
|
+
const beyondBudget = !fetchable.includes(candidate);
|
|
1037
|
+
return {
|
|
1038
|
+
date: candidate.publishedAt,
|
|
1039
|
+
title: candidate.title ?? candidate.url,
|
|
1040
|
+
url: candidate.url,
|
|
1041
|
+
summary: candidate.snippet,
|
|
1042
|
+
snapshotHash,
|
|
1043
|
+
capturedAt: now,
|
|
1044
|
+
topics: candidate.topics,
|
|
1045
|
+
...snapshotHash === null ? { note: failure !== void 0 ? `快照抓取失败:${failure.note}` : beyondBudget ? "超出本次抓取预算,未抓取快照" : "无快照" } : {}
|
|
1046
|
+
};
|
|
1047
|
+
});
|
|
1048
|
+
const { corrupt } = await readTimeline(path);
|
|
1049
|
+
const merge = await mergeTimeline(path, batch, config.timelineMaxEntries);
|
|
1050
|
+
const payload = {
|
|
1051
|
+
industry: name,
|
|
1052
|
+
path,
|
|
1053
|
+
added: merge.added.length,
|
|
1054
|
+
duplicates: merge.duplicates,
|
|
1055
|
+
total: merge.total
|
|
1056
|
+
};
|
|
1057
|
+
ctx.emit("industry-research/track", payload);
|
|
1058
|
+
return {
|
|
1059
|
+
industry: name,
|
|
1060
|
+
path,
|
|
1061
|
+
added: merge.added,
|
|
1062
|
+
duplicates: merge.duplicates,
|
|
1063
|
+
blocked,
|
|
1064
|
+
tooOld,
|
|
1065
|
+
fetchFailed,
|
|
1066
|
+
total: merge.total,
|
|
1067
|
+
truncated: merge.truncated,
|
|
1068
|
+
corruptSkipped: corrupt
|
|
1069
|
+
};
|
|
1070
|
+
}
|
|
1071
|
+
});
|
|
1072
|
+
}
|
|
1073
|
+
//#endregion
|
|
1074
|
+
//#region src/company.ts
|
|
1075
|
+
/**
|
|
1076
|
+
* Company scan support: read user-supplied data files from the workspace,
|
|
1077
|
+
* hash them, extract a lightweight outline (Markdown headings) and
|
|
1078
|
+
* figure-candidate lines (lines carrying digits, so every number the model
|
|
1079
|
+
* cites can point at a file and a line), and persist the scan card
|
|
1080
|
+
* (`card.json` + `card.md`). v1 reads text formats only — no PDF.
|
|
1081
|
+
* @module dsh-industry-research/company
|
|
1082
|
+
*/
|
|
1083
|
+
/** File extensions v1 can read as text. PDF and office formats are out of scope. */
|
|
1084
|
+
const READABLE_EXTENSIONS = /* @__PURE__ */ new Set([
|
|
1085
|
+
".md",
|
|
1086
|
+
".txt",
|
|
1087
|
+
".csv",
|
|
1088
|
+
".tsv",
|
|
1089
|
+
".json"
|
|
1090
|
+
]);
|
|
1091
|
+
/** The research-only disclaimer text shared by cards and reports. */
|
|
1092
|
+
const DISCLAIMER = "仅供研究,不构成投资建议";
|
|
1093
|
+
/** Maximum length of one surfaced figure-candidate line. */
|
|
1094
|
+
const FIGURE_LINE_CAP = 200;
|
|
1095
|
+
/** Maximum headings surfaced per file. */
|
|
1096
|
+
const OUTLINE_CAP = 50;
|
|
1097
|
+
/**
|
|
1098
|
+
* Read one containment-verified data file and extract its outline and figure
|
|
1099
|
+
* candidates. Unsupported extensions and unreadable entries throw — the tool
|
|
1100
|
+
* wraps them into the card's gap list where the contract allows skipping.
|
|
1101
|
+
* @param path - absolute, containment-verified file path.
|
|
1102
|
+
* @param maxBytes - per-file read cap.
|
|
1103
|
+
* @returns the scanned file.
|
|
1104
|
+
*/
|
|
1105
|
+
async function scanFile(path, maxBytes) {
|
|
1106
|
+
const ext = extname(path).toLowerCase();
|
|
1107
|
+
if (!READABLE_EXTENSIONS.has(ext)) throw new Error(`unsupported data-file extension ${JSON.stringify(ext)} (v1 reads ${[...READABLE_EXTENSIONS].join(", ")}; PDF is out of scope)`);
|
|
1108
|
+
const info = await stat(path);
|
|
1109
|
+
if (!info.isFile()) throw new Error(`not a regular file: ${path}`);
|
|
1110
|
+
if (info.size > maxBytes) throw new Error(`file exceeds scan.maxFileBytes (${info.size} > ${maxBytes}): ${path}`);
|
|
1111
|
+
const content = await readFile(path, "utf8");
|
|
1112
|
+
const lines = content.split("\n");
|
|
1113
|
+
const headings = [];
|
|
1114
|
+
const figures = [];
|
|
1115
|
+
if (ext === ".md") for (const line of lines) {
|
|
1116
|
+
const match = /^#{1,3}\s+(.+?)\s*$/u.exec(line);
|
|
1117
|
+
if (match?.[1] !== void 0 && headings.length < OUTLINE_CAP) headings.push(match[1]);
|
|
1118
|
+
}
|
|
1119
|
+
lines.forEach((raw, index) => {
|
|
1120
|
+
const text = raw.trim();
|
|
1121
|
+
if (text.length === 0 || !/\d/u.test(text)) return;
|
|
1122
|
+
figures.push({
|
|
1123
|
+
path,
|
|
1124
|
+
line: index + 1,
|
|
1125
|
+
text: text.length > FIGURE_LINE_CAP ? `${text.slice(0, FIGURE_LINE_CAP)}…` : text
|
|
1126
|
+
});
|
|
1127
|
+
});
|
|
1128
|
+
return {
|
|
1129
|
+
source: {
|
|
1130
|
+
path,
|
|
1131
|
+
sha256: sha256Of(content),
|
|
1132
|
+
bytes: info.size,
|
|
1133
|
+
lines: lines.length
|
|
1134
|
+
},
|
|
1135
|
+
outline: headings.length > 0 ? {
|
|
1136
|
+
path,
|
|
1137
|
+
headings
|
|
1138
|
+
} : void 0,
|
|
1139
|
+
figures,
|
|
1140
|
+
content
|
|
1141
|
+
};
|
|
1142
|
+
}
|
|
1143
|
+
/**
|
|
1144
|
+
* Persist a company card as `card.json` + `card.md` inside its directory.
|
|
1145
|
+
* The Markdown card is a template whose analytical sections stay explicitly
|
|
1146
|
+
* 待补 — the tool surfaces evidence (sources, outline, figure candidates); it
|
|
1147
|
+
* never invents business structure, financials, or risks.
|
|
1148
|
+
* @param dir - absolute company card directory.
|
|
1149
|
+
* @param card - the card to persist.
|
|
1150
|
+
* @returns the written file paths.
|
|
1151
|
+
*/
|
|
1152
|
+
async function writeCard(dir, card) {
|
|
1153
|
+
await mkdir(dir, { recursive: true });
|
|
1154
|
+
const cardJsonPath = join(dir, "card.json");
|
|
1155
|
+
const cardPath = join(dir, "card.md");
|
|
1156
|
+
await writeFile(cardJsonPath, `${JSON.stringify(card, null, 2)}\n`, "utf8");
|
|
1157
|
+
await writeFile(cardPath, renderCardMarkdown(card), "utf8");
|
|
1158
|
+
return {
|
|
1159
|
+
cardJsonPath,
|
|
1160
|
+
cardPath
|
|
1161
|
+
};
|
|
1162
|
+
}
|
|
1163
|
+
/**
|
|
1164
|
+
* Render the human-facing Markdown card. Every section that the scan could
|
|
1165
|
+
* not fill from evidence is an explicit 待补 line, never prose.
|
|
1166
|
+
* @param card - the card to render.
|
|
1167
|
+
* @returns the Markdown text.
|
|
1168
|
+
*/
|
|
1169
|
+
function renderCardMarkdown(card) {
|
|
1170
|
+
const lines = [
|
|
1171
|
+
`# 公司速览卡:${card.name}`,
|
|
1172
|
+
"",
|
|
1173
|
+
`> ${card.disclaimer}。扫描时间(asOf):${card.asOf}`,
|
|
1174
|
+
"",
|
|
1175
|
+
"## 业务结构",
|
|
1176
|
+
""
|
|
1177
|
+
];
|
|
1178
|
+
if (card.outline.length > 0) {
|
|
1179
|
+
lines.push("数据文件目录结构(供研究定位,非分析结论):");
|
|
1180
|
+
for (const outline of card.outline) {
|
|
1181
|
+
lines.push(`- \`${outline.path}\``);
|
|
1182
|
+
for (const heading of outline.headings) lines.push(` - ${heading}`);
|
|
1183
|
+
}
|
|
1184
|
+
} else lines.push("待补:未从数据文件中提取到目录结构。");
|
|
1185
|
+
lines.push("", "## 财务要点", "");
|
|
1186
|
+
if (card.figureCandidates.length > 0) {
|
|
1187
|
+
lines.push(`数字候选行(共 ${card.figureCandidates.length} 行,引用时必须标注文件与行号):`);
|
|
1188
|
+
for (const figure of card.figureCandidates.slice(0, 20)) lines.push(`- \`${figure.path}\`:${figure.line} — ${figure.text}`);
|
|
1189
|
+
if (card.figureCandidates.length > 20) lines.push(`- ……其余 ${card.figureCandidates.length - 20} 行见 card.json`);
|
|
1190
|
+
} else lines.push("待补:数据文件中未发现数字行。");
|
|
1191
|
+
lines.push("", "## 风险点", "", "待补:由研究者基于来源材料归纳;不得凭空列举。", "", "## 来源清单", "");
|
|
1192
|
+
if (card.sources.length > 0) for (const source of card.sources) lines.push(`- \`${source.path}\` — SHA-256 \`${source.sha256}\`,${source.bytes} 字节,${source.lines} 行`);
|
|
1193
|
+
else lines.push("无用户提供的数据文件。");
|
|
1194
|
+
if (card.webSources !== null) {
|
|
1195
|
+
lines.push("", "公开源检索(ctx.web):");
|
|
1196
|
+
for (const source of card.webSources) {
|
|
1197
|
+
const label = source.title ?? source.url;
|
|
1198
|
+
const when = source.publishedAt !== void 0 ? `(${source.publishedAt})` : "";
|
|
1199
|
+
lines.push(`- [${label}](${source.url})${when}`);
|
|
1200
|
+
}
|
|
1201
|
+
}
|
|
1202
|
+
lines.push("", "## 缺口声明", "");
|
|
1203
|
+
if (card.gaps.length > 0) for (const gap of card.gaps) lines.push(`- ${gap}`);
|
|
1204
|
+
else lines.push("无。");
|
|
1205
|
+
lines.push("");
|
|
1206
|
+
return lines.join("\n");
|
|
1207
|
+
}
|
|
1208
|
+
/**
|
|
1209
|
+
* Load a persisted card (for report assembly). Missing/corrupt cards fail
|
|
1210
|
+
* loud to the caller, which decides per-card tolerance.
|
|
1211
|
+
* @param cardJsonPath - absolute path of `card.json`.
|
|
1212
|
+
* @returns the parsed card.
|
|
1213
|
+
*/
|
|
1214
|
+
async function readCard(cardJsonPath) {
|
|
1215
|
+
const text = await readFile(cardJsonPath, "utf8");
|
|
1216
|
+
const parsed = JSON.parse(text);
|
|
1217
|
+
if (typeof parsed.name !== "string" || !Array.isArray(parsed.sources) || !Array.isArray(parsed.gaps)) throw new Error(`company card at ${cardJsonPath} is malformed`);
|
|
1218
|
+
return parsed;
|
|
1219
|
+
}
|
|
1220
|
+
/**
|
|
1221
|
+
* Apply the figure-candidate budget to a scan result set, keeping files in
|
|
1222
|
+
* scan order and noting the cut in the card gaps when it applies.
|
|
1223
|
+
* @param figures - all candidates, in scan order.
|
|
1224
|
+
* @param config - resolved config (budget source).
|
|
1225
|
+
* @returns the bounded candidate list.
|
|
1226
|
+
*/
|
|
1227
|
+
function boundFigures(figures, config) {
|
|
1228
|
+
return figures.slice(0, config.scan.maxFigureCandidates);
|
|
1229
|
+
}
|
|
1230
|
+
//#endregion
|
|
1231
|
+
//#region src/tools/company.ts
|
|
1232
|
+
/**
|
|
1233
|
+
* Build the `company_scan` tool definition.
|
|
1234
|
+
* @param ctx - the plugin context (optional web lookup).
|
|
1235
|
+
* @param config - the resolved plugin config.
|
|
1236
|
+
* @returns the tool definition to register.
|
|
1237
|
+
*/
|
|
1238
|
+
function buildCompanyScanTool(ctx, config) {
|
|
1239
|
+
return defineTool({
|
|
1240
|
+
name: "company_scan",
|
|
1241
|
+
description: "公司研究员的速览卡工具:以用户提供的工作区数据文件(年报摘录、数据表)为主、ctx.web 公开检索为辅,产出公司速览卡(业务结构 / 财务要点 / 风险点框架),所有数字都能标注来源文件与行号。不接付费/需登录数据源;缺口显式声明,禁止编造公司数字。仅供研究,不构成投资建议。",
|
|
1242
|
+
parameters: {
|
|
1243
|
+
name: {
|
|
1244
|
+
type: "string",
|
|
1245
|
+
required: true,
|
|
1246
|
+
description: "公司名(作为目录段,如「样例酒业」)"
|
|
1247
|
+
},
|
|
1248
|
+
dataFiles: {
|
|
1249
|
+
type: "array",
|
|
1250
|
+
items: { type: "string" },
|
|
1251
|
+
description: "工作区内数据文件的相对路径列表(.md/.txt/.csv/.tsv/.json;v1 不解析 PDF)"
|
|
1252
|
+
},
|
|
1253
|
+
web: {
|
|
1254
|
+
type: "boolean",
|
|
1255
|
+
description: "是否做 web 公开源补充检索(默认 true;config.offline 时自动跳过)"
|
|
1256
|
+
}
|
|
1257
|
+
},
|
|
1258
|
+
output: {
|
|
1259
|
+
schema: {
|
|
1260
|
+
type: "object",
|
|
1261
|
+
properties: {
|
|
1262
|
+
name: {
|
|
1263
|
+
type: "string",
|
|
1264
|
+
required: true
|
|
1265
|
+
},
|
|
1266
|
+
slug: {
|
|
1267
|
+
type: "string",
|
|
1268
|
+
required: true
|
|
1269
|
+
},
|
|
1270
|
+
dir: {
|
|
1271
|
+
type: "string",
|
|
1272
|
+
required: true
|
|
1273
|
+
},
|
|
1274
|
+
cardPath: {
|
|
1275
|
+
type: "string",
|
|
1276
|
+
required: true
|
|
1277
|
+
},
|
|
1278
|
+
cardJsonPath: {
|
|
1279
|
+
type: "string",
|
|
1280
|
+
required: true
|
|
1281
|
+
},
|
|
1282
|
+
card: {
|
|
1283
|
+
type: "json",
|
|
1284
|
+
required: true
|
|
1285
|
+
},
|
|
1286
|
+
rejected: {
|
|
1287
|
+
type: "array",
|
|
1288
|
+
items: { type: "json" },
|
|
1289
|
+
required: true
|
|
1290
|
+
}
|
|
1291
|
+
},
|
|
1292
|
+
additionalProperties: false
|
|
1293
|
+
},
|
|
1294
|
+
render: (_args, value) => {
|
|
1295
|
+
const current = value;
|
|
1296
|
+
const card = current.card;
|
|
1297
|
+
const lines = [`公司速览卡「${card.name}」→ ${current.cardPath}`, `数据文件 ${card.sources.length} 份,数字候选行 ${card.figureCandidates.length} 行(引用数字必须标注文件与行号)。`];
|
|
1298
|
+
if (card.webSources !== null && card.webSources.length > 0) lines.push(`公开源 ${card.webSources.length} 条:${card.webSources.map((source) => source.title ?? source.url).join(";")}`);
|
|
1299
|
+
if (current.rejected.length > 0) lines.push(`未采纳文件 ${current.rejected.length} 份:${current.rejected.map((entry) => `${entry.path}(${entry.reason})`).join(";")}`);
|
|
1300
|
+
if (card.gaps.length > 0) lines.push(`缺口声明:${card.gaps.join(";")}`);
|
|
1301
|
+
return [{
|
|
1302
|
+
type: "text",
|
|
1303
|
+
text: lines.join("\n")
|
|
1304
|
+
}];
|
|
1305
|
+
}
|
|
1306
|
+
},
|
|
1307
|
+
timeoutMs: Math.max(3e4, config.fetchTimeoutMs * 3),
|
|
1308
|
+
async execute(args, exec) {
|
|
1309
|
+
const cwd = workspaceOf(exec);
|
|
1310
|
+
const { dir, slug } = companyDirOf(config, cwd, args.name);
|
|
1311
|
+
const now = (/* @__PURE__ */ new Date()).toISOString();
|
|
1312
|
+
const gaps = [];
|
|
1313
|
+
const rejected = [];
|
|
1314
|
+
const scanned = [];
|
|
1315
|
+
for (const file of args.dataFiles ?? []) {
|
|
1316
|
+
let absolute;
|
|
1317
|
+
try {
|
|
1318
|
+
absolute = resolveWorkspaceFile(cwd, file);
|
|
1319
|
+
} catch (error) {
|
|
1320
|
+
throw error;
|
|
1321
|
+
}
|
|
1322
|
+
try {
|
|
1323
|
+
scanned.push(await scanFile(absolute, config.scan.maxFileBytes));
|
|
1324
|
+
} catch (error) {
|
|
1325
|
+
rejected.push({
|
|
1326
|
+
path: file,
|
|
1327
|
+
reason: error instanceof Error ? error.message : String(error)
|
|
1328
|
+
});
|
|
1329
|
+
}
|
|
1330
|
+
}
|
|
1331
|
+
if ((args.dataFiles ?? []).length === 0) gaps.push("未提供数据文件(dataFiles):业务结构 / 财务要点 / 风险点均待补");
|
|
1332
|
+
let webSources = null;
|
|
1333
|
+
if (args.web === false) gaps.push("按调用方要求未做 web 公开源检索");
|
|
1334
|
+
else if (config.offline) gaps.push("config.offline 为 true,未做 web 公开源检索");
|
|
1335
|
+
else {
|
|
1336
|
+
const web = lookupWeb(ctx);
|
|
1337
|
+
if (web === void 0) gaps.push("ctx.web 未挂载,未做 web 公开源检索");
|
|
1338
|
+
else try {
|
|
1339
|
+
webSources = (await web.search({
|
|
1340
|
+
query: `${args.name} 公司 业务 简介`,
|
|
1341
|
+
maxResults: 5
|
|
1342
|
+
}, requestSignal(exec.signal, config.fetchTimeoutMs))).sources.map((source) => ({
|
|
1343
|
+
url: source.url,
|
|
1344
|
+
...source.title !== void 0 ? { title: source.title } : {},
|
|
1345
|
+
...source.snippet !== void 0 ? { snippet: source.snippet } : {},
|
|
1346
|
+
...source.publishedAt !== void 0 ? { publishedAt: source.publishedAt } : {}
|
|
1347
|
+
}));
|
|
1348
|
+
} catch (error) {
|
|
1349
|
+
gaps.push(`web 公开源检索失败:${webErrorMessage(error)}`);
|
|
1350
|
+
}
|
|
1351
|
+
}
|
|
1352
|
+
const figures = boundFigures(scanned.flatMap((file) => file.figures), config);
|
|
1353
|
+
const totalFigures = scanned.reduce((sum, file) => sum + file.figures.length, 0);
|
|
1354
|
+
if (totalFigures > figures.length) gaps.push(`数字候选行超出 scan.maxFigureCandidates(${totalFigures} > ${figures.length}),仅保留前 ${figures.length} 行`);
|
|
1355
|
+
if (scanned.length > 0 && totalFigures === 0) gaps.push("数据文件中未发现数字行:财务要点待补");
|
|
1356
|
+
const card = {
|
|
1357
|
+
name: args.name.trim(),
|
|
1358
|
+
slug,
|
|
1359
|
+
asOf: now,
|
|
1360
|
+
sources: scanned.map((file) => file.source),
|
|
1361
|
+
outline: scanned.flatMap((file) => file.outline !== void 0 ? [file.outline] : []),
|
|
1362
|
+
figureCandidates: figures,
|
|
1363
|
+
webSources,
|
|
1364
|
+
gaps,
|
|
1365
|
+
disclaimer: "仅供研究,不构成投资建议"
|
|
1366
|
+
};
|
|
1367
|
+
const { cardJsonPath, cardPath } = await writeCard(dir, card);
|
|
1368
|
+
return {
|
|
1369
|
+
name: card.name,
|
|
1370
|
+
slug,
|
|
1371
|
+
dir,
|
|
1372
|
+
cardPath,
|
|
1373
|
+
cardJsonPath,
|
|
1374
|
+
card,
|
|
1375
|
+
rejected
|
|
1376
|
+
};
|
|
1377
|
+
}
|
|
1378
|
+
});
|
|
1379
|
+
}
|
|
1380
|
+
//#endregion
|
|
1381
|
+
//#region src/report.ts
|
|
1382
|
+
/**
|
|
1383
|
+
* Report assembly for `industry_report`: load nothing itself — it turns the
|
|
1384
|
+
* already-loaded artifacts (chain map, timeline, company cards) into the
|
|
1385
|
+
* frozen evidence/sections/claims contract, validates model-authored drafts,
|
|
1386
|
+
* auto-drafts when the model supplied none, and renders the builtin-fallback
|
|
1387
|
+
* Markdown + manifest when no `ctx.researchReport` engine is mounted.
|
|
1388
|
+
* @module dsh-industry-research/report
|
|
1389
|
+
*/
|
|
1390
|
+
/** The standard auto-draft section keys selectable via the `sections` argument. */
|
|
1391
|
+
const AUTO_SECTIONS = [
|
|
1392
|
+
"overview",
|
|
1393
|
+
"chain",
|
|
1394
|
+
"timeline",
|
|
1395
|
+
"companies",
|
|
1396
|
+
"gaps"
|
|
1397
|
+
];
|
|
1398
|
+
/**
|
|
1399
|
+
* Build the frozen-contract evidence list from the loaded artifacts: one
|
|
1400
|
+
* entry per artifact file, carrying the verbatim content for byte-level
|
|
1401
|
+
* checks.
|
|
1402
|
+
* @param artifacts - the loaded artifacts.
|
|
1403
|
+
* @param capturedAt - ISO-8601 assembly time recorded on every entry.
|
|
1404
|
+
* @returns the evidence list.
|
|
1405
|
+
*/
|
|
1406
|
+
function buildEvidence(artifacts, capturedAt) {
|
|
1407
|
+
const evidence = [];
|
|
1408
|
+
if (artifacts.chain !== void 0) evidence.push({
|
|
1409
|
+
id: "E-chain",
|
|
1410
|
+
title: "产业链结构图 chain.json",
|
|
1411
|
+
origin: artifacts.chain.path,
|
|
1412
|
+
content: artifacts.chain.content,
|
|
1413
|
+
capturedAt
|
|
1414
|
+
});
|
|
1415
|
+
if (artifacts.timeline !== void 0) evidence.push({
|
|
1416
|
+
id: "E-timeline",
|
|
1417
|
+
title: "政策与动态时间线 timeline.jsonl",
|
|
1418
|
+
origin: artifacts.timeline.path,
|
|
1419
|
+
content: artifacts.timeline.content,
|
|
1420
|
+
capturedAt
|
|
1421
|
+
});
|
|
1422
|
+
for (const { path, content, card } of artifacts.cards) evidence.push({
|
|
1423
|
+
id: `E-company-${card.slug}`,
|
|
1424
|
+
title: `公司速览卡 ${card.name}`,
|
|
1425
|
+
origin: path,
|
|
1426
|
+
content,
|
|
1427
|
+
capturedAt
|
|
1428
|
+
});
|
|
1429
|
+
return evidence;
|
|
1430
|
+
}
|
|
1431
|
+
/**
|
|
1432
|
+
* Validate a draft against the registered evidence: every claim's
|
|
1433
|
+
* `evidenceIds` must reference registered evidence, and every `claimIds`
|
|
1434
|
+
* reference in the sections must resolve to a registered claim.
|
|
1435
|
+
* @param draft - the candidate draft.
|
|
1436
|
+
* @param evidenceIds - registered evidence ids.
|
|
1437
|
+
* @returns human-readable problems (empty when valid).
|
|
1438
|
+
*/
|
|
1439
|
+
function validateDraft(draft, evidenceIds) {
|
|
1440
|
+
const problems = [];
|
|
1441
|
+
if (typeof draft.title !== "string" || draft.title.trim().length === 0) problems.push("draft.title must be a non-empty string");
|
|
1442
|
+
if (draft.sections.length === 0) problems.push("draft.sections must not be empty");
|
|
1443
|
+
const claimIds = /* @__PURE__ */ new Set();
|
|
1444
|
+
for (const claim of draft.claims) {
|
|
1445
|
+
if (typeof claim.id !== "string" || claim.id.trim().length === 0) {
|
|
1446
|
+
problems.push("a claim has an empty id");
|
|
1447
|
+
continue;
|
|
1448
|
+
}
|
|
1449
|
+
if (claimIds.has(claim.id)) problems.push(`duplicate claim id "${claim.id}"`);
|
|
1450
|
+
claimIds.add(claim.id);
|
|
1451
|
+
for (const evidenceId of claim.evidenceIds) if (!evidenceIds.has(evidenceId)) problems.push(`claim "${claim.id}" references unknown evidence "${evidenceId}"`);
|
|
1452
|
+
}
|
|
1453
|
+
for (const section of draft.sections) for (const paragraph of section.paragraphs) for (const claimId of paragraph.claimIds ?? []) if (!claimIds.has(claimId)) problems.push(`section "${section.heading}" references unregistered claim "${claimId}"`);
|
|
1454
|
+
return problems;
|
|
1455
|
+
}
|
|
1456
|
+
/**
|
|
1457
|
+
* Build the mechanical draft when the model supplied none: sourced chain
|
|
1458
|
+
* metrics and recent timeline entries become claims bound to their artifact
|
|
1459
|
+
* evidence; everything unsourced stays a declared gap, never prose.
|
|
1460
|
+
* @param industry - the industry display name.
|
|
1461
|
+
* @param artifacts - the loaded artifacts.
|
|
1462
|
+
* @param sections - which standard sections to include (default all).
|
|
1463
|
+
* @returns the auto-built draft.
|
|
1464
|
+
*/
|
|
1465
|
+
function autoDraft(industry, artifacts, sections = [...AUTO_SECTIONS]) {
|
|
1466
|
+
const wanted = new Set(sections);
|
|
1467
|
+
const claims = [];
|
|
1468
|
+
const draftSections = [];
|
|
1469
|
+
if (wanted.has("overview")) {
|
|
1470
|
+
const presence = [
|
|
1471
|
+
artifacts.chain !== void 0 ? `产业链结构图:${artifacts.chain.map.nodes.length} 节点 / ${artifacts.chain.map.edges.length} 边` : "产业链结构图:缺失",
|
|
1472
|
+
artifacts.timeline !== void 0 ? `政策与动态:${artifacts.timeline.entries.length} 条` : "政策与动态:无",
|
|
1473
|
+
`公司速览卡:${artifacts.cards.length} 张`
|
|
1474
|
+
];
|
|
1475
|
+
draftSections.push({
|
|
1476
|
+
heading: "概览",
|
|
1477
|
+
paragraphs: [{ text: `本报告汇总「${industry}」行业研究工作区内的已有材料:${presence.join(";")}。所有数值均以来源回溯表中的证据为准,缺口见文末清单。` }]
|
|
1478
|
+
});
|
|
1479
|
+
}
|
|
1480
|
+
if (wanted.has("chain")) {
|
|
1481
|
+
const paragraphs = [];
|
|
1482
|
+
if (artifacts.chain !== void 0) {
|
|
1483
|
+
const map = artifacts.chain.map;
|
|
1484
|
+
for (const tier of [
|
|
1485
|
+
"upstream",
|
|
1486
|
+
"midstream",
|
|
1487
|
+
"downstream"
|
|
1488
|
+
]) {
|
|
1489
|
+
const names = map.nodes.filter((node) => node.tier === tier).map((node) => node.name);
|
|
1490
|
+
paragraphs.push({ text: `${tier}:${names.length > 0 ? names.join("、") : "(无节点)"}` });
|
|
1491
|
+
}
|
|
1492
|
+
const edgeText = map.edges.map((edge) => {
|
|
1493
|
+
return `${map.nodes.find((node) => node.id === edge.from)?.name ?? edge.from} → ${map.nodes.find((node) => node.id === edge.to)?.name ?? edge.to}${edge.note !== void 0 ? `(${edge.note})` : ""}`;
|
|
1494
|
+
});
|
|
1495
|
+
if (edgeText.length > 0) paragraphs.push({ text: `链上关系:${edgeText.join(";")}` });
|
|
1496
|
+
const metricClaims = [];
|
|
1497
|
+
for (const node of map.nodes) node.metrics.forEach((metric, index) => {
|
|
1498
|
+
if (metric.value === void 0 || metric.sourceRef === void 0) return;
|
|
1499
|
+
const id = `C-chain-${node.id}-${index}`;
|
|
1500
|
+
claims.push({
|
|
1501
|
+
id,
|
|
1502
|
+
text: `${node.name} 的 ${metric.key} 为 ${metric.value}${metric.unit ?? ""}${metric.asOf !== void 0 ? `(截至 ${metric.asOf},来源 ${metric.sourceRef})` : `(来源 ${metric.sourceRef})`}`,
|
|
1503
|
+
evidenceIds: ["E-chain"]
|
|
1504
|
+
});
|
|
1505
|
+
metricClaims.push(id);
|
|
1506
|
+
});
|
|
1507
|
+
if (metricClaims.length > 0) paragraphs.push({
|
|
1508
|
+
text: "链上关键指标见 claims 清单(逐条绑定来源证据)。",
|
|
1509
|
+
claimIds: metricClaims
|
|
1510
|
+
});
|
|
1511
|
+
else paragraphs.push({ text: "链上暂无有来源的指标数值(均为待补槽位)。" });
|
|
1512
|
+
} else paragraphs.push({ text: "待补:尚无产业链结构图(先运行 industry_map)。" });
|
|
1513
|
+
draftSections.push({
|
|
1514
|
+
heading: "产业链结构",
|
|
1515
|
+
paragraphs
|
|
1516
|
+
});
|
|
1517
|
+
}
|
|
1518
|
+
if (wanted.has("timeline")) {
|
|
1519
|
+
const paragraphs = [];
|
|
1520
|
+
if (artifacts.timeline !== void 0 && artifacts.timeline.entries.length > 0) {
|
|
1521
|
+
const recent = artifacts.timeline.entries.slice(-10);
|
|
1522
|
+
for (const [index, entry] of recent.entries()) {
|
|
1523
|
+
const id = `C-timeline-${index}`;
|
|
1524
|
+
claims.push({
|
|
1525
|
+
id,
|
|
1526
|
+
text: `${entry.date ?? "日期未知"}:${entry.title}(来源 ${entry.url})`,
|
|
1527
|
+
evidenceIds: ["E-timeline"]
|
|
1528
|
+
});
|
|
1529
|
+
paragraphs.push({
|
|
1530
|
+
text: `${entry.date ?? "日期未知"} — ${entry.title}`,
|
|
1531
|
+
claimIds: [id]
|
|
1532
|
+
});
|
|
1533
|
+
}
|
|
1534
|
+
if (artifacts.timeline.entries.length > recent.length) paragraphs.push({ text: `(时间线共 ${artifacts.timeline.entries.length} 条,此处仅列最近 ${recent.length} 条。)` });
|
|
1535
|
+
} else paragraphs.push({ text: "待补:尚无政策与动态条目(先运行 industry_track)。" });
|
|
1536
|
+
draftSections.push({
|
|
1537
|
+
heading: "政策与动态",
|
|
1538
|
+
paragraphs
|
|
1539
|
+
});
|
|
1540
|
+
}
|
|
1541
|
+
if (wanted.has("companies")) {
|
|
1542
|
+
const paragraphs = [];
|
|
1543
|
+
if (artifacts.cards.length > 0) for (const { card } of artifacts.cards) paragraphs.push({ text: `「${card.name}」(asOf ${card.asOf}):数据文件 ${card.sources.length} 份,数字候选行 ${card.figureCandidates.length} 行,缺口 ${card.gaps.length} 项;详见该卡 card.md。` });
|
|
1544
|
+
else paragraphs.push({ text: "待补:尚无公司速览卡(先运行 company_scan)。" });
|
|
1545
|
+
draftSections.push({
|
|
1546
|
+
heading: "公司速览",
|
|
1547
|
+
paragraphs
|
|
1548
|
+
});
|
|
1549
|
+
}
|
|
1550
|
+
if (wanted.has("gaps")) draftSections.push({
|
|
1551
|
+
heading: "缺口与待补",
|
|
1552
|
+
paragraphs: artifacts.gaps.length > 0 ? artifacts.gaps.map((gap) => ({ text: gap })) : [{ text: "本次组装未发现材料级缺口(指标级待补见产业链结构一节)。" }]
|
|
1553
|
+
});
|
|
1554
|
+
return {
|
|
1555
|
+
title: `${industry} 行业研究报告`,
|
|
1556
|
+
sections: draftSections,
|
|
1557
|
+
claims
|
|
1558
|
+
};
|
|
1559
|
+
}
|
|
1560
|
+
/**
|
|
1561
|
+
* Render the builtin-fallback Markdown report: sections with claim footnote
|
|
1562
|
+
* markers, a source-traceability appendix with SHA-256 per evidence, and the
|
|
1563
|
+
* unverified claims appendix. The fallback never claims independent
|
|
1564
|
+
* verification.
|
|
1565
|
+
* @param industry - the industry display name.
|
|
1566
|
+
* @param draft - the validated draft.
|
|
1567
|
+
* @param evidence - the registered evidence.
|
|
1568
|
+
* @param generatedAt - ISO-8601 generation time.
|
|
1569
|
+
* @returns the Markdown text.
|
|
1570
|
+
*/
|
|
1571
|
+
function renderFallbackMarkdown(industry, draft, evidence, generatedAt) {
|
|
1572
|
+
const lines = [
|
|
1573
|
+
`# ${draft.title}`,
|
|
1574
|
+
"",
|
|
1575
|
+
`> ${DISCLAIMER}。`,
|
|
1576
|
+
`> 行业:${industry};生成时间:${generatedAt};引擎:builtin-fallback(未经过独立核查引擎,claims 未做逐条核查)。`,
|
|
1577
|
+
""
|
|
1578
|
+
];
|
|
1579
|
+
for (const section of draft.sections) {
|
|
1580
|
+
lines.push(`## ${section.heading}`, "");
|
|
1581
|
+
for (const paragraph of section.paragraphs) {
|
|
1582
|
+
const markers = (paragraph.claimIds ?? []).map((id) => `[${id}]`).join("");
|
|
1583
|
+
lines.push(`${paragraph.text}${markers}`, "");
|
|
1584
|
+
}
|
|
1585
|
+
}
|
|
1586
|
+
lines.push("## 附录:来源回溯表", "", "| 证据 | 来源 | SHA-256 | 抓取时间 |", "|---|---|---|---|");
|
|
1587
|
+
for (const item of evidence) lines.push(`| ${item.id} | ${item.title}(\`${item.origin}\`) | \`${sha256Of(item.content)}\` | ${item.capturedAt} |`);
|
|
1588
|
+
lines.push("", "## 附录:claims 清单(builtin-fallback,未核查)", "");
|
|
1589
|
+
if (draft.claims.length > 0) {
|
|
1590
|
+
lines.push("| claim | 内容 | 证据 | 状态 |", "|---|---|---|---|");
|
|
1591
|
+
for (const claim of draft.claims) lines.push(`| ${claim.id} | ${claim.text} | ${claim.evidenceIds.join(", ")} | unverified |`);
|
|
1592
|
+
} else lines.push("无 claims。");
|
|
1593
|
+
lines.push("");
|
|
1594
|
+
return lines.join("\n");
|
|
1595
|
+
}
|
|
1596
|
+
/**
|
|
1597
|
+
* Write the fallback report directory: `report.md` + `manifest.json`.
|
|
1598
|
+
* @param reportDir - absolute target directory (created).
|
|
1599
|
+
* @param industry - the industry display name.
|
|
1600
|
+
* @param draft - the validated draft.
|
|
1601
|
+
* @param evidence - the registered evidence.
|
|
1602
|
+
* @param gaps - artifact-level gaps to record in the manifest.
|
|
1603
|
+
* @param generatedAt - ISO-8601 generation time.
|
|
1604
|
+
* @returns the written file paths.
|
|
1605
|
+
*/
|
|
1606
|
+
async function writeFallbackReport(reportDir, industry, draft, evidence, gaps, generatedAt) {
|
|
1607
|
+
await mkdir(reportDir, { recursive: true });
|
|
1608
|
+
const reportPath = join(reportDir, "report.md");
|
|
1609
|
+
const manifestPath = join(reportDir, "manifest.json");
|
|
1610
|
+
const manifest = {
|
|
1611
|
+
engine: "builtin-fallback",
|
|
1612
|
+
industry,
|
|
1613
|
+
title: draft.title,
|
|
1614
|
+
generatedAt,
|
|
1615
|
+
disclaimer: DISCLAIMER,
|
|
1616
|
+
evidence: evidence.map((item) => ({
|
|
1617
|
+
id: item.id,
|
|
1618
|
+
title: item.title,
|
|
1619
|
+
origin: item.origin,
|
|
1620
|
+
sha256: sha256Of(item.content),
|
|
1621
|
+
capturedAt: item.capturedAt,
|
|
1622
|
+
bytes: Buffer.byteLength(item.content, "utf8")
|
|
1623
|
+
})),
|
|
1624
|
+
claims: draft.claims.map((claim) => ({
|
|
1625
|
+
...claim,
|
|
1626
|
+
status: "unverified"
|
|
1627
|
+
})),
|
|
1628
|
+
gaps: [...gaps]
|
|
1629
|
+
};
|
|
1630
|
+
await writeFile(reportPath, renderFallbackMarkdown(industry, draft, evidence, generatedAt), "utf8");
|
|
1631
|
+
await writeFile(manifestPath, `${JSON.stringify(manifest, null, 2)}\n`, "utf8");
|
|
1632
|
+
return {
|
|
1633
|
+
reportPath,
|
|
1634
|
+
manifestPath
|
|
1635
|
+
};
|
|
1636
|
+
}
|
|
1637
|
+
/**
|
|
1638
|
+
* The report directory name for one generation: `<YYYYMMDD-HHmmss>` with a
|
|
1639
|
+
* numeric suffix when the same second collides.
|
|
1640
|
+
* @param at - the generation time.
|
|
1641
|
+
* @param exists - whether a candidate directory already exists.
|
|
1642
|
+
* @returns the directory name (not a path).
|
|
1643
|
+
*/
|
|
1644
|
+
function reportDirName(at, exists) {
|
|
1645
|
+
const pad = (value) => String(value).padStart(2, "0");
|
|
1646
|
+
const base = `${at.getFullYear()}${pad(at.getMonth() + 1)}${pad(at.getDate())}-${pad(at.getHours())}${pad(at.getMinutes())}${pad(at.getSeconds())}`;
|
|
1647
|
+
if (!exists(base)) return base;
|
|
1648
|
+
for (let suffix = 2;; suffix += 1) {
|
|
1649
|
+
const candidate = `${base}-${suffix}`;
|
|
1650
|
+
if (!exists(candidate)) return candidate;
|
|
1651
|
+
}
|
|
1652
|
+
}
|
|
1653
|
+
//#endregion
|
|
1654
|
+
//#region src/engine-bridge.ts
|
|
1655
|
+
/**
|
|
1656
|
+
* Look up the optional report engine.
|
|
1657
|
+
* @param ctx - the plugin context.
|
|
1658
|
+
* @returns the engine surface, or undefined when no engine is mounted.
|
|
1659
|
+
*/
|
|
1660
|
+
function lookupEngine(ctx) {
|
|
1661
|
+
return ctx.get("researchReport");
|
|
1662
|
+
}
|
|
1663
|
+
//#endregion
|
|
1664
|
+
//#region src/tools/report.ts
|
|
1665
|
+
/**
|
|
1666
|
+
* The `industry_report` tool: assemble one industry's report from the
|
|
1667
|
+
* workspace artifacts (chain map, timeline, company cards). With a mounted
|
|
1668
|
+
* `ctx.researchReport` engine the evidence/sections/claims go to its
|
|
1669
|
+
* `assemble` and the sealed directory plus per-claim verdicts come back;
|
|
1670
|
+
* without one the builtin fallback renders versioned Markdown
|
|
1671
|
+
* (`reports/<YYYYMMDD-HHmmss>/report.md` + `manifest.json` with the
|
|
1672
|
+
* source-traceability table) and says so honestly (`engine:
|
|
1673
|
+
* 'builtin-fallback'`). The model may author the draft (sections + claims) or
|
|
1674
|
+
* leave it to the mechanical auto-draft.
|
|
1675
|
+
* @module dsh-industry-research/tools/report
|
|
1676
|
+
*/
|
|
1677
|
+
/** The draft parameter schema (semantic reference checks live in {@link validateDraft}). */
|
|
1678
|
+
const DRAFT_PARAMETER = {
|
|
1679
|
+
type: "object",
|
|
1680
|
+
properties: {
|
|
1681
|
+
title: {
|
|
1682
|
+
type: "string",
|
|
1683
|
+
description: "报告标题;缺省「<行业> 行业研究报告」"
|
|
1684
|
+
},
|
|
1685
|
+
sections: {
|
|
1686
|
+
type: "array",
|
|
1687
|
+
required: true,
|
|
1688
|
+
items: {
|
|
1689
|
+
type: "object",
|
|
1690
|
+
properties: {
|
|
1691
|
+
heading: {
|
|
1692
|
+
type: "string",
|
|
1693
|
+
required: true
|
|
1694
|
+
},
|
|
1695
|
+
paragraphs: {
|
|
1696
|
+
type: "array",
|
|
1697
|
+
required: true,
|
|
1698
|
+
items: {
|
|
1699
|
+
type: "object",
|
|
1700
|
+
properties: {
|
|
1701
|
+
text: {
|
|
1702
|
+
type: "string",
|
|
1703
|
+
required: true
|
|
1704
|
+
},
|
|
1705
|
+
claimIds: {
|
|
1706
|
+
type: "array",
|
|
1707
|
+
items: { type: "string" },
|
|
1708
|
+
description: "本段引用的 claim id 列表"
|
|
1709
|
+
}
|
|
1710
|
+
},
|
|
1711
|
+
additionalProperties: false
|
|
1712
|
+
}
|
|
1713
|
+
}
|
|
1714
|
+
},
|
|
1715
|
+
additionalProperties: false
|
|
1716
|
+
}
|
|
1717
|
+
},
|
|
1718
|
+
claims: {
|
|
1719
|
+
type: "array",
|
|
1720
|
+
required: true,
|
|
1721
|
+
items: {
|
|
1722
|
+
type: "object",
|
|
1723
|
+
properties: {
|
|
1724
|
+
id: {
|
|
1725
|
+
type: "string",
|
|
1726
|
+
required: true,
|
|
1727
|
+
description: "claim id(如 C1)"
|
|
1728
|
+
},
|
|
1729
|
+
text: {
|
|
1730
|
+
type: "string",
|
|
1731
|
+
required: true,
|
|
1732
|
+
description: "断言内容(数字须与证据一致)"
|
|
1733
|
+
},
|
|
1734
|
+
evidenceIds: {
|
|
1735
|
+
type: "array",
|
|
1736
|
+
items: { type: "string" },
|
|
1737
|
+
required: true,
|
|
1738
|
+
description: "支撑证据 id(E-chain / E-timeline / E-company-<slug>)"
|
|
1739
|
+
}
|
|
1740
|
+
},
|
|
1741
|
+
additionalProperties: false
|
|
1742
|
+
}
|
|
1743
|
+
}
|
|
1744
|
+
},
|
|
1745
|
+
additionalProperties: false,
|
|
1746
|
+
description: "模型撰写的报告草稿;缺省时由工具按已有材料机械组装"
|
|
1747
|
+
};
|
|
1748
|
+
/**
|
|
1749
|
+
* Load every artifact a report assembles from. Missing artifacts are gaps,
|
|
1750
|
+
* not failures; unreadable ones are gaps with the reason recorded.
|
|
1751
|
+
* @param config - resolved plugin config.
|
|
1752
|
+
* @param cwd - absolute workspace root.
|
|
1753
|
+
* @param industry - validated industry argument.
|
|
1754
|
+
* @param companies - optional company-name filter for card inclusion.
|
|
1755
|
+
* @returns the loaded artifacts.
|
|
1756
|
+
*/
|
|
1757
|
+
async function loadArtifacts(config, cwd, industry, companies) {
|
|
1758
|
+
const { root, dir } = industryDirOf(config, cwd, industry);
|
|
1759
|
+
const gaps = [];
|
|
1760
|
+
const artifacts = {
|
|
1761
|
+
cards: [],
|
|
1762
|
+
gaps
|
|
1763
|
+
};
|
|
1764
|
+
const chainPath = chainPathOf(dir);
|
|
1765
|
+
try {
|
|
1766
|
+
const content = await readFile(chainPath, "utf8");
|
|
1767
|
+
const map = JSON.parse(content);
|
|
1768
|
+
artifacts.chain = {
|
|
1769
|
+
path: chainPath,
|
|
1770
|
+
content,
|
|
1771
|
+
map
|
|
1772
|
+
};
|
|
1773
|
+
gaps.push(...chainGaps(map));
|
|
1774
|
+
} catch (error) {
|
|
1775
|
+
if (error.code === "ENOENT") gaps.push("缺少产业链结构图 chain.json(先运行 industry_map)");
|
|
1776
|
+
else gaps.push(`chain.json 读取/解析失败:${error instanceof Error ? error.message : String(error)}`);
|
|
1777
|
+
}
|
|
1778
|
+
const timelinePath = timelinePathOf(dir);
|
|
1779
|
+
try {
|
|
1780
|
+
const content = await readFile(timelinePath, "utf8");
|
|
1781
|
+
const { entries, corrupt } = await readTimeline(timelinePath);
|
|
1782
|
+
if (entries.length > 0) artifacts.timeline = {
|
|
1783
|
+
path: timelinePath,
|
|
1784
|
+
content,
|
|
1785
|
+
entries
|
|
1786
|
+
};
|
|
1787
|
+
else gaps.push("timeline.jsonl 为空:尚无政策与动态条目(先运行 industry_track)");
|
|
1788
|
+
if (corrupt > 0) gaps.push(`timeline.jsonl 有 ${corrupt} 行损坏已跳过`);
|
|
1789
|
+
} catch (error) {
|
|
1790
|
+
if (error.code === "ENOENT") gaps.push("缺少政策与动态 timeline.jsonl(先运行 industry_track)");
|
|
1791
|
+
else throw error;
|
|
1792
|
+
}
|
|
1793
|
+
const companiesDir = join(root, "companies");
|
|
1794
|
+
let slugs = [];
|
|
1795
|
+
if (companies !== void 0 && companies.length > 0) slugs = companies.map((company) => companyDirOf(config, cwd, company).slug);
|
|
1796
|
+
else try {
|
|
1797
|
+
slugs = (await readdir(companiesDir, { withFileTypes: true })).filter((entry) => entry.isDirectory()).map((entry) => entry.name);
|
|
1798
|
+
} catch (error) {
|
|
1799
|
+
if (error.code !== "ENOENT") throw error;
|
|
1800
|
+
}
|
|
1801
|
+
for (const slug of slugs) {
|
|
1802
|
+
const cardJsonPath = join(companiesDir, slug, "card.json");
|
|
1803
|
+
try {
|
|
1804
|
+
const content = await readFile(cardJsonPath, "utf8");
|
|
1805
|
+
const card = await readCard(cardJsonPath);
|
|
1806
|
+
artifacts.cards.push({
|
|
1807
|
+
path: cardJsonPath,
|
|
1808
|
+
content,
|
|
1809
|
+
card
|
|
1810
|
+
});
|
|
1811
|
+
} catch (error) {
|
|
1812
|
+
if (error.code === "ENOENT") gaps.push(`公司「${slug}」无速览卡(先运行 company_scan)`);
|
|
1813
|
+
else gaps.push(`公司「${slug}」的 card.json 读取/解析失败:${error instanceof Error ? error.message : String(error)}`);
|
|
1814
|
+
}
|
|
1815
|
+
}
|
|
1816
|
+
if (artifacts.cards.length === 0 && slugs.length === 0) gaps.push("尚无公司速览卡(可选:运行 company_scan 补充公司维度)");
|
|
1817
|
+
return artifacts;
|
|
1818
|
+
}
|
|
1819
|
+
/**
|
|
1820
|
+
* Build the `industry_report` tool definition.
|
|
1821
|
+
* @param ctx - the plugin context (event emission + optional engine lookup).
|
|
1822
|
+
* @param config - the resolved plugin config.
|
|
1823
|
+
* @returns the tool definition to register.
|
|
1824
|
+
*/
|
|
1825
|
+
function buildIndustryReportTool(ctx, config) {
|
|
1826
|
+
return defineTool({
|
|
1827
|
+
name: "industry_report",
|
|
1828
|
+
description: "行业研究员的报告组装工具:汇总产业链结构图、政策时间线与公司速览卡,产出可核查的行业研究报告。挂载 ctx.researchReport 引擎时提交其 assemble 封存并回传逐 claim 核查结论;否则走内置降级路径(版本化 Markdown + 来源回溯表,如实标注 engine: builtin-fallback)。数字均须对应来源证据;仅供研究,不构成投资建议。",
|
|
1829
|
+
parameters: {
|
|
1830
|
+
industry: {
|
|
1831
|
+
type: "string",
|
|
1832
|
+
required: true,
|
|
1833
|
+
description: "行业名(作为目录段,如「白酒」)"
|
|
1834
|
+
},
|
|
1835
|
+
sections: {
|
|
1836
|
+
type: "array",
|
|
1837
|
+
items: {
|
|
1838
|
+
type: "string",
|
|
1839
|
+
enum: [...AUTO_SECTIONS]
|
|
1840
|
+
},
|
|
1841
|
+
description: `自动草稿包含哪些标准小节(${AUTO_SECTIONS.join("/")});提供 draft 时忽略`
|
|
1842
|
+
},
|
|
1843
|
+
companies: {
|
|
1844
|
+
type: "array",
|
|
1845
|
+
items: { type: "string" },
|
|
1846
|
+
description: "纳入报告的公司名列表;缺省纳入 companies/ 下全部速览卡"
|
|
1847
|
+
},
|
|
1848
|
+
draft: DRAFT_PARAMETER
|
|
1849
|
+
},
|
|
1850
|
+
output: {
|
|
1851
|
+
schema: {
|
|
1852
|
+
type: "object",
|
|
1853
|
+
properties: {
|
|
1854
|
+
industry: {
|
|
1855
|
+
type: "string",
|
|
1856
|
+
required: true
|
|
1857
|
+
},
|
|
1858
|
+
engine: {
|
|
1859
|
+
type: "string",
|
|
1860
|
+
enum: ["research-report", "builtin-fallback"],
|
|
1861
|
+
required: true
|
|
1862
|
+
},
|
|
1863
|
+
reportDir: {
|
|
1864
|
+
type: "string",
|
|
1865
|
+
required: true
|
|
1866
|
+
},
|
|
1867
|
+
reportPath: {
|
|
1868
|
+
oneOf: [{ type: "string" }, { type: "null" }],
|
|
1869
|
+
required: true
|
|
1870
|
+
},
|
|
1871
|
+
manifestPath: {
|
|
1872
|
+
oneOf: [{ type: "string" }, { type: "null" }],
|
|
1873
|
+
required: true
|
|
1874
|
+
},
|
|
1875
|
+
sealHash: {
|
|
1876
|
+
oneOf: [{ type: "string" }, { type: "null" }],
|
|
1877
|
+
required: true
|
|
1878
|
+
},
|
|
1879
|
+
verdicts: {
|
|
1880
|
+
oneOf: [{
|
|
1881
|
+
type: "array",
|
|
1882
|
+
items: { type: "json" }
|
|
1883
|
+
}, { type: "null" }],
|
|
1884
|
+
required: true
|
|
1885
|
+
},
|
|
1886
|
+
claims: {
|
|
1887
|
+
type: "number",
|
|
1888
|
+
required: true
|
|
1889
|
+
},
|
|
1890
|
+
evidence: {
|
|
1891
|
+
type: "number",
|
|
1892
|
+
required: true
|
|
1893
|
+
},
|
|
1894
|
+
gaps: {
|
|
1895
|
+
type: "array",
|
|
1896
|
+
items: { type: "string" },
|
|
1897
|
+
required: true
|
|
1898
|
+
},
|
|
1899
|
+
generatedAt: {
|
|
1900
|
+
type: "string",
|
|
1901
|
+
required: true
|
|
1902
|
+
}
|
|
1903
|
+
},
|
|
1904
|
+
additionalProperties: false
|
|
1905
|
+
},
|
|
1906
|
+
render: (_args, value) => {
|
|
1907
|
+
const current = value;
|
|
1908
|
+
const lines = current.engine === "research-report" ? [`行业「${current.industry}」研究报告已经独立核查引擎封存 → ${current.reportDir}`, `sealHash:${current.sealHash ?? ""};claims ${current.claims} 条(verified ${(current.verdicts ?? []).filter((verdict) => verdict.status === "verified").length} 条)。`] : [`行业「${current.industry}」研究报告(builtin-fallback,未经过独立核查引擎)→ ${current.reportPath}`, `claims ${current.claims} 条均标记 unverified;来源回溯表见 report.md 附录与 ${current.manifestPath ?? ""}。`];
|
|
1909
|
+
if (current.gaps.length > 0) lines.push(`缺口声明:${current.gaps.join(";")}`);
|
|
1910
|
+
lines.push("仅供研究,不构成投资建议。");
|
|
1911
|
+
return [{
|
|
1912
|
+
type: "text",
|
|
1913
|
+
text: lines.join("\n")
|
|
1914
|
+
}];
|
|
1915
|
+
}
|
|
1916
|
+
},
|
|
1917
|
+
timeoutMs: 6e4,
|
|
1918
|
+
async execute(args, exec) {
|
|
1919
|
+
const cwd = workspaceOf(exec);
|
|
1920
|
+
const { dir, name } = industryDirOf(config, cwd, args.industry);
|
|
1921
|
+
const generatedAt = (/* @__PURE__ */ new Date()).toISOString();
|
|
1922
|
+
const artifacts = await loadArtifacts(config, cwd, name, args.companies);
|
|
1923
|
+
const evidence = buildEvidence(artifacts, generatedAt);
|
|
1924
|
+
const evidenceIds = new Set(evidence.map((item) => item.id));
|
|
1925
|
+
let draft;
|
|
1926
|
+
if (args.draft !== void 0) {
|
|
1927
|
+
draft = {
|
|
1928
|
+
title: args.draft.title ?? `${name} 行业研究报告`,
|
|
1929
|
+
sections: args.draft.sections,
|
|
1930
|
+
claims: args.draft.claims
|
|
1931
|
+
};
|
|
1932
|
+
const problems = validateDraft(draft, evidenceIds);
|
|
1933
|
+
if (problems.length > 0) throw new Error(`draft 校验失败(${problems.length} 项):${problems.join(";")}(已登记证据:${[...evidenceIds].join(", ") || "无"})`);
|
|
1934
|
+
} else draft = autoDraft(name, artifacts, args.sections ?? [...AUTO_SECTIONS]);
|
|
1935
|
+
const engine = lookupEngine(ctx);
|
|
1936
|
+
let value;
|
|
1937
|
+
if (engine !== void 0) {
|
|
1938
|
+
const result = await engine.assemble({
|
|
1939
|
+
title: draft.title,
|
|
1940
|
+
topic: name,
|
|
1941
|
+
evidence,
|
|
1942
|
+
sections: draft.sections,
|
|
1943
|
+
claims: draft.claims
|
|
1944
|
+
});
|
|
1945
|
+
value = {
|
|
1946
|
+
industry: name,
|
|
1947
|
+
engine: "research-report",
|
|
1948
|
+
reportDir: result.reportDir,
|
|
1949
|
+
reportPath: null,
|
|
1950
|
+
manifestPath: null,
|
|
1951
|
+
sealHash: result.sealHash,
|
|
1952
|
+
verdicts: result.verdicts,
|
|
1953
|
+
claims: draft.claims.length,
|
|
1954
|
+
evidence: evidence.length,
|
|
1955
|
+
gaps: artifacts.gaps,
|
|
1956
|
+
generatedAt
|
|
1957
|
+
};
|
|
1958
|
+
} else {
|
|
1959
|
+
const reportsDir = reportsDirOf(dir);
|
|
1960
|
+
const dirName = reportDirName(/* @__PURE__ */ new Date(), (candidate) => existsSync(join(reportsDir, candidate)));
|
|
1961
|
+
const reportDir = join(reportsDir, dirName);
|
|
1962
|
+
const { reportPath, manifestPath } = await writeFallbackReport(reportDir, name, draft, evidence, artifacts.gaps, generatedAt);
|
|
1963
|
+
value = {
|
|
1964
|
+
industry: name,
|
|
1965
|
+
engine: "builtin-fallback",
|
|
1966
|
+
reportDir,
|
|
1967
|
+
reportPath,
|
|
1968
|
+
manifestPath,
|
|
1969
|
+
sealHash: null,
|
|
1970
|
+
verdicts: null,
|
|
1971
|
+
claims: draft.claims.length,
|
|
1972
|
+
evidence: evidence.length,
|
|
1973
|
+
gaps: artifacts.gaps,
|
|
1974
|
+
generatedAt
|
|
1975
|
+
};
|
|
1976
|
+
}
|
|
1977
|
+
const payload = {
|
|
1978
|
+
industry: name,
|
|
1979
|
+
engine: value.engine,
|
|
1980
|
+
reportDir: value.reportDir,
|
|
1981
|
+
claims: value.claims,
|
|
1982
|
+
evidence: value.evidence
|
|
1983
|
+
};
|
|
1984
|
+
ctx.emit("industry-research/report", payload);
|
|
1985
|
+
return value;
|
|
1986
|
+
}
|
|
1987
|
+
});
|
|
1988
|
+
}
|
|
1989
|
+
//#endregion
|
|
1990
|
+
//#region src/version.ts
|
|
1991
|
+
/**
|
|
1992
|
+
* Single source of truth for the package version (stamped by the release
|
|
1993
|
+
* script together with package.json and CHANGELOG.md).
|
|
1994
|
+
* @module dsh-industry-research/version
|
|
1995
|
+
*/
|
|
1996
|
+
/** The package version. */
|
|
1997
|
+
const VERSION = "0.1.0";
|
|
1998
|
+
//#endregion
|
|
1999
|
+
//#region src/index.ts
|
|
2000
|
+
/**
|
|
2001
|
+
* `dsh-industry-research` — industry and company research domain pack for
|
|
2002
|
+
* DeepSeek Harness. Mounts four workspace-bound research tools
|
|
2003
|
+
* (`industry_map` / `industry_track` / `company_scan` / `industry_report`),
|
|
2004
|
+
* publishes two methodology skills (`industry-research-method`,
|
|
2005
|
+
* `company-research-method`) from the packaged `skills/` directory, and emits
|
|
2006
|
+
* typed Cordis events after each committed artifact. The web capability
|
|
2007
|
+
* (`ctx.web`) and the report engine (`ctx.researchReport`) are optional and
|
|
2008
|
+
* looked up structurally at execution time — never injected. Research only;
|
|
2009
|
+
* not investment advice.
|
|
2010
|
+
*
|
|
2011
|
+
* Function plugin — no default export (the Loader unwraps
|
|
2012
|
+
* `exports.default ?? exports`, and a stray default would discard
|
|
2013
|
+
* `name`/`inject`/`Config`/`apply`).
|
|
2014
|
+
* @module dsh-industry-research
|
|
2015
|
+
*/
|
|
2016
|
+
const name = "industry-research";
|
|
2017
|
+
/** The four tools and the skill provider; web/engine stay optional lookups. */
|
|
2018
|
+
const inject = ["skills", "tools"];
|
|
2019
|
+
/** Directory of this module: `src/` under tsx/vitest or `lib/` when built. */
|
|
2020
|
+
const MODULE_DIR = dirname(fileURLToPath(import.meta.url));
|
|
2021
|
+
/** Whether a directory contains at least one `<skill>/SKILL.md` bundle. */
|
|
2022
|
+
function hasSkillBundles(dir) {
|
|
2023
|
+
if (!existsSync(dir)) return false;
|
|
2024
|
+
let entries;
|
|
2025
|
+
try {
|
|
2026
|
+
entries = readdirSync(dir, { withFileTypes: true });
|
|
2027
|
+
} catch {
|
|
2028
|
+
return false;
|
|
2029
|
+
}
|
|
2030
|
+
return entries.some((entry) => entry.isDirectory() && existsSync(join(dir, entry.name, "SKILL.md")));
|
|
2031
|
+
}
|
|
2032
|
+
/**
|
|
2033
|
+
* Resolve the skills root to publish and fail loud on misconfiguration: an
|
|
2034
|
+
* explicit `skillsDir` must exist and hold bundles, and the packaged default
|
|
2035
|
+
* (`skills/` beside `src/` or `lib/`) must exist — never mount silently with
|
|
2036
|
+
* zero skills.
|
|
2037
|
+
* @param skillsDir - explicit config override, if any.
|
|
2038
|
+
* @returns the validated skills root.
|
|
2039
|
+
*/
|
|
2040
|
+
function resolveSkillsRoot(skillsDir) {
|
|
2041
|
+
if (skillsDir !== void 0) {
|
|
2042
|
+
const root = resolve(skillsDir);
|
|
2043
|
+
if (!hasSkillBundles(root)) throw new Error(`industry-research: config.skillsDir "${skillsDir}" does not exist or contains no <skill>/SKILL.md bundles`);
|
|
2044
|
+
return root;
|
|
2045
|
+
}
|
|
2046
|
+
const root = join(MODULE_DIR, "..", "skills");
|
|
2047
|
+
if (!hasSkillBundles(root)) throw new Error(`industry-research: packaged skills root ${root} is missing; expected skills/<name>/SKILL.md bundles beside the built lib/. Set config.skillsDir to an explicit root.`);
|
|
2048
|
+
return root;
|
|
2049
|
+
}
|
|
2050
|
+
/**
|
|
2051
|
+
* Mount the research pack: validate config (fail loud), publish the packaged
|
|
2052
|
+
* skills, and register the four tools. With `enabled: false` nothing is
|
|
2053
|
+
* registered and the plugin stays inert.
|
|
2054
|
+
* @param ctx - the plugin context (host).
|
|
2055
|
+
* @param config - raw plugin config.
|
|
2056
|
+
*/
|
|
2057
|
+
function apply(ctx, config = {}) {
|
|
2058
|
+
const resolved = resolveConfig(config);
|
|
2059
|
+
const logger = ctx.logger("industry-research");
|
|
2060
|
+
if (!resolved.enabled) {
|
|
2061
|
+
logger.info("disabled: enabled is false — no research capabilities are mounted");
|
|
2062
|
+
return;
|
|
2063
|
+
}
|
|
2064
|
+
const skillsRoot = resolveSkillsRoot(resolved.skillsDir);
|
|
2065
|
+
ctx.effect(function* () {
|
|
2066
|
+
yield ctx.skills.registerProvider((control) => new FileSystemSkillProvider(ctx, control, {
|
|
2067
|
+
providerName: "industry-research",
|
|
2068
|
+
includeDefaultRoots: false,
|
|
2069
|
+
customSkillDirs: [skillsRoot],
|
|
2070
|
+
watch: false
|
|
2071
|
+
}));
|
|
2072
|
+
});
|
|
2073
|
+
ctx.tools.register(buildIndustryMapTool(ctx, resolved));
|
|
2074
|
+
ctx.tools.register(buildIndustryTrackTool(ctx, resolved));
|
|
2075
|
+
ctx.tools.register(buildCompanyScanTool(ctx, resolved));
|
|
2076
|
+
ctx.tools.register(buildIndustryReportTool(ctx, resolved));
|
|
2077
|
+
logger.info(`industry-research ${VERSION} mounted: 4 tools, skills from ${skillsRoot}`);
|
|
2078
|
+
}
|
|
2079
|
+
//#endregion
|
|
2080
|
+
export { AUTO_SECTIONS, CHAIN_TIERS, Config, DISCLAIMER, READABLE_EXTENSIONS, VERSION, apply, autoDraft, boundFigures, buildCompanyScanTool, buildEvidence, buildIndustryMapTool, buildIndustryReportTool, buildIndustryTrackTool, chainGaps, chainPathOf, companyDirOf, industryDirOf, inject, loadArtifacts, loadSources, lookupEngine, lookupWeb, mergeTimeline, name, normalizeUrl, notesDirOf, readCard, readTimeline, registerSource, renderCardMarkdown, renderFallbackMarkdown, reportDirName, reportsDirOf, requestSignal, requireWeb, resolveConfig, resolveContained, resolveIndustryRoot, resolveWorkspaceFile, safeSegment, saveSources, scanFile, sha256Of, sourceAllowed, sourcesPathOf, timelinePathOf, validateChainMap, validateDraft, webErrorMessage, workspaceOf, writeCard, writeFallbackReport };
|