futura-scion 0.2.3 → 0.2.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/bin/scion.js CHANGED
@@ -791,6 +791,98 @@ switch (cmd || '') {
791
791
  console.log(JSON.stringify(out, null, 2));
792
792
  break;
793
793
  }
794
+ case 'scaffold': {
795
+ // CONSTRUCTION: compose a template into a project skeleton, gated by
796
+ // the stack's real verifier recipe before success is reported.
797
+ const { scaffold, listTemplates, loadTemplate } = await import('../src/mind/scaffold.js');
798
+ if (!arg || arg === 'list' || arg === '--list') {
799
+ const tpls = listTemplates();
800
+ if (tpls.length === 0) { console.log('no templates in templates/ — add YAML manifests'); break; }
801
+ for (const t of tpls) console.log(`${t.id.padEnd(16)} ${(t.stack ?? '-').toString().padEnd(8)} ${t.description}`);
802
+ break;
803
+ }
804
+ const flags = restArgs.filter((a) => a.startsWith('--'));
805
+ const positional = restArgs.filter((a) => !a.startsWith('--'));
806
+ const target = positional[0]; // output directory
807
+ if (!target) {
808
+ console.error('usage: scion scaffold <template|list> <target-dir> [--name x] [--title y] [--port N] [--param k=v] [--no-verify]');
809
+ process.exitCode = 2;
810
+ break;
811
+ }
812
+ const tpl = loadTemplate(arg);
813
+ const params = {};
814
+ for (const f of flags) {
815
+ const m = f.match(/^--([A-Za-z_][A-Za-z0-9_]*)=(.*)$/);
816
+ if (m) params[m[1]] = m[2];
817
+ }
818
+ // Default missing params to sensible template-agnostic values instead
819
+ // of a raw throw — the CLI user should see WHICH param is missing in a
820
+ // friendly line, not a stack trace.
821
+ for (const req of ['name', 'title', 'port', 'description']) {
822
+ if (!(req in params)) {
823
+ if (req === 'port') params.port = '3000';
824
+ else if (req === 'title') params.title = params.name ?? arg;
825
+ else if (req === 'name') params.name = params.name ?? 'app';
826
+ else params.description ??= `built with FS scaffold (${arg})`;
827
+ }
828
+ }
829
+ let r;
830
+ try {
831
+ r = await scaffold(arg, { params, cwd: resolve(target), verify: !flags.includes('--no-verify') });
832
+ } catch (err) {
833
+ console.error(`scaffold failed: ${err.message}`);
834
+ process.exitCode = 1;
835
+ break;
836
+ }
837
+ if (r.ok === false && r.error) {
838
+ console.error(`scaffold failed: ${r.error}`);
839
+ process.exitCode = 1;
840
+ break;
841
+ }
842
+ console.log(`scaffold ${r.template} → ${target}`);
843
+ for (const f of r.files) console.log(` + ${f}`);
844
+ if (r.gate) {
845
+ console.log(`gate: ${r.gate.verified ? 'VERIFIED' : 'FAILED'} — [${r.gate.evidence.exit_codes.join(',')}]`);
846
+ if (!r.gate.verified) process.exitCode = 1;
847
+ } else {
848
+ console.log('gate: skipped (no stack recipe)');
849
+ }
850
+ break;
851
+ }
852
+ case 'research': {
853
+ // The research rung, driven from the CLI (webresearcher lineage):
854
+ // deterministic-first web research with a completion gate.
855
+ const q = [arg, ...restArgs.filter((a) => !a.startsWith('--'))].filter(Boolean).join(' ').trim();
856
+ const flags = restArgs.filter((a) => a.startsWith('--')) || [];
857
+ const { runResearch } = await import('../src/mind/research/index.js');
858
+ if (!q) {
859
+ console.error('usage: scion research "<question>" [--no-persist] [--pages N] [--deep] [--learn]');
860
+ process.exitCode = 2;
861
+ break;
862
+ }
863
+ const recipe = flags.includes('--deep')
864
+ ? { id: 'deep-dive', parameters: { max_pages: 16 }, verify: { min_pages: 2, min_items: 5, min_corroborated: 1 } }
865
+ : undefined;
866
+ const report = await runResearch(q, {
867
+ recipe,
868
+ persist: !flags.includes('--no-persist'),
869
+ max_pages: Number((flags.find((f) => f.startsWith('--pages')) ?? '').split('=')[1]) || (flags.includes('--pages') ? Number(restArgs[restArgs.indexOf('--pages') + 1]) : undefined) || undefined,
870
+ });
871
+ console.log(`research: ${report.pages} pages → ${report.items.length} items ` +
872
+ `(${report.census.SUPPORTED} corroborated, ${report.census.CONFLICTING} conflicting)`);
873
+ console.log(`gate: ${report.gate.verdict} — ${report.gate.checks.map((c) => `${c.check}:${c.pass ? 'ok' : 'fail'}(${c.actual})`).join(' ')}`);
874
+ for (const f of report.summary.KEY_FACT.slice(0, 5)) console.log(` ✓ ${f.slice(0, 140)}`);
875
+ for (const c of report.summary.CONTRADICTION.slice(0, 3)) console.log(` ⚡ ${c.slice(0, 140)}`);
876
+ for (const e of report.summary.EVIDENCE.slice(0, 3)) console.log(` · ${e.slice(0, 140)}`);
877
+ if (report.fallback) console.log(` [llm-fallback] ${report.fallback.synthesis.slice(0, 200)}`);
878
+ if (flags.includes('--learn')) {
879
+ const { learnResearchLoops } = await import('../src/mind/research/loop-learner.js');
880
+ const learned = learnResearchLoops();
881
+ console.log(`loop-learner: ${learned.length} lesson(s) persisted`);
882
+ }
883
+ process.exitCode = report.gate.verdict === 'PASS' ? 0 : 1;
884
+ break;
885
+ }
794
886
  case 'conventions': {
795
887
  // Measure + persist project conventions (cortex convention-detector
796
888
  // lineage, as EVIDENCE): discovered, never enforced.
@@ -884,6 +976,10 @@ switch (cmd || '') {
884
976
  leaseMs, pollMs,
885
977
  primaryUrl: followerMode || primaryUrl ? (primaryUrl || null) : null,
886
978
  });
979
+ // Lever 3: the capability tools (web-search, scholar-search, scaffold-*)
980
+ // register at boot so serve/UI/MCP all share the widened hands.
981
+ const { registerBuiltinCapabilities } = await import('../src/mind/builtins-register.js');
982
+ registerBuiltinCapabilities();
887
983
  const api = await startApi({ port: Number(arg) || undefined, leader });
888
984
  leader.start();
889
985
  // FS Desktop support: provider-backed generator (rung G) from config,
@@ -11,12 +11,8 @@ ladder:
11
11
  min_insight_confidence: 0.55 # reasoner rung admission floor
12
12
 
13
13
  llm:
14
- daily_tokens: 500
14
+ daily_tokens: 0 # 0 = LLM rung off (zero-LLM doctrine)
15
15
 
16
- apiKey: null
17
- model: "qwen2.5-coder:7b"
18
- baseUrl: "http://localhost:11434/v1"
19
- provider: "ollama"
20
16
  bounds:
21
17
  max_turns: 25
22
18
  timeout_ms: 300000
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "futura-scion",
3
- "version": "0.2.3",
3
+ "version": "0.2.4",
4
4
  "description": "The fused scion of cortex-os-agent + persona: one zero-LLM-dependent agent stack — Mind proposes, Muscle executes, Gate disposes.",
5
5
  "type": "module",
6
6
  "main": "src/index.js",
@@ -13,6 +13,7 @@
13
13
  "bin/",
14
14
  "src/",
15
15
  "recipes/",
16
+ "templates/",
16
17
  "config/",
17
18
  "knowledge/",
18
19
  "README.md",
@@ -368,6 +368,30 @@ export function list(opts = {}) {
368
368
  return rows.map(r => ({ ...r, tags: JSON.parse(r.tags || '[]') }));
369
369
  }
370
370
 
371
+ /**
372
+ * reinforce — DEDUP-REINFORCE (webresearcher cortex_bridge lineage): when
373
+ * new evidence matches an existing memory, STRENGTHEN it instead of
374
+ * duplicating. Confidence rises by delta (capped at 100), importance to at
375
+ * least the floor, and the access counter ticks. Repeated corroboration is
376
+ * a trust signal — the memory-graph weights read it on every recall.
377
+ * @param {number} id — existing memory id
378
+ * @param {object} [opts] — { confidence_delta=5, importance_floor=null }
379
+ * @returns {{ id, confidence, importance, reinforced: true }}
380
+ */
381
+ export function reinforce(id, opts = {}) {
382
+ const db = getDb();
383
+ const row = db.prepare('SELECT confidence, importance FROM memories WHERE id = ?').get(id);
384
+ if (!row) throw new Error(`memory.reinforce: no memory ${id}`);
385
+ const delta = opts.confidence_delta ?? 5;
386
+ const confidence = Math.min(100, row.confidence + delta);
387
+ const importance = opts.importance_floor != null
388
+ ? Math.max(row.importance, opts.importance_floor)
389
+ : Math.min(10, row.importance + (delta >= 5 ? 1 : 0));
390
+ db.prepare('UPDATE memories SET confidence = ?, importance = ?, access_count = access_count + 1 WHERE id = ?')
391
+ .run(confidence, importance, id);
392
+ return { id, confidence, importance, reinforced: true };
393
+ }
394
+
371
395
  export function stats() {
372
396
  const db = getDb();
373
397
  const row = db.prepare(
package/src/config.js CHANGED
@@ -59,6 +59,22 @@ function parseScalar(raw) {
59
59
  return v.replace(/^['"]|['"]$/g, '');
60
60
  }
61
61
 
62
+ /**
63
+ * Strip a block scalar's BASE indentation while preserving relative
64
+ * indentation (Python code in templates lives or dies by its indents).
65
+ * The base is the indentation of the FIRST content line; subsequent lines
66
+ * are dedented by the same amount. `|` literal blocks are exact; `>` folded
67
+ * blocks joined with spaces downstream, so per-line dedent is harmless.
68
+ */
69
+ function dedentBlockLine(line, blockScalar) {
70
+ if (blockScalar.folded) return line.trim();
71
+ if (blockScalar.baseIndent === undefined) {
72
+ blockScalar.baseIndent = line.match(/^\s*/)[0].length;
73
+ }
74
+ const cut = Math.min(blockScalar.baseIndent, line.match(/^\s*/)[0].length);
75
+ return line.slice(cut);
76
+ }
77
+
62
78
  /**
63
79
  * Minimal YAML-subset parser: nested maps by indentation, scalars, inline
64
80
  * arrays, `#` comments. Deliberately NOT full YAML — unsupported syntax
@@ -144,11 +160,23 @@ export function parseYaml(text, source = CONFIG_PATH) {
144
160
  if (!match) {
145
161
  // Continuation line of a block scalar ("key: >" folded text)?
146
162
  if (blockScalar !== null && /^\s+\S/.test(line)) {
147
- blockScalar.text.push(line.trim());
163
+ blockScalar.text.push(dedentBlockLine(line, blockScalar));
148
164
  continue;
149
165
  }
150
166
  throw new Error(`config: ${source}:${lineno} unsupported syntax — ${JSON.stringify(line.trim())}`);
151
167
  }
168
+ // A block scalar is OPEN and this line is deeper-indented than its key:
169
+ // it is SCALAR TEXT, never a new key — even when it looks like
170
+ // "word: value" (template manifests embed whole files in block scalars;
171
+ // their `port: 3000`-shaped lines are content, not structure). Leading
172
+ // indentation is PRESERVED relative to the block's first content line —
173
+ // Python files in templates live or die by their indentation.
174
+ if (blockScalar !== null
175
+ && match[1].length > (blockScalar.indent ?? 0)
176
+ && !/^\s*-\s/.test(line)) {
177
+ blockScalar.text.push(dedentBlockLine(line, blockScalar));
178
+ continue;
179
+ }
152
180
  const [, ws, key, rest] = match;
153
181
  const indent = ws.length;
154
182
 
@@ -0,0 +1,88 @@
1
+ /**
2
+ * mind/builtins-register.js — LEVER 3 (tool-calling breadth): registers the
3
+ * network + construction capabilities as FIRST-CLASS registry tools so the
4
+ * research rung (`research()`) and the swarm can invoke them by name like
5
+ * any shell tool. Import once at boot/serve; registration is idempotent.
6
+ *
7
+ * These are ASYNC capability tools (functions, not argv templates) — the
8
+ * registry stores their runner and runTool dispatches to it. Same rails
9
+ * apply: journaled, redacted, bounded.
10
+ *
11
+ * @module mind/builtins-register
12
+ */
13
+
14
+ 'use strict';
15
+
16
+ import { registerTool } from './tools.js';
17
+ import { searchWeb } from './research/search.js';
18
+ import { listTemplates, loadTemplate } from './scaffold.js';
19
+
20
+ let registered = false;
21
+
22
+ /** Idempotently register the capability tools. */
23
+ export function registerBuiltinCapabilities() {
24
+ if (registered) return listTemplates();
25
+ registered = true;
26
+
27
+ registerTool({
28
+ name: 'web-search',
29
+ description: 'search the web (DuckDuckGo keyless / Tavily keyed); {topic} is the query',
30
+ tags: ['web', 'research', 'network'],
31
+ capability: async (opts = {}) => {
32
+ const r = await searchWeb(opts.topic ?? '', { max_results: opts.max_results ?? 5 });
33
+ return {
34
+ ok: r.results.length > 0,
35
+ stdout: r.results.map((x) => `- ${x.title}\n ${x.url}\n ${String(x.snippet ?? '').slice(0, 160)}`).join('\n').slice(0, 8000),
36
+ meta: { provider: r.provider, count: r.results.length, attempts: r.attempts.length },
37
+ };
38
+ },
39
+ });
40
+
41
+ registerTool({
42
+ name: 'scholar-search',
43
+ description: 'search scholarship (arXiv/Semantic Scholar/CrossRef); {topic} is the query',
44
+ tags: ['academic', 'research', 'network'],
45
+ capability: async (opts = {}) => {
46
+ const { searchScholar } = await import('./research/academic.js');
47
+ const papers = await searchScholar(opts.topic ?? '', { max_results: 5 });
48
+ return {
49
+ ok: papers.length > 0,
50
+ stdout: papers.map((p) => `- ${p.title} (${p.year ?? 'n.d.'}, cites: ${p.citation_count ?? '?'})\n ${p.url}`).join('\n').slice(0, 8000),
51
+ meta: { count: papers.length },
52
+ };
53
+ },
54
+ });
55
+
56
+ registerTool({
57
+ name: 'scaffold-list',
58
+ description: 'list constructable project templates (what FS can BUILD)',
59
+ tags: ['construction'],
60
+ capability: async () => {
61
+ const tpls = listTemplates();
62
+ return {
63
+ ok: tpls.length > 0,
64
+ stdout: tpls.map((t) => `${t.id.padEnd(16)} ${(t.stack ?? '-').padEnd(8)} ${t.description}`).join('\n'),
65
+ meta: { count: tpls.length },
66
+ };
67
+ },
68
+ });
69
+
70
+ registerTool({
71
+ name: 'scaffold-inspect',
72
+ description: 'show a template manifest (files, post commands, stack) for {topic}=template id',
73
+ tags: ['construction'],
74
+ capability: async (opts = {}) => {
75
+ try {
76
+ const t = loadTemplate(opts.topic ?? '');
77
+ return {
78
+ ok: true,
79
+ stdout: `${t.name} (stack: ${t.stack ?? 'none'})\n${t.description}\nfiles:\n${t.files.map((f) => ` + ${f.path}`).join('\n')}\npost:\n${(t.post ?? []).map((c) => ` $ ${c}`).join('\n') || ' (none)'}`,
80
+ };
81
+ } catch (err) {
82
+ return { ok: false, error: String(err?.message || err).slice(0, 200) };
83
+ }
84
+ },
85
+ });
86
+
87
+ return listTemplates();
88
+ }
@@ -0,0 +1,183 @@
1
+ /**
2
+ * mind/research/academic.js — ACADEMIC SEARCH (webresearcher
3
+ * `academic_search.py` ARCANE lineage): unified access to arXiv, Semantic
4
+ * Scholar, and CrossRef with smart ID resolution (DOI <-> arXivID <-> S2 ID).
5
+ *
6
+ * All APIs are FREE and require NO authentication. Deterministic parsing
7
+ * (regex over XML/JSON), bounded fetches, graceful degradation — a provider
8
+ * that fails returns fewer results, never a throw.
9
+ *
10
+ * @module mind/research/academic
11
+ */
12
+
13
+ 'use strict';
14
+
15
+ import { fetchBounded } from './search.js';
16
+ import * as trail from '../../kernel/trail.js';
17
+
18
+ const S2_API = 'https://api.semanticscholar.org/graph/v1';
19
+ const CROSSREF_API = 'https://api.crossref.org/works';
20
+
21
+ function stripXml(s) {
22
+ return String(s ?? '').replace(/<[^>]+>/g, '').trim();
23
+ }
24
+
25
+ /** One paper record: { id, doi?, arxiv_id?, title, abstract, year, venue, authors, url, source } */
26
+ function paper(p) {
27
+ return { ...p, authors: p.authors ?? [], citation_count: p.citation_count ?? null };
28
+ }
29
+
30
+ /* ------------------------------- arXiv ------------------------------- */
31
+
32
+ function parseArxivXml(xml, max) {
33
+ const papers = [];
34
+ const entries = xml.split(/<entry>/).slice(1);
35
+ for (const e of entries) {
36
+ if (papers.length >= max) break;
37
+ const id = stripXml(e.match(/<id>([\s\S]*?)<\/id>/)?.[1]);
38
+ const title = stripXml(e.match(/<title>([\s\S]*?)<\/title>/)?.[1]).replace(/\s+/g, ' ');
39
+ const abstract = stripXml(e.match(/<summary>([\s\S]*?)<\/summary>/)?.[1]).replace(/\s+/g, ' ');
40
+ const published = stripXml(e.match(/<published>([\s\S]*?)<\/published>/)?.[1]);
41
+ const authors = [...e.matchAll(/<name>([\s\S]*?)<\/name>/g)].map((m) => stripXml(m[1]));
42
+ if (!id || !title) continue;
43
+ papers.push(paper({
44
+ id, arxiv_id: id.replace(/^https?:\/\/arxiv\.org\/abs\//, ''),
45
+ title, abstract, year: published ? Number(published.slice(0, 4)) : null,
46
+ venue: 'arXiv', authors, url: id, source: 'arxiv',
47
+ }));
48
+ }
49
+ return papers;
50
+ }
51
+
52
+ async function arxivSearch(query, opts = {}) {
53
+ const max = opts.max_results ?? 5;
54
+ const url = `https://export.arxiv.org/api/query?search_query=all:${encodeURIComponent(query)}&max_results=${max}&sortBy=relevance`;
55
+ const r = await fetchBounded(url, { timeout_ms: opts.timeout_ms });
56
+ if (!r.ok) return [];
57
+ return parseArxivXml(r.text, max);
58
+ }
59
+
60
+ /* ------------------------- Semantic Scholar -------------------------- */
61
+
62
+ async function s2Search(query, opts = {}) {
63
+ const max = opts.max_results ?? 5;
64
+ const fields = 'title,abstract,year,venue,authors,externalIds,citationCount,url';
65
+ const url = `${S2_API}/paper/search?query=${encodeURIComponent(query)}&limit=${max}&fields=${encodeURIComponent(fields)}`;
66
+ const r = await fetchBounded(url, { timeout_ms: opts.timeout_ms });
67
+ if (!r.ok) return [];
68
+ let body;
69
+ try { body = JSON.parse(r.text); } catch { return []; }
70
+ return (body.data ?? []).map((p) => paper({
71
+ id: p.paperId, doi: p.externalIds?.DOI ?? null, arxiv_id: p.externalIds?.ArXiv ?? null,
72
+ title: p.title, abstract: p.abstract, year: p.year, venue: p.venue || null,
73
+ authors: (p.authors ?? []).map((a) => a.name), citation_count: p.citation_count,
74
+ url: p.url ?? `https://www.semanticscholar.org/paper/${p.paperId}`, source: 's2',
75
+ }));
76
+ }
77
+
78
+ /* ------------------------------ CrossRef ----------------------------- */
79
+
80
+ async function crossrefSearch(query, opts = {}) {
81
+ const max = opts.max_results ?? 5;
82
+ const url = `${CROSSREF_API}?query=${encodeURIComponent(query)}&rows=${max}&select=DOI,title,author,published,container-title,URL,is-referenced-by-count,abstract`;
83
+ const r = await fetchBounded(url, { timeout_ms: opts.timeout_ms });
84
+ if (!r.ok) return [];
85
+ let body;
86
+ try { body = JSON.parse(r.text); } catch { return []; }
87
+ return ((body.message ?? {}).items ?? []).map((p) => paper({
88
+ id: p.DOI, doi: p.DOI,
89
+ title: Array.isArray(p.title) ? p.title[0] : p.title,
90
+ abstract: typeof p.abstract === 'string' ? stripXml(p.abstract).replace(/\s+/g, ' ').slice(0, 2000) : null,
91
+ year: p.published ? Number(String(p.published['date-parts']?.[0]?.[0] ?? '').slice(0, 4)) || null : null,
92
+ venue: Array.isArray(p['container-title']) ? p['container-title'][0] : null,
93
+ authors: (p.author ?? []).map((a) => [a.given, a.family].filter(Boolean).join(' ')),
94
+ citation_count: p['is-referenced-by-count'] ?? null,
95
+ url: p.URL ?? `https://doi.org/${p.DOI}`, source: 'crossref',
96
+ })).filter((p) => p.title);
97
+ }
98
+
99
+ /* ---------------------------- ID resolution --------------------------- */
100
+
101
+ const DOI_RE = /^10\.\d{4,}\//;
102
+ const ARXIV_RE = /^\d{4}\.\d{4,5}(v\d+)?$/;
103
+
104
+ /**
105
+ * Resolve any identifier to a canonical paper record. Accepts a DOI
106
+ * (10.xxxx/...), an arXiv ID (2401.12345), or an S2 paper ID.
107
+ */
108
+ export async function resolveId(id, opts = {}) {
109
+ const raw = String(id).trim();
110
+ if (DOI_RE.test(raw)) {
111
+ const r = await fetchBounded(`${CROSSREF_API}/${encodeURIComponent(raw)}`, { timeout_ms: opts.timeout_ms });
112
+ if (r.ok) {
113
+ try {
114
+ const m = JSON.parse(r.text).message;
115
+ return paper({
116
+ id: m.DOI, doi: m.DOI, title: Array.isArray(m.title) ? m.title[0] : m.title,
117
+ year: m.published ? Number(String(m.published['date-parts']?.[0]?.[0] ?? '').slice(0, 4)) || null : null,
118
+ venue: Array.isArray(m['container-title']) ? m['container-title'][0] : null,
119
+ authors: (m.author ?? []).map((a) => [a.given, a.family].filter(Boolean).join(' ')),
120
+ citation_count: m['is-referenced-by-count'] ?? null,
121
+ url: m.URL ?? `https://doi.org/${m.DOI}`, source: 'crossref',
122
+ });
123
+ } catch { /* fall through */ }
124
+ }
125
+ }
126
+ if (ARXIV_RE.test(raw)) {
127
+ const r = await fetchBounded(`https://export.arxiv.org/api/query?id_list=${encodeURIComponent(raw)}`, { timeout_ms: opts.timeout_ms });
128
+ if (r.ok) {
129
+ const papers = parseArxivXml(r.text, 1);
130
+ if (papers.length) return papers[0];
131
+ }
132
+ }
133
+ // S2 fallback (handles S2 IDs and hashes DOIs/arXiv too).
134
+ const s2 = await fetchBounded(
135
+ `${S2_API}/paper/${encodeURIComponent(raw)}?fields=title,abstract,year,venue,authors,externalIds,citationCount,url`,
136
+ { timeout_ms: opts.timeout_ms }
137
+ );
138
+ if (s2.ok) {
139
+ try {
140
+ const p = JSON.parse(s2.text);
141
+ return paper({
142
+ id: p.paperId, doi: p.externalIds?.DOI ?? null, arxiv_id: p.externalIds?.ArXiv ?? null,
143
+ title: p.title, abstract: p.abstract, year: p.year, venue: p.venue || null,
144
+ authors: (p.authors ?? []).map((a) => a.name), citation_count: p.citation_count,
145
+ url: p.url ?? `https://www.semanticscholar.org/paper/${p.paperId}`, source: 's2',
146
+ });
147
+ } catch { /* fall through */ }
148
+ }
149
+ return null;
150
+ }
151
+
152
+ /* ------------------------------- search ------------------------------- */
153
+
154
+ /**
155
+ * Search scholarship across providers. Order: arXiv (preprints) → S2 →
156
+ * CrossRef; merges and dedups by DOI/arXiv ID/title, then by citation count
157
+ * (descending) — deterministic.
158
+ * @param {string} query
159
+ * @param {object} [opts] — { max_results?, timeout_ms? }
160
+ */
161
+ export async function searchScholar(query, opts = {}) {
162
+ const max = opts.max_results ?? 5;
163
+ const started = Date.now();
164
+ const settled = await Promise.allSettled([
165
+ arxivSearch(query, opts), s2Search(query, opts), crossrefSearch(query, opts),
166
+ ]);
167
+ const byKey = new Map();
168
+ for (const s of settled) {
169
+ if (s.status !== 'fulfilled') continue;
170
+ for (const p of s.value) {
171
+ const key = p.doi ?? (p.arxiv_id ? `arxiv:${p.arxiv_id}` : `t:${String(p.title).toLowerCase().slice(0, 90)}`);
172
+ const prev = byKey.get(key);
173
+ if (!prev || (p.citation_count ?? 0) > (prev.citation_count ?? 0)) byKey.set(key, p);
174
+ }
175
+ }
176
+ const papers = [...byKey.values()]
177
+ .sort((a, b) => (b.citation_count ?? 0) - (a.citation_count ?? 0))
178
+ .slice(0, max);
179
+ trail.journal('research.scholar', {
180
+ query: String(query).slice(0, 120), papers: papers.length, duration_ms: Date.now() - started,
181
+ });
182
+ return papers;
183
+ }
@@ -0,0 +1,110 @@
1
+ /**
2
+ * mind/research/corroborate.js — CLAIM CORROBORATION WITH EPISTEMIC HONESTY
3
+ * (webresearcher `claim_verifier.py` / `validation_loop.py` lineage).
4
+ *
5
+ * Most validators just say yes/no. This says "we don't have enough evidence"
6
+ * and "sources disagree" as FIRST-CLASS outcomes. A claim is corroborated
7
+ * when INDEPENDENT sources (different registrable domains) agree on it;
8
+ * conflicting numbers or explicit negation produce CONFLICTING; a single
9
+ * source is INSUFFICIENT_EVIDENCE — never silently treated as truth.
10
+ *
11
+ * Agreement is deterministic: Jaccard token overlap over the claim's
12
+ * normalized content words, plus numeric equality for statistics.
13
+ *
14
+ * @module mind/research/corroborate
15
+ */
16
+
17
+ 'use strict';
18
+
19
+ const STOP = new Set(['the', 'a', 'an', 'is', 'are', 'was', 'were', 'of', 'to', 'in', 'on', 'for', 'and', 'or', 'it', 'its', 'this', 'that', 'with', 'as', 'by', 'at', 'be', 'can', 'has', 'have', 'had']);
20
+
21
+ function words(s) {
22
+ return String(s).toLowerCase().normalize('NFKD').replace(/[^a-z0-9\s]/g, ' ')
23
+ .split(/\s+/).filter((w) => w.length > 2 && !STOP.has(w));
24
+ }
25
+
26
+ function jaccard(a, b) {
27
+ const A = new Set(a);
28
+ const B = new Set(b);
29
+ if (A.size === 0 || B.size === 0) return 0;
30
+ let inter = 0;
31
+ for (const w of A) if (B.has(w)) inter += 1;
32
+ return inter / (A.size + B.size - inter);
33
+ }
34
+
35
+ function numbers(s) {
36
+ return [...String(s).matchAll(/\d+(?:\.\d+)?/g)].map((m) => Number(m[0]));
37
+ }
38
+
39
+ /** Registrable domain (best-effort, no PSL): last two labels. */
40
+ export function domainOf(url) {
41
+ try {
42
+ const host = new URL(url).hostname.replace(/^www\./, '');
43
+ const parts = host.split('.');
44
+ return parts.length >= 2 ? parts.slice(-2).join('.') : host;
45
+ } catch { return String(url); }
46
+ }
47
+
48
+ /**
49
+ * Corroborate one extracted item against the others from INDEPENDENT sources.
50
+ * @param {object} item — an extract.js item ({ kind, text, source_url })
51
+ * @param {Array} all — every extracted item
52
+ * @param {object} [opts] — { threshold=0.45 }
53
+ * @returns {{ verdict, agreement_count, domains, numeric_match, reason }}
54
+ */
55
+ export function corroborateItem(item, all, opts = {}) {
56
+ const threshold = opts.threshold ?? 0.45;
57
+ const w = words(item.text);
58
+ const nums = numbers(item.text);
59
+ const ownDomain = item.source_url ? domainOf(item.source_url) : null;
60
+ const agreeing = new Map(); // domain → item
61
+ const disagreeing = [];
62
+
63
+ for (const other of all) {
64
+ if (other === item) continue;
65
+ if (other.kind !== item.kind) continue;
66
+ const od = other.source_url ? domainOf(other.source_url) : `__${words(other.text).join('+')}`;
67
+ if (ownDomain && od === ownDomain) continue; // same domain ≠ independence
68
+ const sim = jaccard(w, words(other.text));
69
+ if (sim < threshold) continue;
70
+ const on = numbers(other.text);
71
+ if (nums.length > 0 && on.length > 0) {
72
+ const match = nums.every((n, i) => on[i] === n);
73
+ if (!match) { disagreeing.push({ url: other.source_url, sim, number: on[0] }); continue; }
74
+ }
75
+ if (!agreeing.has(od)) agreeing.set(od, other);
76
+ }
77
+
78
+ const domains = [...agreeing.keys()];
79
+ if (disagreeing.length > 0 && agreeing.size > 0) {
80
+ return { verdict: 'CONFLICTING', agreement_count: agreeing.size, domains, numeric_match: false, reason: `${agreeing.size} source(s) agree but ${disagreeing.length} report different values` };
81
+ }
82
+ if (disagreeing.length > 0) {
83
+ return { verdict: 'CONFLICTING', agreement_count: 0, domains, numeric_match: false, reason: `${disagreeing.length} independent source(s) report different values` };
84
+ }
85
+ if (agreeing.size >= 1) {
86
+ return { verdict: 'SUPPORTED', agreement_count: agreeing.size + 1, domains, numeric_match: true, reason: `${agreeing.size + 1} independent source(s) agree` };
87
+ }
88
+ return { verdict: 'INSUFFICIENT_EVIDENCE', agreement_count: 1, domains: ownDomain ? [ownDomain] : [], numeric_match: null, reason: 'single source only' };
89
+ }
90
+
91
+ /**
92
+ * Corroborate all items; annotate each with its verdict.
93
+ * @param {Array} items — extract.js output
94
+ * @returns {Array} items, each with `.corroboration`
95
+ */
96
+ export function corroborateAll(items, opts = {}) {
97
+ for (const it of items ?? []) {
98
+ it.corroboration = corroborateItem(it, items, opts);
99
+ }
100
+ return items ?? [];
101
+ }
102
+
103
+ /** Verdict census for a run report. */
104
+ export function verdictCensus(items) {
105
+ const census = { SUPPORTED: 0, CONFLICTING: 0, INSUFFICIENT_EVIDENCE: 0 };
106
+ for (const it of items ?? []) {
107
+ if (census[it.corroboration?.verdict] !== undefined) census[it.corroboration.verdict] += 1;
108
+ }
109
+ return census;
110
+ }