@maci0/dsh-feynman 0.0.0-stage → 0.21.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/NOTICE +7 -0
- package/README.md +125 -2
- package/cordis.patch.yml +7 -0
- package/icon.svg +6 -0
- package/index.js +494 -0
- package/lib/client.js +350 -0
- package/locale/en.json +6 -0
- package/locale/zh.json +6 -0
- package/package.json +69 -4
- package/prompts.js +333 -0
package/prompts.js
ADDED
|
@@ -0,0 +1,333 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Pure Feynman workflow catalog: slash-command metadata and the agent prompts
|
|
3
|
+
* each command steers with. No DeepSeek Harness imports here, so this module
|
|
4
|
+
* is unit-testable without the harness.
|
|
5
|
+
*
|
|
6
|
+
* Source of truth for behavior: https://www.feynman.is/docs (workflows, agents,
|
|
7
|
+
* tools, slash-commands, cli-commands references).
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
/** Slugify a topic for `outputs/` artifact names. */
|
|
11
|
+
export function slugify(text) {
|
|
12
|
+
const slug = text.toLowerCase().replace(/[^a-z0-9]+/g, '-').replace(/^-+|-+$/g, '').slice(0, 60)
|
|
13
|
+
return slug || 'untitled'
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
/** Strip every `arxiv:` prefix (`arxiv:1 arxiv:2` lists included), so both spellings share slugs and API calls. */
|
|
17
|
+
export function stripArxivPrefix(text) {
|
|
18
|
+
return text.replace(/arxiv:\s*/gi, '').trim()
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
/** Review-loop rounds when the command names none. */
|
|
22
|
+
const DEFAULT_LOOP_ROUNDS = 3
|
|
23
|
+
/** Upper bound on review-loop rounds; a larger count is clamped to it. */
|
|
24
|
+
const MAX_LOOP_ROUNDS = 10
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* Split `/review-loop <target> [rounds]` trailing round count, clamped to 1..MAX_LOOP_ROUNDS.
|
|
28
|
+
* @param rawInput - text after `/feynman review-loop`.
|
|
29
|
+
*/
|
|
30
|
+
export function parseLoopArgs(rawInput) {
|
|
31
|
+
const match = /^(.*?)\s+(\d+)$/.exec(rawInput.trim())
|
|
32
|
+
if (match) return { target: match[1], rounds: Math.min(Math.max(Number(match[2]), 1), MAX_LOOP_ROUNDS) }
|
|
33
|
+
return { target: rawInput.trim(), rounds: DEFAULT_LOOP_ROUNDS }
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
const TOOL_PRELUDE = `You are a research agent. Your retrieval tools are web_search and web_fetch, plus workspace tools (read, grep, glob, bash) for local files, cloned repos, and code. Route each source to its sanctioned API: arXiv papers via the export.arxiv.org API and arxiv.org/abs pages; paper metadata, citations, and references via the OpenAlex API (api.openalex.org); biomedical papers via Europe PMC; datasets, models, and repo files via the Hugging Face Hub API (huggingface.co/api, read-only). Delegate by role when it helps: researcher (deepresearch, lit, review, audit, replicate, recipe, compare, draft) gathers, reviewer (review, audit, compare) runs the adversarial pass, writer (deepresearch, lit, draft, compare) produces the final document, verifier (deepresearch, audit, replicate, recipe) fact-checks. For broad multi-angle work, fan out with the subagent tool (one description + prompt per angle) and synthesize the returns; keep narrow explainers lead-owned to avoid needless orchestration.`
|
|
37
|
+
const KEY_PRELUDE = (hfTokenEnv, alphaxivTokenEnv) => `Credentials: the Hugging Face key lives in ${hfTokenEnv} (HUGGINGFACE_HUB_TOKEN accepted as fallback; send whichever is present) and the AlphaXiv key in ${alphaxivTokenEnv} when set (ask feynman keys, or the human exports them before launch). Spend the AlphaXiv key via web_fetch against the AlphaXiv API (alphaxiv.org Honk with Authorization bearer): paper search, paper content and section extraction (alpha_get_paper section/sections: abstract, introduction, methodology, experiments, results, discussion, limitations, conclusion), paper Q&A, linked-repo code inspection, annotations; without it fall back to arXiv + OpenAlex and mark citation-metadata/discussion-thread/source-text checks blocked. Treat an unset key as blocked for the calls that need it; gated Hugging Face datasets read as blocked. Never bypass paywalls. If a source is unreachable, mark that check blocked in the output; never invent or infer its content.`
|
|
38
|
+
|
|
39
|
+
const DEEPRESEARCH_PROMPT = (topic) => {
|
|
40
|
+
const slug = slugify(topic)
|
|
41
|
+
return `${TOOL_PRELUDE}
|
|
42
|
+
|
|
43
|
+
Workflow: deep research on "${topic}".
|
|
44
|
+
1. Write a plan to outputs/.plans/${slug}.md (key questions, source strategy, scale decision, task ledger, verification log), summarize it, and WAIT for the human to confirm or request changes before executing.
|
|
45
|
+
2. Gather: diversified queries across papers, web sources, docs, and code. Prefer metadata/abstracts/HTML/official docs over PDF extraction.
|
|
46
|
+
3. Extract claims, methods, results, limitations per source, tagged with source locations.
|
|
47
|
+
4. Synthesize into a research brief at outputs/${slug}-brief.md with inline citations: Summary, Background, Key Findings (by theme), Open Questions, References. Record source accounting, formula, and verification caveats in outputs/${slug}-brief.provenance.md.
|
|
48
|
+
5. Verify claims against cited sources; flag misattributions or unsupported assertions. Record verification caveats in the brief.`
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
const LIT_PROMPT = (topic) => {
|
|
52
|
+
const slug = slugify(topic)
|
|
53
|
+
return `${TOOL_PRELUDE}
|
|
54
|
+
|
|
55
|
+
Workflow: structured literature review on "${topic}". If the input names a lab, PI, author, or lab website, switch to publication-corpus mode (resolve identity, write a reachable-publication log first, map topic trajectories, rank 3-5 papers by contrastive originality, methodology strength, and relationship to prior art).
|
|
56
|
+
1. Search broadly (surveys, foundational work, recent frontier). Note search terms, time window, source types.
|
|
57
|
+
2. Extract claims, results, methodology per paper.
|
|
58
|
+
3. Write outputs/${slug}-lit-review.md: Scope and Methodology, Consensus (with citations), Disagreements, Open Questions, Timeline, References. For biomedical topics, frame the question as PICO/PICOS (population, intervention/exposure, comparator, outcomes, study design) or state the study type directly; group evidence by study design (guidelines, systematic reviews, RCTs, cohorts, case reports, preprints, mechanistic), report effect sizes only when source-backed; never ask for or paste protected health information; use de-identified or fictionalized questions; state that the output is research synthesis, not medical advice.`
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
const REVIEW_PROMPT = (rawArtifact) => {
|
|
62
|
+
const artifact = stripArxivPrefix(rawArtifact)
|
|
63
|
+
const slug = slugify(artifact)
|
|
64
|
+
return `${TOOL_PRELUDE}
|
|
65
|
+
|
|
66
|
+
Workflow: internal research review of "${artifact}" (arXiv ID, URL, or local file; fetch or read it first). This is a pre-trust critique, not a publication decision.
|
|
67
|
+
0. Write a plan to outputs/.plans/${slug}-review-plan.md, then continue immediately into evidence gathering and the final review without waiting for confirmation.
|
|
68
|
+
1. Record evidence notes in outputs/.drafts/${slug}-review-evidence.md as you go.
|
|
69
|
+
2. Evaluate: claims vs evidence, methodology soundness and confounds, experimental design (baselines, ablations), reproducibility, writing clarity, completeness (limitations, related work).
|
|
70
|
+
3. Write exactly one final review to outputs/${slug}-review.md with severity-graded findings (critical: undermines validity; major: should fix; minor: suggestion; nit: style), each with a confidence score: Summary Assessment (revision priority), Strengths, Critical Issues, Major Issues, Minor Issues, Inline Annotations tied to document sections. Flag unverifiable claims as needing evidence. If the artifact cannot be parsed, still write the review and mark affected checks blocked.`
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
const AUDIT_PROMPT = (rawItem) => {
|
|
74
|
+
const item = stripArxivPrefix(rawItem)
|
|
75
|
+
const slug = slugify(item)
|
|
76
|
+
return `${TOOL_PRELUDE}
|
|
77
|
+
|
|
78
|
+
Workflow: code audit of "${item}" (arXiv ID or repo URL plus --paper ID; when given only an arXiv ID, find the repo through paper links, Papers With Code, or GitHub search).
|
|
79
|
+
Pass 1 (researcher): extract concrete claims from the paper (hyperparameters, architecture, training procedure, dataset splits, metrics, reported results), each tagged with its paper location.
|
|
80
|
+
Pass 2 (verifier): find each claim's implementation (configs, training scripts, model definitions, eval code). Document mismatches with paper location plus exact file paths and line numbers; list claims with no corresponding code.
|
|
81
|
+
Also flag reproducibility risks: missing seeds, unpinned deps, hardcoded paths, missing environment specs.
|
|
82
|
+
Write outputs/${slug}-audit.md: Match Summary (% claims matched), Confirmed Claims, Mismatches, Missing Implementations, Reproducibility Risks.`
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
const REPLICATE_PROMPT = (rawTarget) => {
|
|
86
|
+
const target = stripArxivPrefix(rawTarget)
|
|
87
|
+
const slug = slugify(target)
|
|
88
|
+
return `${TOOL_PRELUDE}
|
|
89
|
+
|
|
90
|
+
Workflow: replication plan for "${target}" (paper or specific claim). Plan only: do NOT execute anything until the user chooses an environment (local, container, cloud, or plan-only).
|
|
91
|
+
1. Extract stated details: architecture, hyperparameters, schedule, data prep, eval protocol, hardware. Cross-reference linked/supplied code.
|
|
92
|
+
2. For ML-heavy targets add a recipe pass linking each claimed result to dataset, method, hyperparameters, compute, metric, and code path (verify Hugging Face dataset schema/splits via the huggingface.co/api dataset endpoints when relevant).
|
|
93
|
+
3. Write outputs/${slug}-replication-plan.md: Requirements (hardware/software/data/compute estimate), Recipe Extraction, Step-by-step Plan, Underspecified Details (gap + assumption + divergence risk each), Risk Assessment, Success Criteria (what counts as replicated). Label a result replicated only when the planned checks actually pass.`
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
const RECIPE_PROMPT = (task) => {
|
|
97
|
+
const slug = slugify(task)
|
|
98
|
+
return `${TOOL_PRELUDE}
|
|
99
|
+
|
|
100
|
+
Workflow: ML training recipe for "${task}".
|
|
101
|
+
0. Write a plan to outputs/.plans/${slug}-recipe.md, then continue automatically.
|
|
102
|
+
1. Gather candidates from papers, docs, repos, and Hugging Face Hub metadata (dataset features/splits, repo files, configs; read-only via the huggingface.co/api endpoints; HF_TOKEN may be present for gated resources).
|
|
103
|
+
2. Link each reported result to the recipe that produced it: dataset (+split/schema), method, hyperparameters, compute, benchmark, code path, verification status. A paper without usable data/code/config detail is a risk, not a runnable recipe. Label checks verified / unverified / blocked / inferred; never call a recipe state-of-the-art, replicated, or production-ready without supporting checks.
|
|
104
|
+
3. Write outputs/${slug}-recipe.md: Recommendation (one recipe first + why), Ranked Recipe Table, Dataset Notes, Implementation Plan (minimal steps), Known Gaps, Sources (every URL); plus outputs/${slug}-recipe.provenance.md with source accounting and verification caveats.`
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
const COMPARE_PROMPT = (rawInput) => {
|
|
108
|
+
const input = stripArxivPrefix(rawInput)
|
|
109
|
+
const slug = slugify(input)
|
|
110
|
+
return `${TOOL_PRELUDE}
|
|
111
|
+
|
|
112
|
+
Workflow: source comparison for "${input}" (a topic: find the most relevant contrasting sources; or explicit paper IDs/files: use them directly).
|
|
113
|
+
1. Analyze each source independently: claims, results, methodology, limitations.
|
|
114
|
+
2. Align claims across sources: agreement, genuine disagreement, non-overlapping scope. Note when apparent disagreement may come from different protocols rather than conflicting results.
|
|
115
|
+
3. Write outputs/${slug}-compare.md: Source Summaries (one paragraph each), Agreement Matrix, Disagreement Matrix (with divergence analysis), Methodology Differences, Synthesis (well-supported vs contested).`
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
const DRAFT_PROMPT = (rawInput) => {
|
|
119
|
+
const input = stripArxivPrefix(rawInput)
|
|
120
|
+
const slug = slugify(input)
|
|
121
|
+
return `${TOOL_PRELUDE}
|
|
122
|
+
|
|
123
|
+
Workflow: academic draft on "${input}". If the input is --from-session, skip research and write from this session's vetted findings; otherwise gather sources first.
|
|
124
|
+
Write outputs/${slug}-draft.md following academic structure: Abstract, Introduction (motivation, context, contributions), Body Sections, Discussion, Limitations (honest), References (only works cited, consistent format). Inline-cite factual claims; mark anything unsupported as an explicit TODO/gap; never invent results, figures, tables, or benchmark numbers.`
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
const AUTORESEARCH_PROMPT = (idea) => `${TOOL_PRELUDE}
|
|
128
|
+
|
|
129
|
+
Workflow: bounded autoresearch experiment loop for "${idea}".
|
|
130
|
+
1. Confirm with the user first if any of these are missing: benchmark, metric, environment, files in scope, iteration limit.
|
|
131
|
+
2. Loop: Hypothesis → Experiment → Analysis → Decision (keep / vary / pivot). Log every iteration (params, results) to autoresearch.md + autoresearch.jsonl in the workspace; record milestones in CHANGELOG.md; never repeat a failed approach.
|
|
132
|
+
3. Report: Experiment History, Best Configuration, Ablation Results, Recommendations. This is for hypothesis/benchmark-driven search (prompts, hyperparams, retrieval, architectures), not open-ended Q&A.`
|
|
133
|
+
|
|
134
|
+
const WATCH_PROMPT = (topic) => {
|
|
135
|
+
const slug = slugify(topic)
|
|
136
|
+
return `${TOOL_PRELUDE}
|
|
137
|
+
|
|
138
|
+
Workflow: research watch on "${topic}".
|
|
139
|
+
1. Write the watch plan (topic, monitored signals, meaningful-change criteria, check frequency) to outputs/.plans/${slug}-watch.md.
|
|
140
|
+
2. Run a baseline sweep (papers, articles, docs, releases, code) and save outputs/${slug}-baseline.md: New Papers, New Articles, Relevance Notes.
|
|
141
|
+
3. Schedule follow-ups ONLY with the schedule_create tool when it is visible in this session; otherwise mark scheduling blocked and include the exact refresh prompt to run later. Each follow-up check compares against the baseline so genuinely new material is separable from old findings.`
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
/**
|
|
145
|
+
* Split `/rank <topic> [options]` flags. All upstream PaperRank flags are parsed; unknown --flags are rejected.
|
|
146
|
+
* @param rawInput - text after `/feynman rank`.
|
|
147
|
+
*/
|
|
148
|
+
export function parseRankArgs(rawInput) {
|
|
149
|
+
const parts = rawInput.trim().split(/\s+/).filter(Boolean)
|
|
150
|
+
const out = {
|
|
151
|
+
topic: '', limit: 20, expandCitations: 0, fullTextTop: 0, critiqueTop: 0,
|
|
152
|
+
preferenceFile: null, reproductionNotes: null, synthesize: false, synthesisTop: 7,
|
|
153
|
+
synthesisModel: null, outputDir: 'outputs', json: false, unsupported: [],
|
|
154
|
+
}
|
|
155
|
+
const topicParts = []
|
|
156
|
+
for (let i = 0; i < parts.length; i += 1) {
|
|
157
|
+
const token = parts[i]
|
|
158
|
+
const next = parts[i + 1]
|
|
159
|
+
if (token === '--limit' && /^\d+$/.test(next ?? '')) { out.limit = Math.min(Math.max(Number(next), 1), 100); i += 1 }
|
|
160
|
+
else if (token === '--expand-citations' && /^\d+$/.test(next ?? '')) { out.expandCitations = Math.min(Math.max(Number(next), 0), 5); i += 1 }
|
|
161
|
+
else if (token === '--full-text-top' && /^\d+$/.test(next ?? '')) { out.fullTextTop = Math.max(Number(next), 0); i += 1 }
|
|
162
|
+
else if (token === '--critique-top' && /^\d+$/.test(next ?? '')) { out.critiqueTop = Math.max(Number(next), 0); i += 1 }
|
|
163
|
+
else if (token === '--preference-file' && next !== undefined && !next.startsWith('--')) { out.preferenceFile = next; i += 1 }
|
|
164
|
+
else if (token === '--reproduction-notes' && next !== undefined && !next.startsWith('--')) { out.reproductionNotes = next; i += 1 }
|
|
165
|
+
else if (token === '--synthesis-top' && /^\d+$/.test(next ?? '')) { out.synthesisTop = Math.max(Number(next), 1); i += 1 }
|
|
166
|
+
else if ((token === '--synthesis-model' || token === '--model') && next !== undefined && !next.startsWith('--')) { out.synthesisModel = next; i += 1 }
|
|
167
|
+
else if (token === '--output-dir' && next !== undefined && !next.startsWith('--')) { out.outputDir = next; i += 1 }
|
|
168
|
+
else if (token === '--synthesize') { out.synthesize = true }
|
|
169
|
+
else if (token === '--json') { out.json = true }
|
|
170
|
+
else if (token.startsWith('--')) { out.unsupported.push(token) }
|
|
171
|
+
else { topicParts.push(token) }
|
|
172
|
+
}
|
|
173
|
+
out.topic = topicParts.join(' ')
|
|
174
|
+
return out
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
const RANK_PROMPT = ({ topic, limit = 20, expandCitations = 0, fullTextTop = 0, critiqueTop = 0, preferenceFile = null, reproductionNotes = null, synthesize = false, synthesisTop = 7, synthesisModel = null, outputDir = 'outputs', json = false }) => {
|
|
178
|
+
const slug = slugify(topic)
|
|
179
|
+
const dir = (outputDir || 'outputs').replace(/\/+$/, '')
|
|
180
|
+
const selModel = synthesisModel ?? 'the recommended approved research model'
|
|
181
|
+
return `${TOOL_PRELUDE}
|
|
182
|
+
|
|
183
|
+
Workflow: PaperRank read-first ranking for "${topic}". Scores are a transparent heuristic computed live in this session, not a fitted model; say so in the output.
|
|
184
|
+
1. Fetch up to ${limit} seed candidates via the OpenAlex API (works, cites, references, abstracts, URLs, OA status).${expandCitations > 0 ? ` Expand the citation neighborhood first: add up to ${expandCitations} outgoing cited works (referenced_works) and incoming citing works (cites:<work>) per seed before scoring graph prestige; expansion papers are graph context only; ranked outputs still score the seeds.` : ' Build the local graph from the seed result set only.'}
|
|
185
|
+
2. Score each seed 0-100 as a weighted average over available components (ReadFirstScore): topical relevance 30%, citation impact 20%, graph prestige 20% (PageRank-style over referenced_works edges; when the graph has no local citation edges mark prestige unavailable and exclude it instead of guessing), citation velocity 10% (separate: lifetime counts favor older papers), methodology quality 10%, reproducibility 10%. Methodology and reproducibility are deterministic screening signals over metadata, abstract text, URLs${fullTextTop > 0 ? ', and enriched full text' : ''}. Keep bibliometric influence separate from quality judgments; record unverified checks as gaps, not scores.
|
|
186
|
+
${fullTextTop > 0 ? `3. Full-text enrichment: fetch source-specific full text for the top ${fullTextTop} candidates with a fetchable access route, extract canonical paper sections, attach section-specific paper-body spans, answer checklist rubrics (present / partial / missing / not_evaluated for limitations, reproducibility path, experimental details, statistical significance, compute resources), and rescore. Never write raw full text to papers JSONL; store enrichment status, access candidates, fullTextLength, and section boundaries; score evidence keeps matched spans (source, field, marker, character offsets, section, surrounding text).\n` : ''}Write the default artifacts under ${dir}/ (topic slug ${slug}):
|
|
187
|
+
- ${slug}-research-run.json: typed run manifest of jobs, sources, papers, tools, artifacts, verification state, constraints, and next actions (the machine-readable spine; attach follow-up work here, don't scrape report files).
|
|
188
|
+
- ${slug}-paper-rank.md: readable ranked brief.
|
|
189
|
+
- ${slug}-papers.jsonl: normalized paper records.
|
|
190
|
+
- ${slug}-scores.jsonl: component scores, evidence, matched source spans.
|
|
191
|
+
- ${slug}-score-audit.md: per-paper score math, normalized contribution weights, field roles, evidence gaps, source excerpts.
|
|
192
|
+
- ${slug}-rank-sensitivity.json: rerun the same signals under balanced, influence-heavy, method/reproducibility-heavy, frontier-heavy, and topic-heavy profiles (profile ranks, score range, rank range, stability label; stable = robust, volatile = inspect manually).
|
|
193
|
+
- ${slug}-citation-graph.json: seed/citation-neighborhood graph and PageRank-style values.
|
|
194
|
+
- ${slug}-graph-explorer.html: interactive explorer (search/filter seed and expanded nodes, local citation links, score summaries, field roles, critique judgments, source URLs; no raw full-text bodies).
|
|
195
|
+
- ${slug}-field-map.json: OpenAlex topic/concept clusters with roles foundation, frontier, bridge, methodology-anchor, reproducibility-anchor (local navigation labels, not a global taxonomy).
|
|
196
|
+
- ${slug}-rank.provenance.md: source accounting, formula, verification caveats.
|
|
197
|
+
${critiqueTop > 0 ? `Critique: write ${slug}-critique.md with deterministic research-critique strengths, concerns, and follow-up questions for the top ${critiqueTop} papers, grounded in PaperRank evidence (component scores, warnings, source spans, rubric answers), a triage aid, not an external review decision.\n` : ''}${preferenceFile ? `Calibration: read ${preferenceFile} (rankedPaperIds + pairwise preferences), evaluate whether each preferred paper ranks ahead, report default/profile agreement rates, and write ${slug}-score-calibration.json, ${slug}-calibration-template.json, ${slug}-calibration-guide.md. IDs outside the run count as ignored, never silently dropped.\n` : 'Calibration: no preference file supplied: record that default weights are a transparent product hypothesis, not fitted preferences, and write no calibration files.\n'}${reproductionNotes ? `Reproduction: read ${reproductionNotes} (statuses reproduced / partially_reproduced / failed / not_runnable plus central claim, result, metric, expected/observed values, discrepancy, code/data/environment hints, commands, check date) and write ${slug}-reproduction-ledger.json, ${slug}-reproduction-notes-template.json, ${slug}-replication-plan.md. Notes outside the ranked seed set count as ignored. The ledger records externally supplied notes; it does not execute experiments or embed raw full text.\n` : 'Reproduction: no completed reproduction notes supplied: record that inside the brief and provenance, and write no reproduction files.\n'}${synthesize ? `Synthesis: write ${slug}-synthesis-packet.json and ${slug}-synthesis-prompt.md (ranks, score explanations, field roles, critique summaries, rubric gaps, span excerpts, references for the top ${synthesisTop} papers; omit raw full-text bodies), then ask ${selModel} to write ${slug}-model-synthesis.md from that packet. CLI output, synthesis, JSON summary, and provenance record the actual model plus whether it came from the recommendation path or an explicit override.\n` : ''}${json ? 'Also print a compact JSON summary after writing artifacts.\n' : ''}`
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
/** Split `/paper <id> [--fetch-full-text] [--json]` flags. Unknown --flags are rejected. */
|
|
201
|
+
export function parsePaperArgs(rawInput) {
|
|
202
|
+
const parts = rawInput.trim().split(/\s+/).filter(Boolean)
|
|
203
|
+
const out = { id: '', fetchFullText: false, json: false, unsupported: [] }
|
|
204
|
+
const idParts = []
|
|
205
|
+
for (const token of parts) {
|
|
206
|
+
if (token === '--fetch-full-text') out.fetchFullText = true
|
|
207
|
+
else if (token === '--json') out.json = true
|
|
208
|
+
else if (token.startsWith('--')) out.unsupported.push(token)
|
|
209
|
+
else idParts.push(token)
|
|
210
|
+
}
|
|
211
|
+
out.id = idParts.join(' ')
|
|
212
|
+
return out
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
const PAPER_PROMPT = ({ id: rawId, fetchFullText = false, json = false }) => {
|
|
216
|
+
const id = stripArxivPrefix(rawId)
|
|
217
|
+
const slug = slugify(id)
|
|
218
|
+
return `${TOOL_PRELUDE}
|
|
219
|
+
|
|
220
|
+
Workflow: paper access resolution for "${id}" (DOI, PubMed ID, arXiv ID, or title).
|
|
221
|
+
Resolve access candidates via OpenAlex, DOI, PubMed/PMCID, arXiv, and Europe PMC; for a title, search OpenAlex first. Report candidates without bypassing paywalls.${fetchFullText ? ' With --fetch-full-text, fetch text only through source-sanctioned APIs and write bounded artifacts (summary + access record), never raw full-text dumps.' : ' No full-text fetch requested; access candidates only.'} Write outputs/${slug}-paper-access.md and outputs/${slug}-paper-access.json.${json ? ' Also print a compact JSON access summary after writing artifacts.' : ''}`
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
const PREVIEW_PROMPT = (target) => `${TOOL_PRELUDE}
|
|
225
|
+
|
|
226
|
+
Workflow: preview "${target || 'the most recent file under outputs/'}" (a Markdown, HTML, or PDF research artifact).
|
|
227
|
+
Render with pandoc via bash when it is installed (check first with \`pandoc --version\`): Markdown to HTML with math support, or to PDF for sharing. Open HTML directly; no conversion step. Report the rendered path and, for math-heavy documents, confirm inline ($...$), display ($$...$$), tables, and citations rendered correctly. When pandoc is unavailable, say so and give the exact install command for this machine instead of pretending a preview exists.`
|
|
228
|
+
|
|
229
|
+
/** Compose the full model brief: workflow prompt plus live key refs. */
|
|
230
|
+
export function buildPrompt(kind, args, refs = { hfTokenEnv: 'HF_TOKEN', alphaxivTokenEnv: 'ALPHAXIV_API_KEY' }) {
|
|
231
|
+
return `${WORKFLOWS[kind].prompt(args)}\n\n${KEY_PRELUDE(refs.hfTokenEnv, refs.alphaxivTokenEnv)}`
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
/** All workflow commands: name → spec. `args` = trimmed rawInput or '' . */
|
|
235
|
+
export const WORKFLOWS = {
|
|
236
|
+
deepresearch: {
|
|
237
|
+
description: 'Thorough source-heavy investigation → research brief with inline citations',
|
|
238
|
+
hint: '<topic>',
|
|
239
|
+
prompt: DEEPRESEARCH_PROMPT,
|
|
240
|
+
},
|
|
241
|
+
lit: {
|
|
242
|
+
description: 'Structured literature review: consensus, disagreements, open questions (or lab/PI corpus mode)',
|
|
243
|
+
hint: '<topic-or-lab>',
|
|
244
|
+
prompt: LIT_PROMPT,
|
|
245
|
+
},
|
|
246
|
+
review: {
|
|
247
|
+
description: 'Internal research review with severity-graded feedback and inline annotations',
|
|
248
|
+
hint: '<arXiv-ID | URL | file>',
|
|
249
|
+
prompt: REVIEW_PROMPT,
|
|
250
|
+
},
|
|
251
|
+
'review-loop': {
|
|
252
|
+
description: 'Bounded review→fix→re-review loop for a fixed number of rounds (stop ends it early)',
|
|
253
|
+
hint: '<arXiv-ID | URL | file> [rounds=3] | stop',
|
|
254
|
+
prompt: REVIEW_PROMPT,
|
|
255
|
+
},
|
|
256
|
+
audit: {
|
|
257
|
+
description: "Compare a paper's claims against its codebase: mismatches + reproducibility risks",
|
|
258
|
+
hint: '<arXiv-ID | repo-URL> [--paper <id>]',
|
|
259
|
+
prompt: AUDIT_PROMPT,
|
|
260
|
+
},
|
|
261
|
+
replicate: {
|
|
262
|
+
description: 'Source-backed replication plan (executes only after you pick an environment)',
|
|
263
|
+
hint: '<paper | claim>',
|
|
264
|
+
prompt: REPLICATE_PROMPT,
|
|
265
|
+
},
|
|
266
|
+
recipe: {
|
|
267
|
+
description: 'Ranked implementable ML training recipes backed by papers, data, docs, code',
|
|
268
|
+
hint: '<training-task>',
|
|
269
|
+
prompt: RECIPE_PROMPT,
|
|
270
|
+
},
|
|
271
|
+
compare: {
|
|
272
|
+
description: 'Side-by-side source comparison → agreement/disagreement matrix',
|
|
273
|
+
hint: '<topic | paper-IDs>',
|
|
274
|
+
prompt: COMPARE_PROMPT,
|
|
275
|
+
},
|
|
276
|
+
draft: {
|
|
277
|
+
description: 'Paper-style draft from findings (--from-session reuses this session)',
|
|
278
|
+
hint: '<topic | --from-session>',
|
|
279
|
+
prompt: DRAFT_PROMPT,
|
|
280
|
+
},
|
|
281
|
+
autoresearch: {
|
|
282
|
+
description: 'Bounded hypothesis→experiment→analysis→decision loop against a benchmark',
|
|
283
|
+
hint: '<idea>',
|
|
284
|
+
prompt: AUTORESEARCH_PROMPT,
|
|
285
|
+
},
|
|
286
|
+
watch: {
|
|
287
|
+
description: 'Baseline survey + refresh plan for a fast-moving topic',
|
|
288
|
+
hint: '<topic>',
|
|
289
|
+
prompt: WATCH_PROMPT,
|
|
290
|
+
},
|
|
291
|
+
rank: {
|
|
292
|
+
description: 'Rank papers read-first with transparent relevance/citation/method scoring',
|
|
293
|
+
hint: '<topic> [--limit N] [--expand-citations N] [--full-text-top N] [--critique-top N] [--preference-file F] [--reproduction-notes F] [--synthesize [--synthesis-top N] [--synthesis-model P/M]] [--output-dir D] [--json]',
|
|
294
|
+
prompt: RANK_PROMPT,
|
|
295
|
+
},
|
|
296
|
+
paper: {
|
|
297
|
+
description: 'Resolve legal full-text access candidates for one paper',
|
|
298
|
+
hint: '<DOI | PubMed-ID | arXiv-ID | title> [--fetch-full-text] [--json]',
|
|
299
|
+
prompt: PAPER_PROMPT,
|
|
300
|
+
},
|
|
301
|
+
preview: {
|
|
302
|
+
description: 'Render a research artifact (Markdown/HTML/PDF) via pandoc',
|
|
303
|
+
hint: '[artifact-path]',
|
|
304
|
+
optional: true,
|
|
305
|
+
prompt: PREVIEW_PROMPT,
|
|
306
|
+
},
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
/** Session/utility subcommands of /feynman (handlers live in index.js). */
|
|
310
|
+
export const SESSION_COMMANDS = {
|
|
311
|
+
log: 'Write a durable session log: completed work, findings, open questions, next steps',
|
|
312
|
+
jobs: 'Inspect background-job state and durable watch/experiment artifacts',
|
|
313
|
+
help: 'Show grouped research commands',
|
|
314
|
+
'feynman-model': 'Show how to change the model route for this profile',
|
|
315
|
+
init: 'Bootstrap AGENTS.md and session-log folders for a research project',
|
|
316
|
+
outputs: 'Browse research artifacts under outputs/',
|
|
317
|
+
btw: 'Ask a side question as non-waking context while the main turn runs',
|
|
318
|
+
thinking: 'Note a thinking level (off, minimal, low, medium, high, xhigh, max) for upcoming requests',
|
|
319
|
+
search: 'Search prior session transcripts for past research and findings',
|
|
320
|
+
'web-results': 'List web sources fetched this session with result metadata',
|
|
321
|
+
keys: 'Show or store research API keys (Hugging Face, AlphaXiv)',
|
|
322
|
+
doctor: 'Diagnose key state, mounted seams, pandoc, and the config card',
|
|
323
|
+
status: 'Show the current setup summary (keys, refs, model route)',
|
|
324
|
+
}
|
|
325
|
+
|
|
326
|
+
/** Feynman's thinking levels, accepted by /thinking. */
|
|
327
|
+
export const THINKING_LEVELS = ['off', 'minimal', 'low', 'medium', 'high', 'xhigh', 'max']
|
|
328
|
+
|
|
329
|
+
/** Prompt for one review-loop round after the first. */
|
|
330
|
+
// ponytail: fixed-round loop; add a findings-parser stop condition when early exit matters.
|
|
331
|
+
export function loopFollowupPrompt(target, round, remaining) {
|
|
332
|
+
return `Review-loop round ${round} for "${target}" (${remaining} round(s) left after this one). Address the previous round's critical and major findings first using your tools, then re-review the updated state. Reply with: (1) what you changed or verified since last round, (2) the fresh severity-graded findings (critical/major/minor/nit), (3) whether the artifact is clean enough to stop early.`
|
|
333
|
+
}
|