clearai-dsh 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +26 -0
- package/LICENSE +201 -0
- package/README.md +138 -0
- package/README.zh-CN.md +138 -0
- package/bin/clearai.mjs +224 -0
- package/brand/README.md +41 -0
- package/brand/logo-512-dark.png +0 -0
- package/brand/logo-512.png +0 -0
- package/brand/logo-lockup-dark.png +0 -0
- package/brand/logo-lockup.png +0 -0
- package/brand/logo-lockup.svg +12 -0
- package/brand/logo-wordmark.svg +6 -0
- package/brand/logo.svg +19 -0
- package/cordis.patch.yml +39 -0
- package/lib/client.js +3071 -0
- package/lib/fold.js +1576 -0
- package/lib/host.js +605 -0
- package/package.json +65 -0
- package/presets/clearai/agent.cordis.yml +226 -0
- package/presets/clearai/plugins/brain.js +547 -0
- package/presets/clearai/plugins/clearai-kernel.js +5485 -0
- package/presets/clearai/plugins/ontology.js +306 -0
- package/presets/clearai/plugins/prompts.js +312 -0
- package/presets/clearai/preset.yml +5 -0
- package/presets/clearai/skills/clearai-loop/SKILL.md +89 -0
- package/presets/clearai/template/knowledge/README.md +25 -0
- package/presets/clearai/template/memory/README.md +34 -0
- package/presets/clearai/template/project.md +49 -0
- package/presets/clearai/template/skills/README.md +37 -0
- package/presets/clearai/template/skills/chart-diagram-qa/SKILL.md +43 -0
- package/presets/clearai/template/skills/citation-management/SKILL.md +73 -0
- package/presets/clearai/template/skills/citation-management/references/bibtex_formatting.md +908 -0
- package/presets/clearai/template/skills/citation-management/references/citation_validation.md +794 -0
- package/presets/clearai/template/skills/citation-management/references/google_scholar_search.md +725 -0
- package/presets/clearai/template/skills/citation-management/references/metadata_extraction.md +870 -0
- package/presets/clearai/template/skills/citation-management/references/pubmed_search.md +839 -0
- package/presets/clearai/template/skills/citation-management/scripts/doi_to_bibtex.py +204 -0
- package/presets/clearai/template/skills/citation-management/scripts/extract_metadata.py +569 -0
- package/presets/clearai/template/skills/citation-management/scripts/format_bibtex.py +349 -0
- package/presets/clearai/template/skills/citation-management/scripts/generate_schematic.py +139 -0
- package/presets/clearai/template/skills/citation-management/scripts/generate_schematic_ai.py +817 -0
- package/presets/clearai/template/skills/citation-management/scripts/search_google_scholar.py +282 -0
- package/presets/clearai/template/skills/citation-management/scripts/search_pubmed.py +398 -0
- package/presets/clearai/template/skills/citation-management/scripts/validate_citations.py +497 -0
- package/presets/clearai/template/skills/data-analysis/SKILL.md +92 -0
- package/presets/clearai/template/skills/data-analysis/checklists/readiness_check.md +23 -0
- package/presets/clearai/template/skills/data-analysis/templates/analysis_report.md.tpl +63 -0
- package/presets/clearai/template/skills/data-analysis/templates/cleaning_rules_draft.yaml.tpl +32 -0
- package/presets/clearai/template/skills/data-analysis/templates/data_dictionary.md.tpl +12 -0
- package/presets/clearai/template/skills/data-analysis/templates/domain_knowledge_template.md.tpl +316 -0
- package/presets/clearai/template/skills/data-analysis/templates/feature_candidates.json.tpl +20 -0
- package/presets/clearai/template/skills/data-analysis/templates/quality_scorecard.md.tpl +30 -0
- package/presets/clearai/template/skills/data-analysis/workflows/01-data-profiling.md +42 -0
- package/presets/clearai/template/skills/data-analysis/workflows/02-quality-audit.md +36 -0
- package/presets/clearai/template/skills/data-analysis/workflows/03-physical-correlation.md +25 -0
- package/presets/clearai/template/skills/data-analysis/workflows/04-unstructured-mining.md +26 -0
- package/presets/clearai/template/skills/data-qa-analysis/SKILL.md +102 -0
- package/presets/clearai/template/skills/data-qa-analysis/checklists/readiness_check.md +62 -0
- package/presets/clearai/template/skills/data-qa-analysis/templates/best_in_class_report.md.tpl +56 -0
- package/presets/clearai/template/skills/data-qa-analysis/templates/cleaning_rules_draft.yaml.tpl +56 -0
- package/presets/clearai/template/skills/data-qa-analysis/templates/data_dictionary.md.tpl +13 -0
- package/presets/clearai/template/skills/data-qa-analysis/templates/data_source_inventory_and_lineage.md.tpl +146 -0
- package/presets/clearai/template/skills/data-qa-analysis/templates/data_status_report.md.tpl +60 -0
- package/presets/clearai/template/skills/data-qa-analysis/templates/steady_state_rules.yaml.tpl +41 -0
- package/presets/clearai/template/skills/data-qa-analysis/templates/subsystem_registry.md.tpl +101 -0
- package/presets/clearai/template/skills/data-qa-analysis/templates/unified_execution_plan.md.tpl +100 -0
- package/presets/clearai/template/skills/data-qa-analysis/workflows/01-data-source-inventory-and-lineage.md +194 -0
- package/presets/clearai/template/skills/data-qa-analysis/workflows/02-data-alignment-and-tag-semantics.md +122 -0
- package/presets/clearai/template/skills/data-qa-analysis/workflows/03-steady-state-identification.md +126 -0
- package/presets/clearai/template/skills/data-qa-analysis/workflows/04-consumption-analysis.md +152 -0
- package/presets/clearai/template/skills/data-qa-analysis/workflows/05-best-in-class-and-optimization-space.md +78 -0
- package/presets/clearai/template/skills/domain-presearch/SKILL.md +131 -0
- package/presets/clearai/template/skills/domain-presearch/checklists/domain_checklist.md +24 -0
- package/presets/clearai/template/skills/domain-presearch/references/figure_code.md +78 -0
- package/presets/clearai/template/skills/domain-presearch/references/strategic_frameworks.md +38 -0
- package/presets/clearai/template/skills/exploration-loop/SKILL.md +81 -0
- package/presets/clearai/template/skills/exploratory-data-analysis/SKILL.md +77 -0
- package/presets/clearai/template/skills/exploratory-data-analysis/references/bioinformatics_genomics_formats.md +664 -0
- package/presets/clearai/template/skills/exploratory-data-analysis/references/chemistry_molecular_formats.md +664 -0
- package/presets/clearai/template/skills/exploratory-data-analysis/references/general_scientific_formats.md +518 -0
- package/presets/clearai/template/skills/exploratory-data-analysis/references/microscopy_imaging_formats.md +620 -0
- package/presets/clearai/template/skills/exploratory-data-analysis/references/proteomics_metabolomics_formats.md +517 -0
- package/presets/clearai/template/skills/exploratory-data-analysis/references/spectroscopy_analytical_formats.md +633 -0
- package/presets/clearai/template/skills/exploratory-data-analysis/scripts/eda_analyzer.py +547 -0
- package/presets/clearai/template/skills/hypothesis-generation/SKILL.md +73 -0
- package/presets/clearai/template/skills/hypothesis-generation/references/experimental_design_patterns.md +329 -0
- package/presets/clearai/template/skills/hypothesis-generation/references/hypothesis_quality_criteria.md +198 -0
- package/presets/clearai/template/skills/hypothesis-generation/references/literature_search_strategies.md +622 -0
- package/presets/clearai/template/skills/hypothesis-generation/scripts/generate_schematic.py +139 -0
- package/presets/clearai/template/skills/hypothesis-generation/scripts/generate_schematic_ai.py +817 -0
- package/presets/clearai/template/skills/literature-review/SKILL.md +72 -0
- package/presets/clearai/template/skills/literature-review/references/citation_styles.md +166 -0
- package/presets/clearai/template/skills/literature-review/references/database_strategies.md +455 -0
- package/presets/clearai/template/skills/literature-review/scripts/generate_pdf.py +176 -0
- package/presets/clearai/template/skills/literature-review/scripts/generate_schematic.py +139 -0
- package/presets/clearai/template/skills/literature-review/scripts/generate_schematic_ai.py +817 -0
- package/presets/clearai/template/skills/literature-review/scripts/search_databases.py +303 -0
- package/presets/clearai/template/skills/literature-review/scripts/verify_citations.py +221 -0
- package/presets/clearai/template/skills/paper-lookup/SKILL.md +59 -0
- package/presets/clearai/template/skills/paper-lookup/references/arxiv.md +161 -0
- package/presets/clearai/template/skills/paper-lookup/references/biorxiv.md +118 -0
- package/presets/clearai/template/skills/paper-lookup/references/core.md +150 -0
- package/presets/clearai/template/skills/paper-lookup/references/crossref.md +181 -0
- package/presets/clearai/template/skills/paper-lookup/references/medrxiv.md +104 -0
- package/presets/clearai/template/skills/paper-lookup/references/openalex.md +174 -0
- package/presets/clearai/template/skills/paper-lookup/references/pmc.md +152 -0
- package/presets/clearai/template/skills/paper-lookup/references/pubmed.md +124 -0
- package/presets/clearai/template/skills/paper-lookup/references/semantic-scholar.md +203 -0
- package/presets/clearai/template/skills/paper-lookup/references/unpaywall.md +127 -0
- package/presets/clearai/template/skills/process-presearch/SKILL.md +196 -0
- package/presets/clearai/template/skills/process-presearch/checklists/process_checklist.md +18 -0
- package/presets/clearai/template/skills/process-presearch/references/figure_code.md +107 -0
- package/presets/clearai/template/skills/process-presearch/references/source_attribution_example.md +22 -0
- package/presets/clearai/template/skills/process-understanding-extraction/SKILL.md +69 -0
- package/presets/clearai/template/skills/process-understanding-extraction/checklists/readiness_check.md +34 -0
- package/presets/clearai/template/skills/process-understanding-extraction/templates/docx_raw_dump_extractor.py.tpl +132 -0
- package/presets/clearai/template/skills/process-understanding-extraction/templates/entity_map_unit_topology.json.tpl +86 -0
- package/presets/clearai/template/skills/process-understanding-extraction/templates/process_brief.md.tpl +89 -0
- package/presets/clearai/template/skills/process-understanding-extraction/templates/process_brief_builder_from_raw_dump.py.tpl +203 -0
- package/presets/clearai/template/skills/process-understanding-extraction/templates/process_flow_mermaid.md.tpl +41 -0
- package/presets/clearai/template/skills/process-understanding-extraction/templates/unified_execution_plan.md.tpl +53 -0
- package/presets/clearai/template/skills/process-understanding-extraction/workflows/01-process-doc-discovery.md +173 -0
- package/presets/clearai/template/skills/process-understanding-extraction/workflows/02-process-understanding-and-diagramming.md +106 -0
- package/presets/clearai/template/skills/scientific-brainstorming/SKILL.md +64 -0
- package/presets/clearai/template/skills/scientific-brainstorming/references/brainstorming_methods.md +326 -0
- package/presets/clearai/template/skills/scientific-critical-thinking/SKILL.md +72 -0
- package/presets/clearai/template/skills/scientific-critical-thinking/references/common_biases.md +364 -0
- package/presets/clearai/template/skills/scientific-critical-thinking/references/evidence_hierarchy.md +485 -0
- package/presets/clearai/template/skills/scientific-critical-thinking/references/experimental_design.md +496 -0
- package/presets/clearai/template/skills/scientific-critical-thinking/references/logical_fallacies.md +478 -0
- package/presets/clearai/template/skills/scientific-critical-thinking/references/scientific_method.md +169 -0
- package/presets/clearai/template/skills/scientific-critical-thinking/references/statistical_pitfalls.md +506 -0
- package/presets/clearai/template/skills/skill-creator/SKILL.md +109 -0
- package/presets/clearai/template/skills/skill-creator/references/authoring-guide.md +89 -0
- package/presets/clearai/template/skills/statistical-analysis/SKILL.md +79 -0
- package/presets/clearai/template/skills/statistical-analysis/references/assumptions_and_diagnostics.md +369 -0
- package/presets/clearai/template/skills/statistical-analysis/references/bayesian_statistics.md +653 -0
- package/presets/clearai/template/skills/statistical-analysis/references/effect_sizes_and_power.md +578 -0
- package/presets/clearai/template/skills/statistical-analysis/references/reporting_standards.md +469 -0
- package/presets/clearai/template/skills/statistical-analysis/references/test_selection_guide.md +129 -0
- package/presets/clearai/template/skills/statistical-analysis/scripts/assumption_checks.py +538 -0
- package/presets/clearai/template/skills/web-artifact/SKILL.md +165 -0
- package/presets/clearai/template/skills/web-artifact/assets/renderer/renderer.css +229 -0
- package/presets/clearai/template/skills/web-artifact/assets/renderer/renderer.js +373 -0
- package/presets/clearai/template/skills/web-artifact/assets/vendor/elkjs/LICENSE +263 -0
- package/presets/clearai/template/skills/web-artifact/assets/vendor/elkjs/UPSTREAM.md +26 -0
- package/presets/clearai/template/skills/web-artifact/assets/vendor/elkjs/elk.bundled.js +6605 -0
- package/presets/clearai/template/skills/web-artifact/references/when-drawing-a-topology.md +150 -0
- package/presets/clearai/template/skills/web-artifact/references/when-the-page-must-work-offline.md +62 -0
- package/presets/clearai/template/skills/web-artifact/scripts/check_artifact.py +167 -0
- package/presets/clearai/template/skills/web-artifact/scripts/render_topology.js +272 -0
- package/presets/clearai/template/skills/what-if-oracle/LICENSE.txt +5 -0
- package/presets/clearai/template/skills/what-if-oracle/SKILL.md +72 -0
- package/presets/clearai/template/skills/what-if-oracle/references/scenario-templates.md +154 -0
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
# arXiv API
|
|
2
|
+
|
|
3
|
+
arXiv is a preprint server for physics, mathematics, computer science, quantitative biology, quantitative finance, statistics, electrical engineering, and economics.
|
|
4
|
+
|
|
5
|
+
**Important:** The arXiv API returns **Atom XML**, not JSON. There is no JSON option.
|
|
6
|
+
|
|
7
|
+
## Base URL
|
|
8
|
+
|
|
9
|
+
```
|
|
10
|
+
https://export.arxiv.org/api/query
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
## Authentication
|
|
14
|
+
|
|
15
|
+
None required. Fully public.
|
|
16
|
+
|
|
17
|
+
## Query Parameters
|
|
18
|
+
|
|
19
|
+
```
|
|
20
|
+
GET https://export.arxiv.org/api/query?search_query={query}&start={n}&max_results={n}
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
| Parameter | Required | Default | Description |
|
|
24
|
+
|-----------|----------|---------|-------------|
|
|
25
|
+
| `search_query` | Yes* | -- | Search using field prefixes + boolean operators |
|
|
26
|
+
| `id_list` | Yes* | -- | Comma-separated arXiv IDs (e.g., `2103.15348,2005.14165`) |
|
|
27
|
+
| `start` | No | 0 | Pagination offset (0-based) |
|
|
28
|
+
| `max_results` | No | 10 | Results per request (max 2000; absolute max 30000) |
|
|
29
|
+
| `sortBy` | No | `relevance` | `relevance`, `lastUpdatedDate`, `submittedDate` |
|
|
30
|
+
| `sortOrder` | No | `descending` | `ascending` or `descending` |
|
|
31
|
+
|
|
32
|
+
*At least one of `search_query` or `id_list` must be provided. They can be combined (intersection).
|
|
33
|
+
|
|
34
|
+
## Search Field Prefixes
|
|
35
|
+
|
|
36
|
+
| Prefix | Searches |
|
|
37
|
+
|--------|----------|
|
|
38
|
+
| `ti:` | Title |
|
|
39
|
+
| `au:` | Author |
|
|
40
|
+
| `abs:` | Abstract |
|
|
41
|
+
| `co:` | Comment |
|
|
42
|
+
| `jr:` | Journal reference |
|
|
43
|
+
| `cat:` | Subject category |
|
|
44
|
+
| `rn:` | Report number |
|
|
45
|
+
| `all:` | All fields |
|
|
46
|
+
|
|
47
|
+
## Boolean Operators
|
|
48
|
+
|
|
49
|
+
- `AND` -- both conditions
|
|
50
|
+
- `OR` -- either condition
|
|
51
|
+
- `ANDNOT` -- exclude
|
|
52
|
+
- Parentheses for grouping (URL-encode as `%28` / `%29`)
|
|
53
|
+
- Quoted phrases (URL-encode as `%22`)
|
|
54
|
+
|
|
55
|
+
## Example Queries
|
|
56
|
+
|
|
57
|
+
**Search all fields:**
|
|
58
|
+
```
|
|
59
|
+
https://export.arxiv.org/api/query?search_query=all:transformer+attention&max_results=5
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
**Author + category:**
|
|
63
|
+
```
|
|
64
|
+
https://export.arxiv.org/api/query?search_query=au:hinton+AND+cat:cs.LG&max_results=10
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
**Title search:**
|
|
68
|
+
```
|
|
69
|
+
https://export.arxiv.org/api/query?search_query=ti:%22attention+is+all+you+need%22
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
**By ID:**
|
|
73
|
+
```
|
|
74
|
+
https://export.arxiv.org/api/query?id_list=2103.15348
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
**Multiple IDs:**
|
|
78
|
+
```
|
|
79
|
+
https://export.arxiv.org/api/query?id_list=2103.15348,2005.14165,1706.03762
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
**Date range:**
|
|
83
|
+
```
|
|
84
|
+
https://export.arxiv.org/api/query?search_query=cat:cs.AI+AND+submittedDate:[202401010000+TO+202412312359]
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
## Response Format (Atom XML)
|
|
88
|
+
|
|
89
|
+
```xml
|
|
90
|
+
<feed xmlns="http://www.w3.org/2005/Atom">
|
|
91
|
+
<opensearch:totalResults>1234</opensearch:totalResults>
|
|
92
|
+
<opensearch:startIndex>0</opensearch:startIndex>
|
|
93
|
+
<opensearch:itemsPerPage>10</opensearch:itemsPerPage>
|
|
94
|
+
|
|
95
|
+
<entry>
|
|
96
|
+
<id>http://arxiv.org/abs/1706.03762v7</id>
|
|
97
|
+
<title>Attention Is All You Need</title>
|
|
98
|
+
<summary>The dominant sequence transduction models are based on...</summary>
|
|
99
|
+
<published>2017-06-12T17:57:34Z</published>
|
|
100
|
+
<updated>2023-08-02T00:00:12Z</updated>
|
|
101
|
+
<author><name>Ashish Vaswani</name></author>
|
|
102
|
+
<author><name>Noam Shazeer</name></author>
|
|
103
|
+
<!-- more authors -->
|
|
104
|
+
<category term="cs.CL" scheme="http://arxiv.org/schemas/atom"/>
|
|
105
|
+
<arxiv:primary_category term="cs.CL"/>
|
|
106
|
+
<link rel="alternate" href="http://arxiv.org/abs/1706.03762v7"/>
|
|
107
|
+
<link rel="related" type="application/pdf" href="http://arxiv.org/pdf/1706.03762v7"/>
|
|
108
|
+
<arxiv:doi>10.48550/arXiv.1706.03762</arxiv:doi>
|
|
109
|
+
<arxiv:comment>15 pages, 5 figures</arxiv:comment>
|
|
110
|
+
<arxiv:journal_ref>Advances in Neural Information Processing Systems 30 (NIPS 2017)</arxiv:journal_ref>
|
|
111
|
+
</entry>
|
|
112
|
+
</feed>
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
### Key XML elements per entry
|
|
116
|
+
|
|
117
|
+
| Element | Description |
|
|
118
|
+
|---------|-------------|
|
|
119
|
+
| `<id>` | arXiv URL: `http://arxiv.org/abs/{id}` |
|
|
120
|
+
| `<title>` | Paper title |
|
|
121
|
+
| `<summary>` | Abstract |
|
|
122
|
+
| `<published>` | Original submission date (ISO 8601) |
|
|
123
|
+
| `<updated>` | Date of latest version |
|
|
124
|
+
| `<author><name>` | One per author |
|
|
125
|
+
| `<category term="...">` | Subject categories |
|
|
126
|
+
| `<arxiv:primary_category>` | Primary classification |
|
|
127
|
+
| `<link rel="alternate">` | Abstract page URL |
|
|
128
|
+
| `<link rel="related" title="pdf">` | PDF URL |
|
|
129
|
+
| `<arxiv:doi>` | DOI (when available) |
|
|
130
|
+
| `<arxiv:comment>` | Author comments |
|
|
131
|
+
| `<arxiv:journal_ref>` | Journal reference |
|
|
132
|
+
|
|
133
|
+
## Parsing Tips
|
|
134
|
+
|
|
135
|
+
Since arXiv returns XML, you'll need to parse it. With `curl`, you can pipe the output and extract what you need. The XML namespace is `http://www.w3.org/2005/Atom` with arXiv extensions in `http://arxiv.org/schemas/atom`.
|
|
136
|
+
|
|
137
|
+
For practical extraction, the key data is in `<entry>` elements. Each entry's `<id>` contains the arXiv ID in the URL path.
|
|
138
|
+
|
|
139
|
+
## Common Categories
|
|
140
|
+
|
|
141
|
+
| Category | Field |
|
|
142
|
+
|----------|-------|
|
|
143
|
+
| `cs.AI` | Artificial Intelligence |
|
|
144
|
+
| `cs.CL` | Computation and Language (NLP) |
|
|
145
|
+
| `cs.CV` | Computer Vision |
|
|
146
|
+
| `cs.LG` | Machine Learning |
|
|
147
|
+
| `stat.ML` | Machine Learning (Statistics) |
|
|
148
|
+
| `q-bio` | Quantitative Biology |
|
|
149
|
+
| `physics` | Physics (all subcategories) |
|
|
150
|
+
| `math` | Mathematics (all subcategories) |
|
|
151
|
+
| `econ` | Economics |
|
|
152
|
+
| `eess` | Electrical Engineering and Systems Science |
|
|
153
|
+
|
|
154
|
+
Full list: https://arxiv.org/category_taxonomy
|
|
155
|
+
|
|
156
|
+
## Rate Limits
|
|
157
|
+
|
|
158
|
+
- **1 request every 3 seconds** (hard limit)
|
|
159
|
+
- Single connection at a time
|
|
160
|
+
- Search results are cached daily -- same query won't show new results within 24 hours
|
|
161
|
+
- For bulk data, use the OAI-PMH interface instead
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
# bioRxiv API
|
|
2
|
+
|
|
3
|
+
bioRxiv is a preprint server for biology. The API provides metadata for preprints, including title, authors, abstract, DOI, and publication status.
|
|
4
|
+
|
|
5
|
+
**Important:** The bioRxiv API has **no keyword search**. It supports date-range browsing and DOI lookup only. For keyword search of bioRxiv preprints, use Semantic Scholar, OpenAlex, or CORE instead.
|
|
6
|
+
|
|
7
|
+
## Base URL
|
|
8
|
+
|
|
9
|
+
```
|
|
10
|
+
https://api.biorxiv.org
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
## Authentication
|
|
14
|
+
|
|
15
|
+
None required. Fully public API.
|
|
16
|
+
|
|
17
|
+
## Key Endpoints
|
|
18
|
+
|
|
19
|
+
### 1. Content Detail -- Browse by date range
|
|
20
|
+
|
|
21
|
+
```
|
|
22
|
+
GET /details/biorxiv/{interval}/{cursor}/{format}
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
| Parameter | Values | Description |
|
|
26
|
+
|-----------|--------|-------------|
|
|
27
|
+
| `interval` | `YYYY-MM-DD/YYYY-MM-DD` | Date range (inclusive). Keep ranges narrow (1-3 days) to avoid timeouts. |
|
|
28
|
+
| | `N` (integer) | N most recent preprints |
|
|
29
|
+
| | `Nd` (integer + "d") | Last N days |
|
|
30
|
+
| `cursor` | Integer (default `0`) | Pagination offset (100 results per page) |
|
|
31
|
+
| `format` | `json` (default), `xml` | Response format |
|
|
32
|
+
|
|
33
|
+
Optional query parameter: `?category=neuroscience` (filter by category, use underscores for spaces)
|
|
34
|
+
|
|
35
|
+
**Examples:**
|
|
36
|
+
```
|
|
37
|
+
https://api.biorxiv.org/details/biorxiv/2024-01-01/2024-01-31/0
|
|
38
|
+
https://api.biorxiv.org/details/biorxiv/5
|
|
39
|
+
https://api.biorxiv.org/details/biorxiv/10d
|
|
40
|
+
https://api.biorxiv.org/details/biorxiv/2024-01-01/2024-01-31?category=neuroscience
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
### 2. Content Detail -- DOI lookup
|
|
44
|
+
|
|
45
|
+
```
|
|
46
|
+
GET /details/biorxiv/{doi}/na/{format}
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
**Example:**
|
|
50
|
+
```
|
|
51
|
+
https://api.biorxiv.org/details/biorxiv/10.1101/2024.01.16.575895/na/json
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
### 3. Published Article Links
|
|
55
|
+
|
|
56
|
+
```
|
|
57
|
+
GET /pubs/biorxiv/{interval}/{cursor}
|
|
58
|
+
GET /pubs/biorxiv/{doi}/na
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
Links preprints to their published journal versions. Accepts both preprint DOI and published DOI.
|
|
62
|
+
|
|
63
|
+
### 4. Publisher Filter
|
|
64
|
+
|
|
65
|
+
```
|
|
66
|
+
GET /publisher/{prefix}/{interval}/{cursor}
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
Find bioRxiv papers published by a specific publisher (by DOI prefix).
|
|
70
|
+
|
|
71
|
+
**Example:**
|
|
72
|
+
```
|
|
73
|
+
https://api.biorxiv.org/publisher/10.15252/2024-01-01/2024-06-01/0
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
## Response Format
|
|
77
|
+
|
|
78
|
+
```json
|
|
79
|
+
{
|
|
80
|
+
"messages": [{
|
|
81
|
+
"status": "ok",
|
|
82
|
+
"count": 100,
|
|
83
|
+
"total": "1029",
|
|
84
|
+
"cursor": 0
|
|
85
|
+
}],
|
|
86
|
+
"collection": [{
|
|
87
|
+
"title": "Paper title...",
|
|
88
|
+
"authors": "Surname, A.; Surname, B.",
|
|
89
|
+
"author_corresponding": "Full Name",
|
|
90
|
+
"author_corresponding_institution": "Institution",
|
|
91
|
+
"doi": "10.1101/2024.01.16.575895",
|
|
92
|
+
"date": "2024-01-20",
|
|
93
|
+
"version": "1",
|
|
94
|
+
"type": "new results",
|
|
95
|
+
"license": "cc_no",
|
|
96
|
+
"category": "cancer biology",
|
|
97
|
+
"jatsxml": "https://www.biorxiv.org/content/early/.../source.xml",
|
|
98
|
+
"abstract": "Full abstract text...",
|
|
99
|
+
"published": "10.1158/2159-8290.CD-24-0187",
|
|
100
|
+
"server": "bioRxiv"
|
|
101
|
+
}]
|
|
102
|
+
}
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
- `published` is `"NA"` if not yet published in a journal, or the published DOI if it has been.
|
|
106
|
+
- `type` values: `new results`, `confirmatory results`, `contradictory results`
|
|
107
|
+
|
|
108
|
+
## Pagination
|
|
109
|
+
|
|
110
|
+
All multi-result endpoints return **100 results per page**. Use `cursor` to paginate. The `messages` object tells you the `total` count.
|
|
111
|
+
|
|
112
|
+
## Rate Limits
|
|
113
|
+
|
|
114
|
+
No documented rate limits. No authentication required. Be reasonable with request frequency.
|
|
115
|
+
|
|
116
|
+
## Categories
|
|
117
|
+
|
|
118
|
+
`animal-behavior-and-cognition`, `biochemistry`, `bioengineering`, `bioinformatics`, `biophysics`, `cancer-biology`, `cell-biology`, `clinical-trials`, `developmental-biology`, `ecology`, `epidemiology`, `evolutionary-biology`, `genetics`, `genomics`, `immunology`, `microbiology`, `molecular-biology`, `neuroscience`, `paleontology`, `pathology`, `pharmacology-and-toxicology`, `physiology`, `plant-biology`, `scientific-communication-and-education`, `synthetic-biology`, `systems-biology`, `zoology`
|
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
# CORE API
|
|
2
|
+
|
|
3
|
+
CORE aggregates open access research from 15,000+ repositories worldwide. It provides **full text** for 37M+ articles and metadata for 368M+ papers.
|
|
4
|
+
|
|
5
|
+
## Base URL
|
|
6
|
+
|
|
7
|
+
```
|
|
8
|
+
https://api.core.ac.uk/v3
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
**Important:** GET search paths require a **trailing slash** (e.g., `/v3/search/works/` not `/v3/search/works`).
|
|
12
|
+
|
|
13
|
+
## Authentication
|
|
14
|
+
|
|
15
|
+
- **Header:** `Authorization: Bearer YOUR_API_KEY`
|
|
16
|
+
- **Query param:** `?api_key=YOUR_API_KEY`
|
|
17
|
+
- Register at: https://core.ac.uk/services/api
|
|
18
|
+
|
|
19
|
+
**Without auth:** Basic metadata queries work, but full text is NOT available (returns "Not available for public API users").
|
|
20
|
+
|
|
21
|
+
## Rate Limits (token-based)
|
|
22
|
+
|
|
23
|
+
| User Type | Daily Tokens | Per-Minute Max |
|
|
24
|
+
|-----------|-------------|----------------|
|
|
25
|
+
| Unauthenticated | 100/day | 10/min |
|
|
26
|
+
| Registered Personal | 1,000/day | 25/min |
|
|
27
|
+
| Registered Academic | 5,000/day | 10/min |
|
|
28
|
+
|
|
29
|
+
Simple queries cost 1 token. Downloads and scroll pagination cost 3-5 tokens.
|
|
30
|
+
|
|
31
|
+
## Key Endpoints
|
|
32
|
+
|
|
33
|
+
### 1. Search works
|
|
34
|
+
|
|
35
|
+
```
|
|
36
|
+
GET /v3/search/works/?q={query}&limit={n}&offset={n}
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
| Parameter | Default | Description |
|
|
40
|
+
|-----------|---------|-------------|
|
|
41
|
+
| `q` | required | Search query (supports field lookups, boolean operators) |
|
|
42
|
+
| `limit` | 10 | Results per page (max 100) |
|
|
43
|
+
| `offset` | 0 | Pagination offset |
|
|
44
|
+
| `scroll` | false | Enable scroll pagination for >10,000 results |
|
|
45
|
+
| `sort` | relevance | `relevance` or `recency` |
|
|
46
|
+
|
|
47
|
+
**POST alternative** (for complex queries):
|
|
48
|
+
```
|
|
49
|
+
POST /v3/search/works
|
|
50
|
+
Content-Type: application/json
|
|
51
|
+
|
|
52
|
+
{"q": "machine learning", "limit": 10, "offset": 0}
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
**Example:**
|
|
56
|
+
```
|
|
57
|
+
https://api.core.ac.uk/v3/search/works/?q=CRISPR+gene+therapy&limit=10
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
### 2. Query language
|
|
61
|
+
|
|
62
|
+
| Operator | Example | Description |
|
|
63
|
+
|----------|---------|-------------|
|
|
64
|
+
| AND | `title:"AI" AND authors:"Smith"` | Both conditions |
|
|
65
|
+
| OR | `title:"AI" OR fullText:"Deep Learning"` | Either condition |
|
|
66
|
+
| Grouping | `(title:"AI" OR title:"ML") AND yearPublished>"2020"` | Precedence |
|
|
67
|
+
| Field lookup | `title:"Machine Learning"` | Search specific field |
|
|
68
|
+
| Range | `yearPublished>2018` | Numeric comparison |
|
|
69
|
+
| Exists | `_exists_:fullText` | Field must exist |
|
|
70
|
+
| Phrase | `title:"Attention is all you need"` | Exact phrase |
|
|
71
|
+
|
|
72
|
+
**Searchable fields:** `abstract`, `arxivId`, `authors`, `contributors`, `createdDate`, `dataProviders`, `depositedDate`, `documentType`, `doi`, `fullText`, `id`, `language`, `license`, `oai`, `title`, `yearPublished`
|
|
73
|
+
|
|
74
|
+
### 3. Get work by ID
|
|
75
|
+
|
|
76
|
+
```
|
|
77
|
+
GET /v3/works/{id}
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
`id` is a CORE Work ID (integer). Example: `/v3/works/267312`
|
|
81
|
+
|
|
82
|
+
### 4. Get output by ID
|
|
83
|
+
|
|
84
|
+
```
|
|
85
|
+
GET /v3/outputs/{id}
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
### 5. Download full text
|
|
89
|
+
|
|
90
|
+
```
|
|
91
|
+
GET /v3/outputs/{id}/download
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
Returns binary PDF. Requires authentication.
|
|
95
|
+
|
|
96
|
+
```
|
|
97
|
+
GET /v3/works/tei/{id}
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
Returns TEI XML format.
|
|
101
|
+
|
|
102
|
+
### 6. Search outputs
|
|
103
|
+
|
|
104
|
+
```
|
|
105
|
+
GET /v3/search/outputs/?q={query}&limit={n}&offset={n}
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
Search by DOI: `q=doi:10.1038/nature12373`
|
|
109
|
+
|
|
110
|
+
## Response Format
|
|
111
|
+
|
|
112
|
+
### Search response
|
|
113
|
+
```json
|
|
114
|
+
{
|
|
115
|
+
"totalHits": 2281337,
|
|
116
|
+
"limit": 10,
|
|
117
|
+
"offset": 0,
|
|
118
|
+
"scrollId": null,
|
|
119
|
+
"results": [...]
|
|
120
|
+
}
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
### Work object (key fields)
|
|
124
|
+
```json
|
|
125
|
+
{
|
|
126
|
+
"id": 8848131,
|
|
127
|
+
"title": "Attention Is All You Need",
|
|
128
|
+
"authors": [{"name": "Ashish Vaswani"}, ...],
|
|
129
|
+
"abstract": "The dominant sequence...",
|
|
130
|
+
"doi": "10.48550/arXiv.1706.03762",
|
|
131
|
+
"arxivId": "1706.03762",
|
|
132
|
+
"yearPublished": 2017,
|
|
133
|
+
"downloadUrl": "https://core.ac.uk/download/...",
|
|
134
|
+
"fullText": "Full text content (when authenticated)...",
|
|
135
|
+
"language": {"code": "en", "name": "English"},
|
|
136
|
+
"documentType": "research",
|
|
137
|
+
"citationCount": 145678,
|
|
138
|
+
"dataProviders": [{"name": "arXiv"}],
|
|
139
|
+
"links": [{"type": "download", "url": "..."}]
|
|
140
|
+
}
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
## Pagination
|
|
144
|
+
|
|
145
|
+
- **Standard:** `offset` + `limit` (max 10,000 results)
|
|
146
|
+
- **Scroll:** Set `scroll=true`. Response includes `scrollId`. Use in subsequent requests to page beyond 10,000 (costs more tokens).
|
|
147
|
+
|
|
148
|
+
## Error Handling
|
|
149
|
+
|
|
150
|
+
Under heavy load, the API may return partial shard failure messages. These are transient -- retry after a brief wait.
|
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
# Crossref API
|
|
2
|
+
|
|
3
|
+
Crossref is the DOI registration agency for scholarly content. It provides metadata for 150M+ works including journal articles, books, conference papers, datasets, and preprints.
|
|
4
|
+
|
|
5
|
+
## Base URL
|
|
6
|
+
|
|
7
|
+
```
|
|
8
|
+
https://api.crossref.org
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
## Authentication
|
|
12
|
+
|
|
13
|
+
None required. Add `mailto=you@example.com` to get into the **polite pool** (2x faster rate limits).
|
|
14
|
+
|
|
15
|
+
## Rate Limits
|
|
16
|
+
|
|
17
|
+
| Pool | Rate | Concurrency |
|
|
18
|
+
|------|------|-------------|
|
|
19
|
+
| Public (no mailto) | 5 req/sec | 1 concurrent |
|
|
20
|
+
| Polite (with mailto) | 10 req/sec | 3 concurrent |
|
|
21
|
+
|
|
22
|
+
HTTP 429 = temporarily blocked.
|
|
23
|
+
|
|
24
|
+
## Key Endpoints
|
|
25
|
+
|
|
26
|
+
### 1. Search works
|
|
27
|
+
|
|
28
|
+
```
|
|
29
|
+
GET /works?query={text}&rows={n}&mailto=you@example.com
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
| Parameter | Default | Description |
|
|
33
|
+
|-----------|---------|-------------|
|
|
34
|
+
| `query` | -- | Free-text search across all fields |
|
|
35
|
+
| `query.author` | -- | Search author names |
|
|
36
|
+
| `query.bibliographic` | -- | Search titles, authors, ISSNs, years |
|
|
37
|
+
| `query.affiliation` | -- | Search affiliations |
|
|
38
|
+
| `query.container-title` | -- | Search journal names |
|
|
39
|
+
| `filter` | -- | Comma-separated `name:value` pairs |
|
|
40
|
+
| `sort` | `score` | `score`, `published`, `issued`, `deposited`, `updated`, `is-referenced-by-count`, `references-count` |
|
|
41
|
+
| `order` | `desc` | `asc` or `desc` |
|
|
42
|
+
| `rows` | 20 | Results per page (max 1000) |
|
|
43
|
+
| `offset` | 0 | Skip N results (max 10,000) |
|
|
44
|
+
| `cursor` | -- | Use `*` for cursor-based deep pagination |
|
|
45
|
+
| `select` | -- | Comma-separated field names to return |
|
|
46
|
+
| `facet` | -- | Facet counts, e.g. `type-name:10` |
|
|
47
|
+
| `sample` | -- | Return N random items (max 100) |
|
|
48
|
+
|
|
49
|
+
**Example:**
|
|
50
|
+
```
|
|
51
|
+
https://api.crossref.org/works?query=CRISPR+gene+therapy&filter=from-pub-date:2024-01-01,type:journal-article,has-abstract:true&rows=5&sort=published&order=desc&mailto=you@example.com
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
### 2. Get work by DOI
|
|
55
|
+
|
|
56
|
+
```
|
|
57
|
+
GET /works/{doi}?mailto=you@example.com
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
URL-encode the DOI: `10.1038/nature12373` becomes `10.1038%2Fnature12373`
|
|
61
|
+
|
|
62
|
+
**Example:**
|
|
63
|
+
```
|
|
64
|
+
https://api.crossref.org/works/10.1038%2Fnature12373?mailto=you@example.com
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
### 3. Journals
|
|
68
|
+
|
|
69
|
+
```
|
|
70
|
+
GET /journals?query={name}&rows={n}
|
|
71
|
+
GET /journals/{issn}
|
|
72
|
+
GET /journals/{issn}/works?query={text}&rows={n}
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
### 4. Funders
|
|
76
|
+
|
|
77
|
+
```
|
|
78
|
+
GET /funders?query={name}
|
|
79
|
+
GET /funders/{id}
|
|
80
|
+
GET /funders/{id}/works?rows={n}
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
Funder IDs are from the Funder Registry (e.g., `100000001` for NSF).
|
|
84
|
+
|
|
85
|
+
### 5. Members (publishers)
|
|
86
|
+
|
|
87
|
+
```
|
|
88
|
+
GET /members?query={name}
|
|
89
|
+
GET /members/{id}/works?rows={n}
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
## Key Filters
|
|
93
|
+
|
|
94
|
+
### Date filters (accept `YYYY`, `YYYY-MM`, `YYYY-MM-DD`)
|
|
95
|
+
| Filter | Description |
|
|
96
|
+
|--------|-------------|
|
|
97
|
+
| `from-pub-date` / `until-pub-date` | Publication date |
|
|
98
|
+
| `from-print-pub-date` / `until-print-pub-date` | Print publication date |
|
|
99
|
+
| `from-online-pub-date` / `until-online-pub-date` | Online publication date |
|
|
100
|
+
| `from-posted-date` / `until-posted-date` | Posted date (preprints) |
|
|
101
|
+
|
|
102
|
+
### Boolean filters
|
|
103
|
+
| Filter | Description |
|
|
104
|
+
|--------|-------------|
|
|
105
|
+
| `has-abstract` | Has an abstract |
|
|
106
|
+
| `has-orcid` | Has ORCID IDs |
|
|
107
|
+
| `has-funder` | Has funder info |
|
|
108
|
+
| `has-full-text` | Has full-text links |
|
|
109
|
+
| `has-references` | Has reference list |
|
|
110
|
+
| `has-license` | Has license info |
|
|
111
|
+
|
|
112
|
+
### Value filters
|
|
113
|
+
| Filter | Description |
|
|
114
|
+
|--------|-------------|
|
|
115
|
+
| `type` | `journal-article`, `posted-content`, `book-chapter`, `proceedings-article`, etc. |
|
|
116
|
+
| `issn` | Journal ISSN |
|
|
117
|
+
| `doi` | Specific DOI |
|
|
118
|
+
| `orcid` | Contributor ORCID |
|
|
119
|
+
| `funder` | Funder Registry ID |
|
|
120
|
+
| `member` | Crossref member ID |
|
|
121
|
+
| `prefix` | DOI prefix |
|
|
122
|
+
| `license.url` | License URL |
|
|
123
|
+
| `update-type` | `correction`, `retraction` |
|
|
124
|
+
|
|
125
|
+
**Syntax:** `filter=name1:value1,name2:value2`
|
|
126
|
+
|
|
127
|
+
## Pagination
|
|
128
|
+
|
|
129
|
+
### Offset-based (max 10,000)
|
|
130
|
+
```
|
|
131
|
+
/works?query=cancer&rows=100&offset=200
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
### Cursor-based (unlimited)
|
|
135
|
+
1. First request: `?cursor=*&rows=100`
|
|
136
|
+
2. Response includes `next-cursor`
|
|
137
|
+
3. Next request: `?cursor={next-cursor-value}&rows=100`
|
|
138
|
+
4. Cursors expire after 5 minutes
|
|
139
|
+
|
|
140
|
+
## Response Format
|
|
141
|
+
|
|
142
|
+
### List response
|
|
143
|
+
```json
|
|
144
|
+
{
|
|
145
|
+
"status": "ok",
|
|
146
|
+
"message-type": "work-list",
|
|
147
|
+
"message": {
|
|
148
|
+
"total-results": 2779116,
|
|
149
|
+
"items-per-page": 20,
|
|
150
|
+
"next-cursor": "...",
|
|
151
|
+
"items": [...]
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
### Work object (key fields)
|
|
157
|
+
```json
|
|
158
|
+
{
|
|
159
|
+
"DOI": "10.1038/nature12373",
|
|
160
|
+
"title": ["Nanometre-scale thermometry in a living cell"],
|
|
161
|
+
"author": [{"given": "G.", "family": "Kucsko", "sequence": "first"}],
|
|
162
|
+
"publisher": "Springer Science and Business Media LLC",
|
|
163
|
+
"type": "journal-article",
|
|
164
|
+
"published": {"date-parts": [[2013, 7, 31]]},
|
|
165
|
+
"container-title": ["Nature"],
|
|
166
|
+
"ISSN": ["0028-0836", "1476-4687"],
|
|
167
|
+
"volume": "500",
|
|
168
|
+
"issue": "7460",
|
|
169
|
+
"page": "54-58",
|
|
170
|
+
"is-referenced-by-count": 1745,
|
|
171
|
+
"references-count": 30,
|
|
172
|
+
"abstract": "<p>Abstract text with HTML tags...</p>",
|
|
173
|
+
"license": [{"URL": "...", "content-version": "vor"}],
|
|
174
|
+
"link": [{"URL": "...", "content-type": "application/pdf"}],
|
|
175
|
+
"reference": [{"key": "...", "doi-asserted-by": "crossref", "DOI": "..."}],
|
|
176
|
+
"subject": ["Multidisciplinary"],
|
|
177
|
+
"language": "en"
|
|
178
|
+
}
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
Note: `title` and `container-title` are arrays. `published.date-parts` is `[[year, month, day]]`. Abstract may contain HTML tags.
|