clearai-dsh 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (154) hide show
  1. package/CHANGELOG.md +26 -0
  2. package/LICENSE +201 -0
  3. package/README.md +138 -0
  4. package/README.zh-CN.md +138 -0
  5. package/bin/clearai.mjs +224 -0
  6. package/brand/README.md +41 -0
  7. package/brand/logo-512-dark.png +0 -0
  8. package/brand/logo-512.png +0 -0
  9. package/brand/logo-lockup-dark.png +0 -0
  10. package/brand/logo-lockup.png +0 -0
  11. package/brand/logo-lockup.svg +12 -0
  12. package/brand/logo-wordmark.svg +6 -0
  13. package/brand/logo.svg +19 -0
  14. package/cordis.patch.yml +39 -0
  15. package/lib/client.js +3071 -0
  16. package/lib/fold.js +1576 -0
  17. package/lib/host.js +605 -0
  18. package/package.json +65 -0
  19. package/presets/clearai/agent.cordis.yml +226 -0
  20. package/presets/clearai/plugins/brain.js +547 -0
  21. package/presets/clearai/plugins/clearai-kernel.js +5485 -0
  22. package/presets/clearai/plugins/ontology.js +306 -0
  23. package/presets/clearai/plugins/prompts.js +312 -0
  24. package/presets/clearai/preset.yml +5 -0
  25. package/presets/clearai/skills/clearai-loop/SKILL.md +89 -0
  26. package/presets/clearai/template/knowledge/README.md +25 -0
  27. package/presets/clearai/template/memory/README.md +34 -0
  28. package/presets/clearai/template/project.md +49 -0
  29. package/presets/clearai/template/skills/README.md +37 -0
  30. package/presets/clearai/template/skills/chart-diagram-qa/SKILL.md +43 -0
  31. package/presets/clearai/template/skills/citation-management/SKILL.md +73 -0
  32. package/presets/clearai/template/skills/citation-management/references/bibtex_formatting.md +908 -0
  33. package/presets/clearai/template/skills/citation-management/references/citation_validation.md +794 -0
  34. package/presets/clearai/template/skills/citation-management/references/google_scholar_search.md +725 -0
  35. package/presets/clearai/template/skills/citation-management/references/metadata_extraction.md +870 -0
  36. package/presets/clearai/template/skills/citation-management/references/pubmed_search.md +839 -0
  37. package/presets/clearai/template/skills/citation-management/scripts/doi_to_bibtex.py +204 -0
  38. package/presets/clearai/template/skills/citation-management/scripts/extract_metadata.py +569 -0
  39. package/presets/clearai/template/skills/citation-management/scripts/format_bibtex.py +349 -0
  40. package/presets/clearai/template/skills/citation-management/scripts/generate_schematic.py +139 -0
  41. package/presets/clearai/template/skills/citation-management/scripts/generate_schematic_ai.py +817 -0
  42. package/presets/clearai/template/skills/citation-management/scripts/search_google_scholar.py +282 -0
  43. package/presets/clearai/template/skills/citation-management/scripts/search_pubmed.py +398 -0
  44. package/presets/clearai/template/skills/citation-management/scripts/validate_citations.py +497 -0
  45. package/presets/clearai/template/skills/data-analysis/SKILL.md +92 -0
  46. package/presets/clearai/template/skills/data-analysis/checklists/readiness_check.md +23 -0
  47. package/presets/clearai/template/skills/data-analysis/templates/analysis_report.md.tpl +63 -0
  48. package/presets/clearai/template/skills/data-analysis/templates/cleaning_rules_draft.yaml.tpl +32 -0
  49. package/presets/clearai/template/skills/data-analysis/templates/data_dictionary.md.tpl +12 -0
  50. package/presets/clearai/template/skills/data-analysis/templates/domain_knowledge_template.md.tpl +316 -0
  51. package/presets/clearai/template/skills/data-analysis/templates/feature_candidates.json.tpl +20 -0
  52. package/presets/clearai/template/skills/data-analysis/templates/quality_scorecard.md.tpl +30 -0
  53. package/presets/clearai/template/skills/data-analysis/workflows/01-data-profiling.md +42 -0
  54. package/presets/clearai/template/skills/data-analysis/workflows/02-quality-audit.md +36 -0
  55. package/presets/clearai/template/skills/data-analysis/workflows/03-physical-correlation.md +25 -0
  56. package/presets/clearai/template/skills/data-analysis/workflows/04-unstructured-mining.md +26 -0
  57. package/presets/clearai/template/skills/data-qa-analysis/SKILL.md +102 -0
  58. package/presets/clearai/template/skills/data-qa-analysis/checklists/readiness_check.md +62 -0
  59. package/presets/clearai/template/skills/data-qa-analysis/templates/best_in_class_report.md.tpl +56 -0
  60. package/presets/clearai/template/skills/data-qa-analysis/templates/cleaning_rules_draft.yaml.tpl +56 -0
  61. package/presets/clearai/template/skills/data-qa-analysis/templates/data_dictionary.md.tpl +13 -0
  62. package/presets/clearai/template/skills/data-qa-analysis/templates/data_source_inventory_and_lineage.md.tpl +146 -0
  63. package/presets/clearai/template/skills/data-qa-analysis/templates/data_status_report.md.tpl +60 -0
  64. package/presets/clearai/template/skills/data-qa-analysis/templates/steady_state_rules.yaml.tpl +41 -0
  65. package/presets/clearai/template/skills/data-qa-analysis/templates/subsystem_registry.md.tpl +101 -0
  66. package/presets/clearai/template/skills/data-qa-analysis/templates/unified_execution_plan.md.tpl +100 -0
  67. package/presets/clearai/template/skills/data-qa-analysis/workflows/01-data-source-inventory-and-lineage.md +194 -0
  68. package/presets/clearai/template/skills/data-qa-analysis/workflows/02-data-alignment-and-tag-semantics.md +122 -0
  69. package/presets/clearai/template/skills/data-qa-analysis/workflows/03-steady-state-identification.md +126 -0
  70. package/presets/clearai/template/skills/data-qa-analysis/workflows/04-consumption-analysis.md +152 -0
  71. package/presets/clearai/template/skills/data-qa-analysis/workflows/05-best-in-class-and-optimization-space.md +78 -0
  72. package/presets/clearai/template/skills/domain-presearch/SKILL.md +131 -0
  73. package/presets/clearai/template/skills/domain-presearch/checklists/domain_checklist.md +24 -0
  74. package/presets/clearai/template/skills/domain-presearch/references/figure_code.md +78 -0
  75. package/presets/clearai/template/skills/domain-presearch/references/strategic_frameworks.md +38 -0
  76. package/presets/clearai/template/skills/exploration-loop/SKILL.md +81 -0
  77. package/presets/clearai/template/skills/exploratory-data-analysis/SKILL.md +77 -0
  78. package/presets/clearai/template/skills/exploratory-data-analysis/references/bioinformatics_genomics_formats.md +664 -0
  79. package/presets/clearai/template/skills/exploratory-data-analysis/references/chemistry_molecular_formats.md +664 -0
  80. package/presets/clearai/template/skills/exploratory-data-analysis/references/general_scientific_formats.md +518 -0
  81. package/presets/clearai/template/skills/exploratory-data-analysis/references/microscopy_imaging_formats.md +620 -0
  82. package/presets/clearai/template/skills/exploratory-data-analysis/references/proteomics_metabolomics_formats.md +517 -0
  83. package/presets/clearai/template/skills/exploratory-data-analysis/references/spectroscopy_analytical_formats.md +633 -0
  84. package/presets/clearai/template/skills/exploratory-data-analysis/scripts/eda_analyzer.py +547 -0
  85. package/presets/clearai/template/skills/hypothesis-generation/SKILL.md +73 -0
  86. package/presets/clearai/template/skills/hypothesis-generation/references/experimental_design_patterns.md +329 -0
  87. package/presets/clearai/template/skills/hypothesis-generation/references/hypothesis_quality_criteria.md +198 -0
  88. package/presets/clearai/template/skills/hypothesis-generation/references/literature_search_strategies.md +622 -0
  89. package/presets/clearai/template/skills/hypothesis-generation/scripts/generate_schematic.py +139 -0
  90. package/presets/clearai/template/skills/hypothesis-generation/scripts/generate_schematic_ai.py +817 -0
  91. package/presets/clearai/template/skills/literature-review/SKILL.md +72 -0
  92. package/presets/clearai/template/skills/literature-review/references/citation_styles.md +166 -0
  93. package/presets/clearai/template/skills/literature-review/references/database_strategies.md +455 -0
  94. package/presets/clearai/template/skills/literature-review/scripts/generate_pdf.py +176 -0
  95. package/presets/clearai/template/skills/literature-review/scripts/generate_schematic.py +139 -0
  96. package/presets/clearai/template/skills/literature-review/scripts/generate_schematic_ai.py +817 -0
  97. package/presets/clearai/template/skills/literature-review/scripts/search_databases.py +303 -0
  98. package/presets/clearai/template/skills/literature-review/scripts/verify_citations.py +221 -0
  99. package/presets/clearai/template/skills/paper-lookup/SKILL.md +59 -0
  100. package/presets/clearai/template/skills/paper-lookup/references/arxiv.md +161 -0
  101. package/presets/clearai/template/skills/paper-lookup/references/biorxiv.md +118 -0
  102. package/presets/clearai/template/skills/paper-lookup/references/core.md +150 -0
  103. package/presets/clearai/template/skills/paper-lookup/references/crossref.md +181 -0
  104. package/presets/clearai/template/skills/paper-lookup/references/medrxiv.md +104 -0
  105. package/presets/clearai/template/skills/paper-lookup/references/openalex.md +174 -0
  106. package/presets/clearai/template/skills/paper-lookup/references/pmc.md +152 -0
  107. package/presets/clearai/template/skills/paper-lookup/references/pubmed.md +124 -0
  108. package/presets/clearai/template/skills/paper-lookup/references/semantic-scholar.md +203 -0
  109. package/presets/clearai/template/skills/paper-lookup/references/unpaywall.md +127 -0
  110. package/presets/clearai/template/skills/process-presearch/SKILL.md +196 -0
  111. package/presets/clearai/template/skills/process-presearch/checklists/process_checklist.md +18 -0
  112. package/presets/clearai/template/skills/process-presearch/references/figure_code.md +107 -0
  113. package/presets/clearai/template/skills/process-presearch/references/source_attribution_example.md +22 -0
  114. package/presets/clearai/template/skills/process-understanding-extraction/SKILL.md +69 -0
  115. package/presets/clearai/template/skills/process-understanding-extraction/checklists/readiness_check.md +34 -0
  116. package/presets/clearai/template/skills/process-understanding-extraction/templates/docx_raw_dump_extractor.py.tpl +132 -0
  117. package/presets/clearai/template/skills/process-understanding-extraction/templates/entity_map_unit_topology.json.tpl +86 -0
  118. package/presets/clearai/template/skills/process-understanding-extraction/templates/process_brief.md.tpl +89 -0
  119. package/presets/clearai/template/skills/process-understanding-extraction/templates/process_brief_builder_from_raw_dump.py.tpl +203 -0
  120. package/presets/clearai/template/skills/process-understanding-extraction/templates/process_flow_mermaid.md.tpl +41 -0
  121. package/presets/clearai/template/skills/process-understanding-extraction/templates/unified_execution_plan.md.tpl +53 -0
  122. package/presets/clearai/template/skills/process-understanding-extraction/workflows/01-process-doc-discovery.md +173 -0
  123. package/presets/clearai/template/skills/process-understanding-extraction/workflows/02-process-understanding-and-diagramming.md +106 -0
  124. package/presets/clearai/template/skills/scientific-brainstorming/SKILL.md +64 -0
  125. package/presets/clearai/template/skills/scientific-brainstorming/references/brainstorming_methods.md +326 -0
  126. package/presets/clearai/template/skills/scientific-critical-thinking/SKILL.md +72 -0
  127. package/presets/clearai/template/skills/scientific-critical-thinking/references/common_biases.md +364 -0
  128. package/presets/clearai/template/skills/scientific-critical-thinking/references/evidence_hierarchy.md +485 -0
  129. package/presets/clearai/template/skills/scientific-critical-thinking/references/experimental_design.md +496 -0
  130. package/presets/clearai/template/skills/scientific-critical-thinking/references/logical_fallacies.md +478 -0
  131. package/presets/clearai/template/skills/scientific-critical-thinking/references/scientific_method.md +169 -0
  132. package/presets/clearai/template/skills/scientific-critical-thinking/references/statistical_pitfalls.md +506 -0
  133. package/presets/clearai/template/skills/skill-creator/SKILL.md +109 -0
  134. package/presets/clearai/template/skills/skill-creator/references/authoring-guide.md +89 -0
  135. package/presets/clearai/template/skills/statistical-analysis/SKILL.md +79 -0
  136. package/presets/clearai/template/skills/statistical-analysis/references/assumptions_and_diagnostics.md +369 -0
  137. package/presets/clearai/template/skills/statistical-analysis/references/bayesian_statistics.md +653 -0
  138. package/presets/clearai/template/skills/statistical-analysis/references/effect_sizes_and_power.md +578 -0
  139. package/presets/clearai/template/skills/statistical-analysis/references/reporting_standards.md +469 -0
  140. package/presets/clearai/template/skills/statistical-analysis/references/test_selection_guide.md +129 -0
  141. package/presets/clearai/template/skills/statistical-analysis/scripts/assumption_checks.py +538 -0
  142. package/presets/clearai/template/skills/web-artifact/SKILL.md +165 -0
  143. package/presets/clearai/template/skills/web-artifact/assets/renderer/renderer.css +229 -0
  144. package/presets/clearai/template/skills/web-artifact/assets/renderer/renderer.js +373 -0
  145. package/presets/clearai/template/skills/web-artifact/assets/vendor/elkjs/LICENSE +263 -0
  146. package/presets/clearai/template/skills/web-artifact/assets/vendor/elkjs/UPSTREAM.md +26 -0
  147. package/presets/clearai/template/skills/web-artifact/assets/vendor/elkjs/elk.bundled.js +6605 -0
  148. package/presets/clearai/template/skills/web-artifact/references/when-drawing-a-topology.md +150 -0
  149. package/presets/clearai/template/skills/web-artifact/references/when-the-page-must-work-offline.md +62 -0
  150. package/presets/clearai/template/skills/web-artifact/scripts/check_artifact.py +167 -0
  151. package/presets/clearai/template/skills/web-artifact/scripts/render_topology.js +272 -0
  152. package/presets/clearai/template/skills/what-if-oracle/LICENSE.txt +5 -0
  153. package/presets/clearai/template/skills/what-if-oracle/SKILL.md +72 -0
  154. package/presets/clearai/template/skills/what-if-oracle/references/scenario-templates.md +154 -0
@@ -0,0 +1,161 @@
1
+ # arXiv API
2
+
3
+ arXiv is a preprint server for physics, mathematics, computer science, quantitative biology, quantitative finance, statistics, electrical engineering, and economics.
4
+
5
+ **Important:** The arXiv API returns **Atom XML**, not JSON. There is no JSON option.
6
+
7
+ ## Base URL
8
+
9
+ ```
10
+ https://export.arxiv.org/api/query
11
+ ```
12
+
13
+ ## Authentication
14
+
15
+ None required. Fully public.
16
+
17
+ ## Query Parameters
18
+
19
+ ```
20
+ GET https://export.arxiv.org/api/query?search_query={query}&start={n}&max_results={n}
21
+ ```
22
+
23
+ | Parameter | Required | Default | Description |
24
+ |-----------|----------|---------|-------------|
25
+ | `search_query` | Yes* | -- | Search using field prefixes + boolean operators |
26
+ | `id_list` | Yes* | -- | Comma-separated arXiv IDs (e.g., `2103.15348,2005.14165`) |
27
+ | `start` | No | 0 | Pagination offset (0-based) |
28
+ | `max_results` | No | 10 | Results per request (max 2000; absolute max 30000) |
29
+ | `sortBy` | No | `relevance` | `relevance`, `lastUpdatedDate`, `submittedDate` |
30
+ | `sortOrder` | No | `descending` | `ascending` or `descending` |
31
+
32
+ *At least one of `search_query` or `id_list` must be provided. They can be combined (intersection).
33
+
34
+ ## Search Field Prefixes
35
+
36
+ | Prefix | Searches |
37
+ |--------|----------|
38
+ | `ti:` | Title |
39
+ | `au:` | Author |
40
+ | `abs:` | Abstract |
41
+ | `co:` | Comment |
42
+ | `jr:` | Journal reference |
43
+ | `cat:` | Subject category |
44
+ | `rn:` | Report number |
45
+ | `all:` | All fields |
46
+
47
+ ## Boolean Operators
48
+
49
+ - `AND` -- both conditions
50
+ - `OR` -- either condition
51
+ - `ANDNOT` -- exclude
52
+ - Parentheses for grouping (URL-encode as `%28` / `%29`)
53
+ - Quoted phrases (URL-encode as `%22`)
54
+
55
+ ## Example Queries
56
+
57
+ **Search all fields:**
58
+ ```
59
+ https://export.arxiv.org/api/query?search_query=all:transformer+attention&max_results=5
60
+ ```
61
+
62
+ **Author + category:**
63
+ ```
64
+ https://export.arxiv.org/api/query?search_query=au:hinton+AND+cat:cs.LG&max_results=10
65
+ ```
66
+
67
+ **Title search:**
68
+ ```
69
+ https://export.arxiv.org/api/query?search_query=ti:%22attention+is+all+you+need%22
70
+ ```
71
+
72
+ **By ID:**
73
+ ```
74
+ https://export.arxiv.org/api/query?id_list=2103.15348
75
+ ```
76
+
77
+ **Multiple IDs:**
78
+ ```
79
+ https://export.arxiv.org/api/query?id_list=2103.15348,2005.14165,1706.03762
80
+ ```
81
+
82
+ **Date range:**
83
+ ```
84
+ https://export.arxiv.org/api/query?search_query=cat:cs.AI+AND+submittedDate:[202401010000+TO+202412312359]
85
+ ```
86
+
87
+ ## Response Format (Atom XML)
88
+
89
+ ```xml
90
+ <feed xmlns="http://www.w3.org/2005/Atom">
91
+ <opensearch:totalResults>1234</opensearch:totalResults>
92
+ <opensearch:startIndex>0</opensearch:startIndex>
93
+ <opensearch:itemsPerPage>10</opensearch:itemsPerPage>
94
+
95
+ <entry>
96
+ <id>http://arxiv.org/abs/1706.03762v7</id>
97
+ <title>Attention Is All You Need</title>
98
+ <summary>The dominant sequence transduction models are based on...</summary>
99
+ <published>2017-06-12T17:57:34Z</published>
100
+ <updated>2023-08-02T00:00:12Z</updated>
101
+ <author><name>Ashish Vaswani</name></author>
102
+ <author><name>Noam Shazeer</name></author>
103
+ <!-- more authors -->
104
+ <category term="cs.CL" scheme="http://arxiv.org/schemas/atom"/>
105
+ <arxiv:primary_category term="cs.CL"/>
106
+ <link rel="alternate" href="http://arxiv.org/abs/1706.03762v7"/>
107
+ <link rel="related" type="application/pdf" href="http://arxiv.org/pdf/1706.03762v7"/>
108
+ <arxiv:doi>10.48550/arXiv.1706.03762</arxiv:doi>
109
+ <arxiv:comment>15 pages, 5 figures</arxiv:comment>
110
+ <arxiv:journal_ref>Advances in Neural Information Processing Systems 30 (NIPS 2017)</arxiv:journal_ref>
111
+ </entry>
112
+ </feed>
113
+ ```
114
+
115
+ ### Key XML elements per entry
116
+
117
+ | Element | Description |
118
+ |---------|-------------|
119
+ | `<id>` | arXiv URL: `http://arxiv.org/abs/{id}` |
120
+ | `<title>` | Paper title |
121
+ | `<summary>` | Abstract |
122
+ | `<published>` | Original submission date (ISO 8601) |
123
+ | `<updated>` | Date of latest version |
124
+ | `<author><name>` | One per author |
125
+ | `<category term="...">` | Subject categories |
126
+ | `<arxiv:primary_category>` | Primary classification |
127
+ | `<link rel="alternate">` | Abstract page URL |
128
+ | `<link rel="related" title="pdf">` | PDF URL |
129
+ | `<arxiv:doi>` | DOI (when available) |
130
+ | `<arxiv:comment>` | Author comments |
131
+ | `<arxiv:journal_ref>` | Journal reference |
132
+
133
+ ## Parsing Tips
134
+
135
+ Since arXiv returns XML, you'll need to parse it. With `curl`, you can pipe the output and extract what you need. The XML namespace is `http://www.w3.org/2005/Atom` with arXiv extensions in `http://arxiv.org/schemas/atom`.
136
+
137
+ For practical extraction, the key data is in `<entry>` elements. Each entry's `<id>` contains the arXiv ID in the URL path.
138
+
139
+ ## Common Categories
140
+
141
+ | Category | Field |
142
+ |----------|-------|
143
+ | `cs.AI` | Artificial Intelligence |
144
+ | `cs.CL` | Computation and Language (NLP) |
145
+ | `cs.CV` | Computer Vision |
146
+ | `cs.LG` | Machine Learning |
147
+ | `stat.ML` | Machine Learning (Statistics) |
148
+ | `q-bio` | Quantitative Biology |
149
+ | `physics` | Physics (all subcategories) |
150
+ | `math` | Mathematics (all subcategories) |
151
+ | `econ` | Economics |
152
+ | `eess` | Electrical Engineering and Systems Science |
153
+
154
+ Full list: https://arxiv.org/category_taxonomy
155
+
156
+ ## Rate Limits
157
+
158
+ - **1 request every 3 seconds** (hard limit)
159
+ - Single connection at a time
160
+ - Search results are cached daily -- same query won't show new results within 24 hours
161
+ - For bulk data, use the OAI-PMH interface instead
@@ -0,0 +1,118 @@
1
+ # bioRxiv API
2
+
3
+ bioRxiv is a preprint server for biology. The API provides metadata for preprints, including title, authors, abstract, DOI, and publication status.
4
+
5
+ **Important:** The bioRxiv API has **no keyword search**. It supports date-range browsing and DOI lookup only. For keyword search of bioRxiv preprints, use Semantic Scholar, OpenAlex, or CORE instead.
6
+
7
+ ## Base URL
8
+
9
+ ```
10
+ https://api.biorxiv.org
11
+ ```
12
+
13
+ ## Authentication
14
+
15
+ None required. Fully public API.
16
+
17
+ ## Key Endpoints
18
+
19
+ ### 1. Content Detail -- Browse by date range
20
+
21
+ ```
22
+ GET /details/biorxiv/{interval}/{cursor}/{format}
23
+ ```
24
+
25
+ | Parameter | Values | Description |
26
+ |-----------|--------|-------------|
27
+ | `interval` | `YYYY-MM-DD/YYYY-MM-DD` | Date range (inclusive). Keep ranges narrow (1-3 days) to avoid timeouts. |
28
+ | | `N` (integer) | N most recent preprints |
29
+ | | `Nd` (integer + "d") | Last N days |
30
+ | `cursor` | Integer (default `0`) | Pagination offset (100 results per page) |
31
+ | `format` | `json` (default), `xml` | Response format |
32
+
33
+ Optional query parameter: `?category=neuroscience` (filter by category, use underscores for spaces)
34
+
35
+ **Examples:**
36
+ ```
37
+ https://api.biorxiv.org/details/biorxiv/2024-01-01/2024-01-31/0
38
+ https://api.biorxiv.org/details/biorxiv/5
39
+ https://api.biorxiv.org/details/biorxiv/10d
40
+ https://api.biorxiv.org/details/biorxiv/2024-01-01/2024-01-31?category=neuroscience
41
+ ```
42
+
43
+ ### 2. Content Detail -- DOI lookup
44
+
45
+ ```
46
+ GET /details/biorxiv/{doi}/na/{format}
47
+ ```
48
+
49
+ **Example:**
50
+ ```
51
+ https://api.biorxiv.org/details/biorxiv/10.1101/2024.01.16.575895/na/json
52
+ ```
53
+
54
+ ### 3. Published Article Links
55
+
56
+ ```
57
+ GET /pubs/biorxiv/{interval}/{cursor}
58
+ GET /pubs/biorxiv/{doi}/na
59
+ ```
60
+
61
+ Links preprints to their published journal versions. Accepts both preprint DOI and published DOI.
62
+
63
+ ### 4. Publisher Filter
64
+
65
+ ```
66
+ GET /publisher/{prefix}/{interval}/{cursor}
67
+ ```
68
+
69
+ Find bioRxiv papers published by a specific publisher (by DOI prefix).
70
+
71
+ **Example:**
72
+ ```
73
+ https://api.biorxiv.org/publisher/10.15252/2024-01-01/2024-06-01/0
74
+ ```
75
+
76
+ ## Response Format
77
+
78
+ ```json
79
+ {
80
+ "messages": [{
81
+ "status": "ok",
82
+ "count": 100,
83
+ "total": "1029",
84
+ "cursor": 0
85
+ }],
86
+ "collection": [{
87
+ "title": "Paper title...",
88
+ "authors": "Surname, A.; Surname, B.",
89
+ "author_corresponding": "Full Name",
90
+ "author_corresponding_institution": "Institution",
91
+ "doi": "10.1101/2024.01.16.575895",
92
+ "date": "2024-01-20",
93
+ "version": "1",
94
+ "type": "new results",
95
+ "license": "cc_no",
96
+ "category": "cancer biology",
97
+ "jatsxml": "https://www.biorxiv.org/content/early/.../source.xml",
98
+ "abstract": "Full abstract text...",
99
+ "published": "10.1158/2159-8290.CD-24-0187",
100
+ "server": "bioRxiv"
101
+ }]
102
+ }
103
+ ```
104
+
105
+ - `published` is `"NA"` if not yet published in a journal, or the published DOI if it has been.
106
+ - `type` values: `new results`, `confirmatory results`, `contradictory results`
107
+
108
+ ## Pagination
109
+
110
+ All multi-result endpoints return **100 results per page**. Use `cursor` to paginate. The `messages` object tells you the `total` count.
111
+
112
+ ## Rate Limits
113
+
114
+ No documented rate limits. No authentication required. Be reasonable with request frequency.
115
+
116
+ ## Categories
117
+
118
+ `animal-behavior-and-cognition`, `biochemistry`, `bioengineering`, `bioinformatics`, `biophysics`, `cancer-biology`, `cell-biology`, `clinical-trials`, `developmental-biology`, `ecology`, `epidemiology`, `evolutionary-biology`, `genetics`, `genomics`, `immunology`, `microbiology`, `molecular-biology`, `neuroscience`, `paleontology`, `pathology`, `pharmacology-and-toxicology`, `physiology`, `plant-biology`, `scientific-communication-and-education`, `synthetic-biology`, `systems-biology`, `zoology`
@@ -0,0 +1,150 @@
1
+ # CORE API
2
+
3
+ CORE aggregates open access research from 15,000+ repositories worldwide. It provides **full text** for 37M+ articles and metadata for 368M+ papers.
4
+
5
+ ## Base URL
6
+
7
+ ```
8
+ https://api.core.ac.uk/v3
9
+ ```
10
+
11
+ **Important:** GET search paths require a **trailing slash** (e.g., `/v3/search/works/` not `/v3/search/works`).
12
+
13
+ ## Authentication
14
+
15
+ - **Header:** `Authorization: Bearer YOUR_API_KEY`
16
+ - **Query param:** `?api_key=YOUR_API_KEY`
17
+ - Register at: https://core.ac.uk/services/api
18
+
19
+ **Without auth:** Basic metadata queries work, but full text is NOT available (returns "Not available for public API users").
20
+
21
+ ## Rate Limits (token-based)
22
+
23
+ | User Type | Daily Tokens | Per-Minute Max |
24
+ |-----------|-------------|----------------|
25
+ | Unauthenticated | 100/day | 10/min |
26
+ | Registered Personal | 1,000/day | 25/min |
27
+ | Registered Academic | 5,000/day | 10/min |
28
+
29
+ Simple queries cost 1 token. Downloads and scroll pagination cost 3-5 tokens.
30
+
31
+ ## Key Endpoints
32
+
33
+ ### 1. Search works
34
+
35
+ ```
36
+ GET /v3/search/works/?q={query}&limit={n}&offset={n}
37
+ ```
38
+
39
+ | Parameter | Default | Description |
40
+ |-----------|---------|-------------|
41
+ | `q` | required | Search query (supports field lookups, boolean operators) |
42
+ | `limit` | 10 | Results per page (max 100) |
43
+ | `offset` | 0 | Pagination offset |
44
+ | `scroll` | false | Enable scroll pagination for >10,000 results |
45
+ | `sort` | relevance | `relevance` or `recency` |
46
+
47
+ **POST alternative** (for complex queries):
48
+ ```
49
+ POST /v3/search/works
50
+ Content-Type: application/json
51
+
52
+ {"q": "machine learning", "limit": 10, "offset": 0}
53
+ ```
54
+
55
+ **Example:**
56
+ ```
57
+ https://api.core.ac.uk/v3/search/works/?q=CRISPR+gene+therapy&limit=10
58
+ ```
59
+
60
+ ### 2. Query language
61
+
62
+ | Operator | Example | Description |
63
+ |----------|---------|-------------|
64
+ | AND | `title:"AI" AND authors:"Smith"` | Both conditions |
65
+ | OR | `title:"AI" OR fullText:"Deep Learning"` | Either condition |
66
+ | Grouping | `(title:"AI" OR title:"ML") AND yearPublished>"2020"` | Precedence |
67
+ | Field lookup | `title:"Machine Learning"` | Search specific field |
68
+ | Range | `yearPublished>2018` | Numeric comparison |
69
+ | Exists | `_exists_:fullText` | Field must exist |
70
+ | Phrase | `title:"Attention is all you need"` | Exact phrase |
71
+
72
+ **Searchable fields:** `abstract`, `arxivId`, `authors`, `contributors`, `createdDate`, `dataProviders`, `depositedDate`, `documentType`, `doi`, `fullText`, `id`, `language`, `license`, `oai`, `title`, `yearPublished`
73
+
74
+ ### 3. Get work by ID
75
+
76
+ ```
77
+ GET /v3/works/{id}
78
+ ```
79
+
80
+ `id` is a CORE Work ID (integer). Example: `/v3/works/267312`
81
+
82
+ ### 4. Get output by ID
83
+
84
+ ```
85
+ GET /v3/outputs/{id}
86
+ ```
87
+
88
+ ### 5. Download full text
89
+
90
+ ```
91
+ GET /v3/outputs/{id}/download
92
+ ```
93
+
94
+ Returns binary PDF. Requires authentication.
95
+
96
+ ```
97
+ GET /v3/works/tei/{id}
98
+ ```
99
+
100
+ Returns TEI XML format.
101
+
102
+ ### 6. Search outputs
103
+
104
+ ```
105
+ GET /v3/search/outputs/?q={query}&limit={n}&offset={n}
106
+ ```
107
+
108
+ Search by DOI: `q=doi:10.1038/nature12373`
109
+
110
+ ## Response Format
111
+
112
+ ### Search response
113
+ ```json
114
+ {
115
+ "totalHits": 2281337,
116
+ "limit": 10,
117
+ "offset": 0,
118
+ "scrollId": null,
119
+ "results": [...]
120
+ }
121
+ ```
122
+
123
+ ### Work object (key fields)
124
+ ```json
125
+ {
126
+ "id": 8848131,
127
+ "title": "Attention Is All You Need",
128
+ "authors": [{"name": "Ashish Vaswani"}, ...],
129
+ "abstract": "The dominant sequence...",
130
+ "doi": "10.48550/arXiv.1706.03762",
131
+ "arxivId": "1706.03762",
132
+ "yearPublished": 2017,
133
+ "downloadUrl": "https://core.ac.uk/download/...",
134
+ "fullText": "Full text content (when authenticated)...",
135
+ "language": {"code": "en", "name": "English"},
136
+ "documentType": "research",
137
+ "citationCount": 145678,
138
+ "dataProviders": [{"name": "arXiv"}],
139
+ "links": [{"type": "download", "url": "..."}]
140
+ }
141
+ ```
142
+
143
+ ## Pagination
144
+
145
+ - **Standard:** `offset` + `limit` (max 10,000 results)
146
+ - **Scroll:** Set `scroll=true`. Response includes `scrollId`. Use in subsequent requests to page beyond 10,000 (costs more tokens).
147
+
148
+ ## Error Handling
149
+
150
+ Under heavy load, the API may return partial shard failure messages. These are transient -- retry after a brief wait.
@@ -0,0 +1,181 @@
1
+ # Crossref API
2
+
3
+ Crossref is the DOI registration agency for scholarly content. It provides metadata for 150M+ works including journal articles, books, conference papers, datasets, and preprints.
4
+
5
+ ## Base URL
6
+
7
+ ```
8
+ https://api.crossref.org
9
+ ```
10
+
11
+ ## Authentication
12
+
13
+ None required. Add `mailto=you@example.com` to get into the **polite pool** (2x faster rate limits).
14
+
15
+ ## Rate Limits
16
+
17
+ | Pool | Rate | Concurrency |
18
+ |------|------|-------------|
19
+ | Public (no mailto) | 5 req/sec | 1 concurrent |
20
+ | Polite (with mailto) | 10 req/sec | 3 concurrent |
21
+
22
+ HTTP 429 = temporarily blocked.
23
+
24
+ ## Key Endpoints
25
+
26
+ ### 1. Search works
27
+
28
+ ```
29
+ GET /works?query={text}&rows={n}&mailto=you@example.com
30
+ ```
31
+
32
+ | Parameter | Default | Description |
33
+ |-----------|---------|-------------|
34
+ | `query` | -- | Free-text search across all fields |
35
+ | `query.author` | -- | Search author names |
36
+ | `query.bibliographic` | -- | Search titles, authors, ISSNs, years |
37
+ | `query.affiliation` | -- | Search affiliations |
38
+ | `query.container-title` | -- | Search journal names |
39
+ | `filter` | -- | Comma-separated `name:value` pairs |
40
+ | `sort` | `score` | `score`, `published`, `issued`, `deposited`, `updated`, `is-referenced-by-count`, `references-count` |
41
+ | `order` | `desc` | `asc` or `desc` |
42
+ | `rows` | 20 | Results per page (max 1000) |
43
+ | `offset` | 0 | Skip N results (max 10,000) |
44
+ | `cursor` | -- | Use `*` for cursor-based deep pagination |
45
+ | `select` | -- | Comma-separated field names to return |
46
+ | `facet` | -- | Facet counts, e.g. `type-name:10` |
47
+ | `sample` | -- | Return N random items (max 100) |
48
+
49
+ **Example:**
50
+ ```
51
+ https://api.crossref.org/works?query=CRISPR+gene+therapy&filter=from-pub-date:2024-01-01,type:journal-article,has-abstract:true&rows=5&sort=published&order=desc&mailto=you@example.com
52
+ ```
53
+
54
+ ### 2. Get work by DOI
55
+
56
+ ```
57
+ GET /works/{doi}?mailto=you@example.com
58
+ ```
59
+
60
+ URL-encode the DOI: `10.1038/nature12373` becomes `10.1038%2Fnature12373`
61
+
62
+ **Example:**
63
+ ```
64
+ https://api.crossref.org/works/10.1038%2Fnature12373?mailto=you@example.com
65
+ ```
66
+
67
+ ### 3. Journals
68
+
69
+ ```
70
+ GET /journals?query={name}&rows={n}
71
+ GET /journals/{issn}
72
+ GET /journals/{issn}/works?query={text}&rows={n}
73
+ ```
74
+
75
+ ### 4. Funders
76
+
77
+ ```
78
+ GET /funders?query={name}
79
+ GET /funders/{id}
80
+ GET /funders/{id}/works?rows={n}
81
+ ```
82
+
83
+ Funder IDs are from the Funder Registry (e.g., `100000001` for NSF).
84
+
85
+ ### 5. Members (publishers)
86
+
87
+ ```
88
+ GET /members?query={name}
89
+ GET /members/{id}/works?rows={n}
90
+ ```
91
+
92
+ ## Key Filters
93
+
94
+ ### Date filters (accept `YYYY`, `YYYY-MM`, `YYYY-MM-DD`)
95
+ | Filter | Description |
96
+ |--------|-------------|
97
+ | `from-pub-date` / `until-pub-date` | Publication date |
98
+ | `from-print-pub-date` / `until-print-pub-date` | Print publication date |
99
+ | `from-online-pub-date` / `until-online-pub-date` | Online publication date |
100
+ | `from-posted-date` / `until-posted-date` | Posted date (preprints) |
101
+
102
+ ### Boolean filters
103
+ | Filter | Description |
104
+ |--------|-------------|
105
+ | `has-abstract` | Has an abstract |
106
+ | `has-orcid` | Has ORCID IDs |
107
+ | `has-funder` | Has funder info |
108
+ | `has-full-text` | Has full-text links |
109
+ | `has-references` | Has reference list |
110
+ | `has-license` | Has license info |
111
+
112
+ ### Value filters
113
+ | Filter | Description |
114
+ |--------|-------------|
115
+ | `type` | `journal-article`, `posted-content`, `book-chapter`, `proceedings-article`, etc. |
116
+ | `issn` | Journal ISSN |
117
+ | `doi` | Specific DOI |
118
+ | `orcid` | Contributor ORCID |
119
+ | `funder` | Funder Registry ID |
120
+ | `member` | Crossref member ID |
121
+ | `prefix` | DOI prefix |
122
+ | `license.url` | License URL |
123
+ | `update-type` | `correction`, `retraction` |
124
+
125
+ **Syntax:** `filter=name1:value1,name2:value2`
126
+
127
+ ## Pagination
128
+
129
+ ### Offset-based (max 10,000)
130
+ ```
131
+ /works?query=cancer&rows=100&offset=200
132
+ ```
133
+
134
+ ### Cursor-based (unlimited)
135
+ 1. First request: `?cursor=*&rows=100`
136
+ 2. Response includes `next-cursor`
137
+ 3. Next request: `?cursor={next-cursor-value}&rows=100`
138
+ 4. Cursors expire after 5 minutes
139
+
140
+ ## Response Format
141
+
142
+ ### List response
143
+ ```json
144
+ {
145
+ "status": "ok",
146
+ "message-type": "work-list",
147
+ "message": {
148
+ "total-results": 2779116,
149
+ "items-per-page": 20,
150
+ "next-cursor": "...",
151
+ "items": [...]
152
+ }
153
+ }
154
+ ```
155
+
156
+ ### Work object (key fields)
157
+ ```json
158
+ {
159
+ "DOI": "10.1038/nature12373",
160
+ "title": ["Nanometre-scale thermometry in a living cell"],
161
+ "author": [{"given": "G.", "family": "Kucsko", "sequence": "first"}],
162
+ "publisher": "Springer Science and Business Media LLC",
163
+ "type": "journal-article",
164
+ "published": {"date-parts": [[2013, 7, 31]]},
165
+ "container-title": ["Nature"],
166
+ "ISSN": ["0028-0836", "1476-4687"],
167
+ "volume": "500",
168
+ "issue": "7460",
169
+ "page": "54-58",
170
+ "is-referenced-by-count": 1745,
171
+ "references-count": 30,
172
+ "abstract": "<p>Abstract text with HTML tags...</p>",
173
+ "license": [{"URL": "...", "content-version": "vor"}],
174
+ "link": [{"URL": "...", "content-type": "application/pdf"}],
175
+ "reference": [{"key": "...", "doi-asserted-by": "crossref", "DOI": "..."}],
176
+ "subject": ["Multidisciplinary"],
177
+ "language": "en"
178
+ }
179
+ ```
180
+
181
+ Note: `title` and `container-title` are arrays. `published.date-parts` is `[[year, month, day]]`. Abstract may contain HTML tags.