clearai-dsh 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +26 -0
- package/LICENSE +201 -0
- package/README.md +138 -0
- package/README.zh-CN.md +138 -0
- package/bin/clearai.mjs +224 -0
- package/brand/README.md +41 -0
- package/brand/logo-512-dark.png +0 -0
- package/brand/logo-512.png +0 -0
- package/brand/logo-lockup-dark.png +0 -0
- package/brand/logo-lockup.png +0 -0
- package/brand/logo-lockup.svg +12 -0
- package/brand/logo-wordmark.svg +6 -0
- package/brand/logo.svg +19 -0
- package/cordis.patch.yml +39 -0
- package/lib/client.js +3071 -0
- package/lib/fold.js +1576 -0
- package/lib/host.js +605 -0
- package/package.json +65 -0
- package/presets/clearai/agent.cordis.yml +226 -0
- package/presets/clearai/plugins/brain.js +547 -0
- package/presets/clearai/plugins/clearai-kernel.js +5485 -0
- package/presets/clearai/plugins/ontology.js +306 -0
- package/presets/clearai/plugins/prompts.js +312 -0
- package/presets/clearai/preset.yml +5 -0
- package/presets/clearai/skills/clearai-loop/SKILL.md +89 -0
- package/presets/clearai/template/knowledge/README.md +25 -0
- package/presets/clearai/template/memory/README.md +34 -0
- package/presets/clearai/template/project.md +49 -0
- package/presets/clearai/template/skills/README.md +37 -0
- package/presets/clearai/template/skills/chart-diagram-qa/SKILL.md +43 -0
- package/presets/clearai/template/skills/citation-management/SKILL.md +73 -0
- package/presets/clearai/template/skills/citation-management/references/bibtex_formatting.md +908 -0
- package/presets/clearai/template/skills/citation-management/references/citation_validation.md +794 -0
- package/presets/clearai/template/skills/citation-management/references/google_scholar_search.md +725 -0
- package/presets/clearai/template/skills/citation-management/references/metadata_extraction.md +870 -0
- package/presets/clearai/template/skills/citation-management/references/pubmed_search.md +839 -0
- package/presets/clearai/template/skills/citation-management/scripts/doi_to_bibtex.py +204 -0
- package/presets/clearai/template/skills/citation-management/scripts/extract_metadata.py +569 -0
- package/presets/clearai/template/skills/citation-management/scripts/format_bibtex.py +349 -0
- package/presets/clearai/template/skills/citation-management/scripts/generate_schematic.py +139 -0
- package/presets/clearai/template/skills/citation-management/scripts/generate_schematic_ai.py +817 -0
- package/presets/clearai/template/skills/citation-management/scripts/search_google_scholar.py +282 -0
- package/presets/clearai/template/skills/citation-management/scripts/search_pubmed.py +398 -0
- package/presets/clearai/template/skills/citation-management/scripts/validate_citations.py +497 -0
- package/presets/clearai/template/skills/data-analysis/SKILL.md +92 -0
- package/presets/clearai/template/skills/data-analysis/checklists/readiness_check.md +23 -0
- package/presets/clearai/template/skills/data-analysis/templates/analysis_report.md.tpl +63 -0
- package/presets/clearai/template/skills/data-analysis/templates/cleaning_rules_draft.yaml.tpl +32 -0
- package/presets/clearai/template/skills/data-analysis/templates/data_dictionary.md.tpl +12 -0
- package/presets/clearai/template/skills/data-analysis/templates/domain_knowledge_template.md.tpl +316 -0
- package/presets/clearai/template/skills/data-analysis/templates/feature_candidates.json.tpl +20 -0
- package/presets/clearai/template/skills/data-analysis/templates/quality_scorecard.md.tpl +30 -0
- package/presets/clearai/template/skills/data-analysis/workflows/01-data-profiling.md +42 -0
- package/presets/clearai/template/skills/data-analysis/workflows/02-quality-audit.md +36 -0
- package/presets/clearai/template/skills/data-analysis/workflows/03-physical-correlation.md +25 -0
- package/presets/clearai/template/skills/data-analysis/workflows/04-unstructured-mining.md +26 -0
- package/presets/clearai/template/skills/data-qa-analysis/SKILL.md +102 -0
- package/presets/clearai/template/skills/data-qa-analysis/checklists/readiness_check.md +62 -0
- package/presets/clearai/template/skills/data-qa-analysis/templates/best_in_class_report.md.tpl +56 -0
- package/presets/clearai/template/skills/data-qa-analysis/templates/cleaning_rules_draft.yaml.tpl +56 -0
- package/presets/clearai/template/skills/data-qa-analysis/templates/data_dictionary.md.tpl +13 -0
- package/presets/clearai/template/skills/data-qa-analysis/templates/data_source_inventory_and_lineage.md.tpl +146 -0
- package/presets/clearai/template/skills/data-qa-analysis/templates/data_status_report.md.tpl +60 -0
- package/presets/clearai/template/skills/data-qa-analysis/templates/steady_state_rules.yaml.tpl +41 -0
- package/presets/clearai/template/skills/data-qa-analysis/templates/subsystem_registry.md.tpl +101 -0
- package/presets/clearai/template/skills/data-qa-analysis/templates/unified_execution_plan.md.tpl +100 -0
- package/presets/clearai/template/skills/data-qa-analysis/workflows/01-data-source-inventory-and-lineage.md +194 -0
- package/presets/clearai/template/skills/data-qa-analysis/workflows/02-data-alignment-and-tag-semantics.md +122 -0
- package/presets/clearai/template/skills/data-qa-analysis/workflows/03-steady-state-identification.md +126 -0
- package/presets/clearai/template/skills/data-qa-analysis/workflows/04-consumption-analysis.md +152 -0
- package/presets/clearai/template/skills/data-qa-analysis/workflows/05-best-in-class-and-optimization-space.md +78 -0
- package/presets/clearai/template/skills/domain-presearch/SKILL.md +131 -0
- package/presets/clearai/template/skills/domain-presearch/checklists/domain_checklist.md +24 -0
- package/presets/clearai/template/skills/domain-presearch/references/figure_code.md +78 -0
- package/presets/clearai/template/skills/domain-presearch/references/strategic_frameworks.md +38 -0
- package/presets/clearai/template/skills/exploration-loop/SKILL.md +81 -0
- package/presets/clearai/template/skills/exploratory-data-analysis/SKILL.md +77 -0
- package/presets/clearai/template/skills/exploratory-data-analysis/references/bioinformatics_genomics_formats.md +664 -0
- package/presets/clearai/template/skills/exploratory-data-analysis/references/chemistry_molecular_formats.md +664 -0
- package/presets/clearai/template/skills/exploratory-data-analysis/references/general_scientific_formats.md +518 -0
- package/presets/clearai/template/skills/exploratory-data-analysis/references/microscopy_imaging_formats.md +620 -0
- package/presets/clearai/template/skills/exploratory-data-analysis/references/proteomics_metabolomics_formats.md +517 -0
- package/presets/clearai/template/skills/exploratory-data-analysis/references/spectroscopy_analytical_formats.md +633 -0
- package/presets/clearai/template/skills/exploratory-data-analysis/scripts/eda_analyzer.py +547 -0
- package/presets/clearai/template/skills/hypothesis-generation/SKILL.md +73 -0
- package/presets/clearai/template/skills/hypothesis-generation/references/experimental_design_patterns.md +329 -0
- package/presets/clearai/template/skills/hypothesis-generation/references/hypothesis_quality_criteria.md +198 -0
- package/presets/clearai/template/skills/hypothesis-generation/references/literature_search_strategies.md +622 -0
- package/presets/clearai/template/skills/hypothesis-generation/scripts/generate_schematic.py +139 -0
- package/presets/clearai/template/skills/hypothesis-generation/scripts/generate_schematic_ai.py +817 -0
- package/presets/clearai/template/skills/literature-review/SKILL.md +72 -0
- package/presets/clearai/template/skills/literature-review/references/citation_styles.md +166 -0
- package/presets/clearai/template/skills/literature-review/references/database_strategies.md +455 -0
- package/presets/clearai/template/skills/literature-review/scripts/generate_pdf.py +176 -0
- package/presets/clearai/template/skills/literature-review/scripts/generate_schematic.py +139 -0
- package/presets/clearai/template/skills/literature-review/scripts/generate_schematic_ai.py +817 -0
- package/presets/clearai/template/skills/literature-review/scripts/search_databases.py +303 -0
- package/presets/clearai/template/skills/literature-review/scripts/verify_citations.py +221 -0
- package/presets/clearai/template/skills/paper-lookup/SKILL.md +59 -0
- package/presets/clearai/template/skills/paper-lookup/references/arxiv.md +161 -0
- package/presets/clearai/template/skills/paper-lookup/references/biorxiv.md +118 -0
- package/presets/clearai/template/skills/paper-lookup/references/core.md +150 -0
- package/presets/clearai/template/skills/paper-lookup/references/crossref.md +181 -0
- package/presets/clearai/template/skills/paper-lookup/references/medrxiv.md +104 -0
- package/presets/clearai/template/skills/paper-lookup/references/openalex.md +174 -0
- package/presets/clearai/template/skills/paper-lookup/references/pmc.md +152 -0
- package/presets/clearai/template/skills/paper-lookup/references/pubmed.md +124 -0
- package/presets/clearai/template/skills/paper-lookup/references/semantic-scholar.md +203 -0
- package/presets/clearai/template/skills/paper-lookup/references/unpaywall.md +127 -0
- package/presets/clearai/template/skills/process-presearch/SKILL.md +196 -0
- package/presets/clearai/template/skills/process-presearch/checklists/process_checklist.md +18 -0
- package/presets/clearai/template/skills/process-presearch/references/figure_code.md +107 -0
- package/presets/clearai/template/skills/process-presearch/references/source_attribution_example.md +22 -0
- package/presets/clearai/template/skills/process-understanding-extraction/SKILL.md +69 -0
- package/presets/clearai/template/skills/process-understanding-extraction/checklists/readiness_check.md +34 -0
- package/presets/clearai/template/skills/process-understanding-extraction/templates/docx_raw_dump_extractor.py.tpl +132 -0
- package/presets/clearai/template/skills/process-understanding-extraction/templates/entity_map_unit_topology.json.tpl +86 -0
- package/presets/clearai/template/skills/process-understanding-extraction/templates/process_brief.md.tpl +89 -0
- package/presets/clearai/template/skills/process-understanding-extraction/templates/process_brief_builder_from_raw_dump.py.tpl +203 -0
- package/presets/clearai/template/skills/process-understanding-extraction/templates/process_flow_mermaid.md.tpl +41 -0
- package/presets/clearai/template/skills/process-understanding-extraction/templates/unified_execution_plan.md.tpl +53 -0
- package/presets/clearai/template/skills/process-understanding-extraction/workflows/01-process-doc-discovery.md +173 -0
- package/presets/clearai/template/skills/process-understanding-extraction/workflows/02-process-understanding-and-diagramming.md +106 -0
- package/presets/clearai/template/skills/scientific-brainstorming/SKILL.md +64 -0
- package/presets/clearai/template/skills/scientific-brainstorming/references/brainstorming_methods.md +326 -0
- package/presets/clearai/template/skills/scientific-critical-thinking/SKILL.md +72 -0
- package/presets/clearai/template/skills/scientific-critical-thinking/references/common_biases.md +364 -0
- package/presets/clearai/template/skills/scientific-critical-thinking/references/evidence_hierarchy.md +485 -0
- package/presets/clearai/template/skills/scientific-critical-thinking/references/experimental_design.md +496 -0
- package/presets/clearai/template/skills/scientific-critical-thinking/references/logical_fallacies.md +478 -0
- package/presets/clearai/template/skills/scientific-critical-thinking/references/scientific_method.md +169 -0
- package/presets/clearai/template/skills/scientific-critical-thinking/references/statistical_pitfalls.md +506 -0
- package/presets/clearai/template/skills/skill-creator/SKILL.md +109 -0
- package/presets/clearai/template/skills/skill-creator/references/authoring-guide.md +89 -0
- package/presets/clearai/template/skills/statistical-analysis/SKILL.md +79 -0
- package/presets/clearai/template/skills/statistical-analysis/references/assumptions_and_diagnostics.md +369 -0
- package/presets/clearai/template/skills/statistical-analysis/references/bayesian_statistics.md +653 -0
- package/presets/clearai/template/skills/statistical-analysis/references/effect_sizes_and_power.md +578 -0
- package/presets/clearai/template/skills/statistical-analysis/references/reporting_standards.md +469 -0
- package/presets/clearai/template/skills/statistical-analysis/references/test_selection_guide.md +129 -0
- package/presets/clearai/template/skills/statistical-analysis/scripts/assumption_checks.py +538 -0
- package/presets/clearai/template/skills/web-artifact/SKILL.md +165 -0
- package/presets/clearai/template/skills/web-artifact/assets/renderer/renderer.css +229 -0
- package/presets/clearai/template/skills/web-artifact/assets/renderer/renderer.js +373 -0
- package/presets/clearai/template/skills/web-artifact/assets/vendor/elkjs/LICENSE +263 -0
- package/presets/clearai/template/skills/web-artifact/assets/vendor/elkjs/UPSTREAM.md +26 -0
- package/presets/clearai/template/skills/web-artifact/assets/vendor/elkjs/elk.bundled.js +6605 -0
- package/presets/clearai/template/skills/web-artifact/references/when-drawing-a-topology.md +150 -0
- package/presets/clearai/template/skills/web-artifact/references/when-the-page-must-work-offline.md +62 -0
- package/presets/clearai/template/skills/web-artifact/scripts/check_artifact.py +167 -0
- package/presets/clearai/template/skills/web-artifact/scripts/render_topology.js +272 -0
- package/presets/clearai/template/skills/what-if-oracle/LICENSE.txt +5 -0
- package/presets/clearai/template/skills/what-if-oracle/SKILL.md +72 -0
- package/presets/clearai/template/skills/what-if-oracle/references/scenario-templates.md +154 -0
|
@@ -0,0 +1,282 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
Google Scholar Search Tool
|
|
4
|
+
Search Google Scholar and export results.
|
|
5
|
+
|
|
6
|
+
Note: This script requires the 'scholarly' library.
|
|
7
|
+
Install with: pip install scholarly
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import sys
|
|
11
|
+
import argparse
|
|
12
|
+
import json
|
|
13
|
+
import time
|
|
14
|
+
import random
|
|
15
|
+
from typing import List, Dict, Optional
|
|
16
|
+
|
|
17
|
+
try:
|
|
18
|
+
from scholarly import scholarly, ProxyGenerator
|
|
19
|
+
SCHOLARLY_AVAILABLE = True
|
|
20
|
+
except ImportError:
|
|
21
|
+
SCHOLARLY_AVAILABLE = False
|
|
22
|
+
print('Warning: scholarly library not installed. Install with: pip install scholarly', file=sys.stderr)
|
|
23
|
+
|
|
24
|
+
class GoogleScholarSearcher:
|
|
25
|
+
"""Search Google Scholar using scholarly library."""
|
|
26
|
+
|
|
27
|
+
def __init__(self, use_proxy: bool = False):
|
|
28
|
+
"""
|
|
29
|
+
Initialize searcher.
|
|
30
|
+
|
|
31
|
+
Args:
|
|
32
|
+
use_proxy: Use free proxy (helps avoid rate limiting)
|
|
33
|
+
"""
|
|
34
|
+
if not SCHOLARLY_AVAILABLE:
|
|
35
|
+
raise ImportError('scholarly library required. Install with: pip install scholarly')
|
|
36
|
+
|
|
37
|
+
# Setup proxy if requested
|
|
38
|
+
if use_proxy:
|
|
39
|
+
try:
|
|
40
|
+
pg = ProxyGenerator()
|
|
41
|
+
pg.FreeProxies()
|
|
42
|
+
scholarly.use_proxy(pg)
|
|
43
|
+
print('Using free proxy', file=sys.stderr)
|
|
44
|
+
except Exception as e:
|
|
45
|
+
print(f'Warning: Could not setup proxy: {e}', file=sys.stderr)
|
|
46
|
+
|
|
47
|
+
def search(self, query: str, max_results: int = 50,
|
|
48
|
+
year_start: Optional[int] = None, year_end: Optional[int] = None,
|
|
49
|
+
sort_by: str = 'relevance') -> List[Dict]:
|
|
50
|
+
"""
|
|
51
|
+
Search Google Scholar.
|
|
52
|
+
|
|
53
|
+
Args:
|
|
54
|
+
query: Search query
|
|
55
|
+
max_results: Maximum number of results
|
|
56
|
+
year_start: Start year filter
|
|
57
|
+
year_end: End year filter
|
|
58
|
+
sort_by: Sort order ('relevance' or 'citations')
|
|
59
|
+
|
|
60
|
+
Returns:
|
|
61
|
+
List of result dictionaries
|
|
62
|
+
"""
|
|
63
|
+
if not SCHOLARLY_AVAILABLE:
|
|
64
|
+
print('Error: scholarly library not installed', file=sys.stderr)
|
|
65
|
+
return []
|
|
66
|
+
|
|
67
|
+
print(f'Searching Google Scholar: {query}', file=sys.stderr)
|
|
68
|
+
print(f'Max results: {max_results}', file=sys.stderr)
|
|
69
|
+
|
|
70
|
+
results = []
|
|
71
|
+
|
|
72
|
+
try:
|
|
73
|
+
# Perform search
|
|
74
|
+
search_query = scholarly.search_pubs(query)
|
|
75
|
+
|
|
76
|
+
for i, result in enumerate(search_query):
|
|
77
|
+
if i >= max_results:
|
|
78
|
+
break
|
|
79
|
+
|
|
80
|
+
print(f'Retrieved {i+1}/{max_results}', file=sys.stderr)
|
|
81
|
+
|
|
82
|
+
# Extract metadata
|
|
83
|
+
metadata = {
|
|
84
|
+
'title': result.get('bib', {}).get('title', ''),
|
|
85
|
+
'authors': ', '.join(result.get('bib', {}).get('author', [])),
|
|
86
|
+
'year': result.get('bib', {}).get('pub_year', ''),
|
|
87
|
+
'venue': result.get('bib', {}).get('venue', ''),
|
|
88
|
+
'abstract': result.get('bib', {}).get('abstract', ''),
|
|
89
|
+
'citations': result.get('num_citations', 0),
|
|
90
|
+
'url': result.get('pub_url', ''),
|
|
91
|
+
'eprint_url': result.get('eprint_url', ''),
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
# Filter by year
|
|
95
|
+
if year_start or year_end:
|
|
96
|
+
try:
|
|
97
|
+
pub_year = int(metadata['year']) if metadata['year'] else 0
|
|
98
|
+
if year_start and pub_year < year_start:
|
|
99
|
+
continue
|
|
100
|
+
if year_end and pub_year > year_end:
|
|
101
|
+
continue
|
|
102
|
+
except ValueError:
|
|
103
|
+
pass
|
|
104
|
+
|
|
105
|
+
results.append(metadata)
|
|
106
|
+
|
|
107
|
+
# Rate limiting to avoid blocking
|
|
108
|
+
time.sleep(random.uniform(2, 5))
|
|
109
|
+
|
|
110
|
+
except Exception as e:
|
|
111
|
+
print(f'Error during search: {e}', file=sys.stderr)
|
|
112
|
+
|
|
113
|
+
# Sort if requested
|
|
114
|
+
if sort_by == 'citations' and results:
|
|
115
|
+
results.sort(key=lambda x: x.get('citations', 0), reverse=True)
|
|
116
|
+
|
|
117
|
+
return results
|
|
118
|
+
|
|
119
|
+
def metadata_to_bibtex(self, metadata: Dict) -> str:
|
|
120
|
+
"""Convert metadata to BibTeX format."""
|
|
121
|
+
# Generate citation key
|
|
122
|
+
if metadata.get('authors'):
|
|
123
|
+
first_author = metadata['authors'].split(',')[0].strip()
|
|
124
|
+
last_name = first_author.split()[-1] if first_author else 'Unknown'
|
|
125
|
+
else:
|
|
126
|
+
last_name = 'Unknown'
|
|
127
|
+
|
|
128
|
+
year = metadata.get('year', 'XXXX')
|
|
129
|
+
|
|
130
|
+
# Get keyword from title
|
|
131
|
+
import re
|
|
132
|
+
title = metadata.get('title', '')
|
|
133
|
+
words = re.findall(r'\b[a-zA-Z]{4,}\b', title)
|
|
134
|
+
keyword = words[0].lower() if words else 'paper'
|
|
135
|
+
|
|
136
|
+
citation_key = f'{last_name}{year}{keyword}'
|
|
137
|
+
|
|
138
|
+
# Determine entry type (guess based on venue)
|
|
139
|
+
venue = metadata.get('venue', '').lower()
|
|
140
|
+
if 'proceedings' in venue or 'conference' in venue:
|
|
141
|
+
entry_type = 'inproceedings'
|
|
142
|
+
venue_field = 'booktitle'
|
|
143
|
+
else:
|
|
144
|
+
entry_type = 'article'
|
|
145
|
+
venue_field = 'journal'
|
|
146
|
+
|
|
147
|
+
# Build BibTeX
|
|
148
|
+
lines = [f'@{entry_type}{{{citation_key},']
|
|
149
|
+
|
|
150
|
+
# Convert authors format
|
|
151
|
+
if metadata.get('authors'):
|
|
152
|
+
authors = metadata['authors'].replace(',', ' and')
|
|
153
|
+
lines.append(f' author = {{{authors}}},')
|
|
154
|
+
|
|
155
|
+
if metadata.get('title'):
|
|
156
|
+
lines.append(f' title = {{{metadata["title"]}}},')
|
|
157
|
+
|
|
158
|
+
if metadata.get('venue'):
|
|
159
|
+
lines.append(f' {venue_field} = {{{metadata["venue"]}}},')
|
|
160
|
+
|
|
161
|
+
if metadata.get('year'):
|
|
162
|
+
lines.append(f' year = {{{metadata["year"]}}},')
|
|
163
|
+
|
|
164
|
+
if metadata.get('url'):
|
|
165
|
+
lines.append(f' url = {{{metadata["url"]}}},')
|
|
166
|
+
|
|
167
|
+
if metadata.get('citations'):
|
|
168
|
+
lines.append(f' note = {{Cited by: {metadata["citations"]}}},')
|
|
169
|
+
|
|
170
|
+
# Remove trailing comma
|
|
171
|
+
if lines[-1].endswith(','):
|
|
172
|
+
lines[-1] = lines[-1][:-1]
|
|
173
|
+
|
|
174
|
+
lines.append('}')
|
|
175
|
+
|
|
176
|
+
return '\n'.join(lines)
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def main():
|
|
180
|
+
"""Command-line interface."""
|
|
181
|
+
parser = argparse.ArgumentParser(
|
|
182
|
+
description='Search Google Scholar (requires scholarly library)',
|
|
183
|
+
epilog='Example: python search_google_scholar.py "machine learning" --limit 50'
|
|
184
|
+
)
|
|
185
|
+
|
|
186
|
+
parser.add_argument(
|
|
187
|
+
'query',
|
|
188
|
+
help='Search query'
|
|
189
|
+
)
|
|
190
|
+
|
|
191
|
+
parser.add_argument(
|
|
192
|
+
'--limit',
|
|
193
|
+
type=int,
|
|
194
|
+
default=50,
|
|
195
|
+
help='Maximum number of results (default: 50)'
|
|
196
|
+
)
|
|
197
|
+
|
|
198
|
+
parser.add_argument(
|
|
199
|
+
'--year-start',
|
|
200
|
+
type=int,
|
|
201
|
+
help='Start year for filtering'
|
|
202
|
+
)
|
|
203
|
+
|
|
204
|
+
parser.add_argument(
|
|
205
|
+
'--year-end',
|
|
206
|
+
type=int,
|
|
207
|
+
help='End year for filtering'
|
|
208
|
+
)
|
|
209
|
+
|
|
210
|
+
parser.add_argument(
|
|
211
|
+
'--sort-by',
|
|
212
|
+
choices=['relevance', 'citations'],
|
|
213
|
+
default='relevance',
|
|
214
|
+
help='Sort order (default: relevance)'
|
|
215
|
+
)
|
|
216
|
+
|
|
217
|
+
parser.add_argument(
|
|
218
|
+
'--use-proxy',
|
|
219
|
+
action='store_true',
|
|
220
|
+
help='Use free proxy to avoid rate limiting'
|
|
221
|
+
)
|
|
222
|
+
|
|
223
|
+
parser.add_argument(
|
|
224
|
+
'-o', '--output',
|
|
225
|
+
help='Output file (default: stdout)'
|
|
226
|
+
)
|
|
227
|
+
|
|
228
|
+
parser.add_argument(
|
|
229
|
+
'--format',
|
|
230
|
+
choices=['json', 'bibtex'],
|
|
231
|
+
default='json',
|
|
232
|
+
help='Output format (default: json)'
|
|
233
|
+
)
|
|
234
|
+
|
|
235
|
+
args = parser.parse_args()
|
|
236
|
+
|
|
237
|
+
if not SCHOLARLY_AVAILABLE:
|
|
238
|
+
print('\nError: scholarly library not installed', file=sys.stderr)
|
|
239
|
+
print('Install with: pip install scholarly', file=sys.stderr)
|
|
240
|
+
print('\nAlternatively, use PubMed search for biomedical literature:', file=sys.stderr)
|
|
241
|
+
print(' python search_pubmed.py "your query"', file=sys.stderr)
|
|
242
|
+
sys.exit(1)
|
|
243
|
+
|
|
244
|
+
# Search
|
|
245
|
+
searcher = GoogleScholarSearcher(use_proxy=args.use_proxy)
|
|
246
|
+
results = searcher.search(
|
|
247
|
+
args.query,
|
|
248
|
+
max_results=args.limit,
|
|
249
|
+
year_start=args.year_start,
|
|
250
|
+
year_end=args.year_end,
|
|
251
|
+
sort_by=args.sort_by
|
|
252
|
+
)
|
|
253
|
+
|
|
254
|
+
if not results:
|
|
255
|
+
print('No results found', file=sys.stderr)
|
|
256
|
+
sys.exit(1)
|
|
257
|
+
|
|
258
|
+
# Format output
|
|
259
|
+
if args.format == 'json':
|
|
260
|
+
output = json.dumps({
|
|
261
|
+
'query': args.query,
|
|
262
|
+
'count': len(results),
|
|
263
|
+
'results': results
|
|
264
|
+
}, indent=2)
|
|
265
|
+
else: # bibtex
|
|
266
|
+
bibtex_entries = [searcher.metadata_to_bibtex(r) for r in results]
|
|
267
|
+
output = '\n\n'.join(bibtex_entries) + '\n'
|
|
268
|
+
|
|
269
|
+
# Write output
|
|
270
|
+
if args.output:
|
|
271
|
+
with open(args.output, 'w', encoding='utf-8') as f:
|
|
272
|
+
f.write(output)
|
|
273
|
+
print(f'Wrote {len(results)} results to {args.output}', file=sys.stderr)
|
|
274
|
+
else:
|
|
275
|
+
print(output)
|
|
276
|
+
|
|
277
|
+
print(f'\nRetrieved {len(results)} results', file=sys.stderr)
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
if __name__ == '__main__':
|
|
281
|
+
main()
|
|
282
|
+
|
|
@@ -0,0 +1,398 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
PubMed Search Tool
|
|
4
|
+
Search PubMed using E-utilities API and export results.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import sys
|
|
8
|
+
import os
|
|
9
|
+
import requests
|
|
10
|
+
import argparse
|
|
11
|
+
import json
|
|
12
|
+
import time
|
|
13
|
+
import xml.etree.ElementTree as ET
|
|
14
|
+
from typing import List, Dict, Optional
|
|
15
|
+
from datetime import datetime
|
|
16
|
+
|
|
17
|
+
class PubMedSearcher:
|
|
18
|
+
"""Search PubMed using NCBI E-utilities API."""
|
|
19
|
+
|
|
20
|
+
def __init__(self, api_key: Optional[str] = None, email: Optional[str] = None):
|
|
21
|
+
"""
|
|
22
|
+
Initialize searcher.
|
|
23
|
+
|
|
24
|
+
Args:
|
|
25
|
+
api_key: NCBI API key (optional but recommended)
|
|
26
|
+
email: Email for Entrez (optional but recommended)
|
|
27
|
+
"""
|
|
28
|
+
self.api_key = api_key or os.getenv('NCBI_API_KEY', '')
|
|
29
|
+
self.email = email or os.getenv('NCBI_EMAIL', '')
|
|
30
|
+
self.base_url = 'https://eutils.ncbi.nlm.nih.gov/entrez/eutils/'
|
|
31
|
+
self.session = requests.Session()
|
|
32
|
+
|
|
33
|
+
# Rate limiting
|
|
34
|
+
self.delay = 0.11 if self.api_key else 0.34 # 10/sec with key, 3/sec without
|
|
35
|
+
|
|
36
|
+
def search(self, query: str, max_results: int = 100,
|
|
37
|
+
date_start: Optional[str] = None, date_end: Optional[str] = None,
|
|
38
|
+
publication_types: Optional[List[str]] = None) -> List[str]:
|
|
39
|
+
"""
|
|
40
|
+
Search PubMed and return PMIDs.
|
|
41
|
+
|
|
42
|
+
Args:
|
|
43
|
+
query: Search query
|
|
44
|
+
max_results: Maximum number of results
|
|
45
|
+
date_start: Start date (YYYY/MM/DD or YYYY)
|
|
46
|
+
date_end: End date (YYYY/MM/DD or YYYY)
|
|
47
|
+
publication_types: List of publication types to filter
|
|
48
|
+
|
|
49
|
+
Returns:
|
|
50
|
+
List of PMIDs
|
|
51
|
+
"""
|
|
52
|
+
# Build query with filters
|
|
53
|
+
full_query = query
|
|
54
|
+
|
|
55
|
+
# Add date range
|
|
56
|
+
if date_start or date_end:
|
|
57
|
+
start = date_start or '1900'
|
|
58
|
+
end = date_end or datetime.now().strftime('%Y')
|
|
59
|
+
full_query += f' AND {start}:{end}[Publication Date]'
|
|
60
|
+
|
|
61
|
+
# Add publication types
|
|
62
|
+
if publication_types:
|
|
63
|
+
pub_type_query = ' OR '.join([f'"{pt}"[Publication Type]' for pt in publication_types])
|
|
64
|
+
full_query += f' AND ({pub_type_query})'
|
|
65
|
+
|
|
66
|
+
print(f'Searching PubMed: {full_query}', file=sys.stderr)
|
|
67
|
+
|
|
68
|
+
# ESearch to get PMIDs
|
|
69
|
+
esearch_url = self.base_url + 'esearch.fcgi'
|
|
70
|
+
params = {
|
|
71
|
+
'db': 'pubmed',
|
|
72
|
+
'term': full_query,
|
|
73
|
+
'retmax': max_results,
|
|
74
|
+
'retmode': 'json'
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
if self.email:
|
|
78
|
+
params['email'] = self.email
|
|
79
|
+
if self.api_key:
|
|
80
|
+
params['api_key'] = self.api_key
|
|
81
|
+
|
|
82
|
+
try:
|
|
83
|
+
response = self.session.get(esearch_url, params=params, timeout=30)
|
|
84
|
+
response.raise_for_status()
|
|
85
|
+
|
|
86
|
+
data = response.json()
|
|
87
|
+
pmids = data['esearchresult']['idlist']
|
|
88
|
+
count = int(data['esearchresult']['count'])
|
|
89
|
+
|
|
90
|
+
print(f'Found {count} results, retrieving {len(pmids)}', file=sys.stderr)
|
|
91
|
+
|
|
92
|
+
return pmids
|
|
93
|
+
|
|
94
|
+
except Exception as e:
|
|
95
|
+
print(f'Error searching PubMed: {e}', file=sys.stderr)
|
|
96
|
+
return []
|
|
97
|
+
|
|
98
|
+
def fetch_metadata(self, pmids: List[str]) -> List[Dict]:
|
|
99
|
+
"""
|
|
100
|
+
Fetch metadata for PMIDs.
|
|
101
|
+
|
|
102
|
+
Args:
|
|
103
|
+
pmids: List of PubMed IDs
|
|
104
|
+
|
|
105
|
+
Returns:
|
|
106
|
+
List of metadata dictionaries
|
|
107
|
+
"""
|
|
108
|
+
if not pmids:
|
|
109
|
+
return []
|
|
110
|
+
|
|
111
|
+
metadata_list = []
|
|
112
|
+
|
|
113
|
+
# Fetch in batches of 200
|
|
114
|
+
batch_size = 200
|
|
115
|
+
for i in range(0, len(pmids), batch_size):
|
|
116
|
+
batch = pmids[i:i+batch_size]
|
|
117
|
+
print(f'Fetching metadata for PMIDs {i+1}-{min(i+batch_size, len(pmids))}...', file=sys.stderr)
|
|
118
|
+
|
|
119
|
+
efetch_url = self.base_url + 'efetch.fcgi'
|
|
120
|
+
params = {
|
|
121
|
+
'db': 'pubmed',
|
|
122
|
+
'id': ','.join(batch),
|
|
123
|
+
'retmode': 'xml',
|
|
124
|
+
'rettype': 'abstract'
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
if self.email:
|
|
128
|
+
params['email'] = self.email
|
|
129
|
+
if self.api_key:
|
|
130
|
+
params['api_key'] = self.api_key
|
|
131
|
+
|
|
132
|
+
try:
|
|
133
|
+
response = self.session.get(efetch_url, params=params, timeout=60)
|
|
134
|
+
response.raise_for_status()
|
|
135
|
+
|
|
136
|
+
# Parse XML
|
|
137
|
+
root = ET.fromstring(response.content)
|
|
138
|
+
articles = root.findall('.//PubmedArticle')
|
|
139
|
+
|
|
140
|
+
for article in articles:
|
|
141
|
+
metadata = self._extract_metadata_from_xml(article)
|
|
142
|
+
if metadata:
|
|
143
|
+
metadata_list.append(metadata)
|
|
144
|
+
|
|
145
|
+
# Rate limiting
|
|
146
|
+
time.sleep(self.delay)
|
|
147
|
+
|
|
148
|
+
except Exception as e:
|
|
149
|
+
print(f'Error fetching metadata for batch: {e}', file=sys.stderr)
|
|
150
|
+
continue
|
|
151
|
+
|
|
152
|
+
return metadata_list
|
|
153
|
+
|
|
154
|
+
def _extract_metadata_from_xml(self, article: ET.Element) -> Optional[Dict]:
|
|
155
|
+
"""Extract metadata from PubmedArticle XML element."""
|
|
156
|
+
try:
|
|
157
|
+
medline_citation = article.find('.//MedlineCitation')
|
|
158
|
+
article_elem = medline_citation.find('.//Article')
|
|
159
|
+
journal = article_elem.find('.//Journal')
|
|
160
|
+
|
|
161
|
+
# Get PMID
|
|
162
|
+
pmid = medline_citation.findtext('.//PMID', '')
|
|
163
|
+
|
|
164
|
+
# Get DOI
|
|
165
|
+
doi = None
|
|
166
|
+
article_ids = article.findall('.//ArticleId')
|
|
167
|
+
for article_id in article_ids:
|
|
168
|
+
if article_id.get('IdType') == 'doi':
|
|
169
|
+
doi = article_id.text
|
|
170
|
+
break
|
|
171
|
+
|
|
172
|
+
# Get authors
|
|
173
|
+
authors = []
|
|
174
|
+
author_list = article_elem.find('.//AuthorList')
|
|
175
|
+
if author_list is not None:
|
|
176
|
+
for author in author_list.findall('.//Author'):
|
|
177
|
+
last_name = author.findtext('.//LastName', '')
|
|
178
|
+
fore_name = author.findtext('.//ForeName', '')
|
|
179
|
+
if last_name:
|
|
180
|
+
if fore_name:
|
|
181
|
+
authors.append(f'{last_name}, {fore_name}')
|
|
182
|
+
else:
|
|
183
|
+
authors.append(last_name)
|
|
184
|
+
|
|
185
|
+
# Get year
|
|
186
|
+
year = article_elem.findtext('.//Journal/JournalIssue/PubDate/Year', '')
|
|
187
|
+
if not year:
|
|
188
|
+
medline_date = article_elem.findtext('.//Journal/JournalIssue/PubDate/MedlineDate', '')
|
|
189
|
+
if medline_date:
|
|
190
|
+
import re
|
|
191
|
+
year_match = re.search(r'\d{4}', medline_date)
|
|
192
|
+
if year_match:
|
|
193
|
+
year = year_match.group()
|
|
194
|
+
|
|
195
|
+
metadata = {
|
|
196
|
+
'pmid': pmid,
|
|
197
|
+
'doi': doi,
|
|
198
|
+
'title': article_elem.findtext('.//ArticleTitle', ''),
|
|
199
|
+
'authors': ' and '.join(authors),
|
|
200
|
+
'journal': journal.findtext('.//Title', ''),
|
|
201
|
+
'year': year,
|
|
202
|
+
'volume': journal.findtext('.//JournalIssue/Volume', ''),
|
|
203
|
+
'issue': journal.findtext('.//JournalIssue/Issue', ''),
|
|
204
|
+
'pages': article_elem.findtext('.//Pagination/MedlinePgn', ''),
|
|
205
|
+
'abstract': article_elem.findtext('.//Abstract/AbstractText', '')
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
return metadata
|
|
209
|
+
|
|
210
|
+
except Exception as e:
|
|
211
|
+
print(f'Error extracting metadata: {e}', file=sys.stderr)
|
|
212
|
+
return None
|
|
213
|
+
|
|
214
|
+
def metadata_to_bibtex(self, metadata: Dict) -> str:
|
|
215
|
+
"""Convert metadata to BibTeX format."""
|
|
216
|
+
# Generate citation key
|
|
217
|
+
if metadata.get('authors'):
|
|
218
|
+
first_author = metadata['authors'].split(' and ')[0]
|
|
219
|
+
if ',' in first_author:
|
|
220
|
+
last_name = first_author.split(',')[0].strip()
|
|
221
|
+
else:
|
|
222
|
+
last_name = first_author.split()[0]
|
|
223
|
+
else:
|
|
224
|
+
last_name = 'Unknown'
|
|
225
|
+
|
|
226
|
+
year = metadata.get('year', 'XXXX')
|
|
227
|
+
citation_key = f'{last_name}{year}pmid{metadata.get("pmid", "")}'
|
|
228
|
+
|
|
229
|
+
# Build BibTeX entry
|
|
230
|
+
lines = [f'@article{{{citation_key},']
|
|
231
|
+
|
|
232
|
+
if metadata.get('authors'):
|
|
233
|
+
lines.append(f' author = {{{metadata["authors"]}}},')
|
|
234
|
+
|
|
235
|
+
if metadata.get('title'):
|
|
236
|
+
lines.append(f' title = {{{metadata["title"]}}},')
|
|
237
|
+
|
|
238
|
+
if metadata.get('journal'):
|
|
239
|
+
lines.append(f' journal = {{{metadata["journal"]}}},')
|
|
240
|
+
|
|
241
|
+
if metadata.get('year'):
|
|
242
|
+
lines.append(f' year = {{{metadata["year"]}}},')
|
|
243
|
+
|
|
244
|
+
if metadata.get('volume'):
|
|
245
|
+
lines.append(f' volume = {{{metadata["volume"]}}},')
|
|
246
|
+
|
|
247
|
+
if metadata.get('issue'):
|
|
248
|
+
lines.append(f' number = {{{metadata["issue"]}}},')
|
|
249
|
+
|
|
250
|
+
if metadata.get('pages'):
|
|
251
|
+
pages = metadata['pages'].replace('-', '--')
|
|
252
|
+
lines.append(f' pages = {{{pages}}},')
|
|
253
|
+
|
|
254
|
+
if metadata.get('doi'):
|
|
255
|
+
lines.append(f' doi = {{{metadata["doi"]}}},')
|
|
256
|
+
|
|
257
|
+
if metadata.get('pmid'):
|
|
258
|
+
lines.append(f' note = {{PMID: {metadata["pmid"]}}},')
|
|
259
|
+
|
|
260
|
+
# Remove trailing comma
|
|
261
|
+
if lines[-1].endswith(','):
|
|
262
|
+
lines[-1] = lines[-1][:-1]
|
|
263
|
+
|
|
264
|
+
lines.append('}')
|
|
265
|
+
|
|
266
|
+
return '\n'.join(lines)
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def main():
|
|
270
|
+
"""Command-line interface."""
|
|
271
|
+
parser = argparse.ArgumentParser(
|
|
272
|
+
description='Search PubMed using E-utilities API',
|
|
273
|
+
epilog='Example: python search_pubmed.py "CRISPR gene editing" --limit 100'
|
|
274
|
+
)
|
|
275
|
+
|
|
276
|
+
parser.add_argument(
|
|
277
|
+
'query',
|
|
278
|
+
nargs='?',
|
|
279
|
+
help='Search query (PubMed syntax)'
|
|
280
|
+
)
|
|
281
|
+
|
|
282
|
+
parser.add_argument(
|
|
283
|
+
'--query',
|
|
284
|
+
dest='query_arg',
|
|
285
|
+
help='Search query (alternative to positional argument)'
|
|
286
|
+
)
|
|
287
|
+
|
|
288
|
+
parser.add_argument(
|
|
289
|
+
'--query-file',
|
|
290
|
+
help='File containing search query'
|
|
291
|
+
)
|
|
292
|
+
|
|
293
|
+
parser.add_argument(
|
|
294
|
+
'--limit',
|
|
295
|
+
type=int,
|
|
296
|
+
default=100,
|
|
297
|
+
help='Maximum number of results (default: 100)'
|
|
298
|
+
)
|
|
299
|
+
|
|
300
|
+
parser.add_argument(
|
|
301
|
+
'--date-start',
|
|
302
|
+
help='Start date (YYYY/MM/DD or YYYY)'
|
|
303
|
+
)
|
|
304
|
+
|
|
305
|
+
parser.add_argument(
|
|
306
|
+
'--date-end',
|
|
307
|
+
help='End date (YYYY/MM/DD or YYYY)'
|
|
308
|
+
)
|
|
309
|
+
|
|
310
|
+
parser.add_argument(
|
|
311
|
+
'--publication-types',
|
|
312
|
+
help='Comma-separated publication types (e.g., "Review,Clinical Trial")'
|
|
313
|
+
)
|
|
314
|
+
|
|
315
|
+
parser.add_argument(
|
|
316
|
+
'-o', '--output',
|
|
317
|
+
help='Output file (default: stdout)'
|
|
318
|
+
)
|
|
319
|
+
|
|
320
|
+
parser.add_argument(
|
|
321
|
+
'--format',
|
|
322
|
+
choices=['json', 'bibtex'],
|
|
323
|
+
default='json',
|
|
324
|
+
help='Output format (default: json)'
|
|
325
|
+
)
|
|
326
|
+
|
|
327
|
+
parser.add_argument(
|
|
328
|
+
'--api-key',
|
|
329
|
+
help='NCBI API key (or set NCBI_API_KEY env var)'
|
|
330
|
+
)
|
|
331
|
+
|
|
332
|
+
parser.add_argument(
|
|
333
|
+
'--email',
|
|
334
|
+
help='Email for Entrez (or set NCBI_EMAIL env var)'
|
|
335
|
+
)
|
|
336
|
+
|
|
337
|
+
args = parser.parse_args()
|
|
338
|
+
|
|
339
|
+
# Get query
|
|
340
|
+
query = args.query or args.query_arg
|
|
341
|
+
|
|
342
|
+
if args.query_file:
|
|
343
|
+
try:
|
|
344
|
+
with open(args.query_file, 'r', encoding='utf-8') as f:
|
|
345
|
+
query = f.read().strip()
|
|
346
|
+
except Exception as e:
|
|
347
|
+
print(f'Error reading query file: {e}', file=sys.stderr)
|
|
348
|
+
sys.exit(1)
|
|
349
|
+
|
|
350
|
+
if not query:
|
|
351
|
+
parser.print_help()
|
|
352
|
+
sys.exit(1)
|
|
353
|
+
|
|
354
|
+
# Parse publication types
|
|
355
|
+
pub_types = None
|
|
356
|
+
if args.publication_types:
|
|
357
|
+
pub_types = [pt.strip() for pt in args.publication_types.split(',')]
|
|
358
|
+
|
|
359
|
+
# Search PubMed
|
|
360
|
+
searcher = PubMedSearcher(api_key=args.api_key, email=args.email)
|
|
361
|
+
pmids = searcher.search(
|
|
362
|
+
query,
|
|
363
|
+
max_results=args.limit,
|
|
364
|
+
date_start=args.date_start,
|
|
365
|
+
date_end=args.date_end,
|
|
366
|
+
publication_types=pub_types
|
|
367
|
+
)
|
|
368
|
+
|
|
369
|
+
if not pmids:
|
|
370
|
+
print('No results found', file=sys.stderr)
|
|
371
|
+
sys.exit(1)
|
|
372
|
+
|
|
373
|
+
# Fetch metadata
|
|
374
|
+
metadata_list = searcher.fetch_metadata(pmids)
|
|
375
|
+
|
|
376
|
+
# Format output
|
|
377
|
+
if args.format == 'json':
|
|
378
|
+
output = json.dumps({
|
|
379
|
+
'query': query,
|
|
380
|
+
'count': len(metadata_list),
|
|
381
|
+
'results': metadata_list
|
|
382
|
+
}, indent=2)
|
|
383
|
+
else: # bibtex
|
|
384
|
+
bibtex_entries = [searcher.metadata_to_bibtex(m) for m in metadata_list]
|
|
385
|
+
output = '\n\n'.join(bibtex_entries) + '\n'
|
|
386
|
+
|
|
387
|
+
# Write output
|
|
388
|
+
if args.output:
|
|
389
|
+
with open(args.output, 'w', encoding='utf-8') as f:
|
|
390
|
+
f.write(output)
|
|
391
|
+
print(f'Wrote {len(metadata_list)} results to {args.output}', file=sys.stderr)
|
|
392
|
+
else:
|
|
393
|
+
print(output)
|
|
394
|
+
|
|
395
|
+
|
|
396
|
+
if __name__ == '__main__':
|
|
397
|
+
main()
|
|
398
|
+
|