clearai-dsh 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +26 -0
- package/LICENSE +201 -0
- package/README.md +138 -0
- package/README.zh-CN.md +138 -0
- package/bin/clearai.mjs +224 -0
- package/brand/README.md +41 -0
- package/brand/logo-512-dark.png +0 -0
- package/brand/logo-512.png +0 -0
- package/brand/logo-lockup-dark.png +0 -0
- package/brand/logo-lockup.png +0 -0
- package/brand/logo-lockup.svg +12 -0
- package/brand/logo-wordmark.svg +6 -0
- package/brand/logo.svg +19 -0
- package/cordis.patch.yml +39 -0
- package/lib/client.js +3071 -0
- package/lib/fold.js +1576 -0
- package/lib/host.js +605 -0
- package/package.json +65 -0
- package/presets/clearai/agent.cordis.yml +226 -0
- package/presets/clearai/plugins/brain.js +547 -0
- package/presets/clearai/plugins/clearai-kernel.js +5485 -0
- package/presets/clearai/plugins/ontology.js +306 -0
- package/presets/clearai/plugins/prompts.js +312 -0
- package/presets/clearai/preset.yml +5 -0
- package/presets/clearai/skills/clearai-loop/SKILL.md +89 -0
- package/presets/clearai/template/knowledge/README.md +25 -0
- package/presets/clearai/template/memory/README.md +34 -0
- package/presets/clearai/template/project.md +49 -0
- package/presets/clearai/template/skills/README.md +37 -0
- package/presets/clearai/template/skills/chart-diagram-qa/SKILL.md +43 -0
- package/presets/clearai/template/skills/citation-management/SKILL.md +73 -0
- package/presets/clearai/template/skills/citation-management/references/bibtex_formatting.md +908 -0
- package/presets/clearai/template/skills/citation-management/references/citation_validation.md +794 -0
- package/presets/clearai/template/skills/citation-management/references/google_scholar_search.md +725 -0
- package/presets/clearai/template/skills/citation-management/references/metadata_extraction.md +870 -0
- package/presets/clearai/template/skills/citation-management/references/pubmed_search.md +839 -0
- package/presets/clearai/template/skills/citation-management/scripts/doi_to_bibtex.py +204 -0
- package/presets/clearai/template/skills/citation-management/scripts/extract_metadata.py +569 -0
- package/presets/clearai/template/skills/citation-management/scripts/format_bibtex.py +349 -0
- package/presets/clearai/template/skills/citation-management/scripts/generate_schematic.py +139 -0
- package/presets/clearai/template/skills/citation-management/scripts/generate_schematic_ai.py +817 -0
- package/presets/clearai/template/skills/citation-management/scripts/search_google_scholar.py +282 -0
- package/presets/clearai/template/skills/citation-management/scripts/search_pubmed.py +398 -0
- package/presets/clearai/template/skills/citation-management/scripts/validate_citations.py +497 -0
- package/presets/clearai/template/skills/data-analysis/SKILL.md +92 -0
- package/presets/clearai/template/skills/data-analysis/checklists/readiness_check.md +23 -0
- package/presets/clearai/template/skills/data-analysis/templates/analysis_report.md.tpl +63 -0
- package/presets/clearai/template/skills/data-analysis/templates/cleaning_rules_draft.yaml.tpl +32 -0
- package/presets/clearai/template/skills/data-analysis/templates/data_dictionary.md.tpl +12 -0
- package/presets/clearai/template/skills/data-analysis/templates/domain_knowledge_template.md.tpl +316 -0
- package/presets/clearai/template/skills/data-analysis/templates/feature_candidates.json.tpl +20 -0
- package/presets/clearai/template/skills/data-analysis/templates/quality_scorecard.md.tpl +30 -0
- package/presets/clearai/template/skills/data-analysis/workflows/01-data-profiling.md +42 -0
- package/presets/clearai/template/skills/data-analysis/workflows/02-quality-audit.md +36 -0
- package/presets/clearai/template/skills/data-analysis/workflows/03-physical-correlation.md +25 -0
- package/presets/clearai/template/skills/data-analysis/workflows/04-unstructured-mining.md +26 -0
- package/presets/clearai/template/skills/data-qa-analysis/SKILL.md +102 -0
- package/presets/clearai/template/skills/data-qa-analysis/checklists/readiness_check.md +62 -0
- package/presets/clearai/template/skills/data-qa-analysis/templates/best_in_class_report.md.tpl +56 -0
- package/presets/clearai/template/skills/data-qa-analysis/templates/cleaning_rules_draft.yaml.tpl +56 -0
- package/presets/clearai/template/skills/data-qa-analysis/templates/data_dictionary.md.tpl +13 -0
- package/presets/clearai/template/skills/data-qa-analysis/templates/data_source_inventory_and_lineage.md.tpl +146 -0
- package/presets/clearai/template/skills/data-qa-analysis/templates/data_status_report.md.tpl +60 -0
- package/presets/clearai/template/skills/data-qa-analysis/templates/steady_state_rules.yaml.tpl +41 -0
- package/presets/clearai/template/skills/data-qa-analysis/templates/subsystem_registry.md.tpl +101 -0
- package/presets/clearai/template/skills/data-qa-analysis/templates/unified_execution_plan.md.tpl +100 -0
- package/presets/clearai/template/skills/data-qa-analysis/workflows/01-data-source-inventory-and-lineage.md +194 -0
- package/presets/clearai/template/skills/data-qa-analysis/workflows/02-data-alignment-and-tag-semantics.md +122 -0
- package/presets/clearai/template/skills/data-qa-analysis/workflows/03-steady-state-identification.md +126 -0
- package/presets/clearai/template/skills/data-qa-analysis/workflows/04-consumption-analysis.md +152 -0
- package/presets/clearai/template/skills/data-qa-analysis/workflows/05-best-in-class-and-optimization-space.md +78 -0
- package/presets/clearai/template/skills/domain-presearch/SKILL.md +131 -0
- package/presets/clearai/template/skills/domain-presearch/checklists/domain_checklist.md +24 -0
- package/presets/clearai/template/skills/domain-presearch/references/figure_code.md +78 -0
- package/presets/clearai/template/skills/domain-presearch/references/strategic_frameworks.md +38 -0
- package/presets/clearai/template/skills/exploration-loop/SKILL.md +81 -0
- package/presets/clearai/template/skills/exploratory-data-analysis/SKILL.md +77 -0
- package/presets/clearai/template/skills/exploratory-data-analysis/references/bioinformatics_genomics_formats.md +664 -0
- package/presets/clearai/template/skills/exploratory-data-analysis/references/chemistry_molecular_formats.md +664 -0
- package/presets/clearai/template/skills/exploratory-data-analysis/references/general_scientific_formats.md +518 -0
- package/presets/clearai/template/skills/exploratory-data-analysis/references/microscopy_imaging_formats.md +620 -0
- package/presets/clearai/template/skills/exploratory-data-analysis/references/proteomics_metabolomics_formats.md +517 -0
- package/presets/clearai/template/skills/exploratory-data-analysis/references/spectroscopy_analytical_formats.md +633 -0
- package/presets/clearai/template/skills/exploratory-data-analysis/scripts/eda_analyzer.py +547 -0
- package/presets/clearai/template/skills/hypothesis-generation/SKILL.md +73 -0
- package/presets/clearai/template/skills/hypothesis-generation/references/experimental_design_patterns.md +329 -0
- package/presets/clearai/template/skills/hypothesis-generation/references/hypothesis_quality_criteria.md +198 -0
- package/presets/clearai/template/skills/hypothesis-generation/references/literature_search_strategies.md +622 -0
- package/presets/clearai/template/skills/hypothesis-generation/scripts/generate_schematic.py +139 -0
- package/presets/clearai/template/skills/hypothesis-generation/scripts/generate_schematic_ai.py +817 -0
- package/presets/clearai/template/skills/literature-review/SKILL.md +72 -0
- package/presets/clearai/template/skills/literature-review/references/citation_styles.md +166 -0
- package/presets/clearai/template/skills/literature-review/references/database_strategies.md +455 -0
- package/presets/clearai/template/skills/literature-review/scripts/generate_pdf.py +176 -0
- package/presets/clearai/template/skills/literature-review/scripts/generate_schematic.py +139 -0
- package/presets/clearai/template/skills/literature-review/scripts/generate_schematic_ai.py +817 -0
- package/presets/clearai/template/skills/literature-review/scripts/search_databases.py +303 -0
- package/presets/clearai/template/skills/literature-review/scripts/verify_citations.py +221 -0
- package/presets/clearai/template/skills/paper-lookup/SKILL.md +59 -0
- package/presets/clearai/template/skills/paper-lookup/references/arxiv.md +161 -0
- package/presets/clearai/template/skills/paper-lookup/references/biorxiv.md +118 -0
- package/presets/clearai/template/skills/paper-lookup/references/core.md +150 -0
- package/presets/clearai/template/skills/paper-lookup/references/crossref.md +181 -0
- package/presets/clearai/template/skills/paper-lookup/references/medrxiv.md +104 -0
- package/presets/clearai/template/skills/paper-lookup/references/openalex.md +174 -0
- package/presets/clearai/template/skills/paper-lookup/references/pmc.md +152 -0
- package/presets/clearai/template/skills/paper-lookup/references/pubmed.md +124 -0
- package/presets/clearai/template/skills/paper-lookup/references/semantic-scholar.md +203 -0
- package/presets/clearai/template/skills/paper-lookup/references/unpaywall.md +127 -0
- package/presets/clearai/template/skills/process-presearch/SKILL.md +196 -0
- package/presets/clearai/template/skills/process-presearch/checklists/process_checklist.md +18 -0
- package/presets/clearai/template/skills/process-presearch/references/figure_code.md +107 -0
- package/presets/clearai/template/skills/process-presearch/references/source_attribution_example.md +22 -0
- package/presets/clearai/template/skills/process-understanding-extraction/SKILL.md +69 -0
- package/presets/clearai/template/skills/process-understanding-extraction/checklists/readiness_check.md +34 -0
- package/presets/clearai/template/skills/process-understanding-extraction/templates/docx_raw_dump_extractor.py.tpl +132 -0
- package/presets/clearai/template/skills/process-understanding-extraction/templates/entity_map_unit_topology.json.tpl +86 -0
- package/presets/clearai/template/skills/process-understanding-extraction/templates/process_brief.md.tpl +89 -0
- package/presets/clearai/template/skills/process-understanding-extraction/templates/process_brief_builder_from_raw_dump.py.tpl +203 -0
- package/presets/clearai/template/skills/process-understanding-extraction/templates/process_flow_mermaid.md.tpl +41 -0
- package/presets/clearai/template/skills/process-understanding-extraction/templates/unified_execution_plan.md.tpl +53 -0
- package/presets/clearai/template/skills/process-understanding-extraction/workflows/01-process-doc-discovery.md +173 -0
- package/presets/clearai/template/skills/process-understanding-extraction/workflows/02-process-understanding-and-diagramming.md +106 -0
- package/presets/clearai/template/skills/scientific-brainstorming/SKILL.md +64 -0
- package/presets/clearai/template/skills/scientific-brainstorming/references/brainstorming_methods.md +326 -0
- package/presets/clearai/template/skills/scientific-critical-thinking/SKILL.md +72 -0
- package/presets/clearai/template/skills/scientific-critical-thinking/references/common_biases.md +364 -0
- package/presets/clearai/template/skills/scientific-critical-thinking/references/evidence_hierarchy.md +485 -0
- package/presets/clearai/template/skills/scientific-critical-thinking/references/experimental_design.md +496 -0
- package/presets/clearai/template/skills/scientific-critical-thinking/references/logical_fallacies.md +478 -0
- package/presets/clearai/template/skills/scientific-critical-thinking/references/scientific_method.md +169 -0
- package/presets/clearai/template/skills/scientific-critical-thinking/references/statistical_pitfalls.md +506 -0
- package/presets/clearai/template/skills/skill-creator/SKILL.md +109 -0
- package/presets/clearai/template/skills/skill-creator/references/authoring-guide.md +89 -0
- package/presets/clearai/template/skills/statistical-analysis/SKILL.md +79 -0
- package/presets/clearai/template/skills/statistical-analysis/references/assumptions_and_diagnostics.md +369 -0
- package/presets/clearai/template/skills/statistical-analysis/references/bayesian_statistics.md +653 -0
- package/presets/clearai/template/skills/statistical-analysis/references/effect_sizes_and_power.md +578 -0
- package/presets/clearai/template/skills/statistical-analysis/references/reporting_standards.md +469 -0
- package/presets/clearai/template/skills/statistical-analysis/references/test_selection_guide.md +129 -0
- package/presets/clearai/template/skills/statistical-analysis/scripts/assumption_checks.py +538 -0
- package/presets/clearai/template/skills/web-artifact/SKILL.md +165 -0
- package/presets/clearai/template/skills/web-artifact/assets/renderer/renderer.css +229 -0
- package/presets/clearai/template/skills/web-artifact/assets/renderer/renderer.js +373 -0
- package/presets/clearai/template/skills/web-artifact/assets/vendor/elkjs/LICENSE +263 -0
- package/presets/clearai/template/skills/web-artifact/assets/vendor/elkjs/UPSTREAM.md +26 -0
- package/presets/clearai/template/skills/web-artifact/assets/vendor/elkjs/elk.bundled.js +6605 -0
- package/presets/clearai/template/skills/web-artifact/references/when-drawing-a-topology.md +150 -0
- package/presets/clearai/template/skills/web-artifact/references/when-the-page-must-work-offline.md +62 -0
- package/presets/clearai/template/skills/web-artifact/scripts/check_artifact.py +167 -0
- package/presets/clearai/template/skills/web-artifact/scripts/render_topology.js +272 -0
- package/presets/clearai/template/skills/what-if-oracle/LICENSE.txt +5 -0
- package/presets/clearai/template/skills/what-if-oracle/SKILL.md +72 -0
- package/presets/clearai/template/skills/what-if-oracle/references/scenario-templates.md +154 -0
|
@@ -0,0 +1,349 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
BibTeX Formatter and Cleaner
|
|
4
|
+
Format, clean, sort, and deduplicate BibTeX files.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import sys
|
|
8
|
+
import re
|
|
9
|
+
import argparse
|
|
10
|
+
from typing import List, Dict
|
|
11
|
+
from collections import OrderedDict
|
|
12
|
+
|
|
13
|
+
class BibTeXFormatter:
|
|
14
|
+
"""Format and clean BibTeX entries."""
|
|
15
|
+
|
|
16
|
+
def __init__(self):
|
|
17
|
+
# Standard field order for readability
|
|
18
|
+
self.field_order = [
|
|
19
|
+
'author', 'editor', 'title', 'booktitle', 'journal',
|
|
20
|
+
'year', 'month', 'volume', 'number', 'pages',
|
|
21
|
+
'publisher', 'address', 'edition', 'series',
|
|
22
|
+
'school', 'institution', 'organization',
|
|
23
|
+
'howpublished', 'doi', 'url', 'isbn', 'issn',
|
|
24
|
+
'note', 'abstract', 'keywords'
|
|
25
|
+
]
|
|
26
|
+
|
|
27
|
+
def parse_bibtex_file(self, filepath: str) -> List[Dict]:
|
|
28
|
+
"""
|
|
29
|
+
Parse BibTeX file and extract entries.
|
|
30
|
+
|
|
31
|
+
Args:
|
|
32
|
+
filepath: Path to BibTeX file
|
|
33
|
+
|
|
34
|
+
Returns:
|
|
35
|
+
List of entry dictionaries
|
|
36
|
+
"""
|
|
37
|
+
try:
|
|
38
|
+
with open(filepath, 'r', encoding='utf-8') as f:
|
|
39
|
+
content = f.read()
|
|
40
|
+
except Exception as e:
|
|
41
|
+
print(f'Error reading file: {e}', file=sys.stderr)
|
|
42
|
+
return []
|
|
43
|
+
|
|
44
|
+
entries = []
|
|
45
|
+
|
|
46
|
+
# Match BibTeX entries
|
|
47
|
+
pattern = r'@(\w+)\s*\{\s*([^,\s]+)\s*,(.*?)\n\}'
|
|
48
|
+
matches = re.finditer(pattern, content, re.DOTALL | re.IGNORECASE)
|
|
49
|
+
|
|
50
|
+
for match in matches:
|
|
51
|
+
entry_type = match.group(1).lower()
|
|
52
|
+
citation_key = match.group(2).strip()
|
|
53
|
+
fields_text = match.group(3)
|
|
54
|
+
|
|
55
|
+
# Parse fields
|
|
56
|
+
fields = OrderedDict()
|
|
57
|
+
field_pattern = r'(\w+)\s*=\s*\{([^}]*)\}|(\w+)\s*=\s*"([^"]*)"'
|
|
58
|
+
field_matches = re.finditer(field_pattern, fields_text)
|
|
59
|
+
|
|
60
|
+
for field_match in field_matches:
|
|
61
|
+
if field_match.group(1):
|
|
62
|
+
field_name = field_match.group(1).lower()
|
|
63
|
+
field_value = field_match.group(2)
|
|
64
|
+
else:
|
|
65
|
+
field_name = field_match.group(3).lower()
|
|
66
|
+
field_value = field_match.group(4)
|
|
67
|
+
|
|
68
|
+
fields[field_name] = field_value.strip()
|
|
69
|
+
|
|
70
|
+
entries.append({
|
|
71
|
+
'type': entry_type,
|
|
72
|
+
'key': citation_key,
|
|
73
|
+
'fields': fields
|
|
74
|
+
})
|
|
75
|
+
|
|
76
|
+
return entries
|
|
77
|
+
|
|
78
|
+
def format_entry(self, entry: Dict) -> str:
|
|
79
|
+
"""
|
|
80
|
+
Format a single BibTeX entry.
|
|
81
|
+
|
|
82
|
+
Args:
|
|
83
|
+
entry: Entry dictionary
|
|
84
|
+
|
|
85
|
+
Returns:
|
|
86
|
+
Formatted BibTeX string
|
|
87
|
+
"""
|
|
88
|
+
lines = [f'@{entry["type"]}{{{entry["key"]},']
|
|
89
|
+
|
|
90
|
+
# Order fields according to standard order
|
|
91
|
+
ordered_fields = OrderedDict()
|
|
92
|
+
|
|
93
|
+
# Add fields in standard order
|
|
94
|
+
for field_name in self.field_order:
|
|
95
|
+
if field_name in entry['fields']:
|
|
96
|
+
ordered_fields[field_name] = entry['fields'][field_name]
|
|
97
|
+
|
|
98
|
+
# Add any remaining fields
|
|
99
|
+
for field_name, field_value in entry['fields'].items():
|
|
100
|
+
if field_name not in ordered_fields:
|
|
101
|
+
ordered_fields[field_name] = field_value
|
|
102
|
+
|
|
103
|
+
# Format each field
|
|
104
|
+
max_field_len = max(len(f) for f in ordered_fields.keys()) if ordered_fields else 0
|
|
105
|
+
|
|
106
|
+
for field_name, field_value in ordered_fields.items():
|
|
107
|
+
# Pad field name for alignment
|
|
108
|
+
padded_field = field_name.ljust(max_field_len)
|
|
109
|
+
lines.append(f' {padded_field} = {{{field_value}}},')
|
|
110
|
+
|
|
111
|
+
# Remove trailing comma from last field
|
|
112
|
+
if lines[-1].endswith(','):
|
|
113
|
+
lines[-1] = lines[-1][:-1]
|
|
114
|
+
|
|
115
|
+
lines.append('}')
|
|
116
|
+
|
|
117
|
+
return '\n'.join(lines)
|
|
118
|
+
|
|
119
|
+
def fix_common_issues(self, entry: Dict) -> Dict:
|
|
120
|
+
"""
|
|
121
|
+
Fix common formatting issues in entry.
|
|
122
|
+
|
|
123
|
+
Args:
|
|
124
|
+
entry: Entry dictionary
|
|
125
|
+
|
|
126
|
+
Returns:
|
|
127
|
+
Fixed entry dictionary
|
|
128
|
+
"""
|
|
129
|
+
fixed = entry.copy()
|
|
130
|
+
fields = fixed['fields'].copy()
|
|
131
|
+
|
|
132
|
+
# Fix page ranges (single hyphen to double hyphen)
|
|
133
|
+
if 'pages' in fields:
|
|
134
|
+
pages = fields['pages']
|
|
135
|
+
# Replace single hyphen with double hyphen if it's a range
|
|
136
|
+
if re.search(r'\d-\d', pages) and '--' not in pages:
|
|
137
|
+
pages = re.sub(r'(\d)-(\d)', r'\1--\2', pages)
|
|
138
|
+
fields['pages'] = pages
|
|
139
|
+
|
|
140
|
+
# Remove "pp." from pages
|
|
141
|
+
if 'pages' in fields:
|
|
142
|
+
pages = fields['pages']
|
|
143
|
+
pages = re.sub(r'^pp\.\s*', '', pages, flags=re.IGNORECASE)
|
|
144
|
+
fields['pages'] = pages
|
|
145
|
+
|
|
146
|
+
# Fix DOI (remove URL prefix if present)
|
|
147
|
+
if 'doi' in fields:
|
|
148
|
+
doi = fields['doi']
|
|
149
|
+
doi = doi.replace('https://doi.org/', '')
|
|
150
|
+
doi = doi.replace('http://doi.org/', '')
|
|
151
|
+
doi = doi.replace('doi:', '')
|
|
152
|
+
fields['doi'] = doi
|
|
153
|
+
|
|
154
|
+
# Fix author separators (semicolon or ampersand to 'and')
|
|
155
|
+
if 'author' in fields:
|
|
156
|
+
author = fields['author']
|
|
157
|
+
author = author.replace(';', ' and')
|
|
158
|
+
author = author.replace(' & ', ' and ')
|
|
159
|
+
# Clean up multiple 'and's
|
|
160
|
+
author = re.sub(r'\s+and\s+and\s+', ' and ', author)
|
|
161
|
+
fields['author'] = author
|
|
162
|
+
|
|
163
|
+
fixed['fields'] = fields
|
|
164
|
+
return fixed
|
|
165
|
+
|
|
166
|
+
def deduplicate_entries(self, entries: List[Dict]) -> List[Dict]:
|
|
167
|
+
"""
|
|
168
|
+
Remove duplicate entries based on DOI or citation key.
|
|
169
|
+
|
|
170
|
+
Args:
|
|
171
|
+
entries: List of entry dictionaries
|
|
172
|
+
|
|
173
|
+
Returns:
|
|
174
|
+
List of unique entries
|
|
175
|
+
"""
|
|
176
|
+
seen_dois = set()
|
|
177
|
+
seen_keys = set()
|
|
178
|
+
unique_entries = []
|
|
179
|
+
|
|
180
|
+
for entry in entries:
|
|
181
|
+
doi = entry['fields'].get('doi', '').strip()
|
|
182
|
+
key = entry['key']
|
|
183
|
+
|
|
184
|
+
# Check DOI first (more reliable)
|
|
185
|
+
if doi:
|
|
186
|
+
if doi in seen_dois:
|
|
187
|
+
print(f'Duplicate DOI found: {doi} (skipping {key})', file=sys.stderr)
|
|
188
|
+
continue
|
|
189
|
+
seen_dois.add(doi)
|
|
190
|
+
|
|
191
|
+
# Check citation key
|
|
192
|
+
if key in seen_keys:
|
|
193
|
+
print(f'Duplicate citation key found: {key} (skipping)', file=sys.stderr)
|
|
194
|
+
continue
|
|
195
|
+
seen_keys.add(key)
|
|
196
|
+
|
|
197
|
+
unique_entries.append(entry)
|
|
198
|
+
|
|
199
|
+
return unique_entries
|
|
200
|
+
|
|
201
|
+
def sort_entries(self, entries: List[Dict], sort_by: str = 'key', descending: bool = False) -> List[Dict]:
|
|
202
|
+
"""
|
|
203
|
+
Sort entries by specified field.
|
|
204
|
+
|
|
205
|
+
Args:
|
|
206
|
+
entries: List of entry dictionaries
|
|
207
|
+
sort_by: Field to sort by ('key', 'year', 'author', 'title')
|
|
208
|
+
descending: Sort in descending order
|
|
209
|
+
|
|
210
|
+
Returns:
|
|
211
|
+
Sorted list of entries
|
|
212
|
+
"""
|
|
213
|
+
def get_sort_key(entry: Dict) -> str:
|
|
214
|
+
if sort_by == 'key':
|
|
215
|
+
return entry['key'].lower()
|
|
216
|
+
elif sort_by == 'year':
|
|
217
|
+
year = entry['fields'].get('year', '9999')
|
|
218
|
+
return year
|
|
219
|
+
elif sort_by == 'author':
|
|
220
|
+
author = entry['fields'].get('author', 'ZZZ')
|
|
221
|
+
# Get last name of first author
|
|
222
|
+
if ',' in author:
|
|
223
|
+
return author.split(',')[0].lower()
|
|
224
|
+
else:
|
|
225
|
+
return author.split()[0].lower() if author else 'zzz'
|
|
226
|
+
elif sort_by == 'title':
|
|
227
|
+
return entry['fields'].get('title', '').lower()
|
|
228
|
+
else:
|
|
229
|
+
return entry['key'].lower()
|
|
230
|
+
|
|
231
|
+
return sorted(entries, key=get_sort_key, reverse=descending)
|
|
232
|
+
|
|
233
|
+
def format_file(self, filepath: str, output: str = None,
|
|
234
|
+
deduplicate: bool = False, sort_by: str = None,
|
|
235
|
+
descending: bool = False, fix_issues: bool = True) -> None:
|
|
236
|
+
"""
|
|
237
|
+
Format entire BibTeX file.
|
|
238
|
+
|
|
239
|
+
Args:
|
|
240
|
+
filepath: Input BibTeX file
|
|
241
|
+
output: Output file (None for in-place)
|
|
242
|
+
deduplicate: Remove duplicates
|
|
243
|
+
sort_by: Field to sort by
|
|
244
|
+
descending: Sort in descending order
|
|
245
|
+
fix_issues: Fix common formatting issues
|
|
246
|
+
"""
|
|
247
|
+
print(f'Parsing {filepath}...', file=sys.stderr)
|
|
248
|
+
entries = self.parse_bibtex_file(filepath)
|
|
249
|
+
|
|
250
|
+
if not entries:
|
|
251
|
+
print('No entries found', file=sys.stderr)
|
|
252
|
+
return
|
|
253
|
+
|
|
254
|
+
print(f'Found {len(entries)} entries', file=sys.stderr)
|
|
255
|
+
|
|
256
|
+
# Fix common issues
|
|
257
|
+
if fix_issues:
|
|
258
|
+
print('Fixing common issues...', file=sys.stderr)
|
|
259
|
+
entries = [self.fix_common_issues(e) for e in entries]
|
|
260
|
+
|
|
261
|
+
# Deduplicate
|
|
262
|
+
if deduplicate:
|
|
263
|
+
print('Removing duplicates...', file=sys.stderr)
|
|
264
|
+
original_count = len(entries)
|
|
265
|
+
entries = self.deduplicate_entries(entries)
|
|
266
|
+
removed = original_count - len(entries)
|
|
267
|
+
if removed > 0:
|
|
268
|
+
print(f'Removed {removed} duplicate(s)', file=sys.stderr)
|
|
269
|
+
|
|
270
|
+
# Sort
|
|
271
|
+
if sort_by:
|
|
272
|
+
print(f'Sorting by {sort_by}...', file=sys.stderr)
|
|
273
|
+
entries = self.sort_entries(entries, sort_by, descending)
|
|
274
|
+
|
|
275
|
+
# Format entries
|
|
276
|
+
print('Formatting entries...', file=sys.stderr)
|
|
277
|
+
formatted_entries = [self.format_entry(e) for e in entries]
|
|
278
|
+
|
|
279
|
+
# Write output
|
|
280
|
+
output_content = '\n\n'.join(formatted_entries) + '\n'
|
|
281
|
+
|
|
282
|
+
output_file = output or filepath
|
|
283
|
+
try:
|
|
284
|
+
with open(output_file, 'w', encoding='utf-8') as f:
|
|
285
|
+
f.write(output_content)
|
|
286
|
+
print(f'Successfully wrote {len(entries)} entries to {output_file}', file=sys.stderr)
|
|
287
|
+
except Exception as e:
|
|
288
|
+
print(f'Error writing file: {e}', file=sys.stderr)
|
|
289
|
+
sys.exit(1)
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def main():
|
|
293
|
+
"""Command-line interface."""
|
|
294
|
+
parser = argparse.ArgumentParser(
|
|
295
|
+
description='Format, clean, sort, and deduplicate BibTeX files',
|
|
296
|
+
epilog='Example: python format_bibtex.py references.bib --deduplicate --sort year'
|
|
297
|
+
)
|
|
298
|
+
|
|
299
|
+
parser.add_argument(
|
|
300
|
+
'file',
|
|
301
|
+
help='BibTeX file to format'
|
|
302
|
+
)
|
|
303
|
+
|
|
304
|
+
parser.add_argument(
|
|
305
|
+
'-o', '--output',
|
|
306
|
+
help='Output file (default: overwrite input file)'
|
|
307
|
+
)
|
|
308
|
+
|
|
309
|
+
parser.add_argument(
|
|
310
|
+
'--deduplicate',
|
|
311
|
+
action='store_true',
|
|
312
|
+
help='Remove duplicate entries'
|
|
313
|
+
)
|
|
314
|
+
|
|
315
|
+
parser.add_argument(
|
|
316
|
+
'--sort',
|
|
317
|
+
choices=['key', 'year', 'author', 'title'],
|
|
318
|
+
help='Sort entries by field'
|
|
319
|
+
)
|
|
320
|
+
|
|
321
|
+
parser.add_argument(
|
|
322
|
+
'--descending',
|
|
323
|
+
action='store_true',
|
|
324
|
+
help='Sort in descending order'
|
|
325
|
+
)
|
|
326
|
+
|
|
327
|
+
parser.add_argument(
|
|
328
|
+
'--no-fix',
|
|
329
|
+
action='store_true',
|
|
330
|
+
help='Do not fix common issues'
|
|
331
|
+
)
|
|
332
|
+
|
|
333
|
+
args = parser.parse_args()
|
|
334
|
+
|
|
335
|
+
# Format file
|
|
336
|
+
formatter = BibTeXFormatter()
|
|
337
|
+
formatter.format_file(
|
|
338
|
+
args.file,
|
|
339
|
+
output=args.output,
|
|
340
|
+
deduplicate=args.deduplicate,
|
|
341
|
+
sort_by=args.sort,
|
|
342
|
+
descending=args.descending,
|
|
343
|
+
fix_issues=not args.no_fix
|
|
344
|
+
)
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
if __name__ == '__main__':
|
|
348
|
+
main()
|
|
349
|
+
|
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
Scientific schematic generation using Nano Banana 2.
|
|
4
|
+
|
|
5
|
+
Generate any scientific diagram by describing it in natural language.
|
|
6
|
+
Nano Banana 2 handles everything automatically with smart iterative refinement.
|
|
7
|
+
|
|
8
|
+
Smart iteration: Only regenerates if quality is below threshold for your document type.
|
|
9
|
+
Quality review: Uses Gemini 3.1 Pro Preview for professional scientific evaluation.
|
|
10
|
+
|
|
11
|
+
Usage:
|
|
12
|
+
# Generate for journal paper (highest quality threshold)
|
|
13
|
+
python generate_schematic.py "CONSORT flowchart" -o flowchart.png --doc-type journal
|
|
14
|
+
|
|
15
|
+
# Generate for presentation (lower threshold, faster)
|
|
16
|
+
python generate_schematic.py "Transformer architecture" -o transformer.png --doc-type presentation
|
|
17
|
+
|
|
18
|
+
# Generate for poster
|
|
19
|
+
python generate_schematic.py "MAPK signaling pathway" -o pathway.png --doc-type poster
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
import argparse
|
|
23
|
+
import os
|
|
24
|
+
import subprocess
|
|
25
|
+
import sys
|
|
26
|
+
from pathlib import Path
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def main():
|
|
30
|
+
"""Command-line interface."""
|
|
31
|
+
parser = argparse.ArgumentParser(
|
|
32
|
+
description="Generate scientific schematics using AI with smart iterative refinement",
|
|
33
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
34
|
+
epilog="""
|
|
35
|
+
How it works:
|
|
36
|
+
Simply describe your diagram in natural language
|
|
37
|
+
Nano Banana 2 generates it automatically with:
|
|
38
|
+
- Smart iteration (only regenerates if quality is below threshold)
|
|
39
|
+
- Quality review by Gemini 3.1 Pro Preview
|
|
40
|
+
- Document-type aware quality thresholds
|
|
41
|
+
- Publication-ready output
|
|
42
|
+
|
|
43
|
+
Document Types (quality thresholds):
|
|
44
|
+
journal 8.5/10 - Nature, Science, peer-reviewed journals
|
|
45
|
+
conference 8.0/10 - Conference papers
|
|
46
|
+
thesis 8.0/10 - Dissertations, theses
|
|
47
|
+
grant 8.0/10 - Grant proposals
|
|
48
|
+
preprint 7.5/10 - arXiv, bioRxiv, etc.
|
|
49
|
+
report 7.5/10 - Technical reports
|
|
50
|
+
poster 7.0/10 - Academic posters
|
|
51
|
+
presentation 6.5/10 - Slides, talks
|
|
52
|
+
default 7.5/10 - General purpose
|
|
53
|
+
|
|
54
|
+
Examples:
|
|
55
|
+
# Generate for journal paper (strict quality)
|
|
56
|
+
python generate_schematic.py "CONSORT participant flow" -o flowchart.png --doc-type journal
|
|
57
|
+
|
|
58
|
+
# Generate for poster (moderate quality)
|
|
59
|
+
python generate_schematic.py "Transformer architecture" -o arch.png --doc-type poster
|
|
60
|
+
|
|
61
|
+
# Generate for slides (faster, lower threshold)
|
|
62
|
+
python generate_schematic.py "System diagram" -o system.png --doc-type presentation
|
|
63
|
+
|
|
64
|
+
# Custom max iterations
|
|
65
|
+
python generate_schematic.py "Complex pathway" -o pathway.png --iterations 2
|
|
66
|
+
|
|
67
|
+
# Verbose output
|
|
68
|
+
python generate_schematic.py "Circuit diagram" -o circuit.png -v
|
|
69
|
+
|
|
70
|
+
Environment Variables:
|
|
71
|
+
OPENROUTER_API_KEY Required for AI generation
|
|
72
|
+
"""
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
parser.add_argument("prompt",
|
|
76
|
+
help="Description of the diagram to generate")
|
|
77
|
+
parser.add_argument("-o", "--output", required=True,
|
|
78
|
+
help="Output file path")
|
|
79
|
+
parser.add_argument("--doc-type", default="default",
|
|
80
|
+
choices=["journal", "conference", "poster", "presentation",
|
|
81
|
+
"report", "grant", "thesis", "preprint", "default"],
|
|
82
|
+
help="Document type for quality threshold (default: default)")
|
|
83
|
+
parser.add_argument("--iterations", type=int, default=2,
|
|
84
|
+
help="Maximum refinement iterations (default: 2, max: 2)")
|
|
85
|
+
parser.add_argument("--api-key",
|
|
86
|
+
help="OpenRouter API key (or use OPENROUTER_API_KEY env var)")
|
|
87
|
+
parser.add_argument("-v", "--verbose", action="store_true",
|
|
88
|
+
help="Verbose output")
|
|
89
|
+
|
|
90
|
+
args = parser.parse_args()
|
|
91
|
+
|
|
92
|
+
# Check for API key
|
|
93
|
+
api_key = args.api_key or os.getenv("OPENROUTER_API_KEY")
|
|
94
|
+
if not api_key:
|
|
95
|
+
print("Error: OPENROUTER_API_KEY environment variable not set")
|
|
96
|
+
print("\nFor AI generation, you need an OpenRouter API key.")
|
|
97
|
+
print("Get one at: https://openrouter.ai/keys")
|
|
98
|
+
print("\nSet it with:")
|
|
99
|
+
print(" export OPENROUTER_API_KEY='your_api_key'")
|
|
100
|
+
print("\nOr use --api-key flag")
|
|
101
|
+
sys.exit(1)
|
|
102
|
+
|
|
103
|
+
# Find AI generation script
|
|
104
|
+
script_dir = Path(__file__).parent
|
|
105
|
+
ai_script = script_dir / "generate_schematic_ai.py"
|
|
106
|
+
|
|
107
|
+
if not ai_script.exists():
|
|
108
|
+
print(f"Error: AI generation script not found: {ai_script}")
|
|
109
|
+
sys.exit(1)
|
|
110
|
+
|
|
111
|
+
# Build command
|
|
112
|
+
cmd = [sys.executable, str(ai_script), args.prompt, "-o", args.output]
|
|
113
|
+
|
|
114
|
+
if args.doc_type != "default":
|
|
115
|
+
cmd.extend(["--doc-type", args.doc_type])
|
|
116
|
+
|
|
117
|
+
# Enforce max 2 iterations
|
|
118
|
+
iterations = min(args.iterations, 2)
|
|
119
|
+
if iterations != 2:
|
|
120
|
+
cmd.extend(["--iterations", str(iterations)])
|
|
121
|
+
|
|
122
|
+
if args.verbose:
|
|
123
|
+
cmd.append("-v")
|
|
124
|
+
|
|
125
|
+
# Execute — pass API key via environment to avoid exposure in process listings
|
|
126
|
+
try:
|
|
127
|
+
env = os.environ.copy()
|
|
128
|
+
if api_key:
|
|
129
|
+
env["OPENROUTER_API_KEY"] = api_key
|
|
130
|
+
result = subprocess.run(cmd, check=False, env=env)
|
|
131
|
+
sys.exit(result.returncode)
|
|
132
|
+
except Exception as e:
|
|
133
|
+
print(f"Error executing AI generation: {e}")
|
|
134
|
+
sys.exit(1)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
if __name__ == "__main__":
|
|
138
|
+
main()
|
|
139
|
+
|