foo-py 0.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (242) hide show
  1. agents.py +10690 -0
  2. app.py +16234 -0
  3. boogr/__init__.py +563 -0
  4. boogr/default_icon.ico +0 -0
  5. boogr/enums.py +381 -0
  6. boogr/minion.py +175 -0
  7. boogr/resources/ico/BooIcon.ico +0 -0
  8. boogr/resources/ico/Booger.ico +0 -0
  9. boogr/resources/ico/Save.ico +0 -0
  10. boogr/resources/ico/adobe.ico +0 -0
  11. boogr/resources/ico/atk.ico +0 -0
  12. boogr/resources/ico/b.ico +0 -0
  13. boogr/resources/ico/batch.ico +0 -0
  14. boogr/resources/ico/black_sigma.ico +0 -0
  15. boogr/resources/ico/boo.ico +0 -0
  16. boogr/resources/ico/boogr.ico +0 -0
  17. boogr/resources/ico/browse.ico +0 -0
  18. boogr/resources/ico/chart.ico +0 -0
  19. boogr/resources/ico/copy.ico +0 -0
  20. boogr/resources/ico/csv.ico +0 -0
  21. boogr/resources/ico/dataedit.ico +0 -0
  22. boogr/resources/ico/doc.ico +0 -0
  23. boogr/resources/ico/e_logo.ico +0 -0
  24. boogr/resources/ico/error.ico +0 -0
  25. boogr/resources/ico/euler_circle.ico +0 -0
  26. boogr/resources/ico/excel.ico +0 -0
  27. boogr/resources/ico/file_browse.ico +0 -0
  28. boogr/resources/ico/filter.ico +0 -0
  29. boogr/resources/ico/folder_browse.ico +0 -0
  30. boogr/resources/ico/info.ico +0 -0
  31. boogr/resources/ico/input.ico +0 -0
  32. boogr/resources/ico/koolaid.ico +0 -0
  33. boogr/resources/ico/machinelearning.ico +0 -0
  34. boogr/resources/ico/message.ico +0 -0
  35. boogr/resources/ico/pdf.ico +0 -0
  36. boogr/resources/ico/pi.ico +0 -0
  37. boogr/resources/ico/setting.ico +0 -0
  38. boogr/resources/ico/sword_ninja.ico +0 -0
  39. boogr/resources/ico/textfile.ico +0 -0
  40. boogr/resources/ico/webcam.ico +0 -0
  41. boogr/resources/img/atk.png +0 -0
  42. boogr/resources/img/boogr.png +0 -0
  43. boogr/resources/img/button/Authority.png +0 -0
  44. boogr/resources/img/button/BOC.png +0 -0
  45. boogr/resources/img/button/DERA.png +0 -0
  46. boogr/resources/img/button/DWH.png +0 -0
  47. boogr/resources/img/button/EMD.png +0 -0
  48. boogr/resources/img/button/OAR.png +0 -0
  49. boogr/resources/img/button/OECA.png +0 -0
  50. boogr/resources/img/button/OGC.png +0 -0
  51. boogr/resources/img/button/OMS.png +0 -0
  52. boogr/resources/img/button/ORD.png +0 -0
  53. boogr/resources/img/button/OW.png +0 -0
  54. boogr/resources/img/button/RCRA.png +0 -0
  55. boogr/resources/img/button/TSCA.png +0 -0
  56. boogr/resources/img/button/WIFIA.png +0 -0
  57. boogr/resources/img/button/access.png +0 -0
  58. boogr/resources/img/button/add.png +0 -0
  59. boogr/resources/img/button/adobe.png +0 -0
  60. boogr/resources/img/button/airline.png +0 -0
  61. boogr/resources/img/button/analytics.png +0 -0
  62. boogr/resources/img/button/appropriation.png +0 -0
  63. boogr/resources/img/button/atk.ico +0 -0
  64. boogr/resources/img/button/atk.png +0 -0
  65. boogr/resources/img/button/attachment.png +0 -0
  66. boogr/resources/img/button/bfy.png +0 -0
  67. boogr/resources/img/button/bluetooth.png +0 -0
  68. boogr/resources/img/button/browse.png +0 -0
  69. boogr/resources/img/button/budget.png +0 -0
  70. boogr/resources/img/button/calculator.png +0 -0
  71. boogr/resources/img/button/calendar.png +0 -0
  72. boogr/resources/img/button/cancel.png +0 -0
  73. boogr/resources/img/button/categoricalgrants.png +0 -0
  74. boogr/resources/img/button/chart.png +0 -0
  75. boogr/resources/img/button/chrome.png +0 -0
  76. boogr/resources/img/button/close.png +0 -0
  77. boogr/resources/img/button/columndelete.png +0 -0
  78. boogr/resources/img/button/columnedit.png +0 -0
  79. boogr/resources/img/button/columninsert.png +0 -0
  80. boogr/resources/img/button/commandline.png +0 -0
  81. boogr/resources/img/button/commute.png +0 -0
  82. boogr/resources/img/button/compass.png +0 -0
  83. boogr/resources/img/button/contracts.png +0 -0
  84. boogr/resources/img/button/controlpanel.png +0 -0
  85. boogr/resources/img/button/csv.png +0 -0
  86. boogr/resources/img/button/database.png +0 -0
  87. boogr/resources/img/button/databaseadd.png +0 -0
  88. boogr/resources/img/button/databasedelete.png +0 -0
  89. boogr/resources/img/button/databaserefresh.png +0 -0
  90. boogr/resources/img/button/databasesql.png +0 -0
  91. boogr/resources/img/button/databaseverify.png +0 -0
  92. boogr/resources/img/button/datagrid.png +0 -0
  93. boogr/resources/img/button/delete.png +0 -0
  94. boogr/resources/img/button/division.png +0 -0
  95. boogr/resources/img/button/document.png +0 -0
  96. boogr/resources/img/button/documentadd.png +0 -0
  97. boogr/resources/img/button/documentation.png +0 -0
  98. boogr/resources/img/button/documentdelete.png +0 -0
  99. boogr/resources/img/button/documentedit.png +0 -0
  100. boogr/resources/img/button/documenterror.png +0 -0
  101. boogr/resources/img/button/documentsearch.png +0 -0
  102. boogr/resources/img/button/edge.png +0 -0
  103. boogr/resources/img/button/edit.png +0 -0
  104. boogr/resources/img/button/efy.png +0 -0
  105. boogr/resources/img/button/environment.png +0 -0
  106. boogr/resources/img/button/ev.png +0 -0
  107. boogr/resources/img/button/excel.png +0 -0
  108. boogr/resources/img/button/expenses.png +0 -0
  109. boogr/resources/img/button/export.png +0 -0
  110. boogr/resources/img/button/file.png +0 -0
  111. boogr/resources/img/button/file_word.png +0 -0
  112. boogr/resources/img/button/fileadd.png +0 -0
  113. boogr/resources/img/button/filebrowse.png +0 -0
  114. boogr/resources/img/button/filecopy.png +0 -0
  115. boogr/resources/img/button/filedelete.png +0 -0
  116. boogr/resources/img/button/fileedit.png +0 -0
  117. boogr/resources/img/button/filereader.png +0 -0
  118. boogr/resources/img/button/filesearch.png +0 -0
  119. boogr/resources/img/button/filetransfer.png +0 -0
  120. boogr/resources/img/button/fileverify.png +0 -0
  121. boogr/resources/img/button/filewriter.png +0 -0
  122. boogr/resources/img/button/filter.png +0 -0
  123. boogr/resources/img/button/first.png +0 -0
  124. boogr/resources/img/button/folder.png +0 -0
  125. boogr/resources/img/button/folderbrowse.png +0 -0
  126. boogr/resources/img/button/foldercompress.png +0 -0
  127. boogr/resources/img/button/foldercopy.png +0 -0
  128. boogr/resources/img/button/folderdownload.png +0 -0
  129. boogr/resources/img/button/folderopen.png +0 -0
  130. boogr/resources/img/button/fte.png +0 -0
  131. boogr/resources/img/button/function.png +0 -0
  132. boogr/resources/img/button/gmail.png +0 -0
  133. boogr/resources/img/button/go.png +0 -0
  134. boogr/resources/img/button/google.png +0 -0
  135. boogr/resources/img/button/grants.png +0 -0
  136. boogr/resources/img/button/guidance.png +0 -0
  137. boogr/resources/img/button/home.png +0 -0
  138. boogr/resources/img/button/id.png +0 -0
  139. boogr/resources/img/button/image.png +0 -0
  140. boogr/resources/img/button/import.png +0 -0
  141. boogr/resources/img/button/information.png +0 -0
  142. boogr/resources/img/button/internet.png +0 -0
  143. boogr/resources/img/button/justice.png +0 -0
  144. boogr/resources/img/button/last.png +0 -0
  145. boogr/resources/img/button/ledger.png +0 -0
  146. boogr/resources/img/button/left.png +0 -0
  147. boogr/resources/img/button/levels.png +0 -0
  148. boogr/resources/img/button/logout.png +0 -0
  149. boogr/resources/img/button/lust.png +0 -0
  150. boogr/resources/img/button/menu.png +0 -0
  151. boogr/resources/img/button/metrics.png +0 -0
  152. boogr/resources/img/button/mpg.png +0 -0
  153. boogr/resources/img/button/next.png +0 -0
  154. boogr/resources/img/button/no.png +0 -0
  155. boogr/resources/img/button/oil.png +0 -0
  156. boogr/resources/img/button/ok.png +0 -0
  157. boogr/resources/img/button/omb.png +0 -0
  158. boogr/resources/img/button/onenote.png +0 -0
  159. boogr/resources/img/button/oust.png +0 -0
  160. boogr/resources/img/button/outlay.png +0 -0
  161. boogr/resources/img/button/outlook.png +0 -0
  162. boogr/resources/img/button/pause.png +0 -0
  163. boogr/resources/img/button/payroll.png +0 -0
  164. boogr/resources/img/button/pdf.png +0 -0
  165. boogr/resources/img/button/percentage.png +0 -0
  166. boogr/resources/img/button/play.png +0 -0
  167. boogr/resources/img/button/plusminus.png +0 -0
  168. boogr/resources/img/button/previous.png +0 -0
  169. boogr/resources/img/button/print.png +0 -0
  170. boogr/resources/img/button/recertification.png +0 -0
  171. boogr/resources/img/button/recycle.png +0 -0
  172. boogr/resources/img/button/redo.png +0 -0
  173. boogr/resources/img/button/refresh.png +0 -0
  174. boogr/resources/img/button/remove.png +0 -0
  175. boogr/resources/img/button/reserve.png +0 -0
  176. boogr/resources/img/button/right.png +0 -0
  177. boogr/resources/img/button/row.png +0 -0
  178. boogr/resources/img/button/rowcopy.png +0 -0
  179. boogr/resources/img/button/rowdelete.png +0 -0
  180. boogr/resources/img/button/rowedit.png +0 -0
  181. boogr/resources/img/button/rowinsert.png +0 -0
  182. boogr/resources/img/button/save.png +0 -0
  183. boogr/resources/img/button/scan.png +0 -0
  184. boogr/resources/img/button/sharepoint.png +0 -0
  185. boogr/resources/img/button/sigma.png +0 -0
  186. boogr/resources/img/button/site.png +0 -0
  187. boogr/resources/img/button/sitetravel.png +0 -0
  188. boogr/resources/img/button/sort.png +0 -0
  189. boogr/resources/img/button/spreadsheet.png +0 -0
  190. boogr/resources/img/button/statistics.png +0 -0
  191. boogr/resources/img/button/table.png +0 -0
  192. boogr/resources/img/button/tableadd.png +0 -0
  193. boogr/resources/img/button/tabledelete.png +0 -0
  194. boogr/resources/img/button/tablesettings.png +0 -0
  195. boogr/resources/img/button/text.png +0 -0
  196. boogr/resources/img/button/traffic.png +0 -0
  197. boogr/resources/img/button/travel.png +0 -0
  198. boogr/resources/img/button/undelete.png +0 -0
  199. boogr/resources/img/button/undo.png +0 -0
  200. boogr/resources/img/button/wcf.png +0 -0
  201. boogr/resources/img/button/windows.png +0 -0
  202. boogr/resources/img/button/word.png +0 -0
  203. boogr/resources/img/button/xml.png +0 -0
  204. boogr/resources/img/button/yes.png +0 -0
  205. boogr/resources/img/button/zipfile.png +0 -0
  206. boogr/resources/img/gooey.png +0 -0
  207. boogr/resources/img/web/google.png +0 -0
  208. config.py +1015 -0
  209. core.py +170 -0
  210. data.py +832 -0
  211. desktop.py +173 -0
  212. embedders.py +269 -0
  213. fetchers.py +25142 -0
  214. foo_assets/__init__.py +1 -0
  215. foo_assets/resources/images/favicon.ico +0 -0
  216. foo_assets/resources/images/foo-apikeys.png +0 -0
  217. foo_assets/resources/images/foo-architecture.png +0 -0
  218. foo_assets/resources/images/foo-workflows.png +0 -0
  219. foo_assets/resources/images/foo.ico +0 -0
  220. foo_assets/resources/images/foo.svg +60 -0
  221. foo_assets/resources/images/foo_logo.ico +0 -0
  222. foo_assets/resources/images/foo_logo.png +0 -0
  223. foo_assets/resources/images/foo_logo.svg +63 -0
  224. foo_assets/resources/images/foo_portfolio.png +0 -0
  225. foo_assets/resources/images/foo_project.png +0 -0
  226. foo_assets/resources/images/system_diagram.svg +46 -0
  227. foo_assets/resources/images/uml_class_diagram.svg +65 -0
  228. foo_assets/streamlit_config.toml +85 -0
  229. foo_cli.py +52 -0
  230. foo_py-0.1.1.dist-info/METADATA +596 -0
  231. foo_py-0.1.1.dist-info/RECORD +242 -0
  232. foo_py-0.1.1.dist-info/WHEEL +5 -0
  233. foo_py-0.1.1.dist-info/entry_points.txt +2 -0
  234. foo_py-0.1.1.dist-info/licenses/LICENSE.txt +21 -0
  235. foo_py-0.1.1.dist-info/top_level.txt +18 -0
  236. generators.py +3323 -0
  237. loaders.py +4570 -0
  238. models.py +344 -0
  239. processors.py +3216 -0
  240. scrapers.py +706 -0
  241. stores/vector.py +365 -0
  242. writers.py +268 -0
config.py ADDED
@@ -0,0 +1,1015 @@
1
+ '''
2
+ ******************************************************************************************
3
+ Assembly: Foo
4
+ Filename: config.py
5
+ Author: Terry Eppler
6
+ Created: 05-31-2022
7
+
8
+ Last Modified By: Terry Eppler
9
+ Last Modified On: 05-01-2025
10
+ ******************************************************************************************
11
+ <copyright file="config.py" company="Terry D. Eppler">
12
+
13
+ config.py
14
+ Copyright © 2024 Terry Eppler
15
+
16
+ Permission is hereby granted, free of charge, to any person obtaining a copy
17
+ of this software and associated documentation files (the “Software”),
18
+ to deal in the Software without restriction,
19
+ including without limitation the rights to use,
20
+ copy, modify, merge, publish, distribute, sublicense,
21
+ and/or sell copies of the Software,
22
+ and to permit persons to whom the Software is furnished to do so,
23
+ subject to the following conditions:
24
+
25
+ The above copyright notice and this permission notice shall be included in all
26
+ copies or substantial portions of the Software.
27
+
28
+ THE SOFTWARE IS PROVIDED “AS IS”, WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED,
29
+ INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
30
+ FITNESS FOR A PARTICULAR PURPOSE AND NON-INFRINGEMENT.
31
+ IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM,
32
+ DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE,
33
+ ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER
34
+ DEALINGS IN THE SOFTWARE.
35
+
36
+ You can contact me at: terryeppler@gmail.com or eppler.terry@epa.gov
37
+
38
+ </copyright>
39
+ <summary>
40
+ config.py
41
+ </summary>
42
+ ******************************************************************************************
43
+ '''
44
+ import os
45
+ from pathlib import Path
46
+
47
+ # -------------- APP-LEVEL UTILITIES -------------
48
+
49
+ def throw_if( name: str, value: object ) -> None:
50
+ """Raise ``ValueError`` when a required value is empty.
51
+
52
+ Purpose:
53
+ Provide a small, consistent guard for required arguments and configuration values. The
54
+ function treats falsy values as invalid and raises a ``ValueError`` containing the
55
+ caller-supplied argument or setting name.
56
+
57
+ Args:
58
+ name (str): Name of the argument or configuration value being validated.
59
+ value (object): Value to validate.
60
+
61
+ Returns:
62
+ None.
63
+
64
+ Raises:
65
+ ValueError: Raised when ``value`` is falsy.
66
+ """
67
+ if not value:
68
+ raise ValueError( f'Argument "{name}" cannot be empty!' )
69
+
70
+ def get_bool( name: str, default: bool = False ) -> bool:
71
+ """Read a Boolean environment variable using Fiddy's true-value convention.
72
+
73
+ Purpose:
74
+ Convert environment-variable text into a deterministic Boolean value. Missing variables
75
+ return the caller-provided default. Values of ``1``, ``true``, ``yes``, ``y``, and
76
+ ``on`` are treated as ``True``; all other defined values are treated as ``False``.
77
+
78
+ Args:
79
+ name (str): Environment variable name.
80
+ default (bool): Default value used when the environment variable is not defined.
81
+
82
+ Returns:
83
+ bool: Parsed Boolean value. If parsing fails, the original ``default`` value is returned.
84
+ """
85
+ try:
86
+ throw_if( 'name', name )
87
+ value = os.getenv( name )
88
+ return default if value is None else value.strip( ).lower( ) in (
89
+ '1',
90
+ 'true',
91
+ 'yes',
92
+ 'y',
93
+ 'on'
94
+ )
95
+ except Exception:
96
+ return default
97
+
98
+ def get_int( name: str, default: int ) -> int:
99
+ """Read an integer environment variable with a deterministic fallback.
100
+
101
+ Purpose:
102
+ Parse an optional environment variable as an integer while preserving a safe default when
103
+ the variable is missing, empty, or invalid.
104
+
105
+ Args:
106
+ name (str): Environment variable name.
107
+ default (int): Default integer value used when parsing is not possible.
108
+
109
+ Returns:
110
+ int: Parsed integer value or the supplied default value.
111
+ """
112
+ try:
113
+ throw_if( 'name', name )
114
+ value = os.getenv( name )
115
+ return default if value in (None, '') else int( str( value ).strip( ) )
116
+ except Exception:
117
+ return default
118
+
119
+ def get_float( name: str, default: float ) -> float:
120
+ """Read a floating-point environment variable with a deterministic fallback.
121
+
122
+ Purpose:
123
+ Parse an optional environment variable as a float while preserving a safe default when the
124
+ variable is missing, empty, or invalid.
125
+
126
+ Args:
127
+ name (str): Environment variable name.
128
+ default (float): Default floating-point value used when parsing is not possible.
129
+
130
+ Returns:
131
+ float: Parsed floating-point value or the supplied default value.
132
+ """
133
+ try:
134
+ throw_if( 'name', name )
135
+ value = os.getenv( name )
136
+ return default if value in (None, '') else float( str( value ).strip( ) )
137
+ except Exception:
138
+ return default
139
+
140
+ def get_path( name: str, default: Path ) -> Path:
141
+ """Read a path environment variable and return a resolved ``Path``.
142
+
143
+ Purpose:
144
+ Resolve optional filesystem configuration from the environment. Missing variables return
145
+ the resolved default path. Invalid values return the resolved default path rather than
146
+ interrupting module import.
147
+
148
+ Args:
149
+ name (str): Environment variable name.
150
+ default (Path): Default path used when the environment variable is not defined.
151
+
152
+ Returns:
153
+ Path: Resolved path value or resolved default path.
154
+ """
155
+ try:
156
+ throw_if( 'name', name )
157
+ throw_if( 'default', default )
158
+ value = os.getenv( name )
159
+ return Path( value ).resolve( ) if value else default.resolve( )
160
+ except Exception:
161
+ return default.resolve( )
162
+
163
+ def get_text( name: str, default: str ) -> str:
164
+ """Read a text environment variable with a deterministic fallback.
165
+
166
+ Purpose:
167
+ Return an environment variable as text while preserving the supplied default when the
168
+ variable is missing or empty.
169
+
170
+ Args:
171
+ name (str): Environment variable name.
172
+ default (str): Default text value.
173
+
174
+ Returns:
175
+ str: Environment value or supplied default.
176
+ """
177
+ try:
178
+ throw_if( 'name', name )
179
+ value = os.getenv( name )
180
+ return default if value in (None, '') else str( value )
181
+ except Exception:
182
+ return default
183
+
184
+ # ------ CONSTANTS -------------------
185
+
186
+ BASE_DIR = Path( __file__ ).resolve( ).parent
187
+ ROOT_DIR = Path( __file__ ).resolve( ).parent
188
+ # Frozen desktop builds keep mutable data outside the installed application directory.
189
+ if os.getenv( 'FOO_DESKTOP' ) == '1':
190
+ ROOT_DIR = Path( os.environ[ 'LOCALAPPDATA' ] ) / 'Foo'
191
+ ROOT_DIR.mkdir( parents=True, exist_ok=True )
192
+ ( ROOT_DIR / 'logging' ).mkdir( parents=True, exist_ok=True )
193
+ ( ROOT_DIR / 'stores' / 'sqlite' / 'datamodels' ).mkdir( parents=True, exist_ok=True )
194
+ LOG_DIR = get_path( 'LOG_DIR', ROOT_DIR / 'logging' )
195
+ LOG_PATH = get_text( 'LOG_PATH', str( LOG_DIR / 'Exceptions.db' ) )
196
+ LOG_FILE = get_text( 'LOG_FILE', 'Exceptions' )
197
+
198
+ # ------ ENVIRONMENT API KEYS -------------------
199
+ ACCESS_DRIVER = r'DRIVER={ Microsoft Access Driver (*.mdb, *.accdb) };DBQ='
200
+ AIRNOW_API_KEY = os.getenv( 'AIRNOW_API_KEY' )
201
+ CLAUDE_API_KEY = os.getenv( 'CLAUDE_API_KEY' )
202
+ CONGRESS_API_KEY = os.getenv( 'CONGRESS_API_KEY' )
203
+ CHROMA_API_KEY = os.getenv( 'CHROMA_API_KEY' )
204
+ CHROMA_TENET_ID = os.getenv( 'CHROMA_TENET_ID' )
205
+ GEOAPIFY_API_KEY = os.getenv( 'GEOAPIFY_API_KEY' )
206
+ GEOCODING_API_KEY = os.getenv( 'GEOCODING_API_KEY' )
207
+ GEMINI_API_KEY = os.getenv( 'GEMINI_API_KEY' )
208
+ GOOGLE_API_KEY = os.getenv( 'GOOGLE_API_KEY' )
209
+ GOOGLE_CSE_ID = os.getenv( 'GOOGLE_CSE_ID' )
210
+ GOOGLE_CLOUD_PROJECT_ID = os.getenv( 'GOOGLE_CLOUD_PROJECT_ID' )
211
+ GOOGLE_CLOUD_LOCATION = os.getenv( 'GOOGLE_CLOUD_LOCATION' )
212
+ GOVINFO_API_KEY = os.getenv( 'GOVINFO_API_KEY' )
213
+ GOOGLE_GENAI_USE_VERTEXAI = os.getenv( 'GOOGLE_GENAI_USE_VERTEXAI' )
214
+ GOOGLE_WEATHER_API_KEY = os.getenv( 'GOOGLE_WEATHER_API_KEY' )
215
+ GOOGLE_ACCOUNT_FILE = os.getenv( 'GOOGLE_ACCOUNT_CREDENTIALS' )
216
+ GOOGLE_DRIVE_TOKEN_PATH = os.getenv( 'GOOGLE_DRIVE_TOKEN_PATH' )
217
+ GOOGLE_DRIVE_FOLDER_ID = os.getenv( 'GOOGLE_DRIVE_FOLDER_ID' )
218
+ HUGGINGFACE_API_KEY = os.getenv( 'HUGGINGFACE_API_KEY' )
219
+ IPINFO_API_KEY = os.getenv( 'IPINFO_API_KEY' )
220
+ OPENAI_API_KEY = os.getenv( 'OPENAI_API_KEY' )
221
+ PINECONE_API_KEY = os.getenv( 'PINECONE_API_KEY' )
222
+ LANGSMITH_API_KEY = os.getenv( 'LANGSMITH_API_KEY' )
223
+ LLAMAINDEX_API_KEY = os.getenv( 'LLAMAINDEX_API_KEY' )
224
+ LLAMACLOUD_API_KEY = os.getenv( 'LLAMACLOUD_API_KEY' )
225
+ MISTRAL_API_KEY = os.getenv( 'MISTRAL_API_KEY' )
226
+ NASA_API_KEY = os.getenv( 'NASA_API_KEY' )
227
+ NASA_EARTHDATA_TOKEN = os.getenv( 'NASA_EARTHDATA_TOKEN' )
228
+ NEWS_API_KEY = os.getenv( 'NEWSAPI_API_KEY' )
229
+ THENEWS_API_KEY = os.getenv( 'THENEWSAPI_API_KEY' )
230
+ WEATHERAPI_API_KEY = os.getenv( 'WEATHERAPI_API_KEY' )
231
+ XAI_API_KEY = os.getenv( 'XAI_API_KEY' )
232
+ O365_CLIENT_ID = os.getenv( 'O365_CLIENT_ID' )
233
+ O365_CLIENT_SECRET = os.getenv( 'O365_CLIENT_SECRET' )
234
+ OPENAQ_API_KEY = os.getenv( 'OPENAQ_API_KEY' )
235
+ OPENSKY_API_CLIENT_ID = os.getenv( 'OPENSKY_API_CLIENT_ID' )
236
+ OPENSKY_API_CREDENTIALS = os.getenv( 'OPENSKY_API_CREDENTIALS' )
237
+ OPENSKY_API_CLIENT_SECRET = os.getenv( 'OPENSKY_API_CLIENT_SECRET' )
238
+ CENSUS_API_KEY = os.getenv( 'CENSUS_API_KEY' )
239
+ SOCRATA_API_KEY = os.getenv( 'SOCRATA_API_KEY' )
240
+ HEALTHDATA_API_KEY = os.getenv( 'HEALTHDATA_API_KEY' )
241
+ USGS_WATERDATA_API_KEY = os.getenv( 'USGS_WATERDATA_API_KEY' ) or os.getenv( 'USGS_API_KEY' )
242
+ DATA_GOV_API_KEY = os.getenv( 'DATAGOV_API_KEY' )
243
+ GOVINFO_API_KEY = os.getenv( 'GOVINFO_API_KEY' )
244
+ FIRMS_MAP_KEY = os.getenv( 'FIRMS_MAP_KEY' )
245
+ PURPLEAIR_API_KEY = os.getenv( 'PURPLEAIR_API_KEY' )
246
+ SKYMAP_TOKEN = os.getenv( 'SKY_MAP_TOKEN' )
247
+
248
+ # ---------------- CONSTANTS -----------------------
249
+ APP_TITLE = 'Foo'
250
+ BLUE_DIVIDER = "<div style='height:2px;align:left;background:#0078FC;margin:6px 30px 30px 0;'></div>"
251
+ SQLSERVER_DRIVER = r'DRIVER={ ODBC Driver 17 for SQL Server };SERVER=.\SQLExpress;'
252
+ DB_PATH = ROOT_DIR / 'stores' / 'sqlite' / 'datamodels' / 'Data.db'
253
+ AGENTS ='''Mozilla/5.0 Windows NT 10.0; Win64; x64; AppleWebKit/537.36 (KHTML, like Gecko)
254
+ Chrome/124.0 Safari/537.36'''
255
+ FAVICON = r'resources/images/favicon.ico'
256
+ LOGO = r'resources/images/foo_logo.png'
257
+ DB = r'stores/sqlite/datamodels/Data.db'
258
+
259
+ MODE = [ 'Loading', 'Scraping', 'Retrieval',
260
+ 'Geospatial', 'Demographic', 'Environmental',
261
+ 'Astronomical', 'Generation' ]
262
+
263
+
264
+ MODE_MAP = \
265
+ {
266
+ 'Loading': 'Data Loading',
267
+ 'Scraping': 'Data Scraping',
268
+ 'Retrieval': 'Data Collections & Public Archives',
269
+ 'Geospatial': 'Weather & Geospatial Information',
270
+ 'Demographic': 'Health & Population Data',
271
+ 'Environmental': 'Environmental Information',
272
+ 'Astronomical': 'Physics & Astronomical Data',
273
+ 'Generation': 'AI Generation',
274
+ }
275
+
276
+ CHUNKABLE_LOADERS = {
277
+ 'TextLoader': [ 'chars', 'tokens' ],
278
+ 'CsvLoader': [ 'chars' ],
279
+ 'PdfLoader': [ 'chars' ],
280
+ 'ExcelLoader': [ 'chars' ],
281
+ 'WordLoader': [ 'chars' ],
282
+ 'MarkdownLoader': [ 'chars' ],
283
+ 'HtmlLoader': [ 'chars' ],
284
+ 'JsonLoader': [ 'chars' ],
285
+ 'PowerPointLoader': [ 'chars' ],
286
+ }
287
+
288
+ REQUIRED_CORPORA = [
289
+ 'brown',
290
+ 'gutenberg',
291
+ 'reuters',
292
+ 'webtext',
293
+ 'inaugural',
294
+ 'state_union',
295
+ 'punkt',
296
+ 'punkt_tab',
297
+ 'stopwords',
298
+ ]
299
+
300
+ SESSION_STATE_DEFAULTS = {
301
+ # ------------ Ingestion
302
+ 'documents': None,
303
+ 'raw_documents': None,
304
+ 'active_loader': None,
305
+ # ------------ Input
306
+ 'raw_text': None,
307
+ 'raw_tokens': None,
308
+ 'raw_text_view': None,
309
+ # Processing
310
+ 'parser': None,
311
+ 'processed_text': None,
312
+ 'processed_text_view': None,
313
+ # ------------ Performance
314
+ 'start_time': None,
315
+ 'end_time': None,
316
+ 'total_time': None,
317
+ # ------------ Tokenization / Vocabulary
318
+ 'tokens': None,
319
+ 'vocabulary': None,
320
+ 'token_counts': None,
321
+ 'df_synsets': None,
322
+ # ------------ SQLite / Excel
323
+ # ------------ Chunking
324
+ 'lines': None,
325
+ 'chunks': None,
326
+ 'chunk_modes': None,
327
+ 'chunked_documents': None,
328
+ # ------------ Embeddings
329
+ 'embedder': None,
330
+ 'embeddings': None,
331
+ 'embedding_provider': None,
332
+ 'embedding_model': None,
333
+ 'embedding_source': None,
334
+ 'embedding_documents': None,
335
+ 'df_embedding_input': None,
336
+ 'df_embedding_output': None,
337
+ # ------------ Retrieval / Search
338
+ 'search_results': None,
339
+ # ------------ DataFrames
340
+ 'df_frequency': None,
341
+ 'df_chunks': None,
342
+ # ------------ Data
343
+ # ------------ Sidebar / API Keys
344
+ 'api_keys': {
345
+ 'openai_api_key': None,
346
+ 'groq_api_key': None,
347
+ 'google_api_key': None,
348
+ 'pinecone_api_key': None,
349
+ 'google_credentials_path_api_key': None,
350
+ },
351
+ # ------------ XML Loader (explicit contract)
352
+ 'xml_loader': None,
353
+ 'xml_documents': None,
354
+ 'xml_split_documents': None,
355
+ 'xml_tree_loaded': None,
356
+ 'xml_namespaces': None,
357
+ 'xml_xpath_results': None,
358
+ # ------------ WordNet Caches
359
+ 'wordnet_synsets_sig': None,
360
+ 'df_wordnet_synsets': None,
361
+ 'df_wordnet_lemmas': None,
362
+ }
363
+
364
+ # ----------------- Models
365
+
366
+ GPT_MODELS = [ 'gpt-5.4', 'gpt-5', 'gpt-5-mini', 'gpt-5-nano',
367
+ 'gpt-5.1', 'gpt-5.2', 'gpt-4.1' ]
368
+
369
+ GEMINI_MODELS = [ 'gemini-2.5-flash', 'gemini-2.5-flash-lite',
370
+ 'gemini-2.5-flash-lite' ]
371
+
372
+ GROK_MODELS = [ 'grok-4-1-fast-reasoning', 'grok-4-fast-reasoning', 'grok-4',
373
+ 'grok-code-fast-1', 'grok-3-mini', 'grok-2-image-1212' ]
374
+
375
+ CLAUDE_MODELS = [ 'claude-opus-4-6', 'claude-sonnet-4-6',
376
+ 'claude-haiku-4-5' ]
377
+
378
+ MISTRAL_MODELS = [ 'mistral-large-latest', 'mistral-medium-latest',
379
+ 'mistral-small-latest', 'mistral-ocr-latest' ]
380
+
381
+ # ------------- API DEFINITIONS ------------------
382
+
383
+ ARXIV = r'''arXiv is a free distribution service and an open-access archive for nearly 2.4 million
384
+ scholarly articles in the fields of physics, mathematics, computer science, quantitative
385
+ biology, quantitative finance, statistics, electrical engineering and systems science, and
386
+ economics. Materials on this site are not peer-reviewed by arXiv.
387
+
388
+ https://arxiv.org/
389
+ '''
390
+
391
+ GOOGLE_DRIVE = r'''Google Drive is a secure, cloud-based storage service by Google that allows users
392
+ to store, synchronize, and share files across computers, phones, and tablets. Offering 15GB
393
+ of free storage, it acts as a centralized hub for documents, photos, and videos, enabling
394
+ real-time collaboration with Google Workspace apps like Docs, Sheets, and Slides.
395
+
396
+ https://workspace.google.com/products/drive/
397
+ '''
398
+
399
+ WIKIPEDIA= r'''A multilingual free online encyclopedia written and maintained by a community of
400
+ volunteers, known as Wikipedians, through open collaboration and using a wiki-based editing
401
+ system called MediaWiki. Wikipedia is the largest and most-read reference work in history.
402
+
403
+ https://www.wikipedia.org
404
+ '''
405
+
406
+ PUBMED = r'''The National Center for Biotechnology Information, National Library of Medicine
407
+ comprises more than 35 million citations for biomedical literature from MEDLINE,
408
+ life science journals, and online books. Citations may include links to full text content
409
+ from PubMed Central and publisher web sites.
410
+
411
+ https://pubmed.ncbi.nlm.nih.gov/
412
+ '''
413
+
414
+ THENEWS = r'''News Aggregators: Creating "one-stop-shop" apps or websites that compile headlines
415
+ from multiple sources into a single feed. Market Intelligence: Tracking competitors,
416
+ industry trends, and "sales triggers" (like mergers or funding rounds) to inform business
417
+ decisions. Financial Analysis: Monitoring stock market news and economic shifts to assist
418
+ in trading and risk management. Media Monitoring: Tracking brand mentions and public
419
+ sentiment across the web to manage PR and customer feedback
420
+ '''
421
+
422
+ GOOGLE_CSE = r'''The Cse Service is the endpoint that returns the requested searches.
423
+ You must identify a particular search engine to use in your request
424
+ (using the cx query parameter) as well as the search query (using the q query parameter).
425
+ In addition, you should provide a developer key (using the key query parameter).
426
+
427
+ http://www.google.com/cse/
428
+ '''
429
+
430
+ GOOGLE_WEATHER = r'''The Weather API lets you request real-time, hyperlocal weather data for
431
+ locations around the world. Weather information includes temperature, precipitation,
432
+ humidity, and more.
433
+
434
+ https://developers.google.com/maps/documentation/weather/overview
435
+ '''
436
+
437
+ US_NAVAL_OBSERVATORY = r'''Provides access to APIs from the US Naval Observatory's Celestial Navigation Data for
438
+ Assumed Position and Time: this data service provides all the astronomical information
439
+ necessary to plot navigational lines of position from observations of the altitudes of
440
+ celestial bodies.
441
+
442
+ https://www.cnmoc.usff.navy.mil/usno/
443
+ '''
444
+
445
+ OPEN_SCIENCE = r'''Provides access to APIs from NASA's Open Science Data Repostitory (OSDR).
446
+ NASA OSDR provides a RESTful Application Programming Interface (API) to its
447
+ full-text search, data file retrieval, and metadata retrieval capabilities.
448
+ The API provides a choice of standard web output formats,
449
+ either JavaScript Object Notation (JSON) or Hyper Text Markup Language (HTML),
450
+ of query results.
451
+
452
+ https://science.nasa.gov/biological-physical/data/osdr/
453
+ '''
454
+
455
+ GOV_INFO = r'''The GovInfo Link Service provides services for developers and webmasters to access
456
+ content and metadata on GovInfo. Current and planned services include a link service,
457
+ list service, and search service. The link service is used to create embedded links to
458
+ content and metadata on GovInfo and is currently enabled for the collections below.
459
+ The collection code is listed in parenthesis after each collection name, and the available
460
+ queries are listed below each collection. More information about each query is provided on
461
+ the individual collection page.
462
+
463
+ https://api.govinfo.gov/docs/
464
+ '''
465
+
466
+ CONGRESS = r'''Submit queries against the Congressional Research Service's (CRS) Appropriation
467
+ Status Table.
468
+
469
+ https://api.congress.gov/
470
+ '''
471
+
472
+ INTERNET_ARCHIVE = r'''The Internet Archive is a non-profit, digital library founded in 1996 that
473
+ provides free, public access to vast collections of digitized materials, including over 800
474
+ billion web pages via its "Wayback Machine". It acts as a "time machine" for the web, preserving
475
+ books, movies, music, software, and websites to provide universal access to knowledge.
476
+
477
+ https://archive.org/developers/index-apis.html
478
+ '''
479
+
480
+ THE_SATELLITE_CENTER = r'''Provides access to APIs from NASA's Satellite Situation Center Web (SSCWeb) service
481
+ that is operated jointly by the NASA/GSFC Space Physics Data Facility (SPDF) and the
482
+ National Space Science Data Center (NSSDC) to support a range of NASA science programs
483
+ and to fulfill key international NASA responsibilities including those of NSSDC and the
484
+ World Data Center-A for Rockets and Satellites.
485
+
486
+ https://api.nasa.gov/
487
+ '''
488
+
489
+ NEAR_BY_OBJECTS = r'''Provides access to APIs from JPL’s SSD (Solar System Dynamics) and CNEOS
490
+ (Center for Near-Earth Object Studies) API (Application Program Interface) service.
491
+ This service provides an interface to machine-readable data (JSON-format) related to SSD
492
+ and CNEOS.
493
+
494
+ https://ssd-api.jpl.nasa.gov/
495
+ '''
496
+
497
+ ASTRONOMY_CATALOG = r'''The Open Astronomy Catalog (OAC) API is a RESTful interface designed for
498
+ programmatic access to open-access astronomical data, specifically focusing on transient events.
499
+
500
+ https://astrocats.space/
501
+ '''
502
+
503
+ ASTRO_QUERY = r'''Access to the astropy package that contains key functionality and common tools needed for
504
+ performing astronomy and astrophysics with Python. It is at the core of the Astropy Project,
505
+ which aims to enable the community to develop a robust ecosystem of affiliated packages
506
+ covering a broad range of needs for astronomical research, data processing, and data analysis.
507
+
508
+ https://astroquery.readthedocs.io/en/latest/
509
+ '''
510
+
511
+ OPEN_WEATHER = r'''
512
+ Provides forecast weather retrieval by location name using the Open-Meteo
513
+ Geocoding API and Open-Meteo Forecast API. This class is forecast-only by design and
514
+ intentionally excludes archive / historical date-based retrieval so it does not overlap
515
+ with the separate Historical Weather class.
516
+
517
+ https://open-meteo.com/
518
+ '''
519
+
520
+ HISTORICAL_WEATHER = r'''Provides historical weather retrieval by location name and date using the
521
+ Open-Meteo Geocoding API and Open-Meteo Historical Weather API. This class is intentionally
522
+ designed around the actual user-facing need in the Foo fetcher expander: enter a location
523
+ and a date, resolve that location to coordinates, then retrieve historical weather for that date.
524
+
525
+ https://open-meteo.com/en/docs/historical-weather-api
526
+ '''
527
+
528
+ NASA_EARTH_OBSERVATORY = r'''NASA Earth Observatory's Natural Event Tracker (EONET) allows users to access imagery,
529
+ often in near real-time (NRT), of natural events such as dust storms, forest fires, and
530
+ tropical cyclones—empowering people all across the planet to locate, track, and potentially
531
+ prepare for and manage events that affect communities in their paths.
532
+ Version 3 API for events, categories, sources, and layers.
533
+
534
+ https://eonet.gsfc.nasa.gov/docs/v3
535
+ '''
536
+
537
+ UN_DATA = r'''The UNdata API provides programmatic access to the United Nations' global statistical
538
+ database, allowing developers and researchers to query and retrieve data directly for use
539
+ in applications, websites, or local processing. The core UNdata API provides access to the
540
+ broader UNdata platform, which contains over 60 million data points from across the UN system.
541
+
542
+ https://data.un.org/Host.aspx?Content=API
543
+ '''
544
+
545
+ NASA_EONET = r'''NASA Earth Observatory's Natural Event Tracker (EONET) allows users to access imagery,
546
+ often in near real-time (NRT), of natural events such as dust storms, forest fires, and tropical
547
+ cyclones—empowering people all across the planet to locate, track, and potentially prepare for
548
+ and manage events that affect communities in their paths. The EONET application programming
549
+ interface (API) provides customization of features including curation and direct links to
550
+ image sources.
551
+
552
+ https://eonet.gsfc.nasa.gov/docs/v3
553
+ '''
554
+
555
+ NASA_FIRMS = r'''ASA’s Fire Information for Resource Management System (FIRMS) provides near
556
+ real-time, satellite-derived active fire and hotspot data (within 3 hours of observation)
557
+ to monitor wildfires. Using sensors from MODIS and VIIRS, it offers global coverage through
558
+ an interactive map, email alerts, and GIS data. It is designed for firefighters, scientists,
559
+ and natural resource managers.
560
+
561
+ https://firms.modaps.eosdis.nasa.gov/api/
562
+ '''
563
+
564
+ EPA_ENVIROFACTS = r'''The Envirofacts Data Warehouse contains information from select EPA Environmental
565
+ program office databases and provides access about environmental activities that may affect air,
566
+ water, and land anywhere in the United States.
567
+
568
+ https://www.epa.gov/enviro/envirofacts-data-service-api
569
+ '''
570
+
571
+ EPA_UV_INDEX = r'''The EPA UV Index predicts daily solar UV radiation intensity on a 1–11+ scale,
572
+ helping to gauge sun-safe precautions. Developed with the National Weather Service, it factors
573
+ in ozone, clouds, and elevation to forecast noon intensity. A UV Alert is issued if the
574
+ index is 6+ and unusually high.
575
+
576
+ https://enviro.epa.gov/envirofacts/uv/search
577
+ '''
578
+
579
+ SPACE_WEATHER = r'''NASA DONKI (Space Weather Database Of Notifications, Knowledge, Information) is
580
+ is a comprehensive on-line tool for space weather forecasters, scientists, and the general
581
+ space science community. DONKI chronicles the daily interpretations of space weather observations,
582
+ analysis, models, forecasts, and notifications provided by the Space Weather Research Center (SWRC),
583
+ comprehensive knowledge-base search functionality to support anomaly resolution and space
584
+ science research, intelligent linkages, relationships, cause-and-effects between space weather
585
+ activities and comprehensive webservice API access to information stored in DONKI.
586
+
587
+ https://ccmc.gsfc.nasa.gov/tools/DONKI/#donki-webservice-calls-api
588
+ '''
589
+
590
+ NEAR_BY_OBJECTS = r''''SSD (Solar System Dynamics) and CNEOS (Center for Near-Earth Object Studies)
591
+ API (Application Program Interface) service. This service provides an interface to
592
+ machine-readable data (JSON-format) related to SSD and CNEOS.
593
+
594
+ https://api.nasa.gov/
595
+ '''
596
+
597
+ SKY_MAP = r'''Provides static and link-based star chart generation using the SKY-MAP.ORG
598
+ XML API, Site Linker, and Image Generator interfaces.
599
+
600
+ https://wikisky.org/XML_API_V1.0.html
601
+ '''
602
+
603
+ NASA_OPEN_SCIENCE = r'''NASA’s Open Science Data Repository (OSDR) enables the reuse of comprehensive,
604
+ multi-modal space life science data—including omics, physiological, phenotypic, behavioral,
605
+ and environmental telemetry—to advance basic and applied research as well as operational
606
+ outcomes for human space exploration.
607
+
608
+ https://science.nasa.gov/biological-physical/data/osdr/
609
+ '''
610
+
611
+ OPEN_SKY = r'''The OpenSky Network consists of a multitude of sensors connected to the Internet by
612
+ volunteers, industrial supporters, and academic/governmental organizations. All collected
613
+ raw data is archived in a large historical database. The database is primarily used by
614
+ researchers from different areas to analyze and improve air traffic control technologies
615
+ and processes. The main technologies behind the OpenSky Network are the Automatic Dependent
616
+ Surveillance-Broadcast (ADS-B) and Mode S. These technologies provide detailed (live) aircraft
617
+ information over the publicly accessible 1090 MHz radio frequency channel.
618
+
619
+ https://opensky-network.org/data/api
620
+ '''
621
+
622
+ USGS_EARTHQUAKES = r'''The USGS Earthquake Hazards Program monitors, reports, and researches global
623
+ seismic activity to reduce losses and save lives. It operates the National Earthquake
624
+ Information Center (NEIC) and Advanced National Seismic System (ANSS) to detect magnitude,
625
+ location, and impacts, providing data for public safety, engineering, and hazard assessments.
626
+
627
+ https://earthquake.usgs.gov/fdsnws/event/1/
628
+ '''
629
+
630
+ USGS_WATER = r'''The USGS Water Resources Mission Area monitors, assesses, and conducts research on
631
+ the nation's water, providing data on streamflow, groundwater, water quality, and water use.
632
+ With over 1.9 million sites across all 50 states, they provide real-time information crucial
633
+ for water management, safety, and economic, environmental, and recreational decisions.
634
+
635
+ https://api.waterdata.usgs.gov/
636
+ '''
637
+
638
+ NOAA_TIDES_CURRENTS = r'''NOAA Tides & Currents, managed by the Center for Operational Oceanographic
639
+ Products and Services (CO-OPS), is the authoritative U.S. source for water level, tidal, and
640
+ oceanographic data. It offers real-time monitoring and predictions for over 3,000 stations,
641
+ crucial for navigation, safety, and coastal resilience.
642
+
643
+ https://tidesandcurrents.noaa.gov/web_services_info.html
644
+ '''
645
+
646
+ NOAA_CLIMATE_DATA = r'''NOAA climate data is provided primarily through the National Centers for
647
+ Environmental Information (NCEI), serving as the world's largest archive of atmospheric,
648
+ coastal, and geophysical data. It offers free access to historical weather records, 30-year
649
+ Climate Normals, and datasets on temperature, precipitation, and storms. Data is accessible
650
+ via the NCEI Climate Data Online (CDO) portal and NOAA Climate.gov maps and tools.
651
+
652
+ https://www.ncdc.noaa.gov/cdo-web/webservices/v2
653
+ '''
654
+
655
+ USGS_NATIONAL_MAP = r'''The USGS National Map is a premier collaborative program providing free,
656
+ accurate, and public domain geospatial data for the United States and its territories.
657
+ It offers topographic maps, 3D elevation data, and 8+ data layers (transportation,
658
+ hydrography, structures) via an online viewer, data downloads, and web services for public,
659
+ academic, and government use.
660
+
661
+ https://www.usgs.gov/programs/national-geospatial-program/national-map
662
+ '''
663
+
664
+ JUPYTER_NOTEBOOK = r'''Jupyter Notebook is an open-source, browser-based application used for
665
+ interactive computing, data analysis, and visualization. It allows users to create documents
666
+ containing live code (primarily Python), markdown text, equations, and plots in one place.
667
+ It is widely used in data science for exploratory analysis and data cleaning
668
+
669
+ https://reference.langchain.com/python/langchain-community/document_loaders/notebook/NotebookLoader
670
+ '''
671
+
672
+ AWS_BUCKET = r'''The fundamental container for data in Amazon Simple Storage Service (S3), a highly
673
+ scalable object storage service. These are logical containers used to store data, similar to
674
+ folders but operating on a flat structure rather than a traditional hierarchical file system.
675
+
676
+ https://reference.langchain.com/python/langchain-community/document_loaders/s3_directory/S3DirectoryLoader
677
+ '''
678
+
679
+ AWS_FILE = r'''High-performance, fully managed file system access to S3 object storage for analytics,
680
+ machine learning, and shared workloads, delivering ~1ms latency. It bridges object storage with
681
+ file-based tools, offering intelligent prefetching and NFS compatibility. AWS also offers
682
+ dedicated file services like Amazon EFS, FSx for Lustre, and FSx for Windows.
683
+
684
+ https://reference.langchain.com/python/langchain-community/document_loaders/s3_file/S3FileLoader
685
+ '''
686
+
687
+ GOOGLE_BUCKET = r'''Google Cloud Storage buckets are foundational, globally unique containers holding
688
+ data as objects (files) within Google Cloud, ranging up to 5 TB each. They offer high durability
689
+ and availability, featuring flexible storage classes (Standard, Nearline, Coldline, Archive)
690
+ tailored for various access frequencies, and are managed via IAM policies, console, or API.
691
+
692
+ https://reference.langchain.com/python/langchain-google-community/gcs_directory/GCSDirectoryLoader
693
+ '''
694
+
695
+ GOOGLE_FILE = r'''Highly scalable, secure, and durable storage solutions, primarily using Cloud Storage
696
+ for object storage (unstructured data) and Filestore for managed file shares (structured data).
697
+ It allows storing massive datasets, static website hosting, and data backups, with data organized
698
+ into buckets and accessed via APIs, featuring flexible, tiered pricing based on access frequency.
699
+
700
+ https://reference.langchain.com/python/langchain-google-community/gcs_file/GCSFileLoader
701
+ '''
702
+
703
+ GOOGLE_STT = r'''Google Speech-to-Text is a Google Cloud service that converts audio into text using
704
+ advanced AI models, supporting 125+ languages and dialects. It enables real-time streaming or
705
+ batch processing of audio files, commonly used in voice search, customer service, and transcription.
706
+ The service offers high accuracy, including specialized models for video, phone calls, and, as of 2026,
707
+ the advanced Chirp 3 model, which uses self-supervised learning for better accuracy across accents and languages.
708
+
709
+ https://reference.langchain.com/python/langchain-classic/document_loaders/google_speech_to_text
710
+ '''
711
+
712
+ ONEDRIVE = r'''Microsoft OneDrive is a cloud storage service that allows you to store, sync, and back up
713
+ files (photos, documents, etc.) online, making them accessible from any device, including Windows,
714
+ macOS, Android, and iOS. It integrates with Microsoft 365, enabling real-time collaboration,
715
+ file sharing, and automatic, secure backup, with 5GB of free storage for new users.
716
+
717
+ https://reference.langchain.com/python/langchain-community/document_loaders/onedrive/OneDriveLoader
718
+ '''
719
+
720
+ USGS_SCIENCE = r'''USGS ScienceBase is a collaborative digital repository and information
721
+ management platform used by the U.S. Geological Survey (USGS) to catalog, manage, and
722
+ release scientific data. It serves as a central hub for natural science data, offering tools
723
+ for data stewardship, discovery, and access via a searchable, open-source catalog. It is
724
+ heavily used for releasing public data products and hosting project-specific information.
725
+
726
+ https://www.usgs.gov/sciencebase-instructions-and-documentation/sciencebase-geospatial-services
727
+ '''
728
+
729
+ GROKIPEDIA = r'''Grokipedia is an AI-powered online encyclopedia launched in October 2025 by xAI,
730
+ the artificial intelligence company. It is designed as a direct, "truth-seeking"
731
+ competitor to Wikipedia, which Musk has frequently criticized as being biased.
732
+
733
+ https://github.com/AkeBoss-tech/grokipedia-api
734
+ '''
735
+
736
+ SOCRATA = r'''Socrata is a cloud-based Data-as-a-Service (DaaS) platform that specializes in data
737
+ publishing and visualization for government organizations. It provides tools for open data
738
+ portals, performance management, and data integration to make public data accessible.
739
+
740
+ https://dev.socrata.com/docs/endpoints.html
741
+ '''
742
+
743
+ HEALTH_DATA = r'''Health Data API is a digital interface that allows different software systems—
744
+ such as electronic health record (EHR) platforms, patient portals, and wearable devices—
745
+ to communicate and exchange medical information securely. These APIs are essential for
746
+ "interoperability," enabling a unified view of a patient’s health history across multiple
747
+ providers and services.
748
+
749
+ https://developers.google.com/health/reference/rest
750
+ '''
751
+
752
+ GLOBAL_HEALTH = r'''The Global Health Observatory (GHO) API is the World Health Organization
753
+ (WHO) primary gateway for accessing global health statistics programmatically. It allows
754
+ researchers and developers to retrieve data on over 1,000 health indicators across 194 Member States.
755
+
756
+ https://www.who.int/data/gho/info/gho-odata-api
757
+ '''
758
+
759
+ CENSUS_DATA = r'''The Census Bureau's Application Programming Interface (API) is a free data service
760
+ that allows developers and researchers to programmatically access over 1,600 datasets
761
+ containing raw statistical data on the U.S. population and economy
762
+
763
+ https://www.census.gov/data/developers/data-sets.html
764
+ '''
765
+
766
+ WORLD_POPULATION = r'''World Population APIs provide programmatic access to demographic data,
767
+ including total population, age, sex, and density, often broken down by region and time.
768
+
769
+ https://www.worldpop.org/sdi/introapi/
770
+ '''
771
+
772
+ WONDER = r'''The CDC WONDER API (Wide-ranging Online Data for Epidemiologic Research) is a web
773
+ service provided by the Centers for Disease Control and Prevention that allows for automated,
774
+ programmatic access to public health data. It enables developers and researchers to query
775
+ various CDC WONDER online databases and retrieve data directly into their own applications,
776
+ widgets, or analytical workflows.
777
+
778
+ https://wonder.cdc.gov/wonder/help/wonder-api.html
779
+ '''
780
+
781
+ AIR_NOW = r'''AirNow is the official U.S. government website and app providing real-time,
782
+ local air quality data and forecasts using the color-coded Air Quality Index (AQI).
783
+ It covers ozone and particle pollution ( and ) via a partnership of the EPA, NOAA, and
784
+ local agencies, offering a "Fire and Smoke Map" to monitor smoke impacts
785
+
786
+ https://www.airnow.gov/?
787
+ '''
788
+
789
+ PURPLE_AIR = r'''PurpleAir provides low-cost, real-time air quality monitors and a public,
790
+ crowdsourced map to measure, visualize, and share hyper-local,, particulate matter data.
791
+ Using laser counters, these sensors empower communities to track pollution, particularly
792
+ during wildfire events. Data is accessible via the PurpleAir Map and is used by researchers,
793
+ public agencies, and individuals worldwide.
794
+
795
+ https://api.purpleair.com/
796
+ '''
797
+
798
+ OPEN_AQ = r'''OpenAQ is a non-profit organization that aggregates, harmonizes, and shares open-source,
799
+ global air quality data to fight "air inequality". It provides real-time and historical
800
+ data—primarily on PM2.5, PM10, and other pollutants—from over 48,000 locations across 150 c
801
+ ountries. The platform empowers researchers, journalists, and communities to access, analyze,
802
+ and use air quality data through an open API, fostering collaboration to improve public
803
+ health and policy.
804
+
805
+ https://github.com/openaq/openaq-api
806
+ '''
807
+
808
+ # ------- AI DEFINITIONS --------------------------
809
+
810
+ GROK_AI = r'''Grok is a generative artificial intelligence chatbot developed by xAI, an AI company.
811
+ It is designed to be a "maximum truth-seeking" AI with a "rebellious streak" and a sense of humor,
812
+ intended to compete directly with other AI systems like OpenAI's ChatGPT and Google's Gemini
813
+ '''
814
+
815
+ CHATGPT_AI = r'''ChatGPT is an AI-powered conversational chatbot developed by OpenAI, designed to
816
+ understand and generate human-like text, code, and images. Based on Generative Pre-trained
817
+ Transformer (GPT) technology, it is trained on massive datasets to answer questions, write
818
+ content, and summarize information. It is accessible via web and mobile apps, with free and
819
+ paid "Plus" subscriptions.
820
+ '''
821
+
822
+ GEMINI_AI = r'''Google Gemini is a conversational generative AI chatbot and virtual assistant developed
823
+ by Google, formerly known as Bard. Powered by advanced Large Language Models (LLMs), it acts
824
+ as a personal, proactive assistant that integrates across Google apps (Gmail, Docs, Drive)
825
+ to generate text, images, and code
826
+ '''
827
+
828
+ MISTRAL_AI = r'''Mistral AI is a prominent French artificial intelligence startup founded in April 2023
829
+ by former researchers from Meta and Google DeepMind. Based in Paris, the company has rapidly
830
+ become a leading European AI firm, focusing on creating efficient, open-weight large language
831
+ models (LLMs) that rival established US-based tech giants like OpenAI and Anthropic. As of
832
+ late 2025, the company is valued at over €11.7 billion (approx. $14 billion).
833
+ '''
834
+
835
+ CLAUDE_AI = r'''Claude is an advanced AI assistant developed by Anthropic, designed to be helpful, safe,
836
+ and honest. Known for high-quality coding, summarizing, and reasoning abilities, Claude features a
837
+ large context window (up to 200,000 tokens) and models like Opus, Sonnet, and Haiku. It is used for
838
+ tasks like analyzing data, drafting content, and conversational AI.
839
+ '''
840
+
841
+ # -------- LOADER DEFINITIONS -------------------
842
+
843
+ TEXT_LOADER = r'''Provides LangChain's TextLoader functionality to parse plain-text files
844
+ into Document objects.
845
+
846
+ https://reference.langchain.com/python/langchain-community/document_loaders/text/TextLoader
847
+ '''
848
+
849
+ NLTK_LOADER = r'''The Natural Language Toolkit (NLTK) is a comprehensive, open-source Python library
850
+ used for symbolic and statistical Natural Language Processing (NLP). Developed originally at
851
+ the University of Pennsylvania by Steven Bird and Edward Loper, it has become a standard tool
852
+ in academia for teaching and research in computational linguistics.
853
+
854
+ https://www.nltk.org/
855
+ '''
856
+
857
+ HTML_LOADER = r'''Provides Langchain's UnstructuredHTMLLoader's functionality to parse HTML files
858
+ into Document objects. You can run the loader in one of two modes: "single" and "elements".
859
+ If you use "single" mode, the document will be returned as a single langchain Document object.
860
+ If you use "elements" mode, the unstructured library will split the document into elements
861
+ such as Title and NarrativeText. You can pass in additional unstructured kwargs after mode
862
+ to apply different unstructured settings.
863
+
864
+ https://reference.langchain.com/python/langchain-community/document_loaders/html/UnstructuredHTMLLoader
865
+ '''
866
+
867
+ WEB_CRAWLER = r'''Web fetching with optional Playwright-backed page rendering.
868
+ '''
869
+
870
+ WEB_LOADER = r'''Functionality to load all text from HTML webpages into
871
+ a document format that can be used downstream.
872
+
873
+ https://reference.langchain.com/python/langchain-community/document_loaders/web_base/WebBaseLoader
874
+ '''
875
+
876
+ GITHUB_LOADER = r'''The LangChain GitHub Loader is a suite of integrations designed to ingest data
877
+ from GitHub repositories into a format compatible with Large Language Models (LLMs).
878
+ These loaders are primarily used in Retrieval-Augmented Generation (RAG) pipelines to
879
+ allow AI agents to "chat" with codebases, analyze issues, or summarize pull requests.
880
+
881
+ https://reference.langchain.com/python/langchain-community/document_loaders/github/GithubFileLoader
882
+ '''
883
+
884
+ WIKIPEDIA_LOADER = r'''The LangChain Wikipedia Loader (WikipediaLoader) is a component designed to
885
+ fetch and convert Wikipedia pages into a standardized Document format for use in LLM applications.
886
+
887
+ https://reference.langchain.com/python/langchain-community/document_loaders/wikipedia/WikipediaLoader
888
+ '''
889
+
890
+ ARXIV_LOADER = r'''arXiv is a free distribution service and an open-access archive for nearly 2.4 million
891
+ scholarly articles in the fields of physics, mathematics, computer science, quantitative
892
+ biology, quantitative finance, statistics, electrical engineering and systems science, and
893
+ economics. Materials on this site are not peer-reviewed by arXiv.
894
+
895
+ https://reference.langchain.com/python/langchain-community/document_loaders/arxiv/ArxivLoader
896
+ '''
897
+
898
+ PDF_LOADER = r'''Public, SDK-oriented PDF loader with: Page-aware metadata, Two-stage chunking,
899
+ Configurable chunk profiles, Table isolation, Optional OCR fallback.
900
+
901
+ https://reference.langchain.com/python/langchain-community/document_loaders/pdf/PyPDFLoader
902
+ '''
903
+
904
+ EXCEL_LOADER = r'''Provides LangChain's UnstructuredExcelLoader functionality
905
+ to parse Excel spreadsheets into documents.
906
+
907
+ https://reference.langchain.com/python/langchain-community/document_loaders/excel/UnstructuredExcelLoader
908
+ '''
909
+
910
+ POWERPOINT_LOADER = r'''The UnstructuredPowerPointLoader (within LangChain) is a tool for parsing
911
+ Microsoft PowerPoint (.ppt/.pptx) files to extract text and metadata, enabling AI applications
912
+ to read and process presentations. It supports loading documents in "single" (full text) or
913
+ "elements" (chunked by title/narrative) modes, ideal for Retrieval Augmented Generation (RAG) tasks
914
+
915
+ https://reference.langchain.com/python/langchain-community/document_loaders/powerpoint/UnstructuredPowerPointLoader
916
+ '''
917
+
918
+ JSON_LOADER = r'''The LangChain JSONLoader is a specialized document loader used to transform JSON
919
+ and JSON Lines data into standardized LangChain Document objects. It is a critical component
920
+ for building applications like Retrieval-Augmented Generation (RAG) that need to process structured data.
921
+
922
+ https://reference.langchain.com/python/langchain-community/document_loaders/json_loader/JSONLoader
923
+ '''
924
+
925
+ MARKDOWN_LOADER = r'''LangChain's Markdown document loaders are specialized tools used to convert
926
+ Markdown files into standardized LangChain Document objects. These objects are then used
927
+ for downstream tasks like Retrieval Augmented Generation (RAG), embedding generation,
928
+ or semantic chunking.
929
+
930
+ https://reference.langchain.com/python/langchain-community/document_loaders/markdown/UnstructuredMarkdownLoader
931
+ '''
932
+
933
+ XML_LOADER = r'''The UnstructuredXMLLoader in LangChain is a specialized tool designed to load and
934
+ parse XML files into standardized LangChain Document objects. It leverages the Unstructured.io
935
+ library to extract text content and preserve document structure for use in downstream LLM
936
+ applications like RAG.
937
+
938
+ https://reference.langchain.com/python/langchain-community/document_loaders/xml/UnstructuredXMLLoader
939
+ '''
940
+
941
+ CSV_LOADER = r'''The LangChain CSVLoader is a standard utility within the langchain-community package
942
+ designed to transform structured CSV data into a list of standardized Document objects.
943
+ This process is the foundational step for integrating tabular data into LLM-powered workflows,
944
+ such as Retrieval Augmented Generation (RAG).
945
+
946
+ https://reference.langchain.com/python/langchain-community/document_loaders/csv_loader/CSVLoader
947
+ '''
948
+
949
+ WORD_LOADER = '''Works with both .docx and .doc files. You can run the loader in one of two modes:
950
+ "single" and "elements". If you use "single" mode, the document will be returned as a
951
+ single langchain Document object. If you use "elements" mode, the unstructured library will
952
+ split the document into elements such as Title and NarrativeText.
953
+
954
+ https://reference.langchain.com/python/langchain-community/document_loaders/word_document/UnstructuredWordDocumentLoader
955
+ '''
956
+
957
+ NOTEBOOK_LOADER = '''Loads .ipynb notebook files.
958
+
959
+ https://reference.langchain.com/python/langchain-community/document_loaders/notebook/NotebookLoader
960
+ '''
961
+
962
+ # -------- GENERATION PARAMETER DEFINITIONS -------------------
963
+
964
+ TEMPERATURE = r'''Optional. A number between 0 and 2. Higher values like 0.8 will make the output
965
+ more random, while lower values like 0.2 will make it more focused and deterministic'''
966
+
967
+ TOP_P = r'''Optional. The maximum cumulative probability of tokens to consider when sampling.
968
+ The model uses combined Top-k and Top-p (nucleus) sampling. Tokens are sorted based on
969
+ their assigned probabilities so that only the most likely tokens are considered.
970
+ Top-k sampling directly limits the maximum number of tokens to consider,
971
+ while Nucleus sampling limits the number of tokens based on the cumulative probability.'''
972
+
973
+ TOP_K = r'''Optional. The maximum number of tokens to consider when sampling. Gemini models use
974
+ Top-p (nucleus) sampling or a combination of Top-k and nucleus sampling. Top-k sampling considers
975
+ the set of topK most probable tokens. Models running with nucleus sampling don't allow topK setting.
976
+ Note: The default value varies by Model and is specified by theModel.top_p attribute returned
977
+ from the getModel function. An empty topK attribute indicates that the model doesn't apply
978
+ top-k sampling and doesn't allow setting topK on requests.'''
979
+
980
+ PRESENCE_PENALTY = r'''Optional. Presence penalty applied to the next token's logprobs
981
+ if the token has already been seen in the response. This penalty is binary on/off
982
+ and not dependant on the number of times the token is used (after the first).'''
983
+
984
+ FREQUENCY_PENALTY = r'''Optional. Frequency penalty applied to the next token's logprobs,
985
+ multiplied by the number of times each token has been seen in the respponse so far.
986
+ A positive penalty will discourage the use of tokens that have already been used,
987
+ proportional to the number of times the token has been used: The more a token is used,
988
+ the more difficult it is for the model to use that token again increasing
989
+ the vocabulary of responses.'''
990
+
991
+ MAX_OUTPUT_TOKENS = r'''Optional. The maximum number of tokens used in generating output content'''
992
+
993
+ ALLOWED_DOMAINS = r'''Optional. The allowed domains used in generating output content and
994
+ grounding of generated content to reduce halucinations.'''
995
+
996
+ STOP_SEQUENCE = r'''Optional. Up to 4 string sequences where the API will stop generating further tokens.'''
997
+
998
+ STORE = 'Optional. Whether to maintain state from turn to turn, preserving reasoning and tool context '
999
+
1000
+ STREAM = 'Optional. Whether to return the generated respose in asynchronous chunks'
1001
+
1002
+ TOOLS = '''Optional. An array of tools the model may call while generating a response. You can specify which
1003
+ tool to use by setting the tool_choice parameter. Used by the Reponses API
1004
+ and Reasoning models'''
1005
+
1006
+ INCLUDE = r'''Optional. Specifies additional output data to include in the model response enabling reasoning
1007
+ items to be used in multi-turn conversations when using the Responses API statelessly
1008
+ and Reasoning models.
1009
+ '''
1010
+
1011
+ REASONING = r'''Optional. Reasoning models introduce reasoning tokens in addition to input and output tokens.
1012
+ The models use these reasoning tokens to “think,” breaking down the prompt and
1013
+ considering multiple approaches to generating a response. After generating reasoning tokens,
1014
+ the model produces an answer as visible completion tokens and discards
1015
+ the reasoning tokens from its context. Used by the Reasoning models'''