foo-py 0.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- foo_py-0.1.1/LICENSE.txt +21 -0
- foo_py-0.1.1/PKG-INFO +596 -0
- foo_py-0.1.1/README.md +508 -0
- foo_py-0.1.1/agents.py +10690 -0
- foo_py-0.1.1/app.py +16234 -0
- foo_py-0.1.1/boogr/__init__.py +563 -0
- foo_py-0.1.1/boogr/default_icon.ico +0 -0
- foo_py-0.1.1/boogr/enums.py +381 -0
- foo_py-0.1.1/boogr/minion.py +175 -0
- foo_py-0.1.1/boogr/resources/ico/BooIcon.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/Booger.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/Save.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/adobe.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/atk.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/b.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/batch.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/black_sigma.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/boo.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/boogr.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/browse.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/chart.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/copy.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/csv.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/dataedit.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/doc.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/e_logo.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/error.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/euler_circle.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/excel.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/file_browse.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/filter.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/folder_browse.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/info.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/input.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/koolaid.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/machinelearning.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/message.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/pdf.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/pi.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/setting.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/sword_ninja.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/textfile.ico +0 -0
- foo_py-0.1.1/boogr/resources/ico/webcam.ico +0 -0
- foo_py-0.1.1/boogr/resources/img/atk.png +0 -0
- foo_py-0.1.1/boogr/resources/img/boogr.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/Authority.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/BOC.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/DERA.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/DWH.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/EMD.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/OAR.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/OECA.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/OGC.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/OMS.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/ORD.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/OW.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/RCRA.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/TSCA.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/WIFIA.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/access.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/add.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/adobe.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/airline.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/analytics.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/appropriation.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/atk.ico +0 -0
- foo_py-0.1.1/boogr/resources/img/button/atk.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/attachment.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/bfy.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/bluetooth.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/browse.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/budget.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/calculator.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/calendar.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/cancel.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/categoricalgrants.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/chart.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/chrome.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/close.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/columndelete.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/columnedit.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/columninsert.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/commandline.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/commute.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/compass.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/contracts.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/controlpanel.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/csv.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/database.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/databaseadd.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/databasedelete.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/databaserefresh.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/databasesql.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/databaseverify.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/datagrid.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/delete.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/division.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/document.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/documentadd.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/documentation.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/documentdelete.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/documentedit.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/documenterror.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/documentsearch.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/edge.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/edit.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/efy.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/environment.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/ev.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/excel.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/expenses.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/export.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/file.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/file_word.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/fileadd.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/filebrowse.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/filecopy.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/filedelete.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/fileedit.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/filereader.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/filesearch.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/filetransfer.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/fileverify.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/filewriter.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/filter.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/first.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/folder.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/folderbrowse.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/foldercompress.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/foldercopy.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/folderdownload.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/folderopen.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/fte.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/function.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/gmail.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/go.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/google.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/grants.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/guidance.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/home.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/id.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/image.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/import.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/information.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/internet.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/justice.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/last.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/ledger.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/left.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/levels.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/logout.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/lust.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/menu.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/metrics.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/mpg.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/next.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/no.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/oil.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/ok.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/omb.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/onenote.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/oust.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/outlay.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/outlook.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/pause.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/payroll.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/pdf.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/percentage.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/play.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/plusminus.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/previous.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/print.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/recertification.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/recycle.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/redo.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/refresh.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/remove.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/reserve.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/right.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/row.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/rowcopy.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/rowdelete.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/rowedit.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/rowinsert.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/save.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/scan.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/sharepoint.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/sigma.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/site.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/sitetravel.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/sort.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/spreadsheet.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/statistics.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/table.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/tableadd.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/tabledelete.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/tablesettings.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/text.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/traffic.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/travel.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/undelete.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/undo.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/wcf.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/windows.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/word.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/xml.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/yes.png +0 -0
- foo_py-0.1.1/boogr/resources/img/button/zipfile.png +0 -0
- foo_py-0.1.1/boogr/resources/img/gooey.png +0 -0
- foo_py-0.1.1/boogr/resources/img/web/google.png +0 -0
- foo_py-0.1.1/config.py +1015 -0
- foo_py-0.1.1/core.py +170 -0
- foo_py-0.1.1/data.py +832 -0
- foo_py-0.1.1/desktop.py +173 -0
- foo_py-0.1.1/embedders.py +269 -0
- foo_py-0.1.1/fetchers.py +25142 -0
- foo_py-0.1.1/foo_assets/__init__.py +1 -0
- foo_py-0.1.1/foo_assets/resources/images/favicon.ico +0 -0
- foo_py-0.1.1/foo_assets/resources/images/foo-apikeys.png +0 -0
- foo_py-0.1.1/foo_assets/resources/images/foo-architecture.png +0 -0
- foo_py-0.1.1/foo_assets/resources/images/foo-workflows.png +0 -0
- foo_py-0.1.1/foo_assets/resources/images/foo.ico +0 -0
- foo_py-0.1.1/foo_assets/resources/images/foo.svg +60 -0
- foo_py-0.1.1/foo_assets/resources/images/foo_logo.ico +0 -0
- foo_py-0.1.1/foo_assets/resources/images/foo_logo.png +0 -0
- foo_py-0.1.1/foo_assets/resources/images/foo_logo.svg +63 -0
- foo_py-0.1.1/foo_assets/resources/images/foo_portfolio.png +0 -0
- foo_py-0.1.1/foo_assets/resources/images/foo_project.png +0 -0
- foo_py-0.1.1/foo_assets/resources/images/system_diagram.svg +46 -0
- foo_py-0.1.1/foo_assets/resources/images/uml_class_diagram.svg +65 -0
- foo_py-0.1.1/foo_assets/streamlit_config.toml +85 -0
- foo_py-0.1.1/foo_cli.py +52 -0
- foo_py-0.1.1/foo_py.egg-info/PKG-INFO +596 -0
- foo_py-0.1.1/foo_py.egg-info/SOURCES.txt +247 -0
- foo_py-0.1.1/foo_py.egg-info/dependency_links.txt +1 -0
- foo_py-0.1.1/foo_py.egg-info/entry_points.txt +2 -0
- foo_py-0.1.1/foo_py.egg-info/requires.txt +77 -0
- foo_py-0.1.1/foo_py.egg-info/top_level.txt +18 -0
- foo_py-0.1.1/generators.py +3323 -0
- foo_py-0.1.1/loaders.py +4570 -0
- foo_py-0.1.1/models.py +344 -0
- foo_py-0.1.1/processors.py +3216 -0
- foo_py-0.1.1/pyproject.toml +32 -0
- foo_py-0.1.1/requirements.txt +86 -0
- foo_py-0.1.1/scrapers.py +706 -0
- foo_py-0.1.1/setup.cfg +4 -0
- foo_py-0.1.1/stores/vector.py +365 -0
- foo_py-0.1.1/tests/test_source_processing_contracts.py +96 -0
- foo_py-0.1.1/writers.py +268 -0
foo_py-0.1.1/LICENSE.txt
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2022 Terry D. Eppler
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
foo_py-0.1.1/PKG-INFO
ADDED
|
@@ -0,0 +1,596 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: foo-py
|
|
3
|
+
Version: 0.1.1
|
|
4
|
+
Summary: Foo Streamlit workspace for acquisition, analysis and AI workflows.
|
|
5
|
+
Author: Terry D. Eppler
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Repository, https://github.com/is-leeroy-jenkins/foo
|
|
8
|
+
Project-URL: Documentation, https://is-leeroy-jenkins.github.io/foo/
|
|
9
|
+
Requires-Python: >=3.11
|
|
10
|
+
Description-Content-Type: text/markdown
|
|
11
|
+
License-File: LICENSE.txt
|
|
12
|
+
Requires-Dist: langchain-core==1.2.20
|
|
13
|
+
Requires-Dist: langchain-community==0.4.1
|
|
14
|
+
Requires-Dist: langchain-text-splitters==1.1.1
|
|
15
|
+
Requires-Dist: langchain-google-community==3.0.5
|
|
16
|
+
Requires-Dist: langchain-googledrive==0.3.35
|
|
17
|
+
Requires-Dist: google-api-python-client
|
|
18
|
+
Requires-Dist: google-auth-httplib2
|
|
19
|
+
Requires-Dist: google-auth-oauthlib==1.3.0
|
|
20
|
+
Requires-Dist: langchain-openai==1.1.11
|
|
21
|
+
Requires-Dist: langchain-google-genai
|
|
22
|
+
Requires-Dist: langchain-mistralai
|
|
23
|
+
Requires-Dist: langchain-huggingface
|
|
24
|
+
Requires-Dist: langchain-chroma
|
|
25
|
+
Requires-Dist: langchain-pinecone
|
|
26
|
+
Requires-Dist: langchain-yt-dlp>=0.0.8
|
|
27
|
+
Requires-Dist: pydantic==2.12.5
|
|
28
|
+
Requires-Dist: pydantic-core==2.41.5
|
|
29
|
+
Requires-Dist: packaging<26,>=25.0
|
|
30
|
+
Requires-Dist: openai==2.29.0
|
|
31
|
+
Requires-Dist: anthropic<1,>=0.40
|
|
32
|
+
Requires-Dist: google-genai<2,>=1
|
|
33
|
+
Requires-Dist: xai-sdk==1.8.2
|
|
34
|
+
Requires-Dist: mistralai<2,>=1
|
|
35
|
+
Requires-Dist: googlemaps<5,>=4.10
|
|
36
|
+
Requires-Dist: llama-cpp-python
|
|
37
|
+
Requires-Dist: requests<3,>=2.31.0
|
|
38
|
+
Requires-Dist: urllib3<3,>=2.0.7
|
|
39
|
+
Requires-Dist: beautifulsoup4<5,>=4.12
|
|
40
|
+
Requires-Dist: lxml<6,>=5
|
|
41
|
+
Requires-Dist: numpy==1.26.4
|
|
42
|
+
Requires-Dist: pandas==2.2.2
|
|
43
|
+
Requires-Dist: matplotlib<4,>=3.8
|
|
44
|
+
Requires-Dist: plotly<6,>=5.24
|
|
45
|
+
Requires-Dist: pillow<12,>=10
|
|
46
|
+
Requires-Dist: nltk<4,>=3.9
|
|
47
|
+
Requires-Dist: textstat<1,>=0.7.4
|
|
48
|
+
Requires-Dist: arxiv<3,>=2
|
|
49
|
+
Requires-Dist: wikipedia<2,>=1.4
|
|
50
|
+
Requires-Dist: docx2txt<1,>=0.8
|
|
51
|
+
Requires-Dist: PyMuPDF==1.24.9
|
|
52
|
+
Requires-Dist: openpyxl==3.1.5
|
|
53
|
+
Requires-Dist: unstructured<0.17,>=0.16
|
|
54
|
+
Requires-Dist: extract-msg<1,>=0.48
|
|
55
|
+
Requires-Dist: python-pptx<2,>=1.0
|
|
56
|
+
Requires-Dist: rapidocr-onnxruntime<2,>=1.3
|
|
57
|
+
Requires-Dist: yt-dlp>=2025.1.0
|
|
58
|
+
Requires-Dist: astropy<8,>=6
|
|
59
|
+
Requires-Dist: astroquery<1,>=0.4.9
|
|
60
|
+
Requires-Dist: cartopy<1,>=0.23
|
|
61
|
+
Requires-Dist: owslib<1,>=0.32
|
|
62
|
+
Requires-Dist: sscws==2.4.7
|
|
63
|
+
Requires-Dist: grokipedia-api<1,>=0.1
|
|
64
|
+
Requires-Dist: crawl4ai<1,>=0.4
|
|
65
|
+
Requires-Dist: playwright<2,>=1.50
|
|
66
|
+
Requires-Dist: streamlit==1.55.0
|
|
67
|
+
Requires-Dist: streamlit-extras
|
|
68
|
+
Requires-Dist: streamlit-pdf
|
|
69
|
+
Requires-Dist: openai
|
|
70
|
+
Requires-Dist: gensim
|
|
71
|
+
Requires-Dist: python-docx
|
|
72
|
+
Requires-Dist: scikit-learn
|
|
73
|
+
Requires-Dist: spacy
|
|
74
|
+
Requires-Dist: sentence-transformers
|
|
75
|
+
Requires-Dist: textblob
|
|
76
|
+
Requires-Dist: tiktoken
|
|
77
|
+
Requires-Dist: chromadb
|
|
78
|
+
Requires-Dist: torchvision
|
|
79
|
+
Requires-Dist: mkdocs
|
|
80
|
+
Requires-Dist: mkdocs-material
|
|
81
|
+
Requires-Dist: mkdocstrings[python]
|
|
82
|
+
Requires-Dist: mkdocs-autorefs
|
|
83
|
+
Requires-Dist: pymdown-extensions
|
|
84
|
+
Requires-Dist: black
|
|
85
|
+
Requires-Dist: pyinstaller<7,>=6.10; sys_platform == "win32"
|
|
86
|
+
Requires-Dist: pywebview<7,>=5.4; sys_platform == "win32"
|
|
87
|
+
Dynamic: license-file
|
|
88
|
+
|
|
89
|
+
###### foo
|
|
90
|
+
|
|
91
|
+

|
|
92
|
+
<p align="left">
|
|
93
|
+
<a href="#-features">Features</a> ·
|
|
94
|
+
<a href="#-application-modes">Modes</a> ·
|
|
95
|
+
<a href="#%EF%B8%8F-architecture">Architecture</a> ·
|
|
96
|
+
<a href="#%EF%B8%8F-installation">Install</a> ·
|
|
97
|
+
<a href="docs/windows-installation.md">Windows Installer</a> ·
|
|
98
|
+
<a href="#%EF%B8%8F-running-the-streamlit-app">Run</a> ·
|
|
99
|
+
<a href="#-loaders">Loaders</a> ·
|
|
100
|
+
<a href="#%EF%B8%8F-scraping">Scraping</a> ·
|
|
101
|
+
<a href="#%EF%B8%8F-retrieval-sources">Retrievers</a> ·
|
|
102
|
+
<a href="#-domain-fetchers">Fetchers</a> ·
|
|
103
|
+
<a href="#-generation-providers">AI</a> ·
|
|
104
|
+
<a href="#-requirements">Requirements</a> ·
|
|
105
|
+
<a href="#-example-usage">Examples</a> ·
|
|
106
|
+
</p>
|
|
107
|
+
|
|
108
|
+
___
|
|
109
|
+
|
|
110
|
+
[](https://is-leeroy-jenkins.github.io/foo/)
|
|
111
|
+
|
|
112
|
+
Foo is a data loading, scraping, retrieval, geospatial, environmental,
|
|
113
|
+
astronomical, demographic, generative-AI, and data-processing workspace. It is designed
|
|
114
|
+
to give users explicit, hands-on control over how content is loaded, extracted, queried, fetched,
|
|
115
|
+
cleaned, analyzed, visualized, and routed into downstream machine-learning or agentic workflows.
|
|
116
|
+
|
|
117
|
+
Foo is modular by design. Loaders, scrapers, fetchers, generators, databases, and external APIs can
|
|
118
|
+
operate independently while remaining composable inside one cohesive interface. The application
|
|
119
|
+
supports local files, web pages, public archives, Google services, government data sources,
|
|
120
|
+
geospatial APIs, environmental APIs, astronomical APIs, demographic APIs, and multiple LLM
|
|
121
|
+
providers.
|
|
122
|
+
|
|
123
|
+
## 🎥 Demo
|
|
124
|
+
|
|
125
|
+

|
|
126
|
+
___
|
|
127
|
+
|
|
128
|
+
## ☁️ Cloud
|
|
129
|
+
|
|
130
|
+
<table>
|
|
131
|
+
<tr>
|
|
132
|
+
<td align="center">
|
|
133
|
+
<img width="190" height="1" alt=""><br>
|
|
134
|
+
<a href="https://chonky.nicehill-0e7bfe90.centralus.azurecontainerapps.io">
|
|
135
|
+
<img src="https://img.shields.io/badge/Docker-App-2496ED?logo=docker&logoColor=white" alt="Docker App">
|
|
136
|
+
</a>
|
|
137
|
+
</td>
|
|
138
|
+
|
|
139
|
+
<td align="center">
|
|
140
|
+
<img width="190" height="1" alt=""><br>
|
|
141
|
+
<a href="https://fooo-py.streamlit.app/">
|
|
142
|
+
<img src="https://img.shields.io/badge/Streamlit-App-FF4B4B?logo=streamlit&logoColor=white" alt="Streamlit App">
|
|
143
|
+
</a>
|
|
144
|
+
</td>
|
|
145
|
+
|
|
146
|
+
<td align="center">
|
|
147
|
+
<img width="190" height="1" alt=""><br>
|
|
148
|
+
<a href="https://drive.google.com/file/d/1q3ZhnGBGaJjgfiiRirOrONHv2wF8_CtH/view?usp=sharing">
|
|
149
|
+
<img src="https://colab.research.google.com/assets/colab-badge.svg" alt="Open In Colab">
|
|
150
|
+
</a>
|
|
151
|
+
</td>
|
|
152
|
+
|
|
153
|
+
<td align="center">
|
|
154
|
+
<img width="190" height="1" alt=""><br>
|
|
155
|
+
<a href="https://leeroy.usw-16.palantirfoundry.com/shares/links/eitihztkzx76q">
|
|
156
|
+
<img src="https://img.shields.io/badge/Databricks%20Repo-Foo--Py-FF3621?logo=databricks&logoColor=white" alt="Databricks Notebook">
|
|
157
|
+
</a>
|
|
158
|
+
</td>
|
|
159
|
+
|
|
160
|
+
<td align="center">
|
|
161
|
+
<img width="190" height="1" alt=""><br>
|
|
162
|
+
<a href="https://leeroy.usw-16.palantirfoundry.com/shares/links/r7ukk3ybt65bk">
|
|
163
|
+
<img src="https://img.shields.io/badge/Palantir%20Foundry-Repo-101113?logo=palantir&logoColor=white" alt="Palantir Repo">
|
|
164
|
+
</a>
|
|
165
|
+
</td>
|
|
166
|
+
</tr>
|
|
167
|
+
</table>
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
## ✨ Features
|
|
171
|
+
|
|
172
|
+
| Capability | Description |
|
|
173
|
+
| --------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
174
|
+
| Modular document loading | Load text, CSV, XML, PDF, Markdown, HTML, JSON, PowerPoint, Excel, arXiv, Wikipedia, GitHub, web pages, crawled websites, notebooks, and cloud files. |
|
|
175
|
+
| Web scraping | Extract titles, plain text, raw HTML, headings, paragraphs, lists, tables, articles, sections, divisions, blockquotes, hyperlinks, and image references. |
|
|
176
|
+
| Public retrieval | Query arXiv, Google Drive, Wikipedia, Google Custom Search, NASA Open Science, GovInfo, Congress.gov, Internet Archive, Grokipedia, Jupyter notebooks, cloud files, and cloud buckets. |
|
|
177
|
+
| Geospatial workflows | Query geocoding, Google Maps, Google Weather, OpenWeather, historical weather, USGS earthquakes, NASA Earth Observatory, USGS National Map, USGS ScienceBase, and OpenSky. |
|
|
178
|
+
| Environmental workflows | Query AirNow, NOAA Climate Data, NASA EONET, EPA EnviroFacts, NOAA Tides and Currents, EPA UV Index, PurpleAir, OpenAQ, NASA FIRMS, and USGS Water Data. |
|
|
179
|
+
| Astronomical workflows | Query U.S. Naval Observatory, Satellite Center, Astro Catalog, AstroQuery, StarMap, SIMBAD, Space Weather, Star Chart, and near-Earth object data. |
|
|
180
|
+
| Demographic and health data | Query U.S. Census, CDC Socrata, U.S. HealthData, WHO Global, United Nations, World Population, CDC WONDER, PubMed, and Open City Data. |
|
|
181
|
+
| Generative AI | Use ChatGPT, Grok, Claude, Gemini, and Mistral through a shared prompt and parameter interface. |
|
|
182
|
+
| SQLite management | Import Excel workbooks, browse tables, perform CRUD operations, filter, aggregate, visualize, alter schema, and run read-only SQL. |
|
|
183
|
+
| Text analytics | Compute token counts, vocabulary, type-token ratio, hapax ratio, stopword ratio, lexical density, top tokens, and optional readability metrics. |
|
|
184
|
+
| Document processing | Recursively split loaded or retrieved documents, word-tokenize each chunk, generate embeddings, and store vectors without replacing source-specific result displays. |
|
|
185
|
+
| Vector retrieval | Connect to Chroma or Pinecone for non-destructive writes, filtered similarity search, scored retrieval, deletion, counts, health checks, and LangChain retrievers. |
|
|
186
|
+
|
|
187
|
+
## 🕹️ Application Modes
|
|
188
|
+
|
|
189
|
+
| Mode | Purpose | Major Components |
|
|
190
|
+
| ------------------- | -------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
191
|
+
| **Loading** | Load local, web, corpus, repository, and cloud documents into shared document/session state. | Text, NLTK corpora, CSV, XML, PDF, Markdown, HTML, JSON, PowerPoint, Excel, arXiv, Wikipedia, GitHub, Web Loader, Web Crawler. |
|
|
192
|
+
| **Scraping** | Scrape a target URL or recursively crawl pages and extract structured web content. | Page title, basic text, raw HTML, headings, paragraphs, lists, tables, articles, sections, divisions, blockquotes, hyperlinks, images. |
|
|
193
|
+
| **Retrieval** | Query public collections, archives, search services, cloud files, and cloud buckets. | arXiv, Google Drive, Wikipedia, Google Search, NASA Open Science, GovInfo, U.S. Congress, Internet Archive, Grokipedia, Jupyter Notebook, Google Cloud File, AWS S3 File, OneDrive, Google Speech-to-Text, AWS S3 Bucket, Google Cloud Bucket. |
|
|
194
|
+
| **Geospatial** | Retrieve location, weather, map, flight, and earth-science data. | Geocoding, Google Maps, Google Weather, OpenWeather, Historical Weather, USGS Earthquakes, NASA Earth Observatory, USGS National Map, USGS ScienceBase, OpenSky. |
|
|
195
|
+
| **Environmental** | Retrieve environmental, climate, water, fire, air-quality, UV, and sensor data. | AirNow, NOAA Climate Data, NASA EONET, EPA EnviroFacts, NOAA Tides and Currents, EPA UV Index, PurpleAir, OpenAQ, NASA FIRMS, USGS Water Data. |
|
|
196
|
+
| **Astronomical** | Retrieve astronomical, satellite, star, space-weather, and near-Earth object data. | U.S. Naval Observatory, Satellite Center, Astro Catalog, AstroQuery, StarMap, SIMBAD, Space Weather, Star Chart, Near-Earth Objects. |
|
|
197
|
+
| **Demographic** | Retrieve demographic, health, population, city, and public-health records. | U.S. Census, CDC Socrata, U.S. Health, WHO Global, United Nations, World Population, CDC WONDER, PubMed Search, Open City Data. |
|
|
198
|
+
| **Generation** | Generate or analyze text using multiple AI providers. | ChatGPT, Grok, Claude, Gemini, Mistral. |
|
|
199
|
+
|
|
200
|
+
## 🏛️ Architecture
|
|
201
|
+
|
|
202
|
+
```text
|
|
203
|
+
📥 Loader → 🧹 Text Processing → 🧠 Session State → 🔍 Retrieval / Analysis / Generation
|
|
204
|
+
│ │ │
|
|
205
|
+
├── 🕸️ Scraper ├── 📊 Metrics ├── 🗄️ SQLite
|
|
206
|
+
├── 🌐 Fetcher ├── 🧩 Documents ├── 📈 Visualization
|
|
207
|
+
└── ☁️ Cloud └── 📝 Raw Text └── 🤖 LLM Providers
|
|
208
|
+
```
|
|
209
|
+
|
|
210
|
+
Foo uses a Streamlit UI over modular Python classes. The application imports loader classes from
|
|
211
|
+
`loaders.py`, provider classes from `generators.py`, and API/data-source wrappers from `fetchers.py`.
|
|
212
|
+
Shared working state is coordinated through `st.session_state`, allowing loaded documents, raw text,
|
|
213
|
+
processed text, tokens, metrics, and database results to flow between controls.
|
|
214
|
+
|
|
215
|
+

|
|
216
|
+
|
|
217
|
+
___
|
|
218
|
+
|
|
219
|
+
## 🗂️ Directory Structure
|
|
220
|
+
|
|
221
|
+
```text
|
|
222
|
+
foo/
|
|
223
|
+
├── app.py # Streamlit user interface
|
|
224
|
+
├── config.py # App title, mode map, defaults, labels, API references, and constants
|
|
225
|
+
├── core.py # Optional package core abstractions
|
|
226
|
+
├── data.py # Data helpers and shared data abstractions
|
|
227
|
+
├── embedders.py # Hosted and local LangChain embedding implementations
|
|
228
|
+
├── fetchers.py # External API, archive, geospatial, environmental, and science fetchers
|
|
229
|
+
├── generators.py # ChatGPT, Claude, Grok, Mistral, and Gemini wrappers
|
|
230
|
+
├── loaders.py # File, web, cloud, repository, and corpus loaders
|
|
231
|
+
├── scrapers.py # HTML extraction helpers
|
|
232
|
+
├── requirements.txt # Python dependencies
|
|
233
|
+
├── stores/
|
|
234
|
+
│ ├── vector.py # Chroma and Pinecone lifecycle implementations
|
|
235
|
+
│ └── sqlite/ # SQLite database storage
|
|
236
|
+
└── resources/
|
|
237
|
+
└── images/ # README and UI image assets
|
|
238
|
+
```
|
|
239
|
+
## 🔑 API Set-up
|
|
240
|
+
|
|
241
|
+
- [Science APIs](https://github.com/is-leeroy-jenkins/foo/blob/main/resources/setup/API-Setup.md)
|
|
242
|
+
- [OpenAI](https://github.com/is-leeroy-jenkins/foo/blob/main/resources/setup/environments.md)
|
|
243
|
+
- [Gemini AI](https://github.com/is-leeroy-jenkins/foo/blob/main/resources/setup/gemini.md)
|
|
244
|
+
- [Grok AI](https://github.com/is-leeroy-jenkins/foo/blob/main/resources/setup/xai.md)
|
|
245
|
+
- [Mistral AI](https://github.com/is-leeroy-jenkins/foo/blob/main/resources/setup/mistral.md)
|
|
246
|
+
- [Claude AI](https://github.com/is-leeroy-jenkins/foo/blob/main/resources/setup/claude.md)
|
|
247
|
+
|
|
248
|
+
## 🛡️ Installation
|
|
249
|
+
|
|
250
|
+
```bash
|
|
251
|
+
git clone https://github.com/is-leeroy-jenkins/Foo.git
|
|
252
|
+
cd Foo
|
|
253
|
+
python -m venv .venv
|
|
254
|
+
.venv\Scripts\Activate.ps1
|
|
255
|
+
python -m pip install --upgrade pip
|
|
256
|
+
python -m pip install -r requirements.txt
|
|
257
|
+
```
|
|
258
|
+
|
|
259
|
+
For Linux or macOS:
|
|
260
|
+
|
|
261
|
+
```bash
|
|
262
|
+
git clone https://github.com/is-leeroy-jenkins/Foo.git
|
|
263
|
+
cd Foo
|
|
264
|
+
python3 -m venv .venv
|
|
265
|
+
source .venv/bin/activate
|
|
266
|
+
python -m pip install --upgrade pip
|
|
267
|
+
python -m pip install -r requirements.txt
|
|
268
|
+
```
|
|
269
|
+
|
|
270
|
+
## 🪟 Windows Desktop Installer
|
|
271
|
+
|
|
272
|
+
Foo can be packaged as a **Windows desktop application** using **PyInstaller (`onedir`)** and **Inno Setup**. The installed `Foo.exe` runs Streamlit on the local computer and displays the interface in a dedicated **Microsoft Edge WebView2 window**, without requiring the user to open a browser tab. The installer provides Start Menu shortcuts, an optional desktop shortcut, and an uninstaller. Writable application data is intended to live under `%LOCALAPPDATA%\Foo`.
|
|
273
|
+
|
|
274
|
+
To build an installer, run the [Windows desktop installer workflow](.github/workflows/build-windows-installer.yml) from GitHub Actions, then download its `Foo-Windows-Installer` artifact after a successful run. Windows builds can also be created locally using the [PyInstaller specification](foo.spec) and [Inno Setup script](desktop/foo.iss). Browser runtimes, models, and some native dependencies require separate validation before distribution.
|
|
275
|
+
|
|
276
|
+
**[Windows installation guide](docs/windows-installation.md)** · [Desktop launcher](desktop/launcher.py) · [Installer script](desktop/foo.iss) · [Build workflow](.github/workflows/build-windows-installer.yml) · [Desktop build notes](desktop/README.md)
|
|
277
|
+
|
|
278
|
+
## ▶️ Running the Streamlit App
|
|
279
|
+
|
|
280
|
+
```bash
|
|
281
|
+
streamlit run app.py
|
|
282
|
+
```
|
|
283
|
+
|
|
284
|
+
The application initializes Streamlit in wide layout, loads the configured mode map from `config.py`,
|
|
285
|
+
and displays the active mode selector in the sidebar under **🕹️ Mode**.
|
|
286
|
+
|
|
287
|
+
## 🚀 Quick Start
|
|
288
|
+
|
|
289
|
+
### Run the Application
|
|
290
|
+
|
|
291
|
+
```bash
|
|
292
|
+
streamlit run app.py
|
|
293
|
+
```
|
|
294
|
+
|
|
295
|
+
### Load a Document
|
|
296
|
+
|
|
297
|
+
1. Open **Loading** mode.
|
|
298
|
+
2. Expand a loader such as **PDF Loader**, **Excel Loader**, **Web Loader**, or **GitHub Loader**.
|
|
299
|
+
3. Select or enter the source.
|
|
300
|
+
4. Click **Load**.
|
|
301
|
+
5. Configure recursive chunking, embedding, and vector storage in the same source expander.
|
|
302
|
+
6. Review the **Document**, **Chunks**, and **Embeddings** tabs. Chunking occurs first, followed by
|
|
303
|
+
word tokenization of every resulting chunk; embeddings use the original chunk text.
|
|
304
|
+
|
|
305
|
+
### Scrape a Web Page
|
|
306
|
+
|
|
307
|
+
1. Open **Scraping** mode.
|
|
308
|
+
2. Enter a target URL.
|
|
309
|
+
3. Select core output and structured extraction options.
|
|
310
|
+
4. Optionally enable recursive crawl controls.
|
|
311
|
+
5. Click **Run Scraper**.
|
|
312
|
+
6. Use the adjoining processing controls and right-side tabs to inspect, chunk, tokenize, embed,
|
|
313
|
+
and store the scraped documents.
|
|
314
|
+
|
|
315
|
+
### Query a Public Source
|
|
316
|
+
|
|
317
|
+
1. Open **Retrieval** mode.
|
|
318
|
+
2. Expand a source such as **ArXiv**, **Google Search**, **Gov Info**, or **US Congress**.
|
|
319
|
+
3. Enter the query and parameters.
|
|
320
|
+
4. Click **Submit**.
|
|
321
|
+
5. Review rendered summaries, rows, and raw results.
|
|
322
|
+
|
|
323
|
+
Retrieval, Geospatial, Environmental, Astronomical, and Demographic source expanders use the same
|
|
324
|
+
processing workflow while retaining their provider-specific maps, tables, images, metrics, and raw
|
|
325
|
+
results.
|
|
326
|
+
|
|
327
|
+

|
|
328
|
+
|
|
329
|
+
___
|
|
330
|
+
|
|
331
|
+
## 📤 Loaders
|
|
332
|
+
|
|
333
|
+
| Loader | Input | Purpose |
|
|
334
|
+
| --------------------- | ---------------------------------------------------- | ---------------------------------------------------------------------------------------------- |
|
|
335
|
+
| **Text Loader** | `.txt` files | Loads plain text into document/session state. |
|
|
336
|
+
| **Corpora Loader** | NLTK corpora or local text directory | Loads Brown, Gutenberg, Reuters, WebText, Inaugural, State of the Union, or local text files. |
|
|
337
|
+
| **CSV Loader** | `.csv` files | Loads delimited tabular text as documents. |
|
|
338
|
+
| **XML Loader** | `.xml` files | Supports semantic XML loading, document splitting, structured tree loading, and XPath queries. |
|
|
339
|
+
| **PDF Loader** | `.pdf` files | Loads PDF content in single or element mode, with plain or OCR extraction options. |
|
|
340
|
+
| **Markdown Loader** | `.md`, `.markdown` files | Loads Markdown content into document state. |
|
|
341
|
+
| **HTML Loader** | `.html`, `.htm` files | Loads local HTML files. |
|
|
342
|
+
| **JSON Loader** | `.json` files | Loads JSON or JSON Lines. |
|
|
343
|
+
| **PowerPoint Loader** | `.pptx` files | Loads PowerPoint slide content. |
|
|
344
|
+
| **Excel Loader** | `.xlsx`, `.xls` files | Loads Excel sheets and stores sheet data in SQLite tables. |
|
|
345
|
+
| **ArXiv Loader** | Query text | Retrieves arXiv documents. |
|
|
346
|
+
| **Wikipedia Loader** | Query text | Retrieves Wikipedia documents. |
|
|
347
|
+
| **GitHub Loader** | GitHub API URL, repository, branch, file-type filter | Loads repository files matching the selected filter. |
|
|
348
|
+
| **Web Loader** | One or more URLs | Loads web documents. |
|
|
349
|
+
| **Web Crawler** | Start URL | Recursively crawls web pages with depth/domain controls. |
|
|
350
|
+
|
|
351
|
+
## 🕸️ Scraping
|
|
352
|
+
|
|
353
|
+
| Output Category | Supported Extraction |
|
|
354
|
+
| ------------------------- | ---------------------------------------------------------------------------------------------- |
|
|
355
|
+
| Core output | Page title, basic text, raw HTML. |
|
|
356
|
+
| Structured text | Headings, paragraphs, lists, tables, articles, sections, divisions, blockquotes. |
|
|
357
|
+
| Link and media extraction | Hyperlinks and images. |
|
|
358
|
+
| Crawl controls | Recursive crawl, max depth, max pages, same-domain-only filtering. |
|
|
359
|
+
| Results | Per-page metadata, plain text, raw HTML, discovered links, extracted records, and error lists. |
|
|
360
|
+
|
|
361
|
+
## 🏛️ Retrieval Sources
|
|
362
|
+
|
|
363
|
+
| Source | Purpose |
|
|
364
|
+
| ------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
365
|
+
| **ArXiv** | Retrieve research documents by query or identifier. |
|
|
366
|
+
| **Google Drive** | Retrieve documents or snippets from Google Drive. |
|
|
367
|
+
| **Wikipedia** | Retrieve Wikipedia article content and metadata. |
|
|
368
|
+
| **Google Search** | Use Google Custom Search with exact terms, exclusions, file type, date restriction, site search, image search, country, language, and safe-search controls. |
|
|
369
|
+
| **Open Science** | Query NASA Open Science / OSDR dataset, metadata, assays, and data endpoints. |
|
|
370
|
+
| **Gov Info** | Search GovInfo, retrieve package summaries, or browse collections. |
|
|
371
|
+
| **US Congress** | Query Congress.gov congresses, bills, bill details, laws, law details, reports, and report details. |
|
|
372
|
+
| **Internet Archive** | Search archived media and text collections. |
|
|
373
|
+
| **Grokipedia** | Retrieve Grokipedia pages or search results. |
|
|
374
|
+
| **Jupyter Notebook** | Load notebook content. |
|
|
375
|
+
| **Google Cloud File** | Load a single Google Cloud file. |
|
|
376
|
+
| **AWS S3 File** | Load a single AWS S3 file. |
|
|
377
|
+
| **OneDrive** | Load OneDrive-hosted documents. |
|
|
378
|
+
| **Google Speech-to-Text** | Transcribe audio using Google Speech-to-Text. |
|
|
379
|
+
| **AWS S3 Bucket** | Load records from an S3 bucket. |
|
|
380
|
+
| **Google Cloud Bucket** | Load records from a Google Cloud bucket. |
|
|
381
|
+
|
|
382
|
+
## 🌎 Domain Fetchers
|
|
383
|
+
|
|
384
|
+
### Geospatial
|
|
385
|
+
|
|
386
|
+
| Fetcher | Purpose |
|
|
387
|
+
| -------------------------- | -------------------------------------------------------------------------------- |
|
|
388
|
+
| **Geocoding** | Resolve address/location text into coordinates and normalized location metadata. |
|
|
389
|
+
| **Google Maps** | Query Google Maps functionality such as place/location operations. |
|
|
390
|
+
| **Google Weather** | Retrieve Google Weather data. |
|
|
391
|
+
| **Open Weather** | Retrieve OpenWeather/Open-Meteo style weather data. |
|
|
392
|
+
| **Historical Weather** | Retrieve historical weather data. |
|
|
393
|
+
| **USGS Earthquakes** | Retrieve earthquake events and feature records. |
|
|
394
|
+
| **NASA Earth Observatory** | Retrieve NASA Earth Observatory content. |
|
|
395
|
+
| **The National Map** | Retrieve USGS National Map results. |
|
|
396
|
+
| **USGS ScienceBase** | Retrieve ScienceBase records. |
|
|
397
|
+
| **OpenSky** | Retrieve aviation/open-sky records. |
|
|
398
|
+
|
|
399
|
+
### Environmental
|
|
400
|
+
|
|
401
|
+
| Fetcher | Purpose |
|
|
402
|
+
| --------------------------- | --------------------------------------------------------- |
|
|
403
|
+
| **AirNow** | Retrieve air-quality observations and forecasts. |
|
|
404
|
+
| **NOAA Climate Data** | Retrieve climate data. |
|
|
405
|
+
| **NASA EONET** | Retrieve natural event records. |
|
|
406
|
+
| **EPA EnviroFacts** | Retrieve EPA environmental facility or data records. |
|
|
407
|
+
| **NOAA Tides and Currents** | Retrieve tides, currents, stations, and water-level data. |
|
|
408
|
+
| **EPA UV Index** | Retrieve UV index information. |
|
|
409
|
+
| **PurpleAir** | Retrieve PurpleAir sensor data. |
|
|
410
|
+
| **OpenAQ** | Retrieve open air-quality data. |
|
|
411
|
+
| **NASA FIRMS** | Retrieve fire/hotspot data. |
|
|
412
|
+
| **USGS Water Data** | Retrieve USGS water data. |
|
|
413
|
+
|
|
414
|
+
### Astronomical
|
|
415
|
+
|
|
416
|
+
| Fetcher | Purpose |
|
|
417
|
+
| ------------------------ | ----------------------------------------------------------------- |
|
|
418
|
+
| **US Naval Observatory** | Retrieve celestial navigation/time data for observer coordinates. |
|
|
419
|
+
| **Satellite Center** | Retrieve satellite or ground station data. |
|
|
420
|
+
| **Astro Catalog** | Retrieve astronomical catalog data. |
|
|
421
|
+
| **AstroQuery** | Query astronomical services. |
|
|
422
|
+
| **Star Map** | Generate or retrieve star map data. |
|
|
423
|
+
| **SIMBAD** | Query SIMBAD astronomical objects. |
|
|
424
|
+
| **Space Weather** | Retrieve space weather data. |
|
|
425
|
+
| **Star Chart** | Generate or retrieve star chart information. |
|
|
426
|
+
| **Near Earth Objects** | Retrieve near-Earth object or related object data. |
|
|
427
|
+
|
|
428
|
+
### Demographic and Health
|
|
429
|
+
|
|
430
|
+
| Fetcher | Purpose |
|
|
431
|
+
| ---------------------- | --------------------------------------------------------- |
|
|
432
|
+
| **U.S. Census Bureau** | Retrieve Census records. |
|
|
433
|
+
| **CDC Socrata** | Retrieve CDC Socrata datasets. |
|
|
434
|
+
| **U.S. Health** | Retrieve HealthData.gov or similar public health records. |
|
|
435
|
+
| **WHO Global** | Retrieve WHO Global Health Observatory data. |
|
|
436
|
+
| **United Nations** | Retrieve United Nations data. |
|
|
437
|
+
| **World Population** | Retrieve world population datasets. |
|
|
438
|
+
| **CDC WONDER** | Retrieve CDC WONDER data. |
|
|
439
|
+
| **PubMed Search** | Search PubMed records. |
|
|
440
|
+
| **Open City Data** | Retrieve city/open-data records. |
|
|
441
|
+
|
|
442
|
+
## 🤖 Generation Providers
|
|
443
|
+
|
|
444
|
+
| Provider | Mode Expander | Purpose |
|
|
445
|
+
| --------- | ------------- | -------------------------------------------------------------- |
|
|
446
|
+
| OpenAI | **ChatGPT** | General text generation and analysis through the Chat wrapper. |
|
|
447
|
+
| xAI | **Grok** | Text generation and analysis through the Grok wrapper. |
|
|
448
|
+
| Anthropic | **Claude** | Text generation and analysis through the Claude wrapper. |
|
|
449
|
+
| Google | **Gemini** | Text generation and analysis through the Gemini wrapper. |
|
|
450
|
+
| Mistral | **Mistral** | Text generation and analysis through the Mistral wrapper. |
|
|
451
|
+
|
|
452
|
+
## 📦 Requirements
|
|
453
|
+
|
|
454
|
+
The table below reflects the requirements implied by the active imports, loaders, fetchers, and UI
|
|
455
|
+
surface in `app.py`. Some provider-specific loaders/fetchers may require additional credentials or
|
|
456
|
+
cloud SDKs depending on deployment.
|
|
457
|
+
|
|
458
|
+
| Requirement | Import / Package Name | Purpose | Required By |
|
|
459
|
+
| ------------------------ | ------------------------------------- | --------------------------------------------------------------------- | ------------------------------------------------------- |
|
|
460
|
+
| Python | `python>=3.10` | Runtime for modern typing syntax and Streamlit application execution. | Entire application. |
|
|
461
|
+
| Streamlit | `streamlit` | Web application framework. | UI, sidebar, modes, expanders, controls, session state. |
|
|
462
|
+
| Altair | `altair` | Declarative charting support. | Visualization and chart-compatible workflows. |
|
|
463
|
+
| Pandas | `pandas` | Dataframes, Excel ingestion, SQL result rendering, tabular previews. | Loaders, Data Management, result tables. |
|
|
464
|
+
| NumPy | `numpy` | Numeric arrays and vector calculations. | Text/vector utilities and analysis helpers. |
|
|
465
|
+
| Plotly | `plotly` | Interactive charts and visualizations. | Data Management visualization engine. |
|
|
466
|
+
| BeautifulSoup | `beautifulsoup4` | HTML parsing and link/text extraction. | Scraping mode and HTML preview helpers. |
|
|
467
|
+
| Requests | `requests` | HTTP request support. | Web fetchers and API wrappers. |
|
|
468
|
+
| Crawl4AI | `crawl4ai` | JavaScript-capable or enhanced crawling support. | Web crawling workflows. |
|
|
469
|
+
| LangChain Core | `langchain-core` | `Document` object model for loaded/retrieved records. | Loaders and retrieval result handling. |
|
|
470
|
+
| LXML | `lxml` | XML parsing and XPath operations. | XML Loader. |
|
|
471
|
+
| NLTK | `nltk` | Tokenization, stopwords, WordNet, corpora, text metrics. | Loading metrics and Corpora Loader. |
|
|
472
|
+
| TextStat | `textstat` | Optional readability metrics. | Readability panel. |
|
|
473
|
+
| Astroquery | `astroquery` | Astronomical service access, including SIMBAD. | Astronomical mode. |
|
|
474
|
+
| SQLite | `sqlite3` | Local database storage and SQL execution. | Data Management and local stores. |
|
|
475
|
+
| OpenPyXL | `openpyxl` | Excel `.xlsx` read/write engine. | Excel Loader and Data Management import. |
|
|
476
|
+
| Python PPTX | `python-pptx` | PowerPoint text extraction support. | PowerPoint Loader. |
|
|
477
|
+
| PyMuPDF | `PyMuPDF` | PDF extraction support where used by PDF loader internals. | PDF Loader. |
|
|
478
|
+
| Unstructured | `unstructured` | Optional document extraction for complex files. | PDF/XML/document loader implementations. |
|
|
479
|
+
| Python DOCX / Docx2Txt | `python-docx` / `docx2txt` | Word document extraction support. | WordLoader. |
|
|
480
|
+
| Boto3 | `boto3` | AWS S3 file and bucket access. | AWS S3 File and AWS S3 Bucket loaders. |
|
|
481
|
+
| Google API Client | `google-api-python-client` | Google Drive and Google API access. | Google Drive and cloud workflows. |
|
|
482
|
+
| Google Auth | `google-auth`, `google-auth-oauthlib` | Google credentials and OAuth flows. | Google Drive, Google Cloud, Google Speech-to-Text. |
|
|
483
|
+
| Google Cloud Storage | `google-cloud-storage` | Google Cloud bucket/file access. | Google Cloud File and Google Cloud Bucket loaders. |
|
|
484
|
+
| Google Cloud Speech | `google-cloud-speech` | Speech-to-text transcription. | Google Speech-to-Text loader. |
|
|
485
|
+
| ArXiv | `arxiv` | arXiv search and document retrieval. | ArXiv Loader and Retrieval mode. |
|
|
486
|
+
| Streamlit Runtime Extras | `watchdog` | Optional local development file watching. | Local Streamlit development. |
|
|
487
|
+
| Environment Variables | `python-dotenv` | Optional `.env` loading for API keys. | Local configuration. |
|
|
488
|
+
| Typing Extensions | `typing-extensions` | Backported typing support where needed. | Compatibility support. |
|
|
489
|
+
|
|
490
|
+
|
|
491
|
+
## 🔑 Configuration
|
|
492
|
+
|
|
493
|
+
| Key / Setting | Purpose | Used By |
|
|
494
|
+
| ------------------------- | -------------------------------------------------------------------------------- | ------------------------------------------ |
|
|
495
|
+
| `APP_TITLE` | Streamlit page title. | App setup. |
|
|
496
|
+
| `FAVICON` | Browser/page icon. | App setup. |
|
|
497
|
+
| `LOGO` | Sidebar/application logo. | App setup. |
|
|
498
|
+
| `MODE_MAP` | Mode names displayed in the sidebar. | Sidebar mode selector. |
|
|
499
|
+
| `SESSION_STATE_DEFAULTS` | Shared session-state defaults. | Startup initialization. |
|
|
500
|
+
| `REQUIRED_CORPORA` | NLTK resources to verify/download. | Loading mode and text analytics. |
|
|
501
|
+
| `DB_PATH` | SQLite database path. | Data Management and persistent app tables. |
|
|
502
|
+
| `GOOGLE_API_KEY` | Google service key. | Google Search and related Google fetchers. |
|
|
503
|
+
| `GOOGLE_CSE_ID` | Google Custom Search Engine ID. | Google Search. |
|
|
504
|
+
| `GOOGLE_ACCOUNT_FILE` | Google service account credential file. | Google Drive and Google Cloud workflows. |
|
|
505
|
+
| `GOOGLE_DRIVE_FOLDER_ID` | Default Drive folder identifier. | Google Drive retrieval. |
|
|
506
|
+
| `GOOGLE_DRIVE_TOKEN_PATH` | Optional token persistence path. | Google Drive retrieval. |
|
|
507
|
+
| `LANGSMITH_API_KEY` | Optional LangSmith tracing key. | LangChain tracing where configured. |
|
|
508
|
+
| Provider model lists | `GPT_MODELS`, `GROK_MODELS`, `CLAUDE_MODELS`, `GEMINI_MODELS`, `MISTRAL_MODELS`. | Generation mode model selectors. |
|
|
509
|
+
|
|
510
|
+
## 🔍 Example Usage
|
|
511
|
+
|
|
512
|
+
### Scrape Web Page Paragraphs
|
|
513
|
+
|
|
514
|
+
```python
|
|
515
|
+
from foo.scrapers import WebExtractor
|
|
516
|
+
|
|
517
|
+
extractor = WebExtractor()
|
|
518
|
+
paragraphs = extractor.scrape_paragraphs("https://example.com")
|
|
519
|
+
print(paragraphs)
|
|
520
|
+
```
|
|
521
|
+
|
|
522
|
+
### Load and Chunk a PDF
|
|
523
|
+
|
|
524
|
+
```python
|
|
525
|
+
from foo.loaders import PdfLoader
|
|
526
|
+
|
|
527
|
+
loader = PdfLoader()
|
|
528
|
+
documents = loader.load("docs/report.pdf")
|
|
529
|
+
chunks = loader.split(documents, chunk=1000, overlap=100)
|
|
530
|
+
print(chunks)
|
|
531
|
+
```
|
|
532
|
+
|
|
533
|
+
### Query a Fetcher
|
|
534
|
+
|
|
535
|
+
```python
|
|
536
|
+
from foo.fetchers import Wikipedia
|
|
537
|
+
|
|
538
|
+
fetcher = Wikipedia(language="en", max_documents=5)
|
|
539
|
+
documents = fetcher.fetch("Natural language processing")
|
|
540
|
+
for document in documents:
|
|
541
|
+
print(document.metadata)
|
|
542
|
+
print(document.page_content[:500])
|
|
543
|
+
```
|
|
544
|
+
|
|
545
|
+
### Run a Read-Only SQLite Query
|
|
546
|
+
|
|
547
|
+
```python
|
|
548
|
+
import sqlite3
|
|
549
|
+
import pandas as pd
|
|
550
|
+
|
|
551
|
+
with sqlite3.connect("stores/sqlite/data.db") as connection:
|
|
552
|
+
df_results = pd.read_sql_query("SELECT * FROM Prompts LIMIT 10;", connection)
|
|
553
|
+
|
|
554
|
+
print(df_results)
|
|
555
|
+
```
|
|
556
|
+
|
|
557
|
+
## ⚙️ Technical Notes
|
|
558
|
+
|
|
559
|
+
| Topic | Note |
|
|
560
|
+
| --------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
|
|
561
|
+
| Session state | The application uses `st.session_state` for mode state, model parameters, loaded documents, raw text, processed text, tokens, vocabulary, token counts, API configuration, and database working state. |
|
|
562
|
+
| Safety | SQL execution is intentionally constrained to read-only query forms through `is_safe_query`. |
|
|
563
|
+
| Loader contract | Loaded content is promoted into shared document state with raw text and active loader metadata. |
|
|
564
|
+
| SQLite | Data Management uses local SQLite tables and supports schema operations through guarded helper functions. |
|
|
565
|
+
| Metrics | Text metrics are computed from either processed text or raw text depending on what is available. |
|
|
566
|
+
| Optional dependencies | Some loaders and fetchers are only needed when their corresponding mode/expander is used. |
|
|
567
|
+
| Credentials | API keys are entered through sidebar configuration or loaded from configuration/environment variables. |
|
|
568
|
+
|
|
569
|
+
## 🔑 AI API Key
|
|
570
|
+
|
|
571
|
+
| Provider | Setup Link |
|
|
572
|
+
| -------- | ------------------------------------------------------------------------------------------------ |
|
|
573
|
+
| OpenAI | [OpenAI API Key](https://github.com/is-leeroy-jenkins/Buddy/blob/main/resources/setup/openai.md) |
|
|
574
|
+
| Grok | [Grok API Key](https://github.com/is-leeroy-jenkins/Buddy/blob/main/resources/setup/xai.md) |
|
|
575
|
+
| Gemini | [Gemini API Key](https://github.com/is-leeroy-jenkins/Buddy/blob/main/resources/setup/gemini.md) |
|
|
576
|
+
|
|
577
|
+
#### Data Services
|
|
578
|
+
|
|
579
|
+
| Service | Link | Service | Link |
|
|
580
|
+
| -------------- | ---------------------------------------------------------------------------------------------- | ------------ | --------------------------------------------------------------------------------- |
|
|
581
|
+
| OpenAI | [Platform](https://platform.openai.com/home) | Grok | [Account](https://accounts.x.ai/account) |
|
|
582
|
+
| Gemini | [AI Studio](https://aistudio.google.com/api-keys) | Claude | [API Keys](https://platform.claude.com/docs/en/api/admin/api_keys/retrieve) |
|
|
583
|
+
| Mistral | [Console](https://chat.mistral.ai/1) | NASA | [NASA API](https://api.nasa.gov/) |
|
|
584
|
+
| Geolocation | [Google Geolocation](https://developers.google.com/maps/documentation/geolocation/get-api-key) | Google Maps | [Google Maps](https://developers.google.com/maps/documentation/embed/get-api-key) |
|
|
585
|
+
| Gov Data | [GovInfo API](https://api.govinfo.gov/docs/) | The News API | [Register](https://www.thenewsapi.com/register) |
|
|
586
|
+
| Google Weather | [Weather API](https://developers.google.com/maps/documentation/weather/get-api-key) | Grokipedia | [PyPI](https://pypi.org/project/grokipedia-api/) |
|
|
587
|
+
| CDC | [CDC Data](https://data.cdc.gov/login) | Purple Air | [Developer Portal](https://develop.purpleair.com/) |
|
|
588
|
+
| FIRMS | [NASA FIRMS](https://firms.modaps.eosdis.nasa.gov/usfs/api/map_key/) | CENSUS | [API Key](https://api.census.gov/data/key_signup.html) |
|
|
589
|
+
| Wikipedia | [Wikimedia APIs](https://www.mediawiki.org/wiki/Wikimedia_APIs/Get_started) | | |
|
|
590
|
+
|
|
591
|
+
## 📝 License
|
|
592
|
+
|
|
593
|
+

|
|
594
|
+
|
|
595
|
+
* MIT License [here](https://github.com/is-leeroy-jenkins/Foo/blob/main/LICENSE.txt)
|
|
596
|
+
* Copyright © 2022–2025 Terry D. Eppler
|