octo-agent 0.11.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/.clacky/skills/commit/SKILL.md +423 -0
- data/.clacky/skills/gem-release/SKILL.md +199 -0
- data/.clacky/skills/gem-release/scripts/release.sh +304 -0
- data/.clacky/skills/oss-upload/SKILL.md +47 -0
- data/.octorules +106 -0
- data/.rspec +3 -0
- data/.rubocop.yml +8 -0
- data/CHANGELOG.md +76 -0
- data/CODE_OF_CONDUCT.md +132 -0
- data/CONTRIBUTING.md +92 -0
- data/Dockerfile +28 -0
- data/LICENSE.txt +22 -0
- data/POSITIONING.md +46 -0
- data/README.md +134 -0
- data/README_CN.md +134 -0
- data/Rakefile +34 -0
- data/benchmark/fixtures/sample_project/Gemfile +3 -0
- data/benchmark/fixtures/sample_project/lib/api_handler.rb +32 -0
- data/benchmark/fixtures/sample_project/lib/order_calculator.rb +23 -0
- data/benchmark/fixtures/sample_project/lib/user_renderer.rb +20 -0
- data/benchmark/fixtures/sample_project/spec/order_calculator_spec.rb +20 -0
- data/benchmark/results/EVALUATION_REPORT.md +165 -0
- data/benchmark/results/baseline_20260511_174424.json +128 -0
- data/benchmark/results/report_20260511_175256.json +271 -0
- data/benchmark/results/report_20260511_175444.json +271 -0
- data/benchmark/results/treatment_20260511_175103.json +130 -0
- data/benchmark/runner.rb +441 -0
- data/bin/octo +7 -0
- data/docs/agent-first-ui-design.md +77 -0
- data/docs/billing-system.md +318 -0
- data/docs/channel-architecture.md +235 -0
- data/docs/engineering-article.md +343 -0
- data/docs/session-skill-invocation.md +69 -0
- data/docs/time_machine_design.md +247 -0
- data/docs/ui2-architecture.md +124 -0
- data/homebrew/README.md +96 -0
- data/homebrew/openocto.rb +24 -0
- data/lib/octo/agent/hook_manager.rb +61 -0
- data/lib/octo/agent/llm_caller.rb +800 -0
- data/lib/octo/agent/memory_updater.rb +246 -0
- data/lib/octo/agent/message_compressor.rb +225 -0
- data/lib/octo/agent/message_compressor_helper.rb +869 -0
- data/lib/octo/agent/next_message_suggester.rb +215 -0
- data/lib/octo/agent/session_serializer.rb +685 -0
- data/lib/octo/agent/skill_auto_creator.rb +114 -0
- data/lib/octo/agent/skill_evolution.rb +61 -0
- data/lib/octo/agent/skill_manager.rb +466 -0
- data/lib/octo/agent/skill_reflector.rb +89 -0
- data/lib/octo/agent/system_prompt_builder.rb +101 -0
- data/lib/octo/agent/time_machine.rb +214 -0
- data/lib/octo/agent/tool_executor.rb +454 -0
- data/lib/octo/agent/tool_registry.rb +150 -0
- data/lib/octo/agent.rb +2180 -0
- data/lib/octo/agent_config.rb +989 -0
- data/lib/octo/agent_profile.rb +112 -0
- data/lib/octo/anthropic_stream_aggregator.rb +137 -0
- data/lib/octo/background_task_registry.rb +324 -0
- data/lib/octo/banner.rb +34 -0
- data/lib/octo/bedrock_stream_aggregator.rb +137 -0
- data/lib/octo/block_font.rb +331 -0
- data/lib/octo/cli.rb +968 -0
- data/lib/octo/client.rb +623 -0
- data/lib/octo/default_agents/SOUL.md +3 -0
- data/lib/octo/default_agents/USER.md +1 -0
- data/lib/octo/default_agents/base_prompt.md +66 -0
- data/lib/octo/default_agents/coding/profile.yml +2 -0
- data/lib/octo/default_agents/coding/system_prompt.md +67 -0
- data/lib/octo/default_agents/general/profile.yml +2 -0
- data/lib/octo/default_agents/general/system_prompt.md +16 -0
- data/lib/octo/default_parsers/doc_parser.rb +69 -0
- data/lib/octo/default_parsers/docx_parser.rb +188 -0
- data/lib/octo/default_parsers/pdf_parser.rb +120 -0
- data/lib/octo/default_parsers/pdf_parser_ocr.py +103 -0
- data/lib/octo/default_parsers/pdf_parser_plumber.py +62 -0
- data/lib/octo/default_parsers/pptx_parser.rb +140 -0
- data/lib/octo/default_parsers/xlsx_parser.rb +121 -0
- data/lib/octo/default_skills/browser-setup/SKILL.md +426 -0
- data/lib/octo/default_skills/channel-manager/SKILL.md +623 -0
- data/lib/octo/default_skills/channel-manager/dingtalk_setup.rb +191 -0
- data/lib/octo/default_skills/channel-manager/discord_setup.rb +199 -0
- data/lib/octo/default_skills/channel-manager/feishu_setup.rb +574 -0
- data/lib/octo/default_skills/channel-manager/import_lark_skills.rb +97 -0
- data/lib/octo/default_skills/channel-manager/install_feishu_skills.rb +105 -0
- data/lib/octo/default_skills/channel-manager/weixin_setup.rb +274 -0
- data/lib/octo/default_skills/code-explorer/SKILL.md +36 -0
- data/lib/octo/default_skills/cron-task-creator/SKILL.md +257 -0
- data/lib/octo/default_skills/cron-task-creator/evals/evals.json +38 -0
- data/lib/octo/default_skills/onboard/SKILL.md +578 -0
- data/lib/octo/default_skills/onboard/scripts/import_external_skills.rb +413 -0
- data/lib/octo/default_skills/onboard/scripts/install_builtin_skills.rb +97 -0
- data/lib/octo/default_skills/persist-memory/SKILL.md +59 -0
- data/lib/octo/default_skills/personal-website/SKILL.md +113 -0
- data/lib/octo/default_skills/personal-website/publish.rb +235 -0
- data/lib/octo/default_skills/product-help/SKILL.md +123 -0
- data/lib/octo/default_skills/product-help/docs/agent-config.md +74 -0
- data/lib/octo/default_skills/product-help/docs/best-practices.md +49 -0
- data/lib/octo/default_skills/product-help/docs/browser-tool.md +53 -0
- data/lib/octo/default_skills/product-help/docs/built-in-skills.md +43 -0
- data/lib/octo/default_skills/product-help/docs/cli-reference.md +82 -0
- data/lib/octo/default_skills/product-help/docs/create-your-first-skill.md +47 -0
- data/lib/octo/default_skills/product-help/docs/faq.md +98 -0
- data/lib/octo/default_skills/product-help/docs/how-to-use-a-skill.md +58 -0
- data/lib/octo/default_skills/product-help/docs/installation.md +59 -0
- data/lib/octo/default_skills/product-help/docs/memory-system.md +61 -0
- data/lib/octo/default_skills/product-help/docs/octorules.md +62 -0
- data/lib/octo/default_skills/product-help/docs/session-management.md +63 -0
- data/lib/octo/default_skills/product-help/docs/skill-basics.md +55 -0
- data/lib/octo/default_skills/product-help/docs/skill-frontmatter.md +61 -0
- data/lib/octo/default_skills/product-help/docs/web-server.md +49 -0
- data/lib/octo/default_skills/product-help/docs/what-is-octo.md +37 -0
- data/lib/octo/default_skills/product-help/docs/windows-installation.md +36 -0
- data/lib/octo/default_skills/product-help/docs/writing-tips.md +53 -0
- data/lib/octo/default_skills/recall-memory/SKILL.md +65 -0
- data/lib/octo/default_skills/skill-add/SKILL.md +59 -0
- data/lib/octo/default_skills/skill-add/scripts/install_from_zip.rb +295 -0
- data/lib/octo/default_skills/skill-creator/SKILL.md +602 -0
- data/lib/octo/default_skills/skill-creator/agents/analyzer.md +274 -0
- data/lib/octo/default_skills/skill-creator/agents/comparator.md +202 -0
- data/lib/octo/default_skills/skill-creator/agents/grader.md +223 -0
- data/lib/octo/default_skills/skill-creator/eval-viewer/generate_review.py +471 -0
- data/lib/octo/default_skills/skill-creator/eval-viewer/viewer.html +1325 -0
- data/lib/octo/default_skills/skill-creator/references/schemas.md +430 -0
- data/lib/octo/default_skills/skill-creator/scripts/__init__.py +0 -0
- data/lib/octo/default_skills/skill-creator/scripts/aggregate_benchmark.py +401 -0
- data/lib/octo/default_skills/skill-creator/scripts/generate_report.py +326 -0
- data/lib/octo/default_skills/skill-creator/scripts/improve_description.py +310 -0
- data/lib/octo/default_skills/skill-creator/scripts/quick_validate.py +103 -0
- data/lib/octo/default_skills/skill-creator/scripts/run_eval.py +317 -0
- data/lib/octo/default_skills/skill-creator/scripts/run_loop.py +331 -0
- data/lib/octo/default_skills/skill-creator/scripts/utils.py +47 -0
- data/lib/octo/default_skills/skill-creator/scripts/validate_skill_frontmatter.rb +143 -0
- data/lib/octo/idle_compression_timer.rb +115 -0
- data/lib/octo/json_ui_controller.rb +204 -0
- data/lib/octo/message_format/anthropic.rb +409 -0
- data/lib/octo/message_format/bedrock.rb +361 -0
- data/lib/octo/message_format/open_ai.rb +222 -0
- data/lib/octo/message_history.rb +373 -0
- data/lib/octo/openai_stream_aggregator.rb +130 -0
- data/lib/octo/plain_ui_controller.rb +166 -0
- data/lib/octo/providers.rb +534 -0
- data/lib/octo/server/browser_manager.rb +397 -0
- data/lib/octo/server/channel/adapters/base.rb +82 -0
- data/lib/octo/server/channel/adapters/dingtalk/adapter.rb +314 -0
- data/lib/octo/server/channel/adapters/dingtalk/api_client.rb +391 -0
- data/lib/octo/server/channel/adapters/dingtalk/stream_client.rb +203 -0
- data/lib/octo/server/channel/adapters/discord/adapter.rb +229 -0
- data/lib/octo/server/channel/adapters/discord/api_client.rb +107 -0
- data/lib/octo/server/channel/adapters/discord/gateway_client.rb +270 -0
- data/lib/octo/server/channel/adapters/feishu/adapter.rb +320 -0
- data/lib/octo/server/channel/adapters/feishu/bot.rb +478 -0
- data/lib/octo/server/channel/adapters/feishu/file_processor.rb +36 -0
- data/lib/octo/server/channel/adapters/feishu/message_parser.rb +129 -0
- data/lib/octo/server/channel/adapters/feishu/ws_client.rb +423 -0
- data/lib/octo/server/channel/adapters/telegram/adapter.rb +375 -0
- data/lib/octo/server/channel/adapters/telegram/api_client.rb +205 -0
- data/lib/octo/server/channel/adapters/wecom/adapter.rb +148 -0
- data/lib/octo/server/channel/adapters/wecom/media_downloader.rb +115 -0
- data/lib/octo/server/channel/adapters/wecom/ws_client.rb +395 -0
- data/lib/octo/server/channel/adapters/weixin/adapter.rb +692 -0
- data/lib/octo/server/channel/adapters/weixin/api_client.rb +402 -0
- data/lib/octo/server/channel/channel_config.rb +178 -0
- data/lib/octo/server/channel/channel_manager.rb +468 -0
- data/lib/octo/server/channel/channel_ui_controller.rb +224 -0
- data/lib/octo/server/channel.rb +33 -0
- data/lib/octo/server/discover.rb +77 -0
- data/lib/octo/server/epipe_safe_io.rb +105 -0
- data/lib/octo/server/http_server.rb +3554 -0
- data/lib/octo/server/scheduler.rb +317 -0
- data/lib/octo/server/server_master.rb +325 -0
- data/lib/octo/server/session_registry.rb +431 -0
- data/lib/octo/server/web_ui_controller.rb +487 -0
- data/lib/octo/session_manager.rb +385 -0
- data/lib/octo/skill.rb +466 -0
- data/lib/octo/skill_loader.rb +328 -0
- data/lib/octo/tools/base.rb +118 -0
- data/lib/octo/tools/browser.rb +625 -0
- data/lib/octo/tools/edit.rb +165 -0
- data/lib/octo/tools/file_reader.rb +549 -0
- data/lib/octo/tools/glob.rb +162 -0
- data/lib/octo/tools/grep.rb +356 -0
- data/lib/octo/tools/invoke_skill.rb +96 -0
- data/lib/octo/tools/list_tasks.rb +54 -0
- data/lib/octo/tools/redo_task.rb +41 -0
- data/lib/octo/tools/request_user_feedback.rb +84 -0
- data/lib/octo/tools/security.rb +333 -0
- data/lib/octo/tools/terminal/output_cleaner.rb +63 -0
- data/lib/octo/tools/terminal/persistent_session.rb +268 -0
- data/lib/octo/tools/terminal/safe_rm.sh +106 -0
- data/lib/octo/tools/terminal/session_manager.rb +213 -0
- data/lib/octo/tools/terminal.rb +1828 -0
- data/lib/octo/tools/todo_manager.rb +374 -0
- data/lib/octo/tools/trash_manager.rb +388 -0
- data/lib/octo/tools/undo_task.rb +35 -0
- data/lib/octo/tools/web_fetch.rb +242 -0
- data/lib/octo/tools/web_search.rb +260 -0
- data/lib/octo/tools/write.rb +77 -0
- data/lib/octo/ui2/block_font.rb +10 -0
- data/lib/octo/ui2/components/base_component.rb +163 -0
- data/lib/octo/ui2/components/command_suggestions.rb +290 -0
- data/lib/octo/ui2/components/common_component.rb +96 -0
- data/lib/octo/ui2/components/inline_input.rb +226 -0
- data/lib/octo/ui2/components/input_area.rb +1338 -0
- data/lib/octo/ui2/components/message_component.rb +99 -0
- data/lib/octo/ui2/components/modal_component.rb +419 -0
- data/lib/octo/ui2/components/todo_area.rb +149 -0
- data/lib/octo/ui2/components/tool_component.rb +107 -0
- data/lib/octo/ui2/components/welcome_banner.rb +139 -0
- data/lib/octo/ui2/layout_manager.rb +807 -0
- data/lib/octo/ui2/line_editor.rb +363 -0
- data/lib/octo/ui2/markdown_renderer.rb +100 -0
- data/lib/octo/ui2/output_buffer.rb +370 -0
- data/lib/octo/ui2/progress_handle.rb +362 -0
- data/lib/octo/ui2/progress_indicator.rb +55 -0
- data/lib/octo/ui2/screen_buffer.rb +273 -0
- data/lib/octo/ui2/terminal_detector.rb +119 -0
- data/lib/octo/ui2/theme_manager.rb +85 -0
- data/lib/octo/ui2/themes/base_theme.rb +105 -0
- data/lib/octo/ui2/themes/hacker_theme.rb +62 -0
- data/lib/octo/ui2/themes/minimal_theme.rb +56 -0
- data/lib/octo/ui2/thinking_verbs.rb +26 -0
- data/lib/octo/ui2/ui_controller.rb +1625 -0
- data/lib/octo/ui2/view_renderer.rb +177 -0
- data/lib/octo/ui2.rb +40 -0
- data/lib/octo/ui_interface.rb +154 -0
- data/lib/octo/utils/arguments_parser.rb +191 -0
- data/lib/octo/utils/browser_detector.rb +195 -0
- data/lib/octo/utils/encoding.rb +92 -0
- data/lib/octo/utils/environment_detector.rb +140 -0
- data/lib/octo/utils/file_ignore_helper.rb +170 -0
- data/lib/octo/utils/file_processor.rb +601 -0
- data/lib/octo/utils/gitignore_parser.rb +154 -0
- data/lib/octo/utils/limit_stack.rb +152 -0
- data/lib/octo/utils/logger.rb +124 -0
- data/lib/octo/utils/login_shell.rb +72 -0
- data/lib/octo/utils/model_pricing.rb +646 -0
- data/lib/octo/utils/parser_manager.rb +165 -0
- data/lib/octo/utils/path_helper.rb +15 -0
- data/lib/octo/utils/scripts_manager.rb +59 -0
- data/lib/octo/utils/string_matcher.rb +158 -0
- data/lib/octo/utils/trash_directory.rb +112 -0
- data/lib/octo/utils/workspace_rules.rb +46 -0
- data/lib/octo/version.rb +5 -0
- data/lib/octo/web/app.css +7141 -0
- data/lib/octo/web/app.js +543 -0
- data/lib/octo/web/apple-touch-icon.png +0 -0
- data/lib/octo/web/auth.js +150 -0
- data/lib/octo/web/channels.js +276 -0
- data/lib/octo/web/datepicker.js +205 -0
- data/lib/octo/web/favicon.png +0 -0
- data/lib/octo/web/i18n.js +1073 -0
- data/lib/octo/web/icon-512.png +0 -0
- data/lib/octo/web/icon-dark.svg +25 -0
- data/lib/octo/web/icon.svg +29 -0
- data/lib/octo/web/index.html +871 -0
- data/lib/octo/web/marked.min.js +69 -0
- data/lib/octo/web/onboard.js +491 -0
- data/lib/octo/web/profile.js +442 -0
- data/lib/octo/web/sessions.js +4421 -0
- data/lib/octo/web/settings.js +913 -0
- data/lib/octo/web/sidebar.js +32 -0
- data/lib/octo/web/skills.js +885 -0
- data/lib/octo/web/tasks.js +297 -0
- data/lib/octo/web/theme.js +105 -0
- data/lib/octo/web/trash.js +343 -0
- data/lib/octo/web/vendor/hljs/highlight.min.js +1244 -0
- data/lib/octo/web/vendor/hljs/hljs-theme.css +95 -0
- data/lib/octo/web/vendor/katex/auto-render.min.js +1 -0
- data/lib/octo/web/vendor/katex/fonts/KaTeX_AMS-Regular.woff2 +0 -0
- data/lib/octo/web/vendor/katex/fonts/KaTeX_Caligraphic-Bold.woff2 +0 -0
- data/lib/octo/web/vendor/katex/fonts/KaTeX_Caligraphic-Regular.woff2 +0 -0
- data/lib/octo/web/vendor/katex/fonts/KaTeX_Fraktur-Bold.woff2 +0 -0
- data/lib/octo/web/vendor/katex/fonts/KaTeX_Fraktur-Regular.woff2 +0 -0
- data/lib/octo/web/vendor/katex/fonts/KaTeX_Main-Bold.woff2 +0 -0
- data/lib/octo/web/vendor/katex/fonts/KaTeX_Main-BoldItalic.woff2 +0 -0
- data/lib/octo/web/vendor/katex/fonts/KaTeX_Main-Italic.woff2 +0 -0
- data/lib/octo/web/vendor/katex/fonts/KaTeX_Main-Regular.woff2 +0 -0
- data/lib/octo/web/vendor/katex/fonts/KaTeX_Math-BoldItalic.woff2 +0 -0
- data/lib/octo/web/vendor/katex/fonts/KaTeX_Math-Italic.woff2 +0 -0
- data/lib/octo/web/vendor/katex/fonts/KaTeX_SansSerif-Bold.woff2 +0 -0
- data/lib/octo/web/vendor/katex/fonts/KaTeX_SansSerif-Italic.woff2 +0 -0
- data/lib/octo/web/vendor/katex/fonts/KaTeX_SansSerif-Regular.woff2 +0 -0
- data/lib/octo/web/vendor/katex/fonts/KaTeX_Script-Regular.woff2 +0 -0
- data/lib/octo/web/vendor/katex/fonts/KaTeX_Size1-Regular.woff2 +0 -0
- data/lib/octo/web/vendor/katex/fonts/KaTeX_Size2-Regular.woff2 +0 -0
- data/lib/octo/web/vendor/katex/fonts/KaTeX_Size3-Regular.woff2 +0 -0
- data/lib/octo/web/vendor/katex/fonts/KaTeX_Size4-Regular.woff2 +0 -0
- data/lib/octo/web/vendor/katex/fonts/KaTeX_Typewriter-Regular.woff2 +0 -0
- data/lib/octo/web/vendor/katex/katex.min.css +1 -0
- data/lib/octo/web/vendor/katex/katex.min.js +1 -0
- data/lib/octo/web/version.js +449 -0
- data/lib/octo/web/weixin-qr.html +209 -0
- data/lib/octo/web/ws-dispatcher.js +357 -0
- data/lib/octo/web/ws.js +128 -0
- data/lib/octo.rb +145 -0
- data/scripts/build/build.sh +329 -0
- data/scripts/build/lib/apt.sh +56 -0
- data/scripts/build/lib/brew.sh +89 -0
- data/scripts/build/lib/colors.sh +17 -0
- data/scripts/build/lib/gem.sh +95 -0
- data/scripts/build/lib/mise.sh +125 -0
- data/scripts/build/lib/network.sh +157 -0
- data/scripts/build/lib/os.sh +57 -0
- data/scripts/build/lib/shell.sh +37 -0
- data/scripts/build/src/install.sh.cc +174 -0
- data/scripts/build/src/install_browser.sh.cc +101 -0
- data/scripts/build/src/install_full.sh.cc +290 -0
- data/scripts/build/src/install_rails_deps.sh.cc +145 -0
- data/scripts/build/src/install_system_deps.sh.cc +123 -0
- data/scripts/build/src/uninstall.sh.cc +101 -0
- data/scripts/install.ps1 +532 -0
- data/scripts/install.sh +567 -0
- data/scripts/install_browser.sh +479 -0
- data/scripts/install_full.sh +838 -0
- data/scripts/install_rails_deps.sh +746 -0
- data/scripts/install_system_deps.sh +518 -0
- data/scripts/uninstall.sh +287 -0
- data/sig/octo.rbs +4 -0
- metadata +614 -0
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
pdf_parser_plumber.py — extract text from a PDF using pdfplumber.
|
|
4
|
+
|
|
5
|
+
Usage:
|
|
6
|
+
python3 pdf_parser_plumber.py <file_path>
|
|
7
|
+
|
|
8
|
+
Output:
|
|
9
|
+
stdout — extracted text, one block per page, separated by blank lines
|
|
10
|
+
stderr — error messages
|
|
11
|
+
exit 0 — success (text was extracted)
|
|
12
|
+
exit 1 — failure / no text found
|
|
13
|
+
exit 2 — dependency missing
|
|
14
|
+
|
|
15
|
+
Called from pdf_parser.rb as the second-tier extractor (after pdftotext).
|
|
16
|
+
This script is copied into ~/.octo/parsers/ and can be edited freely by
|
|
17
|
+
the LLM — e.g. to tune table extraction, layout heuristics, or filter out
|
|
18
|
+
boilerplate headers/footers. Edit, then re-run to test.
|
|
19
|
+
|
|
20
|
+
Install:
|
|
21
|
+
pip3 install pdfplumber
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
# VERSION: 1
|
|
25
|
+
|
|
26
|
+
import sys
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def main():
|
|
30
|
+
if len(sys.argv) < 2:
|
|
31
|
+
sys.stderr.write("Usage: pdf_parser_plumber.py <file_path>\n")
|
|
32
|
+
sys.exit(1)
|
|
33
|
+
|
|
34
|
+
path = sys.argv[1]
|
|
35
|
+
|
|
36
|
+
try:
|
|
37
|
+
import pdfplumber
|
|
38
|
+
except ImportError as e:
|
|
39
|
+
sys.stderr.write(f"pdfplumber missing: {e}\n")
|
|
40
|
+
sys.stderr.write("Install with: pip3 install pdfplumber\n")
|
|
41
|
+
sys.exit(2)
|
|
42
|
+
|
|
43
|
+
pages = []
|
|
44
|
+
try:
|
|
45
|
+
with pdfplumber.open(path) as pdf:
|
|
46
|
+
for i, page in enumerate(pdf.pages, 1):
|
|
47
|
+
text = page.extract_text()
|
|
48
|
+
if text and text.strip():
|
|
49
|
+
pages.append(f"--- Page {i} ---\n{text.strip()}")
|
|
50
|
+
except Exception as e:
|
|
51
|
+
sys.stderr.write(f"pdfplumber failed: {e}\n")
|
|
52
|
+
sys.exit(1)
|
|
53
|
+
|
|
54
|
+
if not pages:
|
|
55
|
+
sys.stderr.write("pdfplumber produced no text.\n")
|
|
56
|
+
sys.exit(1)
|
|
57
|
+
|
|
58
|
+
print("\n\n".join(pages))
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
if __name__ == "__main__":
|
|
62
|
+
main()
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
#!/usr/bin/env ruby
|
|
2
|
+
# frozen_string_literal: true
|
|
3
|
+
#
|
|
4
|
+
# Octo PPTX Parser — CLI interface
|
|
5
|
+
#
|
|
6
|
+
# Usage:
|
|
7
|
+
# ruby pptx_parser.rb <file_path>
|
|
8
|
+
#
|
|
9
|
+
# Output:
|
|
10
|
+
# stdout — extracted content in Markdown (UTF-8)
|
|
11
|
+
# stderr — error messages
|
|
12
|
+
# exit 0 — success
|
|
13
|
+
# exit 1 — failure
|
|
14
|
+
#
|
|
15
|
+
# Dependencies: rubyzip gem (gem install rubyzip)
|
|
16
|
+
#
|
|
17
|
+
# This file lives in ~/.octo/parsers/ and can be modified by the LLM.
|
|
18
|
+
#
|
|
19
|
+
# VERSION: 1
|
|
20
|
+
|
|
21
|
+
require "zip"
|
|
22
|
+
require "rexml/document"
|
|
23
|
+
require "stringio"
|
|
24
|
+
|
|
25
|
+
def extract_text(shape_node)
|
|
26
|
+
paras = []
|
|
27
|
+
REXML::XPath.each(shape_node, ".//a:p") do |para|
|
|
28
|
+
text = REXML::XPath.match(para, ".//a:t").map(&:text).compact.join
|
|
29
|
+
paras << text unless text.strip.empty?
|
|
30
|
+
end
|
|
31
|
+
paras.join("\n")
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
def parse_table(tbl_node)
|
|
35
|
+
rows = []
|
|
36
|
+
REXML::XPath.each(tbl_node, ".//a:tr") do |tr|
|
|
37
|
+
cells = REXML::XPath.match(tr, ".//a:tc").map do |tc|
|
|
38
|
+
REXML::XPath.match(tc, ".//a:t").map(&:text).compact.join(" ").strip
|
|
39
|
+
end
|
|
40
|
+
rows << cells
|
|
41
|
+
end
|
|
42
|
+
return "" if rows.empty?
|
|
43
|
+
|
|
44
|
+
col_count = rows.map(&:size).max
|
|
45
|
+
lines = []
|
|
46
|
+
rows.each_with_index do |row, i|
|
|
47
|
+
padded = row + [""] * [col_count - row.size, 0].max
|
|
48
|
+
lines << "| #{padded.join(" | ")} |"
|
|
49
|
+
lines << "|#{" --- |" * col_count}" if i == 0
|
|
50
|
+
end
|
|
51
|
+
lines.join("\n")
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
def parse_slide(doc, slide_num)
|
|
55
|
+
lines = []
|
|
56
|
+
|
|
57
|
+
title_text = nil
|
|
58
|
+
REXML::XPath.each(doc, "//p:sp") do |sp|
|
|
59
|
+
ph = REXML::XPath.first(sp, ".//p:ph")
|
|
60
|
+
next unless ph
|
|
61
|
+
ph_type = ph.attributes["type"]
|
|
62
|
+
if ph_type == "title" || ph_type == "ctrTitle"
|
|
63
|
+
title_text = extract_text(sp).strip
|
|
64
|
+
break
|
|
65
|
+
end
|
|
66
|
+
end
|
|
67
|
+
|
|
68
|
+
lines << "## Slide #{slide_num}#{title_text && !title_text.empty? ? ": #{title_text}" : ""}"
|
|
69
|
+
|
|
70
|
+
REXML::XPath.each(doc, "//p:sp") do |sp|
|
|
71
|
+
ph = REXML::XPath.first(sp, ".//p:ph")
|
|
72
|
+
if ph
|
|
73
|
+
ph_type = ph.attributes["type"]
|
|
74
|
+
next if %w[title ctrTitle sldNum dt ftr].include?(ph_type)
|
|
75
|
+
end
|
|
76
|
+
|
|
77
|
+
text = extract_text(sp).strip
|
|
78
|
+
next if text.empty?
|
|
79
|
+
next if text == title_text
|
|
80
|
+
|
|
81
|
+
text.each_line do |line|
|
|
82
|
+
lines << "- #{line.rstrip}" unless line.strip.empty?
|
|
83
|
+
end
|
|
84
|
+
end
|
|
85
|
+
|
|
86
|
+
REXML::XPath.each(doc, "//a:tbl") do |tbl|
|
|
87
|
+
lines << parse_table(tbl)
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
lines.join("\n")
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
# --- main ---
|
|
94
|
+
|
|
95
|
+
path = ARGV[0]
|
|
96
|
+
|
|
97
|
+
if path.nil? || path.empty?
|
|
98
|
+
warn "Usage: ruby pptx_parser.rb <file_path>"
|
|
99
|
+
exit 1
|
|
100
|
+
end
|
|
101
|
+
|
|
102
|
+
unless File.exist?(path)
|
|
103
|
+
warn "File not found: #{path}"
|
|
104
|
+
exit 1
|
|
105
|
+
end
|
|
106
|
+
|
|
107
|
+
begin
|
|
108
|
+
body = File.binread(path)
|
|
109
|
+
slides = {}
|
|
110
|
+
|
|
111
|
+
Zip::File.open_buffer(StringIO.new(body)) do |zip|
|
|
112
|
+
zip.each do |entry|
|
|
113
|
+
if entry.name =~ %r{ppt/slides/slide(\d+)\.xml}
|
|
114
|
+
slides[$1.to_i] = entry.get_input_stream.read
|
|
115
|
+
end
|
|
116
|
+
end
|
|
117
|
+
end
|
|
118
|
+
|
|
119
|
+
if slides.empty?
|
|
120
|
+
warn "Presentation appears to be empty"
|
|
121
|
+
exit 1
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
sections = slides.keys.sort.map do |num|
|
|
125
|
+
doc = REXML::Document.new(slides[num])
|
|
126
|
+
parse_slide(doc, num)
|
|
127
|
+
end.compact
|
|
128
|
+
|
|
129
|
+
if sections.empty?
|
|
130
|
+
warn "Presentation appears to be empty"
|
|
131
|
+
exit 1
|
|
132
|
+
end
|
|
133
|
+
|
|
134
|
+
print sections.join("\n\n---\n\n")
|
|
135
|
+
exit 0
|
|
136
|
+
rescue => e
|
|
137
|
+
warn "Failed to parse PPTX: #{e.message}"
|
|
138
|
+
warn "Tip: ensure rubyzip is installed: gem install rubyzip"
|
|
139
|
+
exit 1
|
|
140
|
+
end
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
#!/usr/bin/env ruby
|
|
2
|
+
# frozen_string_literal: true
|
|
3
|
+
#
|
|
4
|
+
# Octo XLSX Parser — CLI interface
|
|
5
|
+
#
|
|
6
|
+
# Usage:
|
|
7
|
+
# ruby xlsx_parser.rb <file_path>
|
|
8
|
+
#
|
|
9
|
+
# Output:
|
|
10
|
+
# stdout — extracted content in Markdown tables (UTF-8)
|
|
11
|
+
# stderr — error messages
|
|
12
|
+
# exit 0 — success
|
|
13
|
+
# exit 1 — failure
|
|
14
|
+
#
|
|
15
|
+
# Dependencies: rubyzip gem (gem install rubyzip)
|
|
16
|
+
#
|
|
17
|
+
# This file lives in ~/.octo/parsers/ and can be modified by the LLM.
|
|
18
|
+
#
|
|
19
|
+
# VERSION: 1
|
|
20
|
+
|
|
21
|
+
require "zip"
|
|
22
|
+
require "rexml/document"
|
|
23
|
+
require "stringio"
|
|
24
|
+
|
|
25
|
+
def parse_row(row_node, shared_strings)
|
|
26
|
+
REXML::XPath.match(row_node, ".//c").map do |c|
|
|
27
|
+
v = REXML::XPath.first(c, "v")&.text
|
|
28
|
+
next "" unless v
|
|
29
|
+
c.attributes["t"] == "s" ? (shared_strings[v.to_i] || "") : v
|
|
30
|
+
end
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
def build_markdown_table(rows)
|
|
34
|
+
col_count = rows.map(&:size).max
|
|
35
|
+
lines = []
|
|
36
|
+
rows.each_with_index do |row, i|
|
|
37
|
+
padded = row + [""] * [col_count - row.size, 0].max
|
|
38
|
+
lines << "| #{padded.join(" | ")} |"
|
|
39
|
+
lines << "|#{" --- |" * col_count}" if i == 0
|
|
40
|
+
end
|
|
41
|
+
lines.join("\n")
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
# --- main ---
|
|
45
|
+
|
|
46
|
+
path = ARGV[0]
|
|
47
|
+
|
|
48
|
+
if path.nil? || path.empty?
|
|
49
|
+
warn "Usage: ruby xlsx_parser.rb <file_path>"
|
|
50
|
+
exit 1
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
unless File.exist?(path)
|
|
54
|
+
warn "File not found: #{path}"
|
|
55
|
+
exit 1
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
begin
|
|
59
|
+
body = File.binread(path)
|
|
60
|
+
shared_strings = []
|
|
61
|
+
sheet_names = {}
|
|
62
|
+
sheet_xmls = {}
|
|
63
|
+
|
|
64
|
+
Zip::File.open_buffer(StringIO.new(body)) do |zip|
|
|
65
|
+
ss_entry = zip.find_entry("xl/sharedStrings.xml")
|
|
66
|
+
if ss_entry
|
|
67
|
+
doc = REXML::Document.new(ss_entry.get_input_stream.read)
|
|
68
|
+
REXML::XPath.each(doc, "//si") do |si|
|
|
69
|
+
shared_strings << REXML::XPath.match(si, ".//t").map(&:text).compact.join
|
|
70
|
+
end
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
wb_entry = zip.find_entry("xl/workbook.xml")
|
|
74
|
+
if wb_entry
|
|
75
|
+
doc = REXML::Document.new(wb_entry.get_input_stream.read)
|
|
76
|
+
REXML::XPath.each(doc, "//sheet") do |s|
|
|
77
|
+
idx = s.attributes["sheetId"]
|
|
78
|
+
name = s.attributes["name"]
|
|
79
|
+
sheet_names[idx] = name if idx && name
|
|
80
|
+
end
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
zip.each do |entry|
|
|
84
|
+
if entry.name =~ %r{xl/worksheets/sheet(\d+)\.xml}
|
|
85
|
+
sheet_xmls[$1] = entry.get_input_stream.read
|
|
86
|
+
end
|
|
87
|
+
end
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
if sheet_xmls.empty?
|
|
91
|
+
warn "Spreadsheet appears to be empty"
|
|
92
|
+
exit 1
|
|
93
|
+
end
|
|
94
|
+
|
|
95
|
+
sections = []
|
|
96
|
+
sheet_xmls.keys.sort_by(&:to_i).each do |idx|
|
|
97
|
+
name = sheet_names[idx] || "Sheet#{idx}"
|
|
98
|
+
doc = REXML::Document.new(sheet_xmls[idx])
|
|
99
|
+
|
|
100
|
+
rows = []
|
|
101
|
+
REXML::XPath.each(doc, "//row") do |row|
|
|
102
|
+
cells = parse_row(row, shared_strings)
|
|
103
|
+
rows << cells unless cells.all?(&:empty?)
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
next if rows.empty?
|
|
107
|
+
sections << "### #{name}\n\n#{build_markdown_table(rows)}"
|
|
108
|
+
end
|
|
109
|
+
|
|
110
|
+
if sections.empty?
|
|
111
|
+
warn "Spreadsheet appears to be empty"
|
|
112
|
+
exit 1
|
|
113
|
+
end
|
|
114
|
+
|
|
115
|
+
print sections.join("\n\n")
|
|
116
|
+
exit 0
|
|
117
|
+
rescue => e
|
|
118
|
+
warn "Failed to parse XLSX: #{e.message}"
|
|
119
|
+
warn "Tip: ensure rubyzip is installed: gem install rubyzip"
|
|
120
|
+
exit 1
|
|
121
|
+
end
|