@funnycode/myclaude 0.1.67 → 0.1.68

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (497) hide show
  1. package/LICENSE +21 -21
  2. package/dist/myclaude.js +15090 -12419
  3. package/dist/myclaude.mjs +15090 -12419
  4. package/package.json +125 -125
  5. package/seed/marketplaces/ecc/.claude/commands/add-language-rules.md +39 -0
  6. package/seed/marketplaces/ecc/.claude/commands/database-migration.md +36 -0
  7. package/seed/marketplaces/ecc/.claude/commands/feature-development.md +38 -0
  8. package/seed/marketplaces/ecc/.claude/ecc-tools.json +334 -0
  9. package/seed/marketplaces/ecc/.claude/enterprise/controls.md +15 -0
  10. package/seed/marketplaces/ecc/.claude/homunculus/instincts/inherited/everything-claude-code-instincts.yaml +162 -0
  11. package/seed/marketplaces/ecc/.claude/identity.json +14 -0
  12. package/seed/marketplaces/ecc/.claude/package-manager.json +4 -0
  13. package/seed/marketplaces/ecc/.claude/research/everything-claude-code-research-playbook.md +21 -0
  14. package/seed/marketplaces/ecc/.claude/rules/everything-claude-code-guardrails.md +43 -0
  15. package/seed/marketplaces/ecc/.claude/rules/node.md +56 -0
  16. package/seed/marketplaces/ecc/.claude/skills/everything-claude-code/SKILL.md +442 -0
  17. package/seed/marketplaces/ecc/.claude/team/everything-claude-code-team-config.json +15 -0
  18. package/seed/marketplaces/ecc/.opencode/package-lock.json +169 -169
  19. package/seed/marketplaces/ecc/.vscode/settings.json +17 -0
  20. package/seed/marketplaces/ecc/commands/aside.md +164 -164
  21. package/seed/marketplaces/ecc/commands/auto-update.md +28 -28
  22. package/seed/marketplaces/ecc/commands/build-fix.md +66 -66
  23. package/seed/marketplaces/ecc/commands/checkpoint.md +78 -78
  24. package/seed/marketplaces/ecc/commands/code-review.md +289 -289
  25. package/seed/marketplaces/ecc/commands/cost-report.md +107 -107
  26. package/seed/marketplaces/ecc/commands/cpp-build.md +173 -173
  27. package/seed/marketplaces/ecc/commands/cpp-review.md +132 -132
  28. package/seed/marketplaces/ecc/commands/cpp-test.md +251 -251
  29. package/seed/marketplaces/ecc/commands/ecc-guide.md +93 -93
  30. package/seed/marketplaces/ecc/commands/evolve.md +178 -178
  31. package/seed/marketplaces/ecc/commands/fastapi-review.md +39 -39
  32. package/seed/marketplaces/ecc/commands/feature-dev.md +49 -49
  33. package/seed/marketplaces/ecc/commands/flutter-build.md +164 -164
  34. package/seed/marketplaces/ecc/commands/flutter-review.md +116 -116
  35. package/seed/marketplaces/ecc/commands/flutter-test.md +144 -144
  36. package/seed/marketplaces/ecc/commands/gan-build.md +103 -103
  37. package/seed/marketplaces/ecc/commands/gan-design.md +39 -39
  38. package/seed/marketplaces/ecc/commands/go-build.md +183 -183
  39. package/seed/marketplaces/ecc/commands/go-review.md +148 -148
  40. package/seed/marketplaces/ecc/commands/go-test.md +268 -268
  41. package/seed/marketplaces/ecc/commands/gradle-build.md +70 -70
  42. package/seed/marketplaces/ecc/commands/harness-audit.md +84 -84
  43. package/seed/marketplaces/ecc/commands/hookify-configure.md +14 -14
  44. package/seed/marketplaces/ecc/commands/hookify-help.md +46 -46
  45. package/seed/marketplaces/ecc/commands/hookify-list.md +21 -21
  46. package/seed/marketplaces/ecc/commands/hookify.md +50 -50
  47. package/seed/marketplaces/ecc/commands/instinct-export.md +66 -66
  48. package/seed/marketplaces/ecc/commands/instinct-import.md +114 -114
  49. package/seed/marketplaces/ecc/commands/instinct-status.md +59 -59
  50. package/seed/marketplaces/ecc/commands/jira.md +106 -106
  51. package/seed/marketplaces/ecc/commands/kotlin-build.md +174 -174
  52. package/seed/marketplaces/ecc/commands/kotlin-review.md +140 -140
  53. package/seed/marketplaces/ecc/commands/kotlin-test.md +312 -312
  54. package/seed/marketplaces/ecc/commands/learn-eval.md +116 -116
  55. package/seed/marketplaces/ecc/commands/learn.md +74 -74
  56. package/seed/marketplaces/ecc/commands/loop-start.md +36 -36
  57. package/seed/marketplaces/ecc/commands/loop-status.md +77 -77
  58. package/seed/marketplaces/ecc/commands/marketing-campaign.md +129 -129
  59. package/seed/marketplaces/ecc/commands/model-route.md +30 -30
  60. package/seed/marketplaces/ecc/commands/multi-backend.md +162 -162
  61. package/seed/marketplaces/ecc/commands/multi-execute.md +319 -319
  62. package/seed/marketplaces/ecc/commands/multi-frontend.md +162 -162
  63. package/seed/marketplaces/ecc/commands/multi-plan.md +272 -272
  64. package/seed/marketplaces/ecc/commands/multi-workflow.md +195 -195
  65. package/seed/marketplaces/ecc/commands/plan-prd.md +160 -160
  66. package/seed/marketplaces/ecc/commands/plan.md +200 -200
  67. package/seed/marketplaces/ecc/commands/pm2.md +276 -276
  68. package/seed/marketplaces/ecc/commands/pr.md +184 -184
  69. package/seed/marketplaces/ecc/commands/project-init.md +86 -86
  70. package/seed/marketplaces/ecc/commands/projects.md +39 -39
  71. package/seed/marketplaces/ecc/commands/promote.md +41 -41
  72. package/seed/marketplaces/ecc/commands/prp-commit.md +112 -112
  73. package/seed/marketplaces/ecc/commands/prp-implement.md +385 -385
  74. package/seed/marketplaces/ecc/commands/prp-plan.md +502 -502
  75. package/seed/marketplaces/ecc/commands/prp-pr.md +184 -184
  76. package/seed/marketplaces/ecc/commands/prp-prd.md +447 -447
  77. package/seed/marketplaces/ecc/commands/prune.md +31 -31
  78. package/seed/marketplaces/ecc/commands/python-review.md +297 -297
  79. package/seed/marketplaces/ecc/commands/quality-gate.md +33 -33
  80. package/seed/marketplaces/ecc/commands/refactor-clean.md +84 -84
  81. package/seed/marketplaces/ecc/commands/resume-session.md +156 -156
  82. package/seed/marketplaces/ecc/commands/review-pr.md +37 -37
  83. package/seed/marketplaces/ecc/commands/rust-build.md +187 -187
  84. package/seed/marketplaces/ecc/commands/rust-review.md +142 -142
  85. package/seed/marketplaces/ecc/commands/rust-test.md +308 -308
  86. package/seed/marketplaces/ecc/commands/santa-loop.md +175 -175
  87. package/seed/marketplaces/ecc/commands/save-session.md +275 -275
  88. package/seed/marketplaces/ecc/commands/security-scan.md +92 -92
  89. package/seed/marketplaces/ecc/commands/sessions.md +339 -339
  90. package/seed/marketplaces/ecc/commands/setup-pm.md +80 -80
  91. package/seed/marketplaces/ecc/commands/skill-create.md +174 -174
  92. package/seed/marketplaces/ecc/commands/skill-health.md +54 -54
  93. package/seed/marketplaces/ecc/commands/test-coverage.md +73 -73
  94. package/seed/marketplaces/ecc/commands/update-codemaps.md +76 -76
  95. package/seed/marketplaces/ecc/commands/update-docs.md +88 -88
  96. package/seed/marketplaces/ecc/skills/accessibility/SKILL.md +146 -146
  97. package/seed/marketplaces/ecc/skills/agent-architecture-audit/SKILL.md +256 -256
  98. package/seed/marketplaces/ecc/skills/agent-eval/SKILL.md +145 -145
  99. package/seed/marketplaces/ecc/skills/agent-harness-construction/SKILL.md +73 -73
  100. package/seed/marketplaces/ecc/skills/agent-introspection-debugging/SKILL.md +153 -153
  101. package/seed/marketplaces/ecc/skills/agent-payment-x402/SKILL.md +224 -224
  102. package/seed/marketplaces/ecc/skills/agent-sort/SKILL.md +215 -215
  103. package/seed/marketplaces/ecc/skills/agentic-engineering/SKILL.md +63 -63
  104. package/seed/marketplaces/ecc/skills/agentic-os/SKILL.md +387 -387
  105. package/seed/marketplaces/ecc/skills/ai-first-engineering/SKILL.md +51 -51
  106. package/seed/marketplaces/ecc/skills/ai-regression-testing/SKILL.md +385 -385
  107. package/seed/marketplaces/ecc/skills/android-clean-architecture/SKILL.md +339 -339
  108. package/seed/marketplaces/ecc/skills/angular-developer/SKILL.md +154 -154
  109. package/seed/marketplaces/ecc/skills/angular-developer/references/angular-animations.md +160 -160
  110. package/seed/marketplaces/ecc/skills/angular-developer/references/angular-aria.md +410 -410
  111. package/seed/marketplaces/ecc/skills/angular-developer/references/cli.md +86 -86
  112. package/seed/marketplaces/ecc/skills/angular-developer/references/component-harnesses.md +59 -59
  113. package/seed/marketplaces/ecc/skills/angular-developer/references/component-styling.md +91 -91
  114. package/seed/marketplaces/ecc/skills/angular-developer/references/components.md +117 -117
  115. package/seed/marketplaces/ecc/skills/angular-developer/references/creating-services.md +97 -97
  116. package/seed/marketplaces/ecc/skills/angular-developer/references/data-resolvers.md +69 -69
  117. package/seed/marketplaces/ecc/skills/angular-developer/references/define-routes.md +67 -67
  118. package/seed/marketplaces/ecc/skills/angular-developer/references/defining-providers.md +72 -72
  119. package/seed/marketplaces/ecc/skills/angular-developer/references/di-fundamentals.md +120 -120
  120. package/seed/marketplaces/ecc/skills/angular-developer/references/e2e-testing.md +56 -56
  121. package/seed/marketplaces/ecc/skills/angular-developer/references/effects.md +83 -83
  122. package/seed/marketplaces/ecc/skills/angular-developer/references/hierarchical-injectors.md +43 -43
  123. package/seed/marketplaces/ecc/skills/angular-developer/references/host-elements.md +80 -80
  124. package/seed/marketplaces/ecc/skills/angular-developer/references/injection-context.md +63 -63
  125. package/seed/marketplaces/ecc/skills/angular-developer/references/inputs.md +101 -101
  126. package/seed/marketplaces/ecc/skills/angular-developer/references/linked-signal.md +59 -59
  127. package/seed/marketplaces/ecc/skills/angular-developer/references/loading-strategies.md +61 -61
  128. package/seed/marketplaces/ecc/skills/angular-developer/references/mcp.md +108 -108
  129. package/seed/marketplaces/ecc/skills/angular-developer/references/navigate-to-routes.md +69 -69
  130. package/seed/marketplaces/ecc/skills/angular-developer/references/outputs.md +86 -86
  131. package/seed/marketplaces/ecc/skills/angular-developer/references/reactive-forms.md +122 -122
  132. package/seed/marketplaces/ecc/skills/angular-developer/references/rendering-strategies.md +44 -44
  133. package/seed/marketplaces/ecc/skills/angular-developer/references/resource.md +77 -77
  134. package/seed/marketplaces/ecc/skills/angular-developer/references/route-animations.md +56 -56
  135. package/seed/marketplaces/ecc/skills/angular-developer/references/route-guards.md +52 -52
  136. package/seed/marketplaces/ecc/skills/angular-developer/references/router-lifecycle.md +45 -45
  137. package/seed/marketplaces/ecc/skills/angular-developer/references/router-testing.md +87 -87
  138. package/seed/marketplaces/ecc/skills/angular-developer/references/show-routes-with-outlets.md +68 -68
  139. package/seed/marketplaces/ecc/skills/angular-developer/references/signal-forms.md +795 -795
  140. package/seed/marketplaces/ecc/skills/angular-developer/references/signals-overview.md +94 -94
  141. package/seed/marketplaces/ecc/skills/angular-developer/references/tailwind-css.md +69 -69
  142. package/seed/marketplaces/ecc/skills/angular-developer/references/template-driven-forms.md +114 -114
  143. package/seed/marketplaces/ecc/skills/angular-developer/references/testing-fundamentals.md +65 -65
  144. package/seed/marketplaces/ecc/skills/api-connector-builder/SKILL.md +120 -120
  145. package/seed/marketplaces/ecc/skills/api-design/SKILL.md +523 -523
  146. package/seed/marketplaces/ecc/skills/architecture-decision-records/SKILL.md +179 -179
  147. package/seed/marketplaces/ecc/skills/article-writing/SKILL.md +79 -79
  148. package/seed/marketplaces/ecc/skills/automation-audit-ops/SKILL.md +142 -142
  149. package/seed/marketplaces/ecc/skills/autonomous-agent-harness/SKILL.md +273 -273
  150. package/seed/marketplaces/ecc/skills/autonomous-loops/SKILL.md +610 -610
  151. package/seed/marketplaces/ecc/skills/backend-patterns/SKILL.md +561 -561
  152. package/seed/marketplaces/ecc/skills/benchmark/SKILL.md +93 -93
  153. package/seed/marketplaces/ecc/skills/benchmark-optimization-loop/SKILL.md +69 -69
  154. package/seed/marketplaces/ecc/skills/blender-motion-state-inspection/SKILL.md +164 -164
  155. package/seed/marketplaces/ecc/skills/blueprint/SKILL.md +105 -105
  156. package/seed/marketplaces/ecc/skills/brand-voice/SKILL.md +97 -97
  157. package/seed/marketplaces/ecc/skills/brand-voice/references/voice-profile-schema.md +55 -55
  158. package/seed/marketplaces/ecc/skills/browser-qa/SKILL.md +87 -87
  159. package/seed/marketplaces/ecc/skills/bun-runtime/SKILL.md +84 -84
  160. package/seed/marketplaces/ecc/skills/canary-watch/SKILL.md +107 -107
  161. package/seed/marketplaces/ecc/skills/carrier-relationship-management/SKILL.md +212 -212
  162. package/seed/marketplaces/ecc/skills/cisco-ios-patterns/SKILL.md +163 -163
  163. package/seed/marketplaces/ecc/skills/ck/SKILL.md +147 -147
  164. package/seed/marketplaces/ecc/skills/ck/commands/forget.mjs +44 -44
  165. package/seed/marketplaces/ecc/skills/ck/commands/info.mjs +24 -24
  166. package/seed/marketplaces/ecc/skills/ck/commands/init.mjs +143 -143
  167. package/seed/marketplaces/ecc/skills/ck/commands/list.mjs +40 -40
  168. package/seed/marketplaces/ecc/skills/ck/commands/migrate.mjs +202 -202
  169. package/seed/marketplaces/ecc/skills/ck/commands/resume.mjs +36 -36
  170. package/seed/marketplaces/ecc/skills/ck/commands/save.mjs +210 -210
  171. package/seed/marketplaces/ecc/skills/ck/commands/shared.mjs +387 -387
  172. package/seed/marketplaces/ecc/skills/ck/hooks/session-start.mjs +224 -224
  173. package/seed/marketplaces/ecc/skills/claude-devfleet/SKILL.md +103 -103
  174. package/seed/marketplaces/ecc/skills/click-path-audit/SKILL.md +244 -244
  175. package/seed/marketplaces/ecc/skills/clickhouse-io/SKILL.md +439 -439
  176. package/seed/marketplaces/ecc/skills/code-tour/SKILL.md +236 -236
  177. package/seed/marketplaces/ecc/skills/codebase-onboarding/SKILL.md +233 -233
  178. package/seed/marketplaces/ecc/skills/coding-standards/SKILL.md +549 -549
  179. package/seed/marketplaces/ecc/skills/compose-multiplatform-patterns/SKILL.md +299 -299
  180. package/seed/marketplaces/ecc/skills/configure-ecc/SKILL.md +384 -384
  181. package/seed/marketplaces/ecc/skills/connections-optimizer/SKILL.md +189 -189
  182. package/seed/marketplaces/ecc/skills/content-engine/SKILL.md +131 -131
  183. package/seed/marketplaces/ecc/skills/content-hash-cache-pattern/SKILL.md +161 -161
  184. package/seed/marketplaces/ecc/skills/context-budget/SKILL.md +135 -135
  185. package/seed/marketplaces/ecc/skills/continuous-agent-loop/SKILL.md +45 -45
  186. package/seed/marketplaces/ecc/skills/continuous-learning/SKILL.md +131 -131
  187. package/seed/marketplaces/ecc/skills/continuous-learning/config.json +18 -18
  188. package/seed/marketplaces/ecc/skills/continuous-learning/evaluate-session.sh +69 -69
  189. package/seed/marketplaces/ecc/skills/continuous-learning-v2/SKILL.md +360 -360
  190. package/seed/marketplaces/ecc/skills/continuous-learning-v2/agents/observer-loop.sh +322 -322
  191. package/seed/marketplaces/ecc/skills/continuous-learning-v2/agents/observer.md +198 -198
  192. package/seed/marketplaces/ecc/skills/continuous-learning-v2/agents/session-guardian.sh +150 -150
  193. package/seed/marketplaces/ecc/skills/continuous-learning-v2/agents/start-observer.sh +248 -248
  194. package/seed/marketplaces/ecc/skills/continuous-learning-v2/config.json +8 -8
  195. package/seed/marketplaces/ecc/skills/continuous-learning-v2/hooks/observe.sh +498 -498
  196. package/seed/marketplaces/ecc/skills/continuous-learning-v2/scripts/detect-project.sh +322 -322
  197. package/seed/marketplaces/ecc/skills/continuous-learning-v2/scripts/instinct-cli.py +1826 -1826
  198. package/seed/marketplaces/ecc/skills/continuous-learning-v2/scripts/lib/homunculus-dir.sh +31 -31
  199. package/seed/marketplaces/ecc/skills/continuous-learning-v2/scripts/migrate-homunculus.sh +62 -62
  200. package/seed/marketplaces/ecc/skills/continuous-learning-v2/scripts/test_parse_instinct.py +1018 -1018
  201. package/seed/marketplaces/ecc/skills/cost-aware-llm-pipeline/SKILL.md +183 -183
  202. package/seed/marketplaces/ecc/skills/cost-tracking/SKILL.md +147 -147
  203. package/seed/marketplaces/ecc/skills/council/SKILL.md +203 -203
  204. package/seed/marketplaces/ecc/skills/cpp-coding-standards/SKILL.md +723 -723
  205. package/seed/marketplaces/ecc/skills/cpp-testing/SKILL.md +324 -324
  206. package/seed/marketplaces/ecc/skills/crosspost/SKILL.md +111 -111
  207. package/seed/marketplaces/ecc/skills/csharp-testing/SKILL.md +321 -321
  208. package/seed/marketplaces/ecc/skills/customer-billing-ops/SKILL.md +140 -140
  209. package/seed/marketplaces/ecc/skills/customs-trade-compliance/SKILL.md +263 -263
  210. package/seed/marketplaces/ecc/skills/dart-flutter-patterns/SKILL.md +563 -563
  211. package/seed/marketplaces/ecc/skills/dashboard-builder/SKILL.md +108 -108
  212. package/seed/marketplaces/ecc/skills/data-scraper-agent/SKILL.md +764 -764
  213. package/seed/marketplaces/ecc/skills/data-throughput-accelerator/SKILL.md +72 -72
  214. package/seed/marketplaces/ecc/skills/database-migrations/SKILL.md +429 -429
  215. package/seed/marketplaces/ecc/skills/deep-research/SKILL.md +159 -159
  216. package/seed/marketplaces/ecc/skills/defi-amm-security/SKILL.md +166 -166
  217. package/seed/marketplaces/ecc/skills/deployment-patterns/SKILL.md +427 -427
  218. package/seed/marketplaces/ecc/skills/design-system/SKILL.md +82 -82
  219. package/seed/marketplaces/ecc/skills/django-celery/SKILL.md +457 -457
  220. package/seed/marketplaces/ecc/skills/django-patterns/SKILL.md +734 -734
  221. package/seed/marketplaces/ecc/skills/django-security/SKILL.md +593 -593
  222. package/seed/marketplaces/ecc/skills/django-tdd/SKILL.md +729 -729
  223. package/seed/marketplaces/ecc/skills/django-verification/SKILL.md +469 -469
  224. package/seed/marketplaces/ecc/skills/dmux-workflows/SKILL.md +191 -191
  225. package/seed/marketplaces/ecc/skills/docker-patterns/SKILL.md +364 -364
  226. package/seed/marketplaces/ecc/skills/documentation-lookup/SKILL.md +90 -90
  227. package/seed/marketplaces/ecc/skills/dotnet-patterns/SKILL.md +321 -321
  228. package/seed/marketplaces/ecc/skills/e2e-testing/SKILL.md +326 -326
  229. package/seed/marketplaces/ecc/skills/ecc-guide/SKILL.md +189 -189
  230. package/seed/marketplaces/ecc/skills/ecc-tools-cost-audit/SKILL.md +160 -160
  231. package/seed/marketplaces/ecc/skills/email-ops/SKILL.md +121 -121
  232. package/seed/marketplaces/ecc/skills/energy-procurement/SKILL.md +228 -228
  233. package/seed/marketplaces/ecc/skills/enterprise-agent-ops/SKILL.md +50 -50
  234. package/seed/marketplaces/ecc/skills/error-handling/SKILL.md +376 -376
  235. package/seed/marketplaces/ecc/skills/eval-harness/SKILL.md +270 -270
  236. package/seed/marketplaces/ecc/skills/evm-token-decimals/SKILL.md +130 -130
  237. package/seed/marketplaces/ecc/skills/exa-search/SKILL.md +107 -107
  238. package/seed/marketplaces/ecc/skills/fal-ai-media/SKILL.md +288 -288
  239. package/seed/marketplaces/ecc/skills/fastapi-patterns/SKILL.md +327 -327
  240. package/seed/marketplaces/ecc/skills/finance-billing-ops/SKILL.md +127 -127
  241. package/seed/marketplaces/ecc/skills/flox-environments/SKILL.md +496 -496
  242. package/seed/marketplaces/ecc/skills/flutter-dart-code-review/SKILL.md +435 -435
  243. package/seed/marketplaces/ecc/skills/foundation-models-on-device/SKILL.md +243 -243
  244. package/seed/marketplaces/ecc/skills/frontend-a11y/SKILL.md +446 -446
  245. package/seed/marketplaces/ecc/skills/frontend-design-direction/SKILL.md +92 -92
  246. package/seed/marketplaces/ecc/skills/frontend-patterns/SKILL.md +642 -642
  247. package/seed/marketplaces/ecc/skills/frontend-slides/SKILL.md +184 -184
  248. package/seed/marketplaces/ecc/skills/frontend-slides/STYLE_PRESETS.md +330 -330
  249. package/seed/marketplaces/ecc/skills/frontend-slides/animation-patterns.md +122 -122
  250. package/seed/marketplaces/ecc/skills/frontend-slides/html-template.md +419 -419
  251. package/seed/marketplaces/ecc/skills/frontend-slides/scripts/export-pdf.sh +418 -418
  252. package/seed/marketplaces/ecc/skills/frontend-slides/scripts/extract-pptx.py +96 -96
  253. package/seed/marketplaces/ecc/skills/frontend-slides/viewport-base.css +153 -153
  254. package/seed/marketplaces/ecc/skills/fsharp-testing/SKILL.md +280 -280
  255. package/seed/marketplaces/ecc/skills/gan-style-harness/SKILL.md +278 -278
  256. package/seed/marketplaces/ecc/skills/gateguard/SKILL.md +125 -125
  257. package/seed/marketplaces/ecc/skills/git-workflow/SKILL.md +715 -715
  258. package/seed/marketplaces/ecc/skills/github-ops/SKILL.md +144 -144
  259. package/seed/marketplaces/ecc/skills/golang-patterns/SKILL.md +674 -674
  260. package/seed/marketplaces/ecc/skills/golang-testing/SKILL.md +720 -720
  261. package/seed/marketplaces/ecc/skills/google-workspace-ops/SKILL.md +95 -95
  262. package/seed/marketplaces/ecc/skills/healthcare-cdss-patterns/SKILL.md +245 -245
  263. package/seed/marketplaces/ecc/skills/healthcare-emr-patterns/SKILL.md +159 -159
  264. package/seed/marketplaces/ecc/skills/healthcare-eval-harness/SKILL.md +207 -207
  265. package/seed/marketplaces/ecc/skills/healthcare-phi-compliance/SKILL.md +145 -145
  266. package/seed/marketplaces/ecc/skills/hermes-imports/SKILL.md +88 -88
  267. package/seed/marketplaces/ecc/skills/hexagonal-architecture/SKILL.md +276 -276
  268. package/seed/marketplaces/ecc/skills/hipaa-compliance/SKILL.md +78 -78
  269. package/seed/marketplaces/ecc/skills/homelab-network-readiness/SKILL.md +169 -169
  270. package/seed/marketplaces/ecc/skills/homelab-network-setup/SKILL.md +129 -129
  271. package/seed/marketplaces/ecc/skills/homelab-pihole-dns/SKILL.md +274 -274
  272. package/seed/marketplaces/ecc/skills/homelab-vlan-segmentation/SKILL.md +311 -311
  273. package/seed/marketplaces/ecc/skills/homelab-wireguard-vpn/SKILL.md +305 -305
  274. package/seed/marketplaces/ecc/skills/hookify-rules/SKILL.md +128 -128
  275. package/seed/marketplaces/ecc/skills/inventory-demand-planning/SKILL.md +247 -247
  276. package/seed/marketplaces/ecc/skills/investor-materials/SKILL.md +96 -96
  277. package/seed/marketplaces/ecc/skills/investor-outreach/SKILL.md +91 -91
  278. package/seed/marketplaces/ecc/skills/ios-icon-gen/SKILL.md +157 -157
  279. package/seed/marketplaces/ecc/skills/ios-icon-gen/scripts/generate_icons.swift +258 -258
  280. package/seed/marketplaces/ecc/skills/ios-icon-gen/scripts/iconify_gen.sh +235 -235
  281. package/seed/marketplaces/ecc/skills/iterative-retrieval/SKILL.md +211 -211
  282. package/seed/marketplaces/ecc/skills/ito-basket-compare/SKILL.md +63 -63
  283. package/seed/marketplaces/ecc/skills/ito-data-atlas-agent/SKILL.md +63 -63
  284. package/seed/marketplaces/ecc/skills/ito-market-intelligence/SKILL.md +60 -60
  285. package/seed/marketplaces/ecc/skills/ito-trade-planner/SKILL.md +67 -67
  286. package/seed/marketplaces/ecc/skills/java-coding-standards/SKILL.md +383 -383
  287. package/seed/marketplaces/ecc/skills/jira-integration/SKILL.md +293 -293
  288. package/seed/marketplaces/ecc/skills/jpa-patterns/SKILL.md +151 -151
  289. package/seed/marketplaces/ecc/skills/knowledge-ops/SKILL.md +154 -154
  290. package/seed/marketplaces/ecc/skills/kotlin-coroutines-flows/SKILL.md +284 -284
  291. package/seed/marketplaces/ecc/skills/kotlin-exposed-patterns/SKILL.md +719 -719
  292. package/seed/marketplaces/ecc/skills/kotlin-ktor-patterns/SKILL.md +689 -689
  293. package/seed/marketplaces/ecc/skills/kotlin-patterns/SKILL.md +711 -711
  294. package/seed/marketplaces/ecc/skills/kotlin-testing/SKILL.md +824 -824
  295. package/seed/marketplaces/ecc/skills/laravel-patterns/SKILL.md +415 -415
  296. package/seed/marketplaces/ecc/skills/laravel-plugin-discovery/SKILL.md +229 -229
  297. package/seed/marketplaces/ecc/skills/laravel-security/SKILL.md +285 -285
  298. package/seed/marketplaces/ecc/skills/laravel-tdd/SKILL.md +283 -283
  299. package/seed/marketplaces/ecc/skills/laravel-verification/SKILL.md +179 -179
  300. package/seed/marketplaces/ecc/skills/latency-critical-systems/SKILL.md +73 -73
  301. package/seed/marketplaces/ecc/skills/lead-intelligence/SKILL.md +321 -321
  302. package/seed/marketplaces/ecc/skills/lead-intelligence/agents/enrichment-agent.md +85 -85
  303. package/seed/marketplaces/ecc/skills/lead-intelligence/agents/mutual-mapper.md +75 -75
  304. package/seed/marketplaces/ecc/skills/lead-intelligence/agents/outreach-drafter.md +98 -98
  305. package/seed/marketplaces/ecc/skills/lead-intelligence/agents/signal-scorer.md +60 -60
  306. package/seed/marketplaces/ecc/skills/liquid-glass-design/SKILL.md +279 -279
  307. package/seed/marketplaces/ecc/skills/llm-trading-agent-security/SKILL.md +146 -146
  308. package/seed/marketplaces/ecc/skills/logistics-exception-management/SKILL.md +222 -222
  309. package/seed/marketplaces/ecc/skills/make-interfaces-feel-better/SKILL.md +151 -151
  310. package/seed/marketplaces/ecc/skills/manim-video/SKILL.md +89 -89
  311. package/seed/marketplaces/ecc/skills/manim-video/assets/network_graph_scene.py +52 -52
  312. package/seed/marketplaces/ecc/skills/market-research/SKILL.md +75 -75
  313. package/seed/marketplaces/ecc/skills/marketing-campaign/SKILL.md +113 -113
  314. package/seed/marketplaces/ecc/skills/mcp-server-patterns/SKILL.md +69 -69
  315. package/seed/marketplaces/ecc/skills/messages-ops/SKILL.md +104 -104
  316. package/seed/marketplaces/ecc/skills/mle-workflow/SKILL.md +346 -346
  317. package/seed/marketplaces/ecc/skills/motion-advanced/SKILL.md +596 -596
  318. package/seed/marketplaces/ecc/skills/motion-foundations/SKILL.md +299 -299
  319. package/seed/marketplaces/ecc/skills/motion-patterns/SKILL.md +435 -435
  320. package/seed/marketplaces/ecc/skills/motion-ui/SKILL.md +575 -575
  321. package/seed/marketplaces/ecc/skills/mysql-patterns/SKILL.md +412 -412
  322. package/seed/marketplaces/ecc/skills/nanoclaw-repl/SKILL.md +33 -33
  323. package/seed/marketplaces/ecc/skills/nestjs-patterns/SKILL.md +230 -230
  324. package/seed/marketplaces/ecc/skills/netmiko-ssh-automation/SKILL.md +173 -173
  325. package/seed/marketplaces/ecc/skills/network-bgp-diagnostics/SKILL.md +167 -167
  326. package/seed/marketplaces/ecc/skills/network-config-validation/SKILL.md +210 -210
  327. package/seed/marketplaces/ecc/skills/network-interface-health/SKILL.md +152 -152
  328. package/seed/marketplaces/ecc/skills/nextjs-turbopack/SKILL.md +57 -57
  329. package/seed/marketplaces/ecc/skills/nodejs-keccak256/SKILL.md +102 -102
  330. package/seed/marketplaces/ecc/skills/nutrient-document-processing/SKILL.md +167 -167
  331. package/seed/marketplaces/ecc/skills/nuxt4-patterns/SKILL.md +100 -100
  332. package/seed/marketplaces/ecc/skills/openclaw-persona-forge/SKILL.md +288 -288
  333. package/seed/marketplaces/ecc/skills/openclaw-persona-forge/gacha.py +224 -224
  334. package/seed/marketplaces/ecc/skills/openclaw-persona-forge/gacha.sh +5 -5
  335. package/seed/marketplaces/ecc/skills/openclaw-persona-forge/references/avatar-style.md +124 -124
  336. package/seed/marketplaces/ecc/skills/openclaw-persona-forge/references/boundary-rules.md +53 -53
  337. package/seed/marketplaces/ecc/skills/openclaw-persona-forge/references/error-handling.md +53 -53
  338. package/seed/marketplaces/ecc/skills/openclaw-persona-forge/references/identity-tension.md +48 -48
  339. package/seed/marketplaces/ecc/skills/openclaw-persona-forge/references/naming-system.md +39 -39
  340. package/seed/marketplaces/ecc/skills/openclaw-persona-forge/references/output-template.md +166 -166
  341. package/seed/marketplaces/ecc/skills/opensource-pipeline/SKILL.md +255 -255
  342. package/seed/marketplaces/ecc/skills/parallel-execution-optimizer/SKILL.md +72 -72
  343. package/seed/marketplaces/ecc/skills/perl-patterns/SKILL.md +504 -504
  344. package/seed/marketplaces/ecc/skills/perl-security/SKILL.md +503 -503
  345. package/seed/marketplaces/ecc/skills/perl-testing/SKILL.md +475 -475
  346. package/seed/marketplaces/ecc/skills/plan-orchestrate/SKILL.md +262 -262
  347. package/seed/marketplaces/ecc/skills/plankton-code-quality/SKILL.md +236 -236
  348. package/seed/marketplaces/ecc/skills/postgres-patterns/SKILL.md +147 -147
  349. package/seed/marketplaces/ecc/skills/prediction-market-oracle-research/SKILL.md +63 -63
  350. package/seed/marketplaces/ecc/skills/prediction-market-risk-review/SKILL.md +60 -60
  351. package/seed/marketplaces/ecc/skills/prisma-patterns/SKILL.md +371 -371
  352. package/seed/marketplaces/ecc/skills/product-capability/SKILL.md +141 -141
  353. package/seed/marketplaces/ecc/skills/product-lens/SKILL.md +92 -92
  354. package/seed/marketplaces/ecc/skills/production-audit/SKILL.md +206 -206
  355. package/seed/marketplaces/ecc/skills/production-scheduling/SKILL.md +238 -238
  356. package/seed/marketplaces/ecc/skills/project-flow-ops/SKILL.md +111 -111
  357. package/seed/marketplaces/ecc/skills/prompt-optimizer/SKILL.md +398 -398
  358. package/seed/marketplaces/ecc/skills/python-patterns/SKILL.md +750 -750
  359. package/seed/marketplaces/ecc/skills/python-testing/SKILL.md +816 -816
  360. package/seed/marketplaces/ecc/skills/pytorch-patterns/SKILL.md +396 -396
  361. package/seed/marketplaces/ecc/skills/quality-nonconformance/SKILL.md +260 -260
  362. package/seed/marketplaces/ecc/skills/quarkus-patterns/SKILL.md +722 -722
  363. package/seed/marketplaces/ecc/skills/quarkus-security/SKILL.md +467 -467
  364. package/seed/marketplaces/ecc/skills/quarkus-tdd/SKILL.md +811 -811
  365. package/seed/marketplaces/ecc/skills/quarkus-verification/SKILL.md +479 -479
  366. package/seed/marketplaces/ecc/skills/ralphinho-rfc-pipeline/SKILL.md +67 -67
  367. package/seed/marketplaces/ecc/skills/recsys-pipeline-architect/SKILL.md +114 -114
  368. package/seed/marketplaces/ecc/skills/recursive-decision-ledger/SKILL.md +79 -79
  369. package/seed/marketplaces/ecc/skills/redis-patterns/SKILL.md +403 -403
  370. package/seed/marketplaces/ecc/skills/regex-vs-llm-structured-text/SKILL.md +220 -220
  371. package/seed/marketplaces/ecc/skills/remotion-video-creation/SKILL.md +43 -43
  372. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/3d.md +86 -86
  373. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/animations.md +29 -29
  374. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/assets/charts-bar-chart.tsx +173 -173
  375. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/assets/text-animations-typewriter.tsx +100 -100
  376. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/assets/text-animations-word-highlight.tsx +108 -108
  377. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/assets.md +78 -78
  378. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/audio.md +172 -172
  379. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/calculate-metadata.md +104 -104
  380. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/can-decode.md +75 -75
  381. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/charts.md +58 -58
  382. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/compositions.md +146 -146
  383. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/display-captions.md +126 -126
  384. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/extract-frames.md +229 -229
  385. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/fonts.md +152 -152
  386. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/get-audio-duration.md +58 -58
  387. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/get-video-dimensions.md +68 -68
  388. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/get-video-duration.md +58 -58
  389. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/gifs.md +138 -138
  390. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/images.md +130 -130
  391. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/import-srt-captions.md +67 -67
  392. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/lottie.md +67 -67
  393. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/measuring-dom-nodes.md +34 -34
  394. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/measuring-text.md +143 -143
  395. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/sequencing.md +106 -106
  396. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/tailwind.md +11 -11
  397. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/text-animations.md +20 -20
  398. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/timing.md +179 -179
  399. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/transcribe-captions.md +19 -19
  400. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/transitions.md +122 -122
  401. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/trimming.md +52 -52
  402. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/videos.md +171 -171
  403. package/seed/marketplaces/ecc/skills/repo-scan/SKILL.md +78 -78
  404. package/seed/marketplaces/ecc/skills/research-ops/SKILL.md +112 -112
  405. package/seed/marketplaces/ecc/skills/returns-reverse-logistics/SKILL.md +240 -240
  406. package/seed/marketplaces/ecc/skills/rules-distill/SKILL.md +264 -264
  407. package/seed/marketplaces/ecc/skills/rules-distill/scripts/scan-rules.sh +58 -58
  408. package/seed/marketplaces/ecc/skills/rules-distill/scripts/scan-skills.sh +129 -129
  409. package/seed/marketplaces/ecc/skills/rust-patterns/SKILL.md +499 -499
  410. package/seed/marketplaces/ecc/skills/rust-testing/SKILL.md +500 -500
  411. package/seed/marketplaces/ecc/skills/safety-guard/SKILL.md +75 -75
  412. package/seed/marketplaces/ecc/skills/santa-method/SKILL.md +306 -306
  413. package/seed/marketplaces/ecc/skills/scientific-db-pubmed-database/SKILL.md +175 -175
  414. package/seed/marketplaces/ecc/skills/scientific-db-uspto-database/SKILL.md +177 -177
  415. package/seed/marketplaces/ecc/skills/scientific-pkg-gget/SKILL.md +166 -166
  416. package/seed/marketplaces/ecc/skills/scientific-thinking-literature-review/SKILL.md +192 -192
  417. package/seed/marketplaces/ecc/skills/scientific-thinking-scholar-evaluation/SKILL.md +160 -160
  418. package/seed/marketplaces/ecc/skills/search-first/SKILL.md +182 -182
  419. package/seed/marketplaces/ecc/skills/security-bounty-hunter/SKILL.md +99 -99
  420. package/seed/marketplaces/ecc/skills/security-review/SKILL.md +503 -503
  421. package/seed/marketplaces/ecc/skills/security-review/cloud-infrastructure-security.md +361 -361
  422. package/seed/marketplaces/ecc/skills/security-scan/SKILL.md +165 -165
  423. package/seed/marketplaces/ecc/skills/seo/SKILL.md +154 -154
  424. package/seed/marketplaces/ecc/skills/skill-comply/SKILL.md +58 -58
  425. package/seed/marketplaces/ecc/skills/skill-comply/fixtures/compliant_trace.jsonl +5 -5
  426. package/seed/marketplaces/ecc/skills/skill-comply/fixtures/noncompliant_trace.jsonl +3 -3
  427. package/seed/marketplaces/ecc/skills/skill-comply/fixtures/tdd_spec.yaml +44 -44
  428. package/seed/marketplaces/ecc/skills/skill-comply/prompts/classifier.md +24 -24
  429. package/seed/marketplaces/ecc/skills/skill-comply/prompts/scenario_generator.md +62 -62
  430. package/seed/marketplaces/ecc/skills/skill-comply/prompts/spec_generator.md +42 -42
  431. package/seed/marketplaces/ecc/skills/skill-comply/pyproject.toml +15 -15
  432. package/seed/marketplaces/ecc/skills/skill-comply/scripts/classifier.py +85 -85
  433. package/seed/marketplaces/ecc/skills/skill-comply/scripts/grader.py +124 -124
  434. package/seed/marketplaces/ecc/skills/skill-comply/scripts/parser.py +107 -107
  435. package/seed/marketplaces/ecc/skills/skill-comply/scripts/report.py +170 -170
  436. package/seed/marketplaces/ecc/skills/skill-comply/scripts/run.py +127 -127
  437. package/seed/marketplaces/ecc/skills/skill-comply/scripts/runner.py +186 -186
  438. package/seed/marketplaces/ecc/skills/skill-comply/scripts/scenario_generator.py +70 -70
  439. package/seed/marketplaces/ecc/skills/skill-comply/scripts/spec_generator.py +72 -72
  440. package/seed/marketplaces/ecc/skills/skill-comply/scripts/utils.py +13 -13
  441. package/seed/marketplaces/ecc/skills/skill-comply/tests/test_grader.py +197 -197
  442. package/seed/marketplaces/ecc/skills/skill-comply/tests/test_parser.py +90 -90
  443. package/seed/marketplaces/ecc/skills/skill-comply/tests/test_runner.py +172 -172
  444. package/seed/marketplaces/ecc/skills/skill-scout/SKILL.md +140 -140
  445. package/seed/marketplaces/ecc/skills/skill-stocktake/SKILL.md +194 -194
  446. package/seed/marketplaces/ecc/skills/skill-stocktake/scripts/quick-diff.sh +87 -87
  447. package/seed/marketplaces/ecc/skills/skill-stocktake/scripts/save-results.sh +56 -56
  448. package/seed/marketplaces/ecc/skills/skill-stocktake/scripts/scan.sh +170 -170
  449. package/seed/marketplaces/ecc/skills/social-graph-ranker/SKILL.md +154 -154
  450. package/seed/marketplaces/ecc/skills/social-publisher/SKILL.md +115 -115
  451. package/seed/marketplaces/ecc/skills/springboot-patterns/SKILL.md +314 -314
  452. package/seed/marketplaces/ecc/skills/springboot-security/SKILL.md +272 -272
  453. package/seed/marketplaces/ecc/skills/springboot-tdd/SKILL.md +158 -158
  454. package/seed/marketplaces/ecc/skills/springboot-verification/SKILL.md +231 -231
  455. package/seed/marketplaces/ecc/skills/strategic-compact/SKILL.md +131 -131
  456. package/seed/marketplaces/ecc/skills/swift-actor-persistence/SKILL.md +143 -143
  457. package/seed/marketplaces/ecc/skills/swift-concurrency-6-2/SKILL.md +216 -216
  458. package/seed/marketplaces/ecc/skills/swift-protocol-di-testing/SKILL.md +190 -190
  459. package/seed/marketplaces/ecc/skills/swiftui-patterns/SKILL.md +259 -259
  460. package/seed/marketplaces/ecc/skills/tdd-workflow/SKILL.md +463 -463
  461. package/seed/marketplaces/ecc/skills/team-builder/SKILL.md +168 -168
  462. package/seed/marketplaces/ecc/skills/terminal-ops/SKILL.md +109 -109
  463. package/seed/marketplaces/ecc/skills/tinystruct-patterns/SKILL.md +203 -203
  464. package/seed/marketplaces/ecc/skills/tinystruct-patterns/references/architecture.md +90 -90
  465. package/seed/marketplaces/ecc/skills/tinystruct-patterns/references/data-handling.md +60 -60
  466. package/seed/marketplaces/ecc/skills/tinystruct-patterns/references/database.md +99 -99
  467. package/seed/marketplaces/ecc/skills/tinystruct-patterns/references/routing.md +64 -64
  468. package/seed/marketplaces/ecc/skills/tinystruct-patterns/references/system-usage.md +97 -97
  469. package/seed/marketplaces/ecc/skills/tinystruct-patterns/references/testing.md +72 -72
  470. package/seed/marketplaces/ecc/skills/token-budget-advisor/SKILL.md +133 -133
  471. package/seed/marketplaces/ecc/skills/ui-demo/SKILL.md +465 -465
  472. package/seed/marketplaces/ecc/skills/ui-to-vue/SKILL.md +134 -134
  473. package/seed/marketplaces/ecc/skills/uncloud/SKILL.md +343 -343
  474. package/seed/marketplaces/ecc/skills/unified-notifications-ops/SKILL.md +187 -187
  475. package/seed/marketplaces/ecc/skills/verification-loop/SKILL.md +126 -126
  476. package/seed/marketplaces/ecc/skills/video-editing/SKILL.md +310 -310
  477. package/seed/marketplaces/ecc/skills/videodb/SKILL.md +374 -374
  478. package/seed/marketplaces/ecc/skills/videodb/reference/api-reference.md +550 -550
  479. package/seed/marketplaces/ecc/skills/videodb/reference/capture-reference.md +407 -407
  480. package/seed/marketplaces/ecc/skills/videodb/reference/capture.md +101 -101
  481. package/seed/marketplaces/ecc/skills/videodb/reference/editor.md +443 -443
  482. package/seed/marketplaces/ecc/skills/videodb/reference/generative.md +331 -331
  483. package/seed/marketplaces/ecc/skills/videodb/reference/rtstream-reference.md +564 -564
  484. package/seed/marketplaces/ecc/skills/videodb/reference/rtstream.md +65 -65
  485. package/seed/marketplaces/ecc/skills/videodb/reference/search.md +230 -230
  486. package/seed/marketplaces/ecc/skills/videodb/reference/streaming.md +406 -406
  487. package/seed/marketplaces/ecc/skills/videodb/reference/use-cases.md +118 -118
  488. package/seed/marketplaces/ecc/skills/videodb/scripts/ws_listener.py +282 -282
  489. package/seed/marketplaces/ecc/skills/visa-doc-translate/README.md +86 -86
  490. package/seed/marketplaces/ecc/skills/visa-doc-translate/SKILL.md +117 -117
  491. package/seed/marketplaces/ecc/skills/vite-patterns/SKILL.md +449 -449
  492. package/seed/marketplaces/ecc/skills/windows-desktop-e2e/SKILL.md +887 -887
  493. package/seed/marketplaces/ecc/skills/workspace-surface-audit/SKILL.md +125 -125
  494. package/seed/marketplaces/ecc/skills/x-api/SKILL.md +234 -234
  495. package/dist/SKILL-r7zmg5v7.md +0 -3
  496. package/dist/cli-mvr5k580.md +0 -3
  497. package/dist/server-zyhc2a9z.md +0 -3
@@ -1,764 +1,764 @@
1
- ---
2
- name: data-scraper-agent
3
- description: Build a fully automated AI-powered data collection agent for any public source — job boards, prices, news, GitHub, sports, anything. Scrapes on a schedule, enriches data with a free LLM (Gemini Flash), stores results in Notion/Sheets/Supabase, and learns from user feedback. Runs 100% free on GitHub Actions. Use when the user wants to monitor, collect, or track any public data automatically.
4
- origin: community
5
- ---
6
-
7
- # Data Scraper Agent
8
-
9
- Build a production-ready, AI-powered data collection agent for any public data source.
10
- Runs on a schedule, enriches results with a free LLM, stores to a database, and improves over time.
11
-
12
- **Stack: Python · Gemini Flash (free) · GitHub Actions (free) · Notion / Sheets / Supabase**
13
-
14
- ## When to Activate
15
-
16
- - User wants to scrape or monitor any public website or API
17
- - User says "build a bot that checks...", "monitor X for me", "collect data from..."
18
- - User wants to track jobs, prices, news, repos, sports scores, events, listings
19
- - User asks how to automate data collection without paying for hosting
20
- - User wants an agent that gets smarter over time based on their decisions
21
-
22
- ## Core Concepts
23
-
24
- ### The Three Layers
25
-
26
- Every data scraper agent has three layers:
27
-
28
- ```
29
- COLLECT → ENRICH → STORE
30
- │ │ │
31
- Scraper AI (LLM) Database
32
- runs on scores/ Notion /
33
- schedule summarises Sheets /
34
- & classifies Supabase
35
- ```
36
-
37
- ### Free Stack
38
-
39
- | Layer | Tool | Why |
40
- |---|---|---|
41
- | **Scraping** | `requests` + `BeautifulSoup` | No cost, covers 80% of public sites |
42
- | **JS-rendered sites** | `playwright` (free) | When HTML scraping fails |
43
- | **AI enrichment** | Gemini Flash via REST API | 500 req/day, 1M tokens/day — free |
44
- | **Storage** | Notion API | Free tier, great UI for review |
45
- | **Schedule** | GitHub Actions cron | Free for public repos |
46
- | **Learning** | JSON feedback file in repo | Zero infra, persists in git |
47
-
48
- ### AI Model Fallback Chain
49
-
50
- Build agents to auto-fallback across Gemini models on quota exhaustion:
51
-
52
- ```
53
- gemini-2.0-flash-lite (30 RPM) →
54
- gemini-2.0-flash (15 RPM) →
55
- gemini-2.5-flash (10 RPM) →
56
- gemini-flash-lite-latest (fallback)
57
- ```
58
-
59
- ### Batch API Calls for Efficiency
60
-
61
- Never call the LLM once per item. Always batch:
62
-
63
- ```python
64
- # BAD: 33 API calls for 33 items
65
- for item in items:
66
- result = call_ai(item) # 33 calls → hits rate limit
67
-
68
- # GOOD: 7 API calls for 33 items (batch size 5)
69
- for batch in chunks(items, size=5):
70
- results = call_ai(batch) # 7 calls → stays within free tier
71
- ```
72
-
73
- ---
74
-
75
- ## Workflow
76
-
77
- ### Step 1: Understand the Goal
78
-
79
- Ask the user:
80
-
81
- 1. **What to collect:** "What data source? URL / API / RSS / public endpoint?"
82
- 2. **What to extract:** "What fields matter? Title, price, URL, date, score?"
83
- 3. **How to store:** "Where should results go? Notion, Google Sheets, Supabase, or local file?"
84
- 4. **How to enrich:** "Do you want AI to score, summarise, classify, or match each item?"
85
- 5. **Frequency:** "How often should it run? Every hour, daily, weekly?"
86
-
87
- Common examples to prompt:
88
- - Job boards → score relevance to resume
89
- - Product prices → alert on drops
90
- - GitHub repos → summarise new releases
91
- - News feeds → classify by topic + sentiment
92
- - Sports results → extract stats to tracker
93
- - Events calendar → filter by interest
94
-
95
- ---
96
-
97
- ### Step 2: Design the Agent Architecture
98
-
99
- Generate this directory structure for the user:
100
-
101
- ```
102
- my-agent/
103
- ├── config.yaml # User customises this (keywords, filters, preferences)
104
- ├── profile/
105
- │ └── context.md # User context the AI uses (resume, interests, criteria)
106
- ├── scraper/
107
- │ ├── __init__.py
108
- │ ├── main.py # Orchestrator: scrape → enrich → store
109
- │ ├── filters.py # Rule-based pre-filter (fast, before AI)
110
- │ └── sources/
111
- │ ├── __init__.py
112
- │ └── source_name.py # One file per data source
113
- ├── ai/
114
- │ ├── __init__.py
115
- │ ├── client.py # Gemini REST client with model fallback
116
- │ ├── pipeline.py # Batch AI analysis
117
- │ ├── jd_fetcher.py # Fetch full content from URLs (optional)
118
- │ └── memory.py # Learn from user feedback
119
- ├── storage/
120
- │ ├── __init__.py
121
- │ └── notion_sync.py # Or sheets_sync.py / supabase_sync.py
122
- ├── data/
123
- │ └── feedback.json # User decision history (auto-updated)
124
- ├── .env.example
125
- ├── setup.py # One-time DB/schema creation
126
- ├── enrich_existing.py # Backfill AI scores on old rows
127
- ├── requirements.txt
128
- └── .github/
129
- └── workflows/
130
- └── scraper.yml # GitHub Actions schedule
131
- ```
132
-
133
- ---
134
-
135
- ### Step 3: Build the Scraper Source
136
-
137
- Template for any data source:
138
-
139
- ```python
140
- # scraper/sources/my_source.py
141
- """
142
- [Source Name] — scrapes [what] from [where].
143
- Method: [REST API / HTML scraping / RSS feed]
144
- """
145
- import requests
146
- from bs4 import BeautifulSoup
147
- from datetime import datetime, timezone
148
- from scraper.filters import is_relevant
149
-
150
- HEADERS = {
151
- "User-Agent": "Mozilla/5.0 (compatible; research-bot/1.0)",
152
- }
153
-
154
-
155
- def fetch() -> list[dict]:
156
- """
157
- Returns a list of items with consistent schema.
158
- Each item must have at minimum: name, url, date_found.
159
- """
160
- results = []
161
-
162
- # ---- REST API source ----
163
- resp = requests.get("https://api.example.com/items", headers=HEADERS, timeout=15)
164
- if resp.status_code == 200:
165
- for item in resp.json().get("results", []):
166
- if not is_relevant(item.get("title", "")):
167
- continue
168
- results.append(_normalise(item))
169
-
170
- return results
171
-
172
-
173
- def _normalise(raw: dict) -> dict:
174
- """Convert raw API/HTML data to the standard schema."""
175
- return {
176
- "name": raw.get("title", ""),
177
- "url": raw.get("link", ""),
178
- "source": "MySource",
179
- "date_found": datetime.now(timezone.utc).date().isoformat(),
180
- # add domain-specific fields here
181
- }
182
- ```
183
-
184
- **HTML scraping pattern:**
185
- ```python
186
- soup = BeautifulSoup(resp.text, "lxml")
187
- for card in soup.select("[class*='listing']"):
188
- title = card.select_one("h2, h3").get_text(strip=True)
189
- link = card.select_one("a")["href"]
190
- if not link.startswith("http"):
191
- link = f"https://example.com{link}"
192
- ```
193
-
194
- **RSS feed pattern:**
195
- ```python
196
- import xml.etree.ElementTree as ET
197
- root = ET.fromstring(resp.text)
198
- for item in root.findall(".//item"):
199
- title = item.findtext("title", "")
200
- link = item.findtext("link", "")
201
- ```
202
-
203
- ---
204
-
205
- ### Step 4: Build the Gemini AI Client
206
-
207
- ```python
208
- # ai/client.py
209
- import os, json, time, requests
210
-
211
- _last_call = 0.0
212
-
213
- MODEL_FALLBACK = [
214
- "gemini-2.0-flash-lite",
215
- "gemini-2.0-flash",
216
- "gemini-2.5-flash",
217
- "gemini-flash-lite-latest",
218
- ]
219
-
220
-
221
- def generate(prompt: str, model: str = "", rate_limit: float = 7.0) -> dict:
222
- """Call Gemini with auto-fallback on 429. Returns parsed JSON or {}."""
223
- global _last_call
224
-
225
- api_key = os.environ.get("GEMINI_API_KEY", "")
226
- if not api_key:
227
- return {}
228
-
229
- elapsed = time.time() - _last_call
230
- if elapsed < rate_limit:
231
- time.sleep(rate_limit - elapsed)
232
-
233
- models = [model] + [m for m in MODEL_FALLBACK if m != model] if model else MODEL_FALLBACK
234
- _last_call = time.time()
235
-
236
- for m in models:
237
- url = f"https://generativelanguage.googleapis.com/v1beta/models/{m}:generateContent?key={api_key}"
238
- payload = {
239
- "contents": [{"parts": [{"text": prompt}]}],
240
- "generationConfig": {
241
- "responseMimeType": "application/json",
242
- "temperature": 0.3,
243
- "maxOutputTokens": 2048,
244
- },
245
- }
246
- try:
247
- resp = requests.post(url, json=payload, timeout=30)
248
- if resp.status_code == 200:
249
- return _parse(resp)
250
- if resp.status_code in (429, 404):
251
- time.sleep(1)
252
- continue
253
- return {}
254
- except requests.RequestException:
255
- return {}
256
-
257
- return {}
258
-
259
-
260
- def _parse(resp) -> dict:
261
- try:
262
- text = (
263
- resp.json()
264
- .get("candidates", [{}])[0]
265
- .get("content", {})
266
- .get("parts", [{}])[0]
267
- .get("text", "")
268
- .strip()
269
- )
270
- if text.startswith("```"):
271
- text = text.split("\n", 1)[-1].rsplit("```", 1)[0]
272
- return json.loads(text)
273
- except (json.JSONDecodeError, KeyError):
274
- return {}
275
- ```
276
-
277
- ---
278
-
279
- ### Step 5: Build the AI Pipeline (Batch)
280
-
281
- ```python
282
- # ai/pipeline.py
283
- import json
284
- import yaml
285
- from pathlib import Path
286
- from ai.client import generate
287
-
288
- def analyse_batch(items: list[dict], context: str = "", preference_prompt: str = "") -> list[dict]:
289
- """Analyse items in batches. Returns items enriched with AI fields."""
290
- config = yaml.safe_load((Path(__file__).parent.parent / "config.yaml").read_text())
291
- model = config.get("ai", {}).get("model", "gemini-2.5-flash")
292
- rate_limit = config.get("ai", {}).get("rate_limit_seconds", 7.0)
293
- min_score = config.get("ai", {}).get("min_score", 0)
294
- batch_size = config.get("ai", {}).get("batch_size", 5)
295
-
296
- batches = [items[i:i + batch_size] for i in range(0, len(items), batch_size)]
297
- print(f" [AI] {len(items)} items → {len(batches)} API calls")
298
-
299
- enriched = []
300
- for i, batch in enumerate(batches):
301
- print(f" [AI] Batch {i + 1}/{len(batches)}...")
302
- prompt = _build_prompt(batch, context, preference_prompt, config)
303
- result = generate(prompt, model=model, rate_limit=rate_limit)
304
-
305
- analyses = result.get("analyses", [])
306
- for j, item in enumerate(batch):
307
- ai = analyses[j] if j < len(analyses) else {}
308
- if ai:
309
- score = max(0, min(100, int(ai.get("score", 0))))
310
- if min_score and score < min_score:
311
- continue
312
- enriched.append({**item, "ai_score": score, "ai_summary": ai.get("summary", ""), "ai_notes": ai.get("notes", "")})
313
- else:
314
- enriched.append(item)
315
-
316
- return enriched
317
-
318
-
319
- def _build_prompt(batch, context, preference_prompt, config):
320
- priorities = config.get("priorities", [])
321
- items_text = "\n\n".join(
322
- f"Item {i+1}: {json.dumps({k: v for k, v in item.items() if not k.startswith('_')})}"
323
- for i, item in enumerate(batch)
324
- )
325
-
326
- return f"""Analyse these {len(batch)} items and return a JSON object.
327
-
328
- # Items
329
- {items_text}
330
-
331
- # User Context
332
- {context[:800] if context else "Not provided"}
333
-
334
- # User Priorities
335
- {chr(10).join(f"- {p}" for p in priorities)}
336
-
337
- {preference_prompt}
338
-
339
- # Instructions
340
- Return: {{"analyses": [{{"score": <0-100>, "summary": "<2 sentences>", "notes": "<why this matches or doesn't>"}} for each item in order]}}
341
- Be concise. Score 90+=excellent match, 70-89=good, 50-69=ok, <50=weak."""
342
- ```
343
-
344
- ---
345
-
346
- ### Step 6: Build the Feedback Learning System
347
-
348
- ```python
349
- # ai/memory.py
350
- """Learn from user decisions to improve future scoring."""
351
- import json
352
- from pathlib import Path
353
-
354
- FEEDBACK_PATH = Path(__file__).parent.parent / "data" / "feedback.json"
355
-
356
-
357
- def load_feedback() -> dict:
358
- if FEEDBACK_PATH.exists():
359
- try:
360
- return json.loads(FEEDBACK_PATH.read_text())
361
- except (json.JSONDecodeError, OSError):
362
- pass
363
- return {"positive": [], "negative": []}
364
-
365
-
366
- def save_feedback(fb: dict):
367
- FEEDBACK_PATH.parent.mkdir(parents=True, exist_ok=True)
368
- FEEDBACK_PATH.write_text(json.dumps(fb, indent=2))
369
-
370
-
371
- def build_preference_prompt(feedback: dict, max_examples: int = 15) -> str:
372
- """Convert feedback history into a prompt bias section."""
373
- lines = []
374
- if feedback.get("positive"):
375
- lines.append("# Items the user LIKED (positive signal):")
376
- for e in feedback["positive"][-max_examples:]:
377
- lines.append(f"- {e}")
378
- if feedback.get("negative"):
379
- lines.append("\n# Items the user SKIPPED/REJECTED (negative signal):")
380
- for e in feedback["negative"][-max_examples:]:
381
- lines.append(f"- {e}")
382
- if lines:
383
- lines.append("\nUse these patterns to bias scoring on new items.")
384
- return "\n".join(lines)
385
- ```
386
-
387
- **Integration with your storage layer:** after each run, query your DB for items with positive/negative status and call `save_feedback()` with the extracted patterns.
388
-
389
- ---
390
-
391
- ### Step 7: Build Storage (Notion example)
392
-
393
- ```python
394
- # storage/notion_sync.py
395
- import os
396
- from notion_client import Client
397
- from notion_client.errors import APIResponseError
398
-
399
- _client = None
400
-
401
- def get_client():
402
- global _client
403
- if _client is None:
404
- _client = Client(auth=os.environ["NOTION_TOKEN"])
405
- return _client
406
-
407
- def get_existing_urls(db_id: str) -> set[str]:
408
- """Fetch all URLs already stored — used for deduplication."""
409
- client, seen, cursor = get_client(), set(), None
410
- while True:
411
- resp = client.databases.query(database_id=db_id, page_size=100, **{"start_cursor": cursor} if cursor else {})
412
- for page in resp["results"]:
413
- url = page["properties"].get("URL", {}).get("url", "")
414
- if url: seen.add(url)
415
- if not resp["has_more"]: break
416
- cursor = resp["next_cursor"]
417
- return seen
418
-
419
- def push_item(db_id: str, item: dict) -> bool:
420
- """Push one item to Notion. Returns True on success."""
421
- props = {
422
- "Name": {"title": [{"text": {"content": item.get("name", "")[:100]}}]},
423
- "URL": {"url": item.get("url")},
424
- "Source": {"select": {"name": item.get("source", "Unknown")}},
425
- "Date Found": {"date": {"start": item.get("date_found")}},
426
- "Status": {"select": {"name": "New"}},
427
- }
428
- # AI fields
429
- if item.get("ai_score") is not None:
430
- props["AI Score"] = {"number": item["ai_score"]}
431
- if item.get("ai_summary"):
432
- props["Summary"] = {"rich_text": [{"text": {"content": item["ai_summary"][:2000]}}]}
433
- if item.get("ai_notes"):
434
- props["Notes"] = {"rich_text": [{"text": {"content": item["ai_notes"][:2000]}}]}
435
-
436
- try:
437
- get_client().pages.create(parent={"database_id": db_id}, properties=props)
438
- return True
439
- except APIResponseError as e:
440
- print(f"[notion] Push failed: {e}")
441
- return False
442
-
443
- def sync(db_id: str, items: list[dict]) -> tuple[int, int]:
444
- existing = get_existing_urls(db_id)
445
- added = skipped = 0
446
- for item in items:
447
- if item.get("url") in existing:
448
- skipped += 1; continue
449
- if push_item(db_id, item):
450
- added += 1; existing.add(item["url"])
451
- else:
452
- skipped += 1
453
- return added, skipped
454
- ```
455
-
456
- ---
457
-
458
- ### Step 8: Orchestrate in main.py
459
-
460
- ```python
461
- # scraper/main.py
462
- import os, sys, yaml
463
- from pathlib import Path
464
- from dotenv import load_dotenv
465
-
466
- load_dotenv()
467
-
468
- from scraper.sources import my_source # add your sources
469
-
470
- # NOTE: This example uses Notion. If storage.provider is "sheets" or "supabase",
471
- # replace this import with storage.sheets_sync or storage.supabase_sync and update
472
- # the env var and sync() call accordingly.
473
- from storage.notion_sync import sync
474
-
475
- SOURCES = [
476
- ("My Source", my_source.fetch),
477
- ]
478
-
479
- def ai_enabled():
480
- return bool(os.environ.get("GEMINI_API_KEY"))
481
-
482
- def main():
483
- config = yaml.safe_load((Path(__file__).parent.parent / "config.yaml").read_text())
484
- provider = config.get("storage", {}).get("provider", "notion")
485
-
486
- # Resolve the storage target identifier from env based on provider
487
- if provider == "notion":
488
- db_id = os.environ.get("NOTION_DATABASE_ID")
489
- if not db_id:
490
- print("ERROR: NOTION_DATABASE_ID not set"); sys.exit(1)
491
- else:
492
- # Extend here for sheets (SHEET_ID) or supabase (SUPABASE_TABLE) etc.
493
- print(f"ERROR: provider '{provider}' not yet wired in main.py"); sys.exit(1)
494
-
495
- config = yaml.safe_load((Path(__file__).parent.parent / "config.yaml").read_text())
496
- all_items = []
497
-
498
- for name, fetch_fn in SOURCES:
499
- try:
500
- items = fetch_fn()
501
- print(f"[{name}] {len(items)} items")
502
- all_items.extend(items)
503
- except Exception as e:
504
- print(f"[{name}] FAILED: {e}")
505
-
506
- # Deduplicate by URL
507
- seen, deduped = set(), []
508
- for item in all_items:
509
- if (url := item.get("url", "")) and url not in seen:
510
- seen.add(url); deduped.append(item)
511
-
512
- print(f"Unique items: {len(deduped)}")
513
-
514
- if ai_enabled() and deduped:
515
- from ai.memory import load_feedback, build_preference_prompt
516
- from ai.pipeline import analyse_batch
517
-
518
- # load_feedback() reads data/feedback.json written by your feedback sync script.
519
- # To keep it current, implement a separate feedback_sync.py that queries your
520
- # storage provider for items with positive/negative statuses and calls save_feedback().
521
- feedback = load_feedback()
522
- preference = build_preference_prompt(feedback)
523
- context_path = Path(__file__).parent.parent / "profile" / "context.md"
524
- context = context_path.read_text() if context_path.exists() else ""
525
- deduped = analyse_batch(deduped, context=context, preference_prompt=preference)
526
- else:
527
- print("[AI] Skipped — GEMINI_API_KEY not set")
528
-
529
- added, skipped = sync(db_id, deduped)
530
- print(f"Done — {added} new, {skipped} existing")
531
-
532
- if __name__ == "__main__":
533
- main()
534
- ```
535
-
536
- ---
537
-
538
- ### Step 9: GitHub Actions Workflow
539
-
540
- ```yaml
541
- # .github/workflows/scraper.yml
542
- name: Data Scraper Agent
543
-
544
- on:
545
- schedule:
546
- - cron: "0 */3 * * *" # every 3 hours — adjust to your needs
547
- workflow_dispatch: # allow manual trigger
548
-
549
- permissions:
550
- contents: write # required for the feedback-history commit step
551
-
552
- jobs:
553
- scrape:
554
- runs-on: ubuntu-latest
555
- timeout-minutes: 20
556
-
557
- steps:
558
- - uses: actions/checkout@v4
559
-
560
- - uses: actions/setup-python@v5
561
- with:
562
- python-version: "3.11"
563
- cache: "pip"
564
-
565
- - run: pip install -r requirements.txt
566
-
567
- # Uncomment if Playwright is enabled in requirements.txt
568
- # - name: Install Playwright browsers
569
- # run: python -m playwright install chromium --with-deps
570
-
571
- - name: Run agent
572
- env:
573
- NOTION_TOKEN: ${{ secrets.NOTION_TOKEN }}
574
- NOTION_DATABASE_ID: ${{ secrets.NOTION_DATABASE_ID }}
575
- GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }}
576
- run: python -m scraper.main
577
-
578
- - name: Commit feedback history
579
- run: |
580
- git config user.name "github-actions[bot]"
581
- git config user.email "github-actions[bot]@users.noreply.github.com"
582
- git add data/feedback.json || true
583
- git diff --cached --quiet || git commit -m "chore: update feedback history"
584
- git push
585
- ```
586
-
587
- ---
588
-
589
- ### Step 10: config.yaml Template
590
-
591
- ```yaml
592
- # Customise this file — no code changes needed
593
-
594
- # What to collect (pre-filter before AI)
595
- filters:
596
- required_keywords: [] # item must contain at least one
597
- blocked_keywords: [] # item must not contain any
598
-
599
- # Your priorities — AI uses these for scoring
600
- priorities:
601
- - "example priority 1"
602
- - "example priority 2"
603
-
604
- # Storage
605
- storage:
606
- provider: "notion" # notion | sheets | supabase | sqlite
607
-
608
- # Feedback learning
609
- feedback:
610
- positive_statuses: ["Saved", "Applied", "Interested"]
611
- negative_statuses: ["Skip", "Rejected", "Not relevant"]
612
-
613
- # AI settings
614
- ai:
615
- enabled: true
616
- model: "gemini-2.5-flash"
617
- min_score: 0 # filter out items below this score
618
- rate_limit_seconds: 7 # seconds between API calls
619
- batch_size: 5 # items per API call
620
- ```
621
-
622
- ---
623
-
624
- ## Common Scraping Patterns
625
-
626
- ### Pattern 1: REST API (easiest)
627
- ```python
628
- resp = requests.get(url, params={"q": query}, headers=HEADERS, timeout=15)
629
- items = resp.json().get("results", [])
630
- ```
631
-
632
- ### Pattern 2: HTML Scraping
633
- ```python
634
- soup = BeautifulSoup(resp.text, "lxml")
635
- for card in soup.select(".listing-card"):
636
- title = card.select_one("h2").get_text(strip=True)
637
- href = card.select_one("a")["href"]
638
- ```
639
-
640
- ### Pattern 3: RSS Feed
641
- ```python
642
- import xml.etree.ElementTree as ET
643
- root = ET.fromstring(resp.text)
644
- for item in root.findall(".//item"):
645
- title = item.findtext("title", "")
646
- link = item.findtext("link", "")
647
- pub_date = item.findtext("pubDate", "")
648
- ```
649
-
650
- ### Pattern 4: Paginated API
651
- ```python
652
- page = 1
653
- while True:
654
- resp = requests.get(url, params={"page": page, "limit": 50}, timeout=15)
655
- data = resp.json()
656
- items = data.get("results", [])
657
- if not items:
658
- break
659
- for item in items:
660
- results.append(_normalise(item))
661
- if not data.get("has_more"):
662
- break
663
- page += 1
664
- ```
665
-
666
- ### Pattern 5: JS-Rendered Pages (Playwright)
667
- ```python
668
- from playwright.sync_api import sync_playwright
669
-
670
- with sync_playwright() as p:
671
- browser = p.chromium.launch()
672
- page = browser.new_page()
673
- page.goto(url)
674
- page.wait_for_selector(".listing")
675
- html = page.content()
676
- browser.close()
677
-
678
- soup = BeautifulSoup(html, "lxml")
679
- ```
680
-
681
- ---
682
-
683
- ## Anti-Patterns to Avoid
684
-
685
- | Anti-pattern | Problem | Fix |
686
- |---|---|---|
687
- | One LLM call per item | Hits rate limits instantly | Batch 5 items per call |
688
- | Hardcoded keywords in code | Not reusable | Move all config to `config.yaml` |
689
- | Scraping without rate limit | IP ban | Add `time.sleep(1)` between requests |
690
- | Storing secrets in code | Security risk | Always use `.env` + GitHub Secrets |
691
- | No deduplication | Duplicate rows pile up | Always check URL before pushing |
692
- | Ignoring `robots.txt` | Legal/ethical risk | Respect crawl rules; use public APIs when available |
693
- | JS-rendered sites with `requests` | Empty response | Use Playwright or look for the underlying API |
694
- | `maxOutputTokens` too low | Truncated JSON, parse error | Use 2048+ for batch responses |
695
-
696
- ---
697
-
698
- ## Free Tier Limits Reference
699
-
700
- | Service | Free Limit | Typical Usage |
701
- |---|---|---|
702
- | Gemini Flash Lite | 30 RPM, 1500 RPD | ~56 req/day at 3-hr intervals |
703
- | Gemini 2.0 Flash | 15 RPM, 1500 RPD | Good fallback |
704
- | Gemini 2.5 Flash | 10 RPM, 500 RPD | Use sparingly |
705
- | GitHub Actions | Unlimited (public repos) | ~20 min/day |
706
- | Notion API | Unlimited | ~200 writes/day |
707
- | Supabase | 500MB DB, 2GB transfer | Fine for most agents |
708
- | Google Sheets API | 300 req/min | Works for small agents |
709
-
710
- ---
711
-
712
- ## Requirements Template
713
-
714
- ```
715
- requests==2.31.0
716
- beautifulsoup4==4.12.3
717
- lxml==5.1.0
718
- python-dotenv==1.0.1
719
- pyyaml==6.0.2
720
- notion-client==2.2.1 # if using Notion
721
- # playwright==1.40.0 # uncomment for JS-rendered sites
722
- ```
723
-
724
- ---
725
-
726
- ## Quality Checklist
727
-
728
- Before marking the agent complete:
729
-
730
- - [ ] `config.yaml` controls all user-facing settings — no hardcoded values
731
- - [ ] `profile/context.md` holds user-specific context for AI matching
732
- - [ ] Deduplication by URL before every storage push
733
- - [ ] Gemini client has model fallback chain (4 models)
734
- - [ ] Batch size ≤ 5 items per API call
735
- - [ ] `maxOutputTokens` ≥ 2048
736
- - [ ] `.env` is in `.gitignore`
737
- - [ ] `.env.example` provided for onboarding
738
- - [ ] `setup.py` creates DB schema on first run
739
- - [ ] `enrich_existing.py` backfills AI scores on old rows
740
- - [ ] GitHub Actions workflow commits `feedback.json` after each run
741
- - [ ] README covers: setup in < 5 minutes, required secrets, customisation
742
-
743
- ---
744
-
745
- ## Real-World Examples
746
-
747
- ```
748
- "Build me an agent that monitors Hacker News for AI startup funding news"
749
- "Scrape product prices from 3 e-commerce sites and alert when they drop"
750
- "Track new GitHub repos tagged with 'llm' or 'agents' — summarise each one"
751
- "Collect Chief of Staff job listings from LinkedIn and Cutshort into Notion"
752
- "Monitor a subreddit for posts mentioning my company — classify sentiment"
753
- "Scrape new academic papers from arXiv on a topic I care about daily"
754
- "Track sports fixture results and keep a running table in Google Sheets"
755
- "Build a real estate listing watcher — alert on new properties under ₹1 Cr"
756
- ```
757
-
758
- ---
759
-
760
- ## Reference Implementation
761
-
762
- A complete working agent built with this exact architecture would scrape 4+ sources,
763
- batch Gemini calls, learn from Applied/Rejected decisions stored in Notion, and run
764
- 100% free on GitHub Actions. Follow Steps 1–9 above to build your own.
1
+ ---
2
+ name: data-scraper-agent
3
+ description: Build a fully automated AI-powered data collection agent for any public source — job boards, prices, news, GitHub, sports, anything. Scrapes on a schedule, enriches data with a free LLM (Gemini Flash), stores results in Notion/Sheets/Supabase, and learns from user feedback. Runs 100% free on GitHub Actions. Use when the user wants to monitor, collect, or track any public data automatically.
4
+ origin: community
5
+ ---
6
+
7
+ # Data Scraper Agent
8
+
9
+ Build a production-ready, AI-powered data collection agent for any public data source.
10
+ Runs on a schedule, enriches results with a free LLM, stores to a database, and improves over time.
11
+
12
+ **Stack: Python · Gemini Flash (free) · GitHub Actions (free) · Notion / Sheets / Supabase**
13
+
14
+ ## When to Activate
15
+
16
+ - User wants to scrape or monitor any public website or API
17
+ - User says "build a bot that checks...", "monitor X for me", "collect data from..."
18
+ - User wants to track jobs, prices, news, repos, sports scores, events, listings
19
+ - User asks how to automate data collection without paying for hosting
20
+ - User wants an agent that gets smarter over time based on their decisions
21
+
22
+ ## Core Concepts
23
+
24
+ ### The Three Layers
25
+
26
+ Every data scraper agent has three layers:
27
+
28
+ ```
29
+ COLLECT → ENRICH → STORE
30
+ │ │ │
31
+ Scraper AI (LLM) Database
32
+ runs on scores/ Notion /
33
+ schedule summarises Sheets /
34
+ & classifies Supabase
35
+ ```
36
+
37
+ ### Free Stack
38
+
39
+ | Layer | Tool | Why |
40
+ |---|---|---|
41
+ | **Scraping** | `requests` + `BeautifulSoup` | No cost, covers 80% of public sites |
42
+ | **JS-rendered sites** | `playwright` (free) | When HTML scraping fails |
43
+ | **AI enrichment** | Gemini Flash via REST API | 500 req/day, 1M tokens/day — free |
44
+ | **Storage** | Notion API | Free tier, great UI for review |
45
+ | **Schedule** | GitHub Actions cron | Free for public repos |
46
+ | **Learning** | JSON feedback file in repo | Zero infra, persists in git |
47
+
48
+ ### AI Model Fallback Chain
49
+
50
+ Build agents to auto-fallback across Gemini models on quota exhaustion:
51
+
52
+ ```
53
+ gemini-2.0-flash-lite (30 RPM) →
54
+ gemini-2.0-flash (15 RPM) →
55
+ gemini-2.5-flash (10 RPM) →
56
+ gemini-flash-lite-latest (fallback)
57
+ ```
58
+
59
+ ### Batch API Calls for Efficiency
60
+
61
+ Never call the LLM once per item. Always batch:
62
+
63
+ ```python
64
+ # BAD: 33 API calls for 33 items
65
+ for item in items:
66
+ result = call_ai(item) # 33 calls → hits rate limit
67
+
68
+ # GOOD: 7 API calls for 33 items (batch size 5)
69
+ for batch in chunks(items, size=5):
70
+ results = call_ai(batch) # 7 calls → stays within free tier
71
+ ```
72
+
73
+ ---
74
+
75
+ ## Workflow
76
+
77
+ ### Step 1: Understand the Goal
78
+
79
+ Ask the user:
80
+
81
+ 1. **What to collect:** "What data source? URL / API / RSS / public endpoint?"
82
+ 2. **What to extract:** "What fields matter? Title, price, URL, date, score?"
83
+ 3. **How to store:** "Where should results go? Notion, Google Sheets, Supabase, or local file?"
84
+ 4. **How to enrich:** "Do you want AI to score, summarise, classify, or match each item?"
85
+ 5. **Frequency:** "How often should it run? Every hour, daily, weekly?"
86
+
87
+ Common examples to prompt:
88
+ - Job boards → score relevance to resume
89
+ - Product prices → alert on drops
90
+ - GitHub repos → summarise new releases
91
+ - News feeds → classify by topic + sentiment
92
+ - Sports results → extract stats to tracker
93
+ - Events calendar → filter by interest
94
+
95
+ ---
96
+
97
+ ### Step 2: Design the Agent Architecture
98
+
99
+ Generate this directory structure for the user:
100
+
101
+ ```
102
+ my-agent/
103
+ ├── config.yaml # User customises this (keywords, filters, preferences)
104
+ ├── profile/
105
+ │ └── context.md # User context the AI uses (resume, interests, criteria)
106
+ ├── scraper/
107
+ │ ├── __init__.py
108
+ │ ├── main.py # Orchestrator: scrape → enrich → store
109
+ │ ├── filters.py # Rule-based pre-filter (fast, before AI)
110
+ │ └── sources/
111
+ │ ├── __init__.py
112
+ │ └── source_name.py # One file per data source
113
+ ├── ai/
114
+ │ ├── __init__.py
115
+ │ ├── client.py # Gemini REST client with model fallback
116
+ │ ├── pipeline.py # Batch AI analysis
117
+ │ ├── jd_fetcher.py # Fetch full content from URLs (optional)
118
+ │ └── memory.py # Learn from user feedback
119
+ ├── storage/
120
+ │ ├── __init__.py
121
+ │ └── notion_sync.py # Or sheets_sync.py / supabase_sync.py
122
+ ├── data/
123
+ │ └── feedback.json # User decision history (auto-updated)
124
+ ├── .env.example
125
+ ├── setup.py # One-time DB/schema creation
126
+ ├── enrich_existing.py # Backfill AI scores on old rows
127
+ ├── requirements.txt
128
+ └── .github/
129
+ └── workflows/
130
+ └── scraper.yml # GitHub Actions schedule
131
+ ```
132
+
133
+ ---
134
+
135
+ ### Step 3: Build the Scraper Source
136
+
137
+ Template for any data source:
138
+
139
+ ```python
140
+ # scraper/sources/my_source.py
141
+ """
142
+ [Source Name] — scrapes [what] from [where].
143
+ Method: [REST API / HTML scraping / RSS feed]
144
+ """
145
+ import requests
146
+ from bs4 import BeautifulSoup
147
+ from datetime import datetime, timezone
148
+ from scraper.filters import is_relevant
149
+
150
+ HEADERS = {
151
+ "User-Agent": "Mozilla/5.0 (compatible; research-bot/1.0)",
152
+ }
153
+
154
+
155
+ def fetch() -> list[dict]:
156
+ """
157
+ Returns a list of items with consistent schema.
158
+ Each item must have at minimum: name, url, date_found.
159
+ """
160
+ results = []
161
+
162
+ # ---- REST API source ----
163
+ resp = requests.get("https://api.example.com/items", headers=HEADERS, timeout=15)
164
+ if resp.status_code == 200:
165
+ for item in resp.json().get("results", []):
166
+ if not is_relevant(item.get("title", "")):
167
+ continue
168
+ results.append(_normalise(item))
169
+
170
+ return results
171
+
172
+
173
+ def _normalise(raw: dict) -> dict:
174
+ """Convert raw API/HTML data to the standard schema."""
175
+ return {
176
+ "name": raw.get("title", ""),
177
+ "url": raw.get("link", ""),
178
+ "source": "MySource",
179
+ "date_found": datetime.now(timezone.utc).date().isoformat(),
180
+ # add domain-specific fields here
181
+ }
182
+ ```
183
+
184
+ **HTML scraping pattern:**
185
+ ```python
186
+ soup = BeautifulSoup(resp.text, "lxml")
187
+ for card in soup.select("[class*='listing']"):
188
+ title = card.select_one("h2, h3").get_text(strip=True)
189
+ link = card.select_one("a")["href"]
190
+ if not link.startswith("http"):
191
+ link = f"https://example.com{link}"
192
+ ```
193
+
194
+ **RSS feed pattern:**
195
+ ```python
196
+ import xml.etree.ElementTree as ET
197
+ root = ET.fromstring(resp.text)
198
+ for item in root.findall(".//item"):
199
+ title = item.findtext("title", "")
200
+ link = item.findtext("link", "")
201
+ ```
202
+
203
+ ---
204
+
205
+ ### Step 4: Build the Gemini AI Client
206
+
207
+ ```python
208
+ # ai/client.py
209
+ import os, json, time, requests
210
+
211
+ _last_call = 0.0
212
+
213
+ MODEL_FALLBACK = [
214
+ "gemini-2.0-flash-lite",
215
+ "gemini-2.0-flash",
216
+ "gemini-2.5-flash",
217
+ "gemini-flash-lite-latest",
218
+ ]
219
+
220
+
221
+ def generate(prompt: str, model: str = "", rate_limit: float = 7.0) -> dict:
222
+ """Call Gemini with auto-fallback on 429. Returns parsed JSON or {}."""
223
+ global _last_call
224
+
225
+ api_key = os.environ.get("GEMINI_API_KEY", "")
226
+ if not api_key:
227
+ return {}
228
+
229
+ elapsed = time.time() - _last_call
230
+ if elapsed < rate_limit:
231
+ time.sleep(rate_limit - elapsed)
232
+
233
+ models = [model] + [m for m in MODEL_FALLBACK if m != model] if model else MODEL_FALLBACK
234
+ _last_call = time.time()
235
+
236
+ for m in models:
237
+ url = f"https://generativelanguage.googleapis.com/v1beta/models/{m}:generateContent?key={api_key}"
238
+ payload = {
239
+ "contents": [{"parts": [{"text": prompt}]}],
240
+ "generationConfig": {
241
+ "responseMimeType": "application/json",
242
+ "temperature": 0.3,
243
+ "maxOutputTokens": 2048,
244
+ },
245
+ }
246
+ try:
247
+ resp = requests.post(url, json=payload, timeout=30)
248
+ if resp.status_code == 200:
249
+ return _parse(resp)
250
+ if resp.status_code in (429, 404):
251
+ time.sleep(1)
252
+ continue
253
+ return {}
254
+ except requests.RequestException:
255
+ return {}
256
+
257
+ return {}
258
+
259
+
260
+ def _parse(resp) -> dict:
261
+ try:
262
+ text = (
263
+ resp.json()
264
+ .get("candidates", [{}])[0]
265
+ .get("content", {})
266
+ .get("parts", [{}])[0]
267
+ .get("text", "")
268
+ .strip()
269
+ )
270
+ if text.startswith("```"):
271
+ text = text.split("\n", 1)[-1].rsplit("```", 1)[0]
272
+ return json.loads(text)
273
+ except (json.JSONDecodeError, KeyError):
274
+ return {}
275
+ ```
276
+
277
+ ---
278
+
279
+ ### Step 5: Build the AI Pipeline (Batch)
280
+
281
+ ```python
282
+ # ai/pipeline.py
283
+ import json
284
+ import yaml
285
+ from pathlib import Path
286
+ from ai.client import generate
287
+
288
+ def analyse_batch(items: list[dict], context: str = "", preference_prompt: str = "") -> list[dict]:
289
+ """Analyse items in batches. Returns items enriched with AI fields."""
290
+ config = yaml.safe_load((Path(__file__).parent.parent / "config.yaml").read_text())
291
+ model = config.get("ai", {}).get("model", "gemini-2.5-flash")
292
+ rate_limit = config.get("ai", {}).get("rate_limit_seconds", 7.0)
293
+ min_score = config.get("ai", {}).get("min_score", 0)
294
+ batch_size = config.get("ai", {}).get("batch_size", 5)
295
+
296
+ batches = [items[i:i + batch_size] for i in range(0, len(items), batch_size)]
297
+ print(f" [AI] {len(items)} items → {len(batches)} API calls")
298
+
299
+ enriched = []
300
+ for i, batch in enumerate(batches):
301
+ print(f" [AI] Batch {i + 1}/{len(batches)}...")
302
+ prompt = _build_prompt(batch, context, preference_prompt, config)
303
+ result = generate(prompt, model=model, rate_limit=rate_limit)
304
+
305
+ analyses = result.get("analyses", [])
306
+ for j, item in enumerate(batch):
307
+ ai = analyses[j] if j < len(analyses) else {}
308
+ if ai:
309
+ score = max(0, min(100, int(ai.get("score", 0))))
310
+ if min_score and score < min_score:
311
+ continue
312
+ enriched.append({**item, "ai_score": score, "ai_summary": ai.get("summary", ""), "ai_notes": ai.get("notes", "")})
313
+ else:
314
+ enriched.append(item)
315
+
316
+ return enriched
317
+
318
+
319
+ def _build_prompt(batch, context, preference_prompt, config):
320
+ priorities = config.get("priorities", [])
321
+ items_text = "\n\n".join(
322
+ f"Item {i+1}: {json.dumps({k: v for k, v in item.items() if not k.startswith('_')})}"
323
+ for i, item in enumerate(batch)
324
+ )
325
+
326
+ return f"""Analyse these {len(batch)} items and return a JSON object.
327
+
328
+ # Items
329
+ {items_text}
330
+
331
+ # User Context
332
+ {context[:800] if context else "Not provided"}
333
+
334
+ # User Priorities
335
+ {chr(10).join(f"- {p}" for p in priorities)}
336
+
337
+ {preference_prompt}
338
+
339
+ # Instructions
340
+ Return: {{"analyses": [{{"score": <0-100>, "summary": "<2 sentences>", "notes": "<why this matches or doesn't>"}} for each item in order]}}
341
+ Be concise. Score 90+=excellent match, 70-89=good, 50-69=ok, <50=weak."""
342
+ ```
343
+
344
+ ---
345
+
346
+ ### Step 6: Build the Feedback Learning System
347
+
348
+ ```python
349
+ # ai/memory.py
350
+ """Learn from user decisions to improve future scoring."""
351
+ import json
352
+ from pathlib import Path
353
+
354
+ FEEDBACK_PATH = Path(__file__).parent.parent / "data" / "feedback.json"
355
+
356
+
357
+ def load_feedback() -> dict:
358
+ if FEEDBACK_PATH.exists():
359
+ try:
360
+ return json.loads(FEEDBACK_PATH.read_text())
361
+ except (json.JSONDecodeError, OSError):
362
+ pass
363
+ return {"positive": [], "negative": []}
364
+
365
+
366
+ def save_feedback(fb: dict):
367
+ FEEDBACK_PATH.parent.mkdir(parents=True, exist_ok=True)
368
+ FEEDBACK_PATH.write_text(json.dumps(fb, indent=2))
369
+
370
+
371
+ def build_preference_prompt(feedback: dict, max_examples: int = 15) -> str:
372
+ """Convert feedback history into a prompt bias section."""
373
+ lines = []
374
+ if feedback.get("positive"):
375
+ lines.append("# Items the user LIKED (positive signal):")
376
+ for e in feedback["positive"][-max_examples:]:
377
+ lines.append(f"- {e}")
378
+ if feedback.get("negative"):
379
+ lines.append("\n# Items the user SKIPPED/REJECTED (negative signal):")
380
+ for e in feedback["negative"][-max_examples:]:
381
+ lines.append(f"- {e}")
382
+ if lines:
383
+ lines.append("\nUse these patterns to bias scoring on new items.")
384
+ return "\n".join(lines)
385
+ ```
386
+
387
+ **Integration with your storage layer:** after each run, query your DB for items with positive/negative status and call `save_feedback()` with the extracted patterns.
388
+
389
+ ---
390
+
391
+ ### Step 7: Build Storage (Notion example)
392
+
393
+ ```python
394
+ # storage/notion_sync.py
395
+ import os
396
+ from notion_client import Client
397
+ from notion_client.errors import APIResponseError
398
+
399
+ _client = None
400
+
401
+ def get_client():
402
+ global _client
403
+ if _client is None:
404
+ _client = Client(auth=os.environ["NOTION_TOKEN"])
405
+ return _client
406
+
407
+ def get_existing_urls(db_id: str) -> set[str]:
408
+ """Fetch all URLs already stored — used for deduplication."""
409
+ client, seen, cursor = get_client(), set(), None
410
+ while True:
411
+ resp = client.databases.query(database_id=db_id, page_size=100, **{"start_cursor": cursor} if cursor else {})
412
+ for page in resp["results"]:
413
+ url = page["properties"].get("URL", {}).get("url", "")
414
+ if url: seen.add(url)
415
+ if not resp["has_more"]: break
416
+ cursor = resp["next_cursor"]
417
+ return seen
418
+
419
+ def push_item(db_id: str, item: dict) -> bool:
420
+ """Push one item to Notion. Returns True on success."""
421
+ props = {
422
+ "Name": {"title": [{"text": {"content": item.get("name", "")[:100]}}]},
423
+ "URL": {"url": item.get("url")},
424
+ "Source": {"select": {"name": item.get("source", "Unknown")}},
425
+ "Date Found": {"date": {"start": item.get("date_found")}},
426
+ "Status": {"select": {"name": "New"}},
427
+ }
428
+ # AI fields
429
+ if item.get("ai_score") is not None:
430
+ props["AI Score"] = {"number": item["ai_score"]}
431
+ if item.get("ai_summary"):
432
+ props["Summary"] = {"rich_text": [{"text": {"content": item["ai_summary"][:2000]}}]}
433
+ if item.get("ai_notes"):
434
+ props["Notes"] = {"rich_text": [{"text": {"content": item["ai_notes"][:2000]}}]}
435
+
436
+ try:
437
+ get_client().pages.create(parent={"database_id": db_id}, properties=props)
438
+ return True
439
+ except APIResponseError as e:
440
+ print(f"[notion] Push failed: {e}")
441
+ return False
442
+
443
+ def sync(db_id: str, items: list[dict]) -> tuple[int, int]:
444
+ existing = get_existing_urls(db_id)
445
+ added = skipped = 0
446
+ for item in items:
447
+ if item.get("url") in existing:
448
+ skipped += 1; continue
449
+ if push_item(db_id, item):
450
+ added += 1; existing.add(item["url"])
451
+ else:
452
+ skipped += 1
453
+ return added, skipped
454
+ ```
455
+
456
+ ---
457
+
458
+ ### Step 8: Orchestrate in main.py
459
+
460
+ ```python
461
+ # scraper/main.py
462
+ import os, sys, yaml
463
+ from pathlib import Path
464
+ from dotenv import load_dotenv
465
+
466
+ load_dotenv()
467
+
468
+ from scraper.sources import my_source # add your sources
469
+
470
+ # NOTE: This example uses Notion. If storage.provider is "sheets" or "supabase",
471
+ # replace this import with storage.sheets_sync or storage.supabase_sync and update
472
+ # the env var and sync() call accordingly.
473
+ from storage.notion_sync import sync
474
+
475
+ SOURCES = [
476
+ ("My Source", my_source.fetch),
477
+ ]
478
+
479
+ def ai_enabled():
480
+ return bool(os.environ.get("GEMINI_API_KEY"))
481
+
482
+ def main():
483
+ config = yaml.safe_load((Path(__file__).parent.parent / "config.yaml").read_text())
484
+ provider = config.get("storage", {}).get("provider", "notion")
485
+
486
+ # Resolve the storage target identifier from env based on provider
487
+ if provider == "notion":
488
+ db_id = os.environ.get("NOTION_DATABASE_ID")
489
+ if not db_id:
490
+ print("ERROR: NOTION_DATABASE_ID not set"); sys.exit(1)
491
+ else:
492
+ # Extend here for sheets (SHEET_ID) or supabase (SUPABASE_TABLE) etc.
493
+ print(f"ERROR: provider '{provider}' not yet wired in main.py"); sys.exit(1)
494
+
495
+ config = yaml.safe_load((Path(__file__).parent.parent / "config.yaml").read_text())
496
+ all_items = []
497
+
498
+ for name, fetch_fn in SOURCES:
499
+ try:
500
+ items = fetch_fn()
501
+ print(f"[{name}] {len(items)} items")
502
+ all_items.extend(items)
503
+ except Exception as e:
504
+ print(f"[{name}] FAILED: {e}")
505
+
506
+ # Deduplicate by URL
507
+ seen, deduped = set(), []
508
+ for item in all_items:
509
+ if (url := item.get("url", "")) and url not in seen:
510
+ seen.add(url); deduped.append(item)
511
+
512
+ print(f"Unique items: {len(deduped)}")
513
+
514
+ if ai_enabled() and deduped:
515
+ from ai.memory import load_feedback, build_preference_prompt
516
+ from ai.pipeline import analyse_batch
517
+
518
+ # load_feedback() reads data/feedback.json written by your feedback sync script.
519
+ # To keep it current, implement a separate feedback_sync.py that queries your
520
+ # storage provider for items with positive/negative statuses and calls save_feedback().
521
+ feedback = load_feedback()
522
+ preference = build_preference_prompt(feedback)
523
+ context_path = Path(__file__).parent.parent / "profile" / "context.md"
524
+ context = context_path.read_text() if context_path.exists() else ""
525
+ deduped = analyse_batch(deduped, context=context, preference_prompt=preference)
526
+ else:
527
+ print("[AI] Skipped — GEMINI_API_KEY not set")
528
+
529
+ added, skipped = sync(db_id, deduped)
530
+ print(f"Done — {added} new, {skipped} existing")
531
+
532
+ if __name__ == "__main__":
533
+ main()
534
+ ```
535
+
536
+ ---
537
+
538
+ ### Step 9: GitHub Actions Workflow
539
+
540
+ ```yaml
541
+ # .github/workflows/scraper.yml
542
+ name: Data Scraper Agent
543
+
544
+ on:
545
+ schedule:
546
+ - cron: "0 */3 * * *" # every 3 hours — adjust to your needs
547
+ workflow_dispatch: # allow manual trigger
548
+
549
+ permissions:
550
+ contents: write # required for the feedback-history commit step
551
+
552
+ jobs:
553
+ scrape:
554
+ runs-on: ubuntu-latest
555
+ timeout-minutes: 20
556
+
557
+ steps:
558
+ - uses: actions/checkout@v4
559
+
560
+ - uses: actions/setup-python@v5
561
+ with:
562
+ python-version: "3.11"
563
+ cache: "pip"
564
+
565
+ - run: pip install -r requirements.txt
566
+
567
+ # Uncomment if Playwright is enabled in requirements.txt
568
+ # - name: Install Playwright browsers
569
+ # run: python -m playwright install chromium --with-deps
570
+
571
+ - name: Run agent
572
+ env:
573
+ NOTION_TOKEN: ${{ secrets.NOTION_TOKEN }}
574
+ NOTION_DATABASE_ID: ${{ secrets.NOTION_DATABASE_ID }}
575
+ GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }}
576
+ run: python -m scraper.main
577
+
578
+ - name: Commit feedback history
579
+ run: |
580
+ git config user.name "github-actions[bot]"
581
+ git config user.email "github-actions[bot]@users.noreply.github.com"
582
+ git add data/feedback.json || true
583
+ git diff --cached --quiet || git commit -m "chore: update feedback history"
584
+ git push
585
+ ```
586
+
587
+ ---
588
+
589
+ ### Step 10: config.yaml Template
590
+
591
+ ```yaml
592
+ # Customise this file — no code changes needed
593
+
594
+ # What to collect (pre-filter before AI)
595
+ filters:
596
+ required_keywords: [] # item must contain at least one
597
+ blocked_keywords: [] # item must not contain any
598
+
599
+ # Your priorities — AI uses these for scoring
600
+ priorities:
601
+ - "example priority 1"
602
+ - "example priority 2"
603
+
604
+ # Storage
605
+ storage:
606
+ provider: "notion" # notion | sheets | supabase | sqlite
607
+
608
+ # Feedback learning
609
+ feedback:
610
+ positive_statuses: ["Saved", "Applied", "Interested"]
611
+ negative_statuses: ["Skip", "Rejected", "Not relevant"]
612
+
613
+ # AI settings
614
+ ai:
615
+ enabled: true
616
+ model: "gemini-2.5-flash"
617
+ min_score: 0 # filter out items below this score
618
+ rate_limit_seconds: 7 # seconds between API calls
619
+ batch_size: 5 # items per API call
620
+ ```
621
+
622
+ ---
623
+
624
+ ## Common Scraping Patterns
625
+
626
+ ### Pattern 1: REST API (easiest)
627
+ ```python
628
+ resp = requests.get(url, params={"q": query}, headers=HEADERS, timeout=15)
629
+ items = resp.json().get("results", [])
630
+ ```
631
+
632
+ ### Pattern 2: HTML Scraping
633
+ ```python
634
+ soup = BeautifulSoup(resp.text, "lxml")
635
+ for card in soup.select(".listing-card"):
636
+ title = card.select_one("h2").get_text(strip=True)
637
+ href = card.select_one("a")["href"]
638
+ ```
639
+
640
+ ### Pattern 3: RSS Feed
641
+ ```python
642
+ import xml.etree.ElementTree as ET
643
+ root = ET.fromstring(resp.text)
644
+ for item in root.findall(".//item"):
645
+ title = item.findtext("title", "")
646
+ link = item.findtext("link", "")
647
+ pub_date = item.findtext("pubDate", "")
648
+ ```
649
+
650
+ ### Pattern 4: Paginated API
651
+ ```python
652
+ page = 1
653
+ while True:
654
+ resp = requests.get(url, params={"page": page, "limit": 50}, timeout=15)
655
+ data = resp.json()
656
+ items = data.get("results", [])
657
+ if not items:
658
+ break
659
+ for item in items:
660
+ results.append(_normalise(item))
661
+ if not data.get("has_more"):
662
+ break
663
+ page += 1
664
+ ```
665
+
666
+ ### Pattern 5: JS-Rendered Pages (Playwright)
667
+ ```python
668
+ from playwright.sync_api import sync_playwright
669
+
670
+ with sync_playwright() as p:
671
+ browser = p.chromium.launch()
672
+ page = browser.new_page()
673
+ page.goto(url)
674
+ page.wait_for_selector(".listing")
675
+ html = page.content()
676
+ browser.close()
677
+
678
+ soup = BeautifulSoup(html, "lxml")
679
+ ```
680
+
681
+ ---
682
+
683
+ ## Anti-Patterns to Avoid
684
+
685
+ | Anti-pattern | Problem | Fix |
686
+ |---|---|---|
687
+ | One LLM call per item | Hits rate limits instantly | Batch 5 items per call |
688
+ | Hardcoded keywords in code | Not reusable | Move all config to `config.yaml` |
689
+ | Scraping without rate limit | IP ban | Add `time.sleep(1)` between requests |
690
+ | Storing secrets in code | Security risk | Always use `.env` + GitHub Secrets |
691
+ | No deduplication | Duplicate rows pile up | Always check URL before pushing |
692
+ | Ignoring `robots.txt` | Legal/ethical risk | Respect crawl rules; use public APIs when available |
693
+ | JS-rendered sites with `requests` | Empty response | Use Playwright or look for the underlying API |
694
+ | `maxOutputTokens` too low | Truncated JSON, parse error | Use 2048+ for batch responses |
695
+
696
+ ---
697
+
698
+ ## Free Tier Limits Reference
699
+
700
+ | Service | Free Limit | Typical Usage |
701
+ |---|---|---|
702
+ | Gemini Flash Lite | 30 RPM, 1500 RPD | ~56 req/day at 3-hr intervals |
703
+ | Gemini 2.0 Flash | 15 RPM, 1500 RPD | Good fallback |
704
+ | Gemini 2.5 Flash | 10 RPM, 500 RPD | Use sparingly |
705
+ | GitHub Actions | Unlimited (public repos) | ~20 min/day |
706
+ | Notion API | Unlimited | ~200 writes/day |
707
+ | Supabase | 500MB DB, 2GB transfer | Fine for most agents |
708
+ | Google Sheets API | 300 req/min | Works for small agents |
709
+
710
+ ---
711
+
712
+ ## Requirements Template
713
+
714
+ ```
715
+ requests==2.31.0
716
+ beautifulsoup4==4.12.3
717
+ lxml==5.1.0
718
+ python-dotenv==1.0.1
719
+ pyyaml==6.0.2
720
+ notion-client==2.2.1 # if using Notion
721
+ # playwright==1.40.0 # uncomment for JS-rendered sites
722
+ ```
723
+
724
+ ---
725
+
726
+ ## Quality Checklist
727
+
728
+ Before marking the agent complete:
729
+
730
+ - [ ] `config.yaml` controls all user-facing settings — no hardcoded values
731
+ - [ ] `profile/context.md` holds user-specific context for AI matching
732
+ - [ ] Deduplication by URL before every storage push
733
+ - [ ] Gemini client has model fallback chain (4 models)
734
+ - [ ] Batch size ≤ 5 items per API call
735
+ - [ ] `maxOutputTokens` ≥ 2048
736
+ - [ ] `.env` is in `.gitignore`
737
+ - [ ] `.env.example` provided for onboarding
738
+ - [ ] `setup.py` creates DB schema on first run
739
+ - [ ] `enrich_existing.py` backfills AI scores on old rows
740
+ - [ ] GitHub Actions workflow commits `feedback.json` after each run
741
+ - [ ] README covers: setup in < 5 minutes, required secrets, customisation
742
+
743
+ ---
744
+
745
+ ## Real-World Examples
746
+
747
+ ```
748
+ "Build me an agent that monitors Hacker News for AI startup funding news"
749
+ "Scrape product prices from 3 e-commerce sites and alert when they drop"
750
+ "Track new GitHub repos tagged with 'llm' or 'agents' — summarise each one"
751
+ "Collect Chief of Staff job listings from LinkedIn and Cutshort into Notion"
752
+ "Monitor a subreddit for posts mentioning my company — classify sentiment"
753
+ "Scrape new academic papers from arXiv on a topic I care about daily"
754
+ "Track sports fixture results and keep a running table in Google Sheets"
755
+ "Build a real estate listing watcher — alert on new properties under ₹1 Cr"
756
+ ```
757
+
758
+ ---
759
+
760
+ ## Reference Implementation
761
+
762
+ A complete working agent built with this exact architecture would scrape 4+ sources,
763
+ batch Gemini calls, learn from Applied/Rejected decisions stored in Notion, and run
764
+ 100% free on GitHub Actions. Follow Steps 1–9 above to build your own.