@funnycode/myclaude 0.1.55 → 0.1.60

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (499) hide show
  1. package/LICENSE +1 -1
  2. package/README.md +63 -0
  3. package/README.zh-CN.md +63 -0
  4. package/dist/SKILL-r7zmg5v7.md +3 -0
  5. package/dist/cli-mvr5k580.md +3 -0
  6. package/dist/myclaude.js +14230 -8799
  7. package/dist/myclaude.mjs +14230 -8799
  8. package/dist/server-zyhc2a9z.md +3 -0
  9. package/package.json +128 -128
  10. package/seed/marketplaces/ecc/.opencode/package-lock.json +169 -169
  11. package/seed/marketplaces/ecc/commands/aside.md +164 -164
  12. package/seed/marketplaces/ecc/commands/auto-update.md +28 -28
  13. package/seed/marketplaces/ecc/commands/build-fix.md +66 -66
  14. package/seed/marketplaces/ecc/commands/checkpoint.md +78 -78
  15. package/seed/marketplaces/ecc/commands/code-review.md +289 -289
  16. package/seed/marketplaces/ecc/commands/cost-report.md +107 -107
  17. package/seed/marketplaces/ecc/commands/cpp-build.md +173 -173
  18. package/seed/marketplaces/ecc/commands/cpp-review.md +132 -132
  19. package/seed/marketplaces/ecc/commands/cpp-test.md +251 -251
  20. package/seed/marketplaces/ecc/commands/ecc-guide.md +93 -93
  21. package/seed/marketplaces/ecc/commands/evolve.md +178 -178
  22. package/seed/marketplaces/ecc/commands/fastapi-review.md +39 -39
  23. package/seed/marketplaces/ecc/commands/feature-dev.md +49 -49
  24. package/seed/marketplaces/ecc/commands/flutter-build.md +164 -164
  25. package/seed/marketplaces/ecc/commands/flutter-review.md +116 -116
  26. package/seed/marketplaces/ecc/commands/flutter-test.md +144 -144
  27. package/seed/marketplaces/ecc/commands/gan-build.md +103 -103
  28. package/seed/marketplaces/ecc/commands/gan-design.md +39 -39
  29. package/seed/marketplaces/ecc/commands/go-build.md +183 -183
  30. package/seed/marketplaces/ecc/commands/go-review.md +148 -148
  31. package/seed/marketplaces/ecc/commands/go-test.md +268 -268
  32. package/seed/marketplaces/ecc/commands/gradle-build.md +70 -70
  33. package/seed/marketplaces/ecc/commands/harness-audit.md +84 -84
  34. package/seed/marketplaces/ecc/commands/hookify-configure.md +14 -14
  35. package/seed/marketplaces/ecc/commands/hookify-help.md +46 -46
  36. package/seed/marketplaces/ecc/commands/hookify-list.md +21 -21
  37. package/seed/marketplaces/ecc/commands/hookify.md +50 -50
  38. package/seed/marketplaces/ecc/commands/instinct-export.md +66 -66
  39. package/seed/marketplaces/ecc/commands/instinct-import.md +114 -114
  40. package/seed/marketplaces/ecc/commands/instinct-status.md +59 -59
  41. package/seed/marketplaces/ecc/commands/jira.md +106 -106
  42. package/seed/marketplaces/ecc/commands/kotlin-build.md +174 -174
  43. package/seed/marketplaces/ecc/commands/kotlin-review.md +140 -140
  44. package/seed/marketplaces/ecc/commands/kotlin-test.md +312 -312
  45. package/seed/marketplaces/ecc/commands/learn-eval.md +116 -116
  46. package/seed/marketplaces/ecc/commands/learn.md +74 -74
  47. package/seed/marketplaces/ecc/commands/loop-start.md +36 -36
  48. package/seed/marketplaces/ecc/commands/loop-status.md +77 -77
  49. package/seed/marketplaces/ecc/commands/marketing-campaign.md +129 -129
  50. package/seed/marketplaces/ecc/commands/model-route.md +30 -30
  51. package/seed/marketplaces/ecc/commands/multi-backend.md +162 -162
  52. package/seed/marketplaces/ecc/commands/multi-execute.md +319 -319
  53. package/seed/marketplaces/ecc/commands/multi-frontend.md +162 -162
  54. package/seed/marketplaces/ecc/commands/multi-plan.md +272 -272
  55. package/seed/marketplaces/ecc/commands/multi-workflow.md +195 -195
  56. package/seed/marketplaces/ecc/commands/plan-prd.md +160 -160
  57. package/seed/marketplaces/ecc/commands/plan.md +200 -200
  58. package/seed/marketplaces/ecc/commands/pm2.md +276 -276
  59. package/seed/marketplaces/ecc/commands/pr.md +184 -184
  60. package/seed/marketplaces/ecc/commands/project-init.md +86 -86
  61. package/seed/marketplaces/ecc/commands/projects.md +39 -39
  62. package/seed/marketplaces/ecc/commands/promote.md +41 -41
  63. package/seed/marketplaces/ecc/commands/prp-commit.md +112 -112
  64. package/seed/marketplaces/ecc/commands/prp-implement.md +385 -385
  65. package/seed/marketplaces/ecc/commands/prp-plan.md +502 -502
  66. package/seed/marketplaces/ecc/commands/prp-pr.md +184 -184
  67. package/seed/marketplaces/ecc/commands/prp-prd.md +447 -447
  68. package/seed/marketplaces/ecc/commands/prune.md +31 -31
  69. package/seed/marketplaces/ecc/commands/python-review.md +297 -297
  70. package/seed/marketplaces/ecc/commands/quality-gate.md +33 -33
  71. package/seed/marketplaces/ecc/commands/refactor-clean.md +84 -84
  72. package/seed/marketplaces/ecc/commands/resume-session.md +156 -156
  73. package/seed/marketplaces/ecc/commands/review-pr.md +37 -37
  74. package/seed/marketplaces/ecc/commands/rust-build.md +187 -187
  75. package/seed/marketplaces/ecc/commands/rust-review.md +142 -142
  76. package/seed/marketplaces/ecc/commands/rust-test.md +308 -308
  77. package/seed/marketplaces/ecc/commands/santa-loop.md +175 -175
  78. package/seed/marketplaces/ecc/commands/save-session.md +275 -275
  79. package/seed/marketplaces/ecc/commands/security-scan.md +92 -92
  80. package/seed/marketplaces/ecc/commands/sessions.md +339 -339
  81. package/seed/marketplaces/ecc/commands/setup-pm.md +80 -80
  82. package/seed/marketplaces/ecc/commands/skill-create.md +174 -174
  83. package/seed/marketplaces/ecc/commands/skill-health.md +54 -54
  84. package/seed/marketplaces/ecc/commands/test-coverage.md +73 -73
  85. package/seed/marketplaces/ecc/commands/update-codemaps.md +76 -76
  86. package/seed/marketplaces/ecc/commands/update-docs.md +88 -88
  87. package/seed/marketplaces/ecc/skills/accessibility/SKILL.md +146 -146
  88. package/seed/marketplaces/ecc/skills/agent-architecture-audit/SKILL.md +256 -256
  89. package/seed/marketplaces/ecc/skills/agent-eval/SKILL.md +145 -145
  90. package/seed/marketplaces/ecc/skills/agent-harness-construction/SKILL.md +73 -73
  91. package/seed/marketplaces/ecc/skills/agent-introspection-debugging/SKILL.md +153 -153
  92. package/seed/marketplaces/ecc/skills/agent-payment-x402/SKILL.md +224 -224
  93. package/seed/marketplaces/ecc/skills/agent-sort/SKILL.md +215 -215
  94. package/seed/marketplaces/ecc/skills/agentic-engineering/SKILL.md +63 -63
  95. package/seed/marketplaces/ecc/skills/agentic-os/SKILL.md +387 -387
  96. package/seed/marketplaces/ecc/skills/ai-first-engineering/SKILL.md +51 -51
  97. package/seed/marketplaces/ecc/skills/ai-regression-testing/SKILL.md +385 -385
  98. package/seed/marketplaces/ecc/skills/android-clean-architecture/SKILL.md +339 -339
  99. package/seed/marketplaces/ecc/skills/angular-developer/SKILL.md +154 -154
  100. package/seed/marketplaces/ecc/skills/angular-developer/references/angular-animations.md +160 -160
  101. package/seed/marketplaces/ecc/skills/angular-developer/references/angular-aria.md +410 -410
  102. package/seed/marketplaces/ecc/skills/angular-developer/references/cli.md +86 -86
  103. package/seed/marketplaces/ecc/skills/angular-developer/references/component-harnesses.md +59 -59
  104. package/seed/marketplaces/ecc/skills/angular-developer/references/component-styling.md +91 -91
  105. package/seed/marketplaces/ecc/skills/angular-developer/references/components.md +117 -117
  106. package/seed/marketplaces/ecc/skills/angular-developer/references/creating-services.md +97 -97
  107. package/seed/marketplaces/ecc/skills/angular-developer/references/data-resolvers.md +69 -69
  108. package/seed/marketplaces/ecc/skills/angular-developer/references/define-routes.md +67 -67
  109. package/seed/marketplaces/ecc/skills/angular-developer/references/defining-providers.md +72 -72
  110. package/seed/marketplaces/ecc/skills/angular-developer/references/di-fundamentals.md +120 -120
  111. package/seed/marketplaces/ecc/skills/angular-developer/references/e2e-testing.md +56 -56
  112. package/seed/marketplaces/ecc/skills/angular-developer/references/effects.md +83 -83
  113. package/seed/marketplaces/ecc/skills/angular-developer/references/hierarchical-injectors.md +43 -43
  114. package/seed/marketplaces/ecc/skills/angular-developer/references/host-elements.md +80 -80
  115. package/seed/marketplaces/ecc/skills/angular-developer/references/injection-context.md +63 -63
  116. package/seed/marketplaces/ecc/skills/angular-developer/references/inputs.md +101 -101
  117. package/seed/marketplaces/ecc/skills/angular-developer/references/linked-signal.md +59 -59
  118. package/seed/marketplaces/ecc/skills/angular-developer/references/loading-strategies.md +61 -61
  119. package/seed/marketplaces/ecc/skills/angular-developer/references/mcp.md +108 -108
  120. package/seed/marketplaces/ecc/skills/angular-developer/references/navigate-to-routes.md +69 -69
  121. package/seed/marketplaces/ecc/skills/angular-developer/references/outputs.md +86 -86
  122. package/seed/marketplaces/ecc/skills/angular-developer/references/reactive-forms.md +122 -122
  123. package/seed/marketplaces/ecc/skills/angular-developer/references/rendering-strategies.md +44 -44
  124. package/seed/marketplaces/ecc/skills/angular-developer/references/resource.md +77 -77
  125. package/seed/marketplaces/ecc/skills/angular-developer/references/route-animations.md +56 -56
  126. package/seed/marketplaces/ecc/skills/angular-developer/references/route-guards.md +52 -52
  127. package/seed/marketplaces/ecc/skills/angular-developer/references/router-lifecycle.md +45 -45
  128. package/seed/marketplaces/ecc/skills/angular-developer/references/router-testing.md +87 -87
  129. package/seed/marketplaces/ecc/skills/angular-developer/references/show-routes-with-outlets.md +68 -68
  130. package/seed/marketplaces/ecc/skills/angular-developer/references/signal-forms.md +795 -795
  131. package/seed/marketplaces/ecc/skills/angular-developer/references/signals-overview.md +94 -94
  132. package/seed/marketplaces/ecc/skills/angular-developer/references/tailwind-css.md +69 -69
  133. package/seed/marketplaces/ecc/skills/angular-developer/references/template-driven-forms.md +114 -114
  134. package/seed/marketplaces/ecc/skills/angular-developer/references/testing-fundamentals.md +65 -65
  135. package/seed/marketplaces/ecc/skills/api-connector-builder/SKILL.md +120 -120
  136. package/seed/marketplaces/ecc/skills/api-design/SKILL.md +523 -523
  137. package/seed/marketplaces/ecc/skills/architecture-decision-records/SKILL.md +179 -179
  138. package/seed/marketplaces/ecc/skills/article-writing/SKILL.md +79 -79
  139. package/seed/marketplaces/ecc/skills/automation-audit-ops/SKILL.md +142 -142
  140. package/seed/marketplaces/ecc/skills/autonomous-agent-harness/SKILL.md +273 -273
  141. package/seed/marketplaces/ecc/skills/autonomous-loops/SKILL.md +610 -610
  142. package/seed/marketplaces/ecc/skills/backend-patterns/SKILL.md +561 -561
  143. package/seed/marketplaces/ecc/skills/benchmark/SKILL.md +93 -93
  144. package/seed/marketplaces/ecc/skills/benchmark-optimization-loop/SKILL.md +69 -69
  145. package/seed/marketplaces/ecc/skills/blender-motion-state-inspection/SKILL.md +164 -164
  146. package/seed/marketplaces/ecc/skills/blueprint/SKILL.md +105 -105
  147. package/seed/marketplaces/ecc/skills/brand-voice/SKILL.md +97 -97
  148. package/seed/marketplaces/ecc/skills/brand-voice/references/voice-profile-schema.md +55 -55
  149. package/seed/marketplaces/ecc/skills/browser-qa/SKILL.md +87 -87
  150. package/seed/marketplaces/ecc/skills/bun-runtime/SKILL.md +84 -84
  151. package/seed/marketplaces/ecc/skills/canary-watch/SKILL.md +107 -107
  152. package/seed/marketplaces/ecc/skills/carrier-relationship-management/SKILL.md +212 -212
  153. package/seed/marketplaces/ecc/skills/cisco-ios-patterns/SKILL.md +163 -163
  154. package/seed/marketplaces/ecc/skills/ck/SKILL.md +147 -147
  155. package/seed/marketplaces/ecc/skills/ck/commands/forget.mjs +44 -44
  156. package/seed/marketplaces/ecc/skills/ck/commands/info.mjs +24 -24
  157. package/seed/marketplaces/ecc/skills/ck/commands/init.mjs +143 -143
  158. package/seed/marketplaces/ecc/skills/ck/commands/list.mjs +40 -40
  159. package/seed/marketplaces/ecc/skills/ck/commands/migrate.mjs +202 -202
  160. package/seed/marketplaces/ecc/skills/ck/commands/resume.mjs +36 -36
  161. package/seed/marketplaces/ecc/skills/ck/commands/save.mjs +210 -210
  162. package/seed/marketplaces/ecc/skills/ck/commands/shared.mjs +387 -387
  163. package/seed/marketplaces/ecc/skills/ck/hooks/session-start.mjs +224 -224
  164. package/seed/marketplaces/ecc/skills/claude-devfleet/SKILL.md +103 -103
  165. package/seed/marketplaces/ecc/skills/click-path-audit/SKILL.md +244 -244
  166. package/seed/marketplaces/ecc/skills/clickhouse-io/SKILL.md +439 -439
  167. package/seed/marketplaces/ecc/skills/code-tour/SKILL.md +236 -236
  168. package/seed/marketplaces/ecc/skills/codebase-onboarding/SKILL.md +233 -233
  169. package/seed/marketplaces/ecc/skills/coding-standards/SKILL.md +549 -549
  170. package/seed/marketplaces/ecc/skills/compose-multiplatform-patterns/SKILL.md +299 -299
  171. package/seed/marketplaces/ecc/skills/configure-ecc/SKILL.md +384 -384
  172. package/seed/marketplaces/ecc/skills/connections-optimizer/SKILL.md +189 -189
  173. package/seed/marketplaces/ecc/skills/content-engine/SKILL.md +131 -131
  174. package/seed/marketplaces/ecc/skills/content-hash-cache-pattern/SKILL.md +161 -161
  175. package/seed/marketplaces/ecc/skills/context-budget/SKILL.md +135 -135
  176. package/seed/marketplaces/ecc/skills/continuous-agent-loop/SKILL.md +45 -45
  177. package/seed/marketplaces/ecc/skills/continuous-learning/SKILL.md +131 -131
  178. package/seed/marketplaces/ecc/skills/continuous-learning/config.json +18 -18
  179. package/seed/marketplaces/ecc/skills/continuous-learning/evaluate-session.sh +69 -69
  180. package/seed/marketplaces/ecc/skills/continuous-learning-v2/SKILL.md +360 -360
  181. package/seed/marketplaces/ecc/skills/continuous-learning-v2/agents/observer-loop.sh +322 -322
  182. package/seed/marketplaces/ecc/skills/continuous-learning-v2/agents/observer.md +198 -198
  183. package/seed/marketplaces/ecc/skills/continuous-learning-v2/agents/session-guardian.sh +150 -150
  184. package/seed/marketplaces/ecc/skills/continuous-learning-v2/agents/start-observer.sh +248 -248
  185. package/seed/marketplaces/ecc/skills/continuous-learning-v2/config.json +8 -8
  186. package/seed/marketplaces/ecc/skills/continuous-learning-v2/hooks/observe.sh +498 -498
  187. package/seed/marketplaces/ecc/skills/continuous-learning-v2/scripts/detect-project.sh +322 -322
  188. package/seed/marketplaces/ecc/skills/continuous-learning-v2/scripts/instinct-cli.py +1826 -1826
  189. package/seed/marketplaces/ecc/skills/continuous-learning-v2/scripts/lib/homunculus-dir.sh +31 -31
  190. package/seed/marketplaces/ecc/skills/continuous-learning-v2/scripts/migrate-homunculus.sh +62 -62
  191. package/seed/marketplaces/ecc/skills/continuous-learning-v2/scripts/test_parse_instinct.py +1018 -1018
  192. package/seed/marketplaces/ecc/skills/cost-aware-llm-pipeline/SKILL.md +183 -183
  193. package/seed/marketplaces/ecc/skills/cost-tracking/SKILL.md +147 -147
  194. package/seed/marketplaces/ecc/skills/council/SKILL.md +203 -203
  195. package/seed/marketplaces/ecc/skills/cpp-coding-standards/SKILL.md +723 -723
  196. package/seed/marketplaces/ecc/skills/cpp-testing/SKILL.md +324 -324
  197. package/seed/marketplaces/ecc/skills/crosspost/SKILL.md +111 -111
  198. package/seed/marketplaces/ecc/skills/csharp-testing/SKILL.md +321 -321
  199. package/seed/marketplaces/ecc/skills/customer-billing-ops/SKILL.md +140 -140
  200. package/seed/marketplaces/ecc/skills/customs-trade-compliance/SKILL.md +263 -263
  201. package/seed/marketplaces/ecc/skills/dart-flutter-patterns/SKILL.md +563 -563
  202. package/seed/marketplaces/ecc/skills/dashboard-builder/SKILL.md +108 -108
  203. package/seed/marketplaces/ecc/skills/data-scraper-agent/SKILL.md +764 -764
  204. package/seed/marketplaces/ecc/skills/data-throughput-accelerator/SKILL.md +72 -72
  205. package/seed/marketplaces/ecc/skills/database-migrations/SKILL.md +429 -429
  206. package/seed/marketplaces/ecc/skills/deep-research/SKILL.md +159 -159
  207. package/seed/marketplaces/ecc/skills/defi-amm-security/SKILL.md +166 -166
  208. package/seed/marketplaces/ecc/skills/deployment-patterns/SKILL.md +427 -427
  209. package/seed/marketplaces/ecc/skills/design-system/SKILL.md +82 -82
  210. package/seed/marketplaces/ecc/skills/django-celery/SKILL.md +457 -457
  211. package/seed/marketplaces/ecc/skills/django-patterns/SKILL.md +734 -734
  212. package/seed/marketplaces/ecc/skills/django-security/SKILL.md +593 -593
  213. package/seed/marketplaces/ecc/skills/django-tdd/SKILL.md +729 -729
  214. package/seed/marketplaces/ecc/skills/django-verification/SKILL.md +469 -469
  215. package/seed/marketplaces/ecc/skills/dmux-workflows/SKILL.md +191 -191
  216. package/seed/marketplaces/ecc/skills/docker-patterns/SKILL.md +364 -364
  217. package/seed/marketplaces/ecc/skills/documentation-lookup/SKILL.md +90 -90
  218. package/seed/marketplaces/ecc/skills/dotnet-patterns/SKILL.md +321 -321
  219. package/seed/marketplaces/ecc/skills/e2e-testing/SKILL.md +326 -326
  220. package/seed/marketplaces/ecc/skills/ecc-guide/SKILL.md +189 -189
  221. package/seed/marketplaces/ecc/skills/ecc-tools-cost-audit/SKILL.md +160 -160
  222. package/seed/marketplaces/ecc/skills/email-ops/SKILL.md +121 -121
  223. package/seed/marketplaces/ecc/skills/energy-procurement/SKILL.md +228 -228
  224. package/seed/marketplaces/ecc/skills/enterprise-agent-ops/SKILL.md +50 -50
  225. package/seed/marketplaces/ecc/skills/error-handling/SKILL.md +376 -376
  226. package/seed/marketplaces/ecc/skills/eval-harness/SKILL.md +270 -270
  227. package/seed/marketplaces/ecc/skills/evm-token-decimals/SKILL.md +130 -130
  228. package/seed/marketplaces/ecc/skills/exa-search/SKILL.md +107 -107
  229. package/seed/marketplaces/ecc/skills/fal-ai-media/SKILL.md +288 -288
  230. package/seed/marketplaces/ecc/skills/fastapi-patterns/SKILL.md +327 -327
  231. package/seed/marketplaces/ecc/skills/finance-billing-ops/SKILL.md +127 -127
  232. package/seed/marketplaces/ecc/skills/flox-environments/SKILL.md +496 -496
  233. package/seed/marketplaces/ecc/skills/flutter-dart-code-review/SKILL.md +435 -435
  234. package/seed/marketplaces/ecc/skills/foundation-models-on-device/SKILL.md +243 -243
  235. package/seed/marketplaces/ecc/skills/frontend-a11y/SKILL.md +446 -446
  236. package/seed/marketplaces/ecc/skills/frontend-design-direction/SKILL.md +92 -92
  237. package/seed/marketplaces/ecc/skills/frontend-patterns/SKILL.md +642 -642
  238. package/seed/marketplaces/ecc/skills/frontend-slides/SKILL.md +184 -184
  239. package/seed/marketplaces/ecc/skills/frontend-slides/STYLE_PRESETS.md +330 -330
  240. package/seed/marketplaces/ecc/skills/frontend-slides/animation-patterns.md +122 -122
  241. package/seed/marketplaces/ecc/skills/frontend-slides/html-template.md +419 -419
  242. package/seed/marketplaces/ecc/skills/frontend-slides/scripts/export-pdf.sh +418 -418
  243. package/seed/marketplaces/ecc/skills/frontend-slides/scripts/extract-pptx.py +96 -96
  244. package/seed/marketplaces/ecc/skills/frontend-slides/viewport-base.css +153 -153
  245. package/seed/marketplaces/ecc/skills/fsharp-testing/SKILL.md +280 -280
  246. package/seed/marketplaces/ecc/skills/gan-style-harness/SKILL.md +278 -278
  247. package/seed/marketplaces/ecc/skills/gateguard/SKILL.md +125 -125
  248. package/seed/marketplaces/ecc/skills/git-workflow/SKILL.md +715 -715
  249. package/seed/marketplaces/ecc/skills/github-ops/SKILL.md +144 -144
  250. package/seed/marketplaces/ecc/skills/golang-patterns/SKILL.md +674 -674
  251. package/seed/marketplaces/ecc/skills/golang-testing/SKILL.md +720 -720
  252. package/seed/marketplaces/ecc/skills/google-workspace-ops/SKILL.md +95 -95
  253. package/seed/marketplaces/ecc/skills/healthcare-cdss-patterns/SKILL.md +245 -245
  254. package/seed/marketplaces/ecc/skills/healthcare-emr-patterns/SKILL.md +159 -159
  255. package/seed/marketplaces/ecc/skills/healthcare-eval-harness/SKILL.md +207 -207
  256. package/seed/marketplaces/ecc/skills/healthcare-phi-compliance/SKILL.md +145 -145
  257. package/seed/marketplaces/ecc/skills/hermes-imports/SKILL.md +88 -88
  258. package/seed/marketplaces/ecc/skills/hexagonal-architecture/SKILL.md +276 -276
  259. package/seed/marketplaces/ecc/skills/hipaa-compliance/SKILL.md +78 -78
  260. package/seed/marketplaces/ecc/skills/homelab-network-readiness/SKILL.md +169 -169
  261. package/seed/marketplaces/ecc/skills/homelab-network-setup/SKILL.md +129 -129
  262. package/seed/marketplaces/ecc/skills/homelab-pihole-dns/SKILL.md +274 -274
  263. package/seed/marketplaces/ecc/skills/homelab-vlan-segmentation/SKILL.md +311 -311
  264. package/seed/marketplaces/ecc/skills/homelab-wireguard-vpn/SKILL.md +305 -305
  265. package/seed/marketplaces/ecc/skills/hookify-rules/SKILL.md +128 -128
  266. package/seed/marketplaces/ecc/skills/inventory-demand-planning/SKILL.md +247 -247
  267. package/seed/marketplaces/ecc/skills/investor-materials/SKILL.md +96 -96
  268. package/seed/marketplaces/ecc/skills/investor-outreach/SKILL.md +91 -91
  269. package/seed/marketplaces/ecc/skills/ios-icon-gen/SKILL.md +157 -157
  270. package/seed/marketplaces/ecc/skills/ios-icon-gen/scripts/generate_icons.swift +258 -258
  271. package/seed/marketplaces/ecc/skills/ios-icon-gen/scripts/iconify_gen.sh +235 -235
  272. package/seed/marketplaces/ecc/skills/iterative-retrieval/SKILL.md +211 -211
  273. package/seed/marketplaces/ecc/skills/ito-basket-compare/SKILL.md +63 -63
  274. package/seed/marketplaces/ecc/skills/ito-data-atlas-agent/SKILL.md +63 -63
  275. package/seed/marketplaces/ecc/skills/ito-market-intelligence/SKILL.md +60 -60
  276. package/seed/marketplaces/ecc/skills/ito-trade-planner/SKILL.md +67 -67
  277. package/seed/marketplaces/ecc/skills/java-coding-standards/SKILL.md +383 -383
  278. package/seed/marketplaces/ecc/skills/jira-integration/SKILL.md +293 -293
  279. package/seed/marketplaces/ecc/skills/jpa-patterns/SKILL.md +151 -151
  280. package/seed/marketplaces/ecc/skills/knowledge-ops/SKILL.md +154 -154
  281. package/seed/marketplaces/ecc/skills/kotlin-coroutines-flows/SKILL.md +284 -284
  282. package/seed/marketplaces/ecc/skills/kotlin-exposed-patterns/SKILL.md +719 -719
  283. package/seed/marketplaces/ecc/skills/kotlin-ktor-patterns/SKILL.md +689 -689
  284. package/seed/marketplaces/ecc/skills/kotlin-patterns/SKILL.md +711 -711
  285. package/seed/marketplaces/ecc/skills/kotlin-testing/SKILL.md +824 -824
  286. package/seed/marketplaces/ecc/skills/laravel-patterns/SKILL.md +415 -415
  287. package/seed/marketplaces/ecc/skills/laravel-plugin-discovery/SKILL.md +229 -229
  288. package/seed/marketplaces/ecc/skills/laravel-security/SKILL.md +285 -285
  289. package/seed/marketplaces/ecc/skills/laravel-tdd/SKILL.md +283 -283
  290. package/seed/marketplaces/ecc/skills/laravel-verification/SKILL.md +179 -179
  291. package/seed/marketplaces/ecc/skills/latency-critical-systems/SKILL.md +73 -73
  292. package/seed/marketplaces/ecc/skills/lead-intelligence/SKILL.md +321 -321
  293. package/seed/marketplaces/ecc/skills/lead-intelligence/agents/enrichment-agent.md +85 -85
  294. package/seed/marketplaces/ecc/skills/lead-intelligence/agents/mutual-mapper.md +75 -75
  295. package/seed/marketplaces/ecc/skills/lead-intelligence/agents/outreach-drafter.md +98 -98
  296. package/seed/marketplaces/ecc/skills/lead-intelligence/agents/signal-scorer.md +60 -60
  297. package/seed/marketplaces/ecc/skills/liquid-glass-design/SKILL.md +279 -279
  298. package/seed/marketplaces/ecc/skills/llm-trading-agent-security/SKILL.md +146 -146
  299. package/seed/marketplaces/ecc/skills/logistics-exception-management/SKILL.md +222 -222
  300. package/seed/marketplaces/ecc/skills/make-interfaces-feel-better/SKILL.md +151 -151
  301. package/seed/marketplaces/ecc/skills/manim-video/SKILL.md +89 -89
  302. package/seed/marketplaces/ecc/skills/manim-video/assets/network_graph_scene.py +52 -52
  303. package/seed/marketplaces/ecc/skills/market-research/SKILL.md +75 -75
  304. package/seed/marketplaces/ecc/skills/marketing-campaign/SKILL.md +113 -113
  305. package/seed/marketplaces/ecc/skills/mcp-server-patterns/SKILL.md +69 -69
  306. package/seed/marketplaces/ecc/skills/messages-ops/SKILL.md +104 -104
  307. package/seed/marketplaces/ecc/skills/mle-workflow/SKILL.md +346 -346
  308. package/seed/marketplaces/ecc/skills/motion-advanced/SKILL.md +596 -596
  309. package/seed/marketplaces/ecc/skills/motion-foundations/SKILL.md +299 -299
  310. package/seed/marketplaces/ecc/skills/motion-patterns/SKILL.md +435 -435
  311. package/seed/marketplaces/ecc/skills/motion-ui/SKILL.md +575 -575
  312. package/seed/marketplaces/ecc/skills/mysql-patterns/SKILL.md +412 -412
  313. package/seed/marketplaces/ecc/skills/nanoclaw-repl/SKILL.md +33 -33
  314. package/seed/marketplaces/ecc/skills/nestjs-patterns/SKILL.md +230 -230
  315. package/seed/marketplaces/ecc/skills/netmiko-ssh-automation/SKILL.md +173 -173
  316. package/seed/marketplaces/ecc/skills/network-bgp-diagnostics/SKILL.md +167 -167
  317. package/seed/marketplaces/ecc/skills/network-config-validation/SKILL.md +210 -210
  318. package/seed/marketplaces/ecc/skills/network-interface-health/SKILL.md +152 -152
  319. package/seed/marketplaces/ecc/skills/nextjs-turbopack/SKILL.md +57 -57
  320. package/seed/marketplaces/ecc/skills/nodejs-keccak256/SKILL.md +102 -102
  321. package/seed/marketplaces/ecc/skills/nutrient-document-processing/SKILL.md +167 -167
  322. package/seed/marketplaces/ecc/skills/nuxt4-patterns/SKILL.md +100 -100
  323. package/seed/marketplaces/ecc/skills/openclaw-persona-forge/SKILL.md +288 -288
  324. package/seed/marketplaces/ecc/skills/openclaw-persona-forge/gacha.py +224 -224
  325. package/seed/marketplaces/ecc/skills/openclaw-persona-forge/gacha.sh +5 -5
  326. package/seed/marketplaces/ecc/skills/openclaw-persona-forge/references/avatar-style.md +124 -124
  327. package/seed/marketplaces/ecc/skills/openclaw-persona-forge/references/boundary-rules.md +53 -53
  328. package/seed/marketplaces/ecc/skills/openclaw-persona-forge/references/error-handling.md +53 -53
  329. package/seed/marketplaces/ecc/skills/openclaw-persona-forge/references/identity-tension.md +48 -48
  330. package/seed/marketplaces/ecc/skills/openclaw-persona-forge/references/naming-system.md +39 -39
  331. package/seed/marketplaces/ecc/skills/openclaw-persona-forge/references/output-template.md +166 -166
  332. package/seed/marketplaces/ecc/skills/opensource-pipeline/SKILL.md +255 -255
  333. package/seed/marketplaces/ecc/skills/parallel-execution-optimizer/SKILL.md +72 -72
  334. package/seed/marketplaces/ecc/skills/perl-patterns/SKILL.md +504 -504
  335. package/seed/marketplaces/ecc/skills/perl-security/SKILL.md +503 -503
  336. package/seed/marketplaces/ecc/skills/perl-testing/SKILL.md +475 -475
  337. package/seed/marketplaces/ecc/skills/plan-orchestrate/SKILL.md +262 -262
  338. package/seed/marketplaces/ecc/skills/plankton-code-quality/SKILL.md +236 -236
  339. package/seed/marketplaces/ecc/skills/postgres-patterns/SKILL.md +147 -147
  340. package/seed/marketplaces/ecc/skills/prediction-market-oracle-research/SKILL.md +63 -63
  341. package/seed/marketplaces/ecc/skills/prediction-market-risk-review/SKILL.md +60 -60
  342. package/seed/marketplaces/ecc/skills/prisma-patterns/SKILL.md +371 -371
  343. package/seed/marketplaces/ecc/skills/product-capability/SKILL.md +141 -141
  344. package/seed/marketplaces/ecc/skills/product-lens/SKILL.md +92 -92
  345. package/seed/marketplaces/ecc/skills/production-audit/SKILL.md +206 -206
  346. package/seed/marketplaces/ecc/skills/production-scheduling/SKILL.md +238 -238
  347. package/seed/marketplaces/ecc/skills/project-flow-ops/SKILL.md +111 -111
  348. package/seed/marketplaces/ecc/skills/prompt-optimizer/SKILL.md +398 -398
  349. package/seed/marketplaces/ecc/skills/python-patterns/SKILL.md +750 -750
  350. package/seed/marketplaces/ecc/skills/python-testing/SKILL.md +816 -816
  351. package/seed/marketplaces/ecc/skills/pytorch-patterns/SKILL.md +396 -396
  352. package/seed/marketplaces/ecc/skills/quality-nonconformance/SKILL.md +260 -260
  353. package/seed/marketplaces/ecc/skills/quarkus-patterns/SKILL.md +722 -722
  354. package/seed/marketplaces/ecc/skills/quarkus-security/SKILL.md +467 -467
  355. package/seed/marketplaces/ecc/skills/quarkus-tdd/SKILL.md +811 -811
  356. package/seed/marketplaces/ecc/skills/quarkus-verification/SKILL.md +479 -479
  357. package/seed/marketplaces/ecc/skills/ralphinho-rfc-pipeline/SKILL.md +67 -67
  358. package/seed/marketplaces/ecc/skills/recsys-pipeline-architect/SKILL.md +114 -114
  359. package/seed/marketplaces/ecc/skills/recursive-decision-ledger/SKILL.md +79 -79
  360. package/seed/marketplaces/ecc/skills/redis-patterns/SKILL.md +403 -403
  361. package/seed/marketplaces/ecc/skills/regex-vs-llm-structured-text/SKILL.md +220 -220
  362. package/seed/marketplaces/ecc/skills/remotion-video-creation/SKILL.md +43 -43
  363. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/3d.md +86 -86
  364. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/animations.md +29 -29
  365. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/assets/charts-bar-chart.tsx +173 -173
  366. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/assets/text-animations-typewriter.tsx +100 -100
  367. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/assets/text-animations-word-highlight.tsx +108 -108
  368. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/assets.md +78 -78
  369. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/audio.md +172 -172
  370. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/calculate-metadata.md +104 -104
  371. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/can-decode.md +75 -75
  372. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/charts.md +58 -58
  373. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/compositions.md +146 -146
  374. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/display-captions.md +126 -126
  375. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/extract-frames.md +229 -229
  376. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/fonts.md +152 -152
  377. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/get-audio-duration.md +58 -58
  378. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/get-video-dimensions.md +68 -68
  379. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/get-video-duration.md +58 -58
  380. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/gifs.md +138 -138
  381. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/images.md +130 -130
  382. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/import-srt-captions.md +67 -67
  383. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/lottie.md +67 -67
  384. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/measuring-dom-nodes.md +34 -34
  385. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/measuring-text.md +143 -143
  386. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/sequencing.md +106 -106
  387. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/tailwind.md +11 -11
  388. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/text-animations.md +20 -20
  389. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/timing.md +179 -179
  390. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/transcribe-captions.md +19 -19
  391. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/transitions.md +122 -122
  392. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/trimming.md +52 -52
  393. package/seed/marketplaces/ecc/skills/remotion-video-creation/rules/videos.md +171 -171
  394. package/seed/marketplaces/ecc/skills/repo-scan/SKILL.md +78 -78
  395. package/seed/marketplaces/ecc/skills/research-ops/SKILL.md +112 -112
  396. package/seed/marketplaces/ecc/skills/returns-reverse-logistics/SKILL.md +240 -240
  397. package/seed/marketplaces/ecc/skills/rules-distill/SKILL.md +264 -264
  398. package/seed/marketplaces/ecc/skills/rules-distill/scripts/scan-rules.sh +58 -58
  399. package/seed/marketplaces/ecc/skills/rules-distill/scripts/scan-skills.sh +129 -129
  400. package/seed/marketplaces/ecc/skills/rust-patterns/SKILL.md +499 -499
  401. package/seed/marketplaces/ecc/skills/rust-testing/SKILL.md +500 -500
  402. package/seed/marketplaces/ecc/skills/safety-guard/SKILL.md +75 -75
  403. package/seed/marketplaces/ecc/skills/santa-method/SKILL.md +306 -306
  404. package/seed/marketplaces/ecc/skills/scientific-db-pubmed-database/SKILL.md +175 -175
  405. package/seed/marketplaces/ecc/skills/scientific-db-uspto-database/SKILL.md +177 -177
  406. package/seed/marketplaces/ecc/skills/scientific-pkg-gget/SKILL.md +166 -166
  407. package/seed/marketplaces/ecc/skills/scientific-thinking-literature-review/SKILL.md +192 -192
  408. package/seed/marketplaces/ecc/skills/scientific-thinking-scholar-evaluation/SKILL.md +160 -160
  409. package/seed/marketplaces/ecc/skills/search-first/SKILL.md +182 -182
  410. package/seed/marketplaces/ecc/skills/security-bounty-hunter/SKILL.md +99 -99
  411. package/seed/marketplaces/ecc/skills/security-review/SKILL.md +503 -503
  412. package/seed/marketplaces/ecc/skills/security-review/cloud-infrastructure-security.md +361 -361
  413. package/seed/marketplaces/ecc/skills/security-scan/SKILL.md +165 -165
  414. package/seed/marketplaces/ecc/skills/seo/SKILL.md +154 -154
  415. package/seed/marketplaces/ecc/skills/skill-comply/SKILL.md +58 -58
  416. package/seed/marketplaces/ecc/skills/skill-comply/fixtures/compliant_trace.jsonl +5 -5
  417. package/seed/marketplaces/ecc/skills/skill-comply/fixtures/noncompliant_trace.jsonl +3 -3
  418. package/seed/marketplaces/ecc/skills/skill-comply/fixtures/tdd_spec.yaml +44 -44
  419. package/seed/marketplaces/ecc/skills/skill-comply/prompts/classifier.md +24 -24
  420. package/seed/marketplaces/ecc/skills/skill-comply/prompts/scenario_generator.md +62 -62
  421. package/seed/marketplaces/ecc/skills/skill-comply/prompts/spec_generator.md +42 -42
  422. package/seed/marketplaces/ecc/skills/skill-comply/pyproject.toml +15 -15
  423. package/seed/marketplaces/ecc/skills/skill-comply/scripts/classifier.py +85 -85
  424. package/seed/marketplaces/ecc/skills/skill-comply/scripts/grader.py +124 -124
  425. package/seed/marketplaces/ecc/skills/skill-comply/scripts/parser.py +107 -107
  426. package/seed/marketplaces/ecc/skills/skill-comply/scripts/report.py +170 -170
  427. package/seed/marketplaces/ecc/skills/skill-comply/scripts/run.py +127 -127
  428. package/seed/marketplaces/ecc/skills/skill-comply/scripts/runner.py +186 -186
  429. package/seed/marketplaces/ecc/skills/skill-comply/scripts/scenario_generator.py +70 -70
  430. package/seed/marketplaces/ecc/skills/skill-comply/scripts/spec_generator.py +72 -72
  431. package/seed/marketplaces/ecc/skills/skill-comply/scripts/utils.py +13 -13
  432. package/seed/marketplaces/ecc/skills/skill-comply/tests/test_grader.py +197 -197
  433. package/seed/marketplaces/ecc/skills/skill-comply/tests/test_parser.py +90 -90
  434. package/seed/marketplaces/ecc/skills/skill-comply/tests/test_runner.py +172 -172
  435. package/seed/marketplaces/ecc/skills/skill-scout/SKILL.md +140 -140
  436. package/seed/marketplaces/ecc/skills/skill-stocktake/SKILL.md +194 -194
  437. package/seed/marketplaces/ecc/skills/skill-stocktake/scripts/quick-diff.sh +87 -87
  438. package/seed/marketplaces/ecc/skills/skill-stocktake/scripts/save-results.sh +56 -56
  439. package/seed/marketplaces/ecc/skills/skill-stocktake/scripts/scan.sh +170 -170
  440. package/seed/marketplaces/ecc/skills/social-graph-ranker/SKILL.md +154 -154
  441. package/seed/marketplaces/ecc/skills/social-publisher/SKILL.md +115 -115
  442. package/seed/marketplaces/ecc/skills/springboot-patterns/SKILL.md +314 -314
  443. package/seed/marketplaces/ecc/skills/springboot-security/SKILL.md +272 -272
  444. package/seed/marketplaces/ecc/skills/springboot-tdd/SKILL.md +158 -158
  445. package/seed/marketplaces/ecc/skills/springboot-verification/SKILL.md +231 -231
  446. package/seed/marketplaces/ecc/skills/strategic-compact/SKILL.md +131 -131
  447. package/seed/marketplaces/ecc/skills/swift-actor-persistence/SKILL.md +143 -143
  448. package/seed/marketplaces/ecc/skills/swift-concurrency-6-2/SKILL.md +216 -216
  449. package/seed/marketplaces/ecc/skills/swift-protocol-di-testing/SKILL.md +190 -190
  450. package/seed/marketplaces/ecc/skills/swiftui-patterns/SKILL.md +259 -259
  451. package/seed/marketplaces/ecc/skills/tdd-workflow/SKILL.md +463 -463
  452. package/seed/marketplaces/ecc/skills/team-builder/SKILL.md +168 -168
  453. package/seed/marketplaces/ecc/skills/terminal-ops/SKILL.md +109 -109
  454. package/seed/marketplaces/ecc/skills/tinystruct-patterns/SKILL.md +203 -203
  455. package/seed/marketplaces/ecc/skills/tinystruct-patterns/references/architecture.md +90 -90
  456. package/seed/marketplaces/ecc/skills/tinystruct-patterns/references/data-handling.md +60 -60
  457. package/seed/marketplaces/ecc/skills/tinystruct-patterns/references/database.md +99 -99
  458. package/seed/marketplaces/ecc/skills/tinystruct-patterns/references/routing.md +64 -64
  459. package/seed/marketplaces/ecc/skills/tinystruct-patterns/references/system-usage.md +97 -97
  460. package/seed/marketplaces/ecc/skills/tinystruct-patterns/references/testing.md +72 -72
  461. package/seed/marketplaces/ecc/skills/token-budget-advisor/SKILL.md +133 -133
  462. package/seed/marketplaces/ecc/skills/ui-demo/SKILL.md +465 -465
  463. package/seed/marketplaces/ecc/skills/ui-to-vue/SKILL.md +134 -134
  464. package/seed/marketplaces/ecc/skills/uncloud/SKILL.md +343 -343
  465. package/seed/marketplaces/ecc/skills/unified-notifications-ops/SKILL.md +187 -187
  466. package/seed/marketplaces/ecc/skills/verification-loop/SKILL.md +126 -126
  467. package/seed/marketplaces/ecc/skills/video-editing/SKILL.md +310 -310
  468. package/seed/marketplaces/ecc/skills/videodb/SKILL.md +374 -374
  469. package/seed/marketplaces/ecc/skills/videodb/reference/api-reference.md +550 -550
  470. package/seed/marketplaces/ecc/skills/videodb/reference/capture-reference.md +407 -407
  471. package/seed/marketplaces/ecc/skills/videodb/reference/capture.md +101 -101
  472. package/seed/marketplaces/ecc/skills/videodb/reference/editor.md +443 -443
  473. package/seed/marketplaces/ecc/skills/videodb/reference/generative.md +331 -331
  474. package/seed/marketplaces/ecc/skills/videodb/reference/rtstream-reference.md +564 -564
  475. package/seed/marketplaces/ecc/skills/videodb/reference/rtstream.md +65 -65
  476. package/seed/marketplaces/ecc/skills/videodb/reference/search.md +230 -230
  477. package/seed/marketplaces/ecc/skills/videodb/reference/streaming.md +406 -406
  478. package/seed/marketplaces/ecc/skills/videodb/reference/use-cases.md +118 -118
  479. package/seed/marketplaces/ecc/skills/videodb/scripts/ws_listener.py +282 -282
  480. package/seed/marketplaces/ecc/skills/visa-doc-translate/README.md +86 -86
  481. package/seed/marketplaces/ecc/skills/visa-doc-translate/SKILL.md +117 -117
  482. package/seed/marketplaces/ecc/skills/vite-patterns/SKILL.md +449 -449
  483. package/seed/marketplaces/ecc/skills/windows-desktop-e2e/SKILL.md +887 -887
  484. package/seed/marketplaces/ecc/skills/workspace-surface-audit/SKILL.md +125 -125
  485. package/seed/marketplaces/ecc/skills/x-api/SKILL.md +234 -234
  486. package/seed/marketplaces/ecc/.claude/commands/add-language-rules.md +0 -39
  487. package/seed/marketplaces/ecc/.claude/commands/database-migration.md +0 -36
  488. package/seed/marketplaces/ecc/.claude/commands/feature-development.md +0 -38
  489. package/seed/marketplaces/ecc/.claude/ecc-tools.json +0 -334
  490. package/seed/marketplaces/ecc/.claude/enterprise/controls.md +0 -15
  491. package/seed/marketplaces/ecc/.claude/homunculus/instincts/inherited/everything-claude-code-instincts.yaml +0 -162
  492. package/seed/marketplaces/ecc/.claude/identity.json +0 -14
  493. package/seed/marketplaces/ecc/.claude/package-manager.json +0 -4
  494. package/seed/marketplaces/ecc/.claude/research/everything-claude-code-research-playbook.md +0 -21
  495. package/seed/marketplaces/ecc/.claude/rules/everything-claude-code-guardrails.md +0 -43
  496. package/seed/marketplaces/ecc/.claude/rules/node.md +0 -56
  497. package/seed/marketplaces/ecc/.claude/skills/everything-claude-code/SKILL.md +0 -442
  498. package/seed/marketplaces/ecc/.claude/team/everything-claude-code-team-config.json +0 -15
  499. package/seed/marketplaces/ecc/.vscode/settings.json +0 -17
@@ -1,764 +1,764 @@
1
- ---
2
- name: data-scraper-agent
3
- description: Build a fully automated AI-powered data collection agent for any public source — job boards, prices, news, GitHub, sports, anything. Scrapes on a schedule, enriches data with a free LLM (Gemini Flash), stores results in Notion/Sheets/Supabase, and learns from user feedback. Runs 100% free on GitHub Actions. Use when the user wants to monitor, collect, or track any public data automatically.
4
- origin: community
5
- ---
6
-
7
- # Data Scraper Agent
8
-
9
- Build a production-ready, AI-powered data collection agent for any public data source.
10
- Runs on a schedule, enriches results with a free LLM, stores to a database, and improves over time.
11
-
12
- **Stack: Python · Gemini Flash (free) · GitHub Actions (free) · Notion / Sheets / Supabase**
13
-
14
- ## When to Activate
15
-
16
- - User wants to scrape or monitor any public website or API
17
- - User says "build a bot that checks...", "monitor X for me", "collect data from..."
18
- - User wants to track jobs, prices, news, repos, sports scores, events, listings
19
- - User asks how to automate data collection without paying for hosting
20
- - User wants an agent that gets smarter over time based on their decisions
21
-
22
- ## Core Concepts
23
-
24
- ### The Three Layers
25
-
26
- Every data scraper agent has three layers:
27
-
28
- ```
29
- COLLECT → ENRICH → STORE
30
- │ │ │
31
- Scraper AI (LLM) Database
32
- runs on scores/ Notion /
33
- schedule summarises Sheets /
34
- & classifies Supabase
35
- ```
36
-
37
- ### Free Stack
38
-
39
- | Layer | Tool | Why |
40
- |---|---|---|
41
- | **Scraping** | `requests` + `BeautifulSoup` | No cost, covers 80% of public sites |
42
- | **JS-rendered sites** | `playwright` (free) | When HTML scraping fails |
43
- | **AI enrichment** | Gemini Flash via REST API | 500 req/day, 1M tokens/day — free |
44
- | **Storage** | Notion API | Free tier, great UI for review |
45
- | **Schedule** | GitHub Actions cron | Free for public repos |
46
- | **Learning** | JSON feedback file in repo | Zero infra, persists in git |
47
-
48
- ### AI Model Fallback Chain
49
-
50
- Build agents to auto-fallback across Gemini models on quota exhaustion:
51
-
52
- ```
53
- gemini-2.0-flash-lite (30 RPM) →
54
- gemini-2.0-flash (15 RPM) →
55
- gemini-2.5-flash (10 RPM) →
56
- gemini-flash-lite-latest (fallback)
57
- ```
58
-
59
- ### Batch API Calls for Efficiency
60
-
61
- Never call the LLM once per item. Always batch:
62
-
63
- ```python
64
- # BAD: 33 API calls for 33 items
65
- for item in items:
66
- result = call_ai(item) # 33 calls → hits rate limit
67
-
68
- # GOOD: 7 API calls for 33 items (batch size 5)
69
- for batch in chunks(items, size=5):
70
- results = call_ai(batch) # 7 calls → stays within free tier
71
- ```
72
-
73
- ---
74
-
75
- ## Workflow
76
-
77
- ### Step 1: Understand the Goal
78
-
79
- Ask the user:
80
-
81
- 1. **What to collect:** "What data source? URL / API / RSS / public endpoint?"
82
- 2. **What to extract:** "What fields matter? Title, price, URL, date, score?"
83
- 3. **How to store:** "Where should results go? Notion, Google Sheets, Supabase, or local file?"
84
- 4. **How to enrich:** "Do you want AI to score, summarise, classify, or match each item?"
85
- 5. **Frequency:** "How often should it run? Every hour, daily, weekly?"
86
-
87
- Common examples to prompt:
88
- - Job boards → score relevance to resume
89
- - Product prices → alert on drops
90
- - GitHub repos → summarise new releases
91
- - News feeds → classify by topic + sentiment
92
- - Sports results → extract stats to tracker
93
- - Events calendar → filter by interest
94
-
95
- ---
96
-
97
- ### Step 2: Design the Agent Architecture
98
-
99
- Generate this directory structure for the user:
100
-
101
- ```
102
- my-agent/
103
- ├── config.yaml # User customises this (keywords, filters, preferences)
104
- ├── profile/
105
- │ └── context.md # User context the AI uses (resume, interests, criteria)
106
- ├── scraper/
107
- │ ├── __init__.py
108
- │ ├── main.py # Orchestrator: scrape → enrich → store
109
- │ ├── filters.py # Rule-based pre-filter (fast, before AI)
110
- │ └── sources/
111
- │ ├── __init__.py
112
- │ └── source_name.py # One file per data source
113
- ├── ai/
114
- │ ├── __init__.py
115
- │ ├── client.py # Gemini REST client with model fallback
116
- │ ├── pipeline.py # Batch AI analysis
117
- │ ├── jd_fetcher.py # Fetch full content from URLs (optional)
118
- │ └── memory.py # Learn from user feedback
119
- ├── storage/
120
- │ ├── __init__.py
121
- │ └── notion_sync.py # Or sheets_sync.py / supabase_sync.py
122
- ├── data/
123
- │ └── feedback.json # User decision history (auto-updated)
124
- ├── .env.example
125
- ├── setup.py # One-time DB/schema creation
126
- ├── enrich_existing.py # Backfill AI scores on old rows
127
- ├── requirements.txt
128
- └── .github/
129
- └── workflows/
130
- └── scraper.yml # GitHub Actions schedule
131
- ```
132
-
133
- ---
134
-
135
- ### Step 3: Build the Scraper Source
136
-
137
- Template for any data source:
138
-
139
- ```python
140
- # scraper/sources/my_source.py
141
- """
142
- [Source Name] — scrapes [what] from [where].
143
- Method: [REST API / HTML scraping / RSS feed]
144
- """
145
- import requests
146
- from bs4 import BeautifulSoup
147
- from datetime import datetime, timezone
148
- from scraper.filters import is_relevant
149
-
150
- HEADERS = {
151
- "User-Agent": "Mozilla/5.0 (compatible; research-bot/1.0)",
152
- }
153
-
154
-
155
- def fetch() -> list[dict]:
156
- """
157
- Returns a list of items with consistent schema.
158
- Each item must have at minimum: name, url, date_found.
159
- """
160
- results = []
161
-
162
- # ---- REST API source ----
163
- resp = requests.get("https://api.example.com/items", headers=HEADERS, timeout=15)
164
- if resp.status_code == 200:
165
- for item in resp.json().get("results", []):
166
- if not is_relevant(item.get("title", "")):
167
- continue
168
- results.append(_normalise(item))
169
-
170
- return results
171
-
172
-
173
- def _normalise(raw: dict) -> dict:
174
- """Convert raw API/HTML data to the standard schema."""
175
- return {
176
- "name": raw.get("title", ""),
177
- "url": raw.get("link", ""),
178
- "source": "MySource",
179
- "date_found": datetime.now(timezone.utc).date().isoformat(),
180
- # add domain-specific fields here
181
- }
182
- ```
183
-
184
- **HTML scraping pattern:**
185
- ```python
186
- soup = BeautifulSoup(resp.text, "lxml")
187
- for card in soup.select("[class*='listing']"):
188
- title = card.select_one("h2, h3").get_text(strip=True)
189
- link = card.select_one("a")["href"]
190
- if not link.startswith("http"):
191
- link = f"https://example.com{link}"
192
- ```
193
-
194
- **RSS feed pattern:**
195
- ```python
196
- import xml.etree.ElementTree as ET
197
- root = ET.fromstring(resp.text)
198
- for item in root.findall(".//item"):
199
- title = item.findtext("title", "")
200
- link = item.findtext("link", "")
201
- ```
202
-
203
- ---
204
-
205
- ### Step 4: Build the Gemini AI Client
206
-
207
- ```python
208
- # ai/client.py
209
- import os, json, time, requests
210
-
211
- _last_call = 0.0
212
-
213
- MODEL_FALLBACK = [
214
- "gemini-2.0-flash-lite",
215
- "gemini-2.0-flash",
216
- "gemini-2.5-flash",
217
- "gemini-flash-lite-latest",
218
- ]
219
-
220
-
221
- def generate(prompt: str, model: str = "", rate_limit: float = 7.0) -> dict:
222
- """Call Gemini with auto-fallback on 429. Returns parsed JSON or {}."""
223
- global _last_call
224
-
225
- api_key = os.environ.get("GEMINI_API_KEY", "")
226
- if not api_key:
227
- return {}
228
-
229
- elapsed = time.time() - _last_call
230
- if elapsed < rate_limit:
231
- time.sleep(rate_limit - elapsed)
232
-
233
- models = [model] + [m for m in MODEL_FALLBACK if m != model] if model else MODEL_FALLBACK
234
- _last_call = time.time()
235
-
236
- for m in models:
237
- url = f"https://generativelanguage.googleapis.com/v1beta/models/{m}:generateContent?key={api_key}"
238
- payload = {
239
- "contents": [{"parts": [{"text": prompt}]}],
240
- "generationConfig": {
241
- "responseMimeType": "application/json",
242
- "temperature": 0.3,
243
- "maxOutputTokens": 2048,
244
- },
245
- }
246
- try:
247
- resp = requests.post(url, json=payload, timeout=30)
248
- if resp.status_code == 200:
249
- return _parse(resp)
250
- if resp.status_code in (429, 404):
251
- time.sleep(1)
252
- continue
253
- return {}
254
- except requests.RequestException:
255
- return {}
256
-
257
- return {}
258
-
259
-
260
- def _parse(resp) -> dict:
261
- try:
262
- text = (
263
- resp.json()
264
- .get("candidates", [{}])[0]
265
- .get("content", {})
266
- .get("parts", [{}])[0]
267
- .get("text", "")
268
- .strip()
269
- )
270
- if text.startswith("```"):
271
- text = text.split("\n", 1)[-1].rsplit("```", 1)[0]
272
- return json.loads(text)
273
- except (json.JSONDecodeError, KeyError):
274
- return {}
275
- ```
276
-
277
- ---
278
-
279
- ### Step 5: Build the AI Pipeline (Batch)
280
-
281
- ```python
282
- # ai/pipeline.py
283
- import json
284
- import yaml
285
- from pathlib import Path
286
- from ai.client import generate
287
-
288
- def analyse_batch(items: list[dict], context: str = "", preference_prompt: str = "") -> list[dict]:
289
- """Analyse items in batches. Returns items enriched with AI fields."""
290
- config = yaml.safe_load((Path(__file__).parent.parent / "config.yaml").read_text())
291
- model = config.get("ai", {}).get("model", "gemini-2.5-flash")
292
- rate_limit = config.get("ai", {}).get("rate_limit_seconds", 7.0)
293
- min_score = config.get("ai", {}).get("min_score", 0)
294
- batch_size = config.get("ai", {}).get("batch_size", 5)
295
-
296
- batches = [items[i:i + batch_size] for i in range(0, len(items), batch_size)]
297
- print(f" [AI] {len(items)} items → {len(batches)} API calls")
298
-
299
- enriched = []
300
- for i, batch in enumerate(batches):
301
- print(f" [AI] Batch {i + 1}/{len(batches)}...")
302
- prompt = _build_prompt(batch, context, preference_prompt, config)
303
- result = generate(prompt, model=model, rate_limit=rate_limit)
304
-
305
- analyses = result.get("analyses", [])
306
- for j, item in enumerate(batch):
307
- ai = analyses[j] if j < len(analyses) else {}
308
- if ai:
309
- score = max(0, min(100, int(ai.get("score", 0))))
310
- if min_score and score < min_score:
311
- continue
312
- enriched.append({**item, "ai_score": score, "ai_summary": ai.get("summary", ""), "ai_notes": ai.get("notes", "")})
313
- else:
314
- enriched.append(item)
315
-
316
- return enriched
317
-
318
-
319
- def _build_prompt(batch, context, preference_prompt, config):
320
- priorities = config.get("priorities", [])
321
- items_text = "\n\n".join(
322
- f"Item {i+1}: {json.dumps({k: v for k, v in item.items() if not k.startswith('_')})}"
323
- for i, item in enumerate(batch)
324
- )
325
-
326
- return f"""Analyse these {len(batch)} items and return a JSON object.
327
-
328
- # Items
329
- {items_text}
330
-
331
- # User Context
332
- {context[:800] if context else "Not provided"}
333
-
334
- # User Priorities
335
- {chr(10).join(f"- {p}" for p in priorities)}
336
-
337
- {preference_prompt}
338
-
339
- # Instructions
340
- Return: {{"analyses": [{{"score": <0-100>, "summary": "<2 sentences>", "notes": "<why this matches or doesn't>"}} for each item in order]}}
341
- Be concise. Score 90+=excellent match, 70-89=good, 50-69=ok, <50=weak."""
342
- ```
343
-
344
- ---
345
-
346
- ### Step 6: Build the Feedback Learning System
347
-
348
- ```python
349
- # ai/memory.py
350
- """Learn from user decisions to improve future scoring."""
351
- import json
352
- from pathlib import Path
353
-
354
- FEEDBACK_PATH = Path(__file__).parent.parent / "data" / "feedback.json"
355
-
356
-
357
- def load_feedback() -> dict:
358
- if FEEDBACK_PATH.exists():
359
- try:
360
- return json.loads(FEEDBACK_PATH.read_text())
361
- except (json.JSONDecodeError, OSError):
362
- pass
363
- return {"positive": [], "negative": []}
364
-
365
-
366
- def save_feedback(fb: dict):
367
- FEEDBACK_PATH.parent.mkdir(parents=True, exist_ok=True)
368
- FEEDBACK_PATH.write_text(json.dumps(fb, indent=2))
369
-
370
-
371
- def build_preference_prompt(feedback: dict, max_examples: int = 15) -> str:
372
- """Convert feedback history into a prompt bias section."""
373
- lines = []
374
- if feedback.get("positive"):
375
- lines.append("# Items the user LIKED (positive signal):")
376
- for e in feedback["positive"][-max_examples:]:
377
- lines.append(f"- {e}")
378
- if feedback.get("negative"):
379
- lines.append("\n# Items the user SKIPPED/REJECTED (negative signal):")
380
- for e in feedback["negative"][-max_examples:]:
381
- lines.append(f"- {e}")
382
- if lines:
383
- lines.append("\nUse these patterns to bias scoring on new items.")
384
- return "\n".join(lines)
385
- ```
386
-
387
- **Integration with your storage layer:** after each run, query your DB for items with positive/negative status and call `save_feedback()` with the extracted patterns.
388
-
389
- ---
390
-
391
- ### Step 7: Build Storage (Notion example)
392
-
393
- ```python
394
- # storage/notion_sync.py
395
- import os
396
- from notion_client import Client
397
- from notion_client.errors import APIResponseError
398
-
399
- _client = None
400
-
401
- def get_client():
402
- global _client
403
- if _client is None:
404
- _client = Client(auth=os.environ["NOTION_TOKEN"])
405
- return _client
406
-
407
- def get_existing_urls(db_id: str) -> set[str]:
408
- """Fetch all URLs already stored — used for deduplication."""
409
- client, seen, cursor = get_client(), set(), None
410
- while True:
411
- resp = client.databases.query(database_id=db_id, page_size=100, **{"start_cursor": cursor} if cursor else {})
412
- for page in resp["results"]:
413
- url = page["properties"].get("URL", {}).get("url", "")
414
- if url: seen.add(url)
415
- if not resp["has_more"]: break
416
- cursor = resp["next_cursor"]
417
- return seen
418
-
419
- def push_item(db_id: str, item: dict) -> bool:
420
- """Push one item to Notion. Returns True on success."""
421
- props = {
422
- "Name": {"title": [{"text": {"content": item.get("name", "")[:100]}}]},
423
- "URL": {"url": item.get("url")},
424
- "Source": {"select": {"name": item.get("source", "Unknown")}},
425
- "Date Found": {"date": {"start": item.get("date_found")}},
426
- "Status": {"select": {"name": "New"}},
427
- }
428
- # AI fields
429
- if item.get("ai_score") is not None:
430
- props["AI Score"] = {"number": item["ai_score"]}
431
- if item.get("ai_summary"):
432
- props["Summary"] = {"rich_text": [{"text": {"content": item["ai_summary"][:2000]}}]}
433
- if item.get("ai_notes"):
434
- props["Notes"] = {"rich_text": [{"text": {"content": item["ai_notes"][:2000]}}]}
435
-
436
- try:
437
- get_client().pages.create(parent={"database_id": db_id}, properties=props)
438
- return True
439
- except APIResponseError as e:
440
- print(f"[notion] Push failed: {e}")
441
- return False
442
-
443
- def sync(db_id: str, items: list[dict]) -> tuple[int, int]:
444
- existing = get_existing_urls(db_id)
445
- added = skipped = 0
446
- for item in items:
447
- if item.get("url") in existing:
448
- skipped += 1; continue
449
- if push_item(db_id, item):
450
- added += 1; existing.add(item["url"])
451
- else:
452
- skipped += 1
453
- return added, skipped
454
- ```
455
-
456
- ---
457
-
458
- ### Step 8: Orchestrate in main.py
459
-
460
- ```python
461
- # scraper/main.py
462
- import os, sys, yaml
463
- from pathlib import Path
464
- from dotenv import load_dotenv
465
-
466
- load_dotenv()
467
-
468
- from scraper.sources import my_source # add your sources
469
-
470
- # NOTE: This example uses Notion. If storage.provider is "sheets" or "supabase",
471
- # replace this import with storage.sheets_sync or storage.supabase_sync and update
472
- # the env var and sync() call accordingly.
473
- from storage.notion_sync import sync
474
-
475
- SOURCES = [
476
- ("My Source", my_source.fetch),
477
- ]
478
-
479
- def ai_enabled():
480
- return bool(os.environ.get("GEMINI_API_KEY"))
481
-
482
- def main():
483
- config = yaml.safe_load((Path(__file__).parent.parent / "config.yaml").read_text())
484
- provider = config.get("storage", {}).get("provider", "notion")
485
-
486
- # Resolve the storage target identifier from env based on provider
487
- if provider == "notion":
488
- db_id = os.environ.get("NOTION_DATABASE_ID")
489
- if not db_id:
490
- print("ERROR: NOTION_DATABASE_ID not set"); sys.exit(1)
491
- else:
492
- # Extend here for sheets (SHEET_ID) or supabase (SUPABASE_TABLE) etc.
493
- print(f"ERROR: provider '{provider}' not yet wired in main.py"); sys.exit(1)
494
-
495
- config = yaml.safe_load((Path(__file__).parent.parent / "config.yaml").read_text())
496
- all_items = []
497
-
498
- for name, fetch_fn in SOURCES:
499
- try:
500
- items = fetch_fn()
501
- print(f"[{name}] {len(items)} items")
502
- all_items.extend(items)
503
- except Exception as e:
504
- print(f"[{name}] FAILED: {e}")
505
-
506
- # Deduplicate by URL
507
- seen, deduped = set(), []
508
- for item in all_items:
509
- if (url := item.get("url", "")) and url not in seen:
510
- seen.add(url); deduped.append(item)
511
-
512
- print(f"Unique items: {len(deduped)}")
513
-
514
- if ai_enabled() and deduped:
515
- from ai.memory import load_feedback, build_preference_prompt
516
- from ai.pipeline import analyse_batch
517
-
518
- # load_feedback() reads data/feedback.json written by your feedback sync script.
519
- # To keep it current, implement a separate feedback_sync.py that queries your
520
- # storage provider for items with positive/negative statuses and calls save_feedback().
521
- feedback = load_feedback()
522
- preference = build_preference_prompt(feedback)
523
- context_path = Path(__file__).parent.parent / "profile" / "context.md"
524
- context = context_path.read_text() if context_path.exists() else ""
525
- deduped = analyse_batch(deduped, context=context, preference_prompt=preference)
526
- else:
527
- print("[AI] Skipped — GEMINI_API_KEY not set")
528
-
529
- added, skipped = sync(db_id, deduped)
530
- print(f"Done — {added} new, {skipped} existing")
531
-
532
- if __name__ == "__main__":
533
- main()
534
- ```
535
-
536
- ---
537
-
538
- ### Step 9: GitHub Actions Workflow
539
-
540
- ```yaml
541
- # .github/workflows/scraper.yml
542
- name: Data Scraper Agent
543
-
544
- on:
545
- schedule:
546
- - cron: "0 */3 * * *" # every 3 hours — adjust to your needs
547
- workflow_dispatch: # allow manual trigger
548
-
549
- permissions:
550
- contents: write # required for the feedback-history commit step
551
-
552
- jobs:
553
- scrape:
554
- runs-on: ubuntu-latest
555
- timeout-minutes: 20
556
-
557
- steps:
558
- - uses: actions/checkout@v4
559
-
560
- - uses: actions/setup-python@v5
561
- with:
562
- python-version: "3.11"
563
- cache: "pip"
564
-
565
- - run: pip install -r requirements.txt
566
-
567
- # Uncomment if Playwright is enabled in requirements.txt
568
- # - name: Install Playwright browsers
569
- # run: python -m playwright install chromium --with-deps
570
-
571
- - name: Run agent
572
- env:
573
- NOTION_TOKEN: ${{ secrets.NOTION_TOKEN }}
574
- NOTION_DATABASE_ID: ${{ secrets.NOTION_DATABASE_ID }}
575
- GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }}
576
- run: python -m scraper.main
577
-
578
- - name: Commit feedback history
579
- run: |
580
- git config user.name "github-actions[bot]"
581
- git config user.email "github-actions[bot]@users.noreply.github.com"
582
- git add data/feedback.json || true
583
- git diff --cached --quiet || git commit -m "chore: update feedback history"
584
- git push
585
- ```
586
-
587
- ---
588
-
589
- ### Step 10: config.yaml Template
590
-
591
- ```yaml
592
- # Customise this file — no code changes needed
593
-
594
- # What to collect (pre-filter before AI)
595
- filters:
596
- required_keywords: [] # item must contain at least one
597
- blocked_keywords: [] # item must not contain any
598
-
599
- # Your priorities — AI uses these for scoring
600
- priorities:
601
- - "example priority 1"
602
- - "example priority 2"
603
-
604
- # Storage
605
- storage:
606
- provider: "notion" # notion | sheets | supabase | sqlite
607
-
608
- # Feedback learning
609
- feedback:
610
- positive_statuses: ["Saved", "Applied", "Interested"]
611
- negative_statuses: ["Skip", "Rejected", "Not relevant"]
612
-
613
- # AI settings
614
- ai:
615
- enabled: true
616
- model: "gemini-2.5-flash"
617
- min_score: 0 # filter out items below this score
618
- rate_limit_seconds: 7 # seconds between API calls
619
- batch_size: 5 # items per API call
620
- ```
621
-
622
- ---
623
-
624
- ## Common Scraping Patterns
625
-
626
- ### Pattern 1: REST API (easiest)
627
- ```python
628
- resp = requests.get(url, params={"q": query}, headers=HEADERS, timeout=15)
629
- items = resp.json().get("results", [])
630
- ```
631
-
632
- ### Pattern 2: HTML Scraping
633
- ```python
634
- soup = BeautifulSoup(resp.text, "lxml")
635
- for card in soup.select(".listing-card"):
636
- title = card.select_one("h2").get_text(strip=True)
637
- href = card.select_one("a")["href"]
638
- ```
639
-
640
- ### Pattern 3: RSS Feed
641
- ```python
642
- import xml.etree.ElementTree as ET
643
- root = ET.fromstring(resp.text)
644
- for item in root.findall(".//item"):
645
- title = item.findtext("title", "")
646
- link = item.findtext("link", "")
647
- pub_date = item.findtext("pubDate", "")
648
- ```
649
-
650
- ### Pattern 4: Paginated API
651
- ```python
652
- page = 1
653
- while True:
654
- resp = requests.get(url, params={"page": page, "limit": 50}, timeout=15)
655
- data = resp.json()
656
- items = data.get("results", [])
657
- if not items:
658
- break
659
- for item in items:
660
- results.append(_normalise(item))
661
- if not data.get("has_more"):
662
- break
663
- page += 1
664
- ```
665
-
666
- ### Pattern 5: JS-Rendered Pages (Playwright)
667
- ```python
668
- from playwright.sync_api import sync_playwright
669
-
670
- with sync_playwright() as p:
671
- browser = p.chromium.launch()
672
- page = browser.new_page()
673
- page.goto(url)
674
- page.wait_for_selector(".listing")
675
- html = page.content()
676
- browser.close()
677
-
678
- soup = BeautifulSoup(html, "lxml")
679
- ```
680
-
681
- ---
682
-
683
- ## Anti-Patterns to Avoid
684
-
685
- | Anti-pattern | Problem | Fix |
686
- |---|---|---|
687
- | One LLM call per item | Hits rate limits instantly | Batch 5 items per call |
688
- | Hardcoded keywords in code | Not reusable | Move all config to `config.yaml` |
689
- | Scraping without rate limit | IP ban | Add `time.sleep(1)` between requests |
690
- | Storing secrets in code | Security risk | Always use `.env` + GitHub Secrets |
691
- | No deduplication | Duplicate rows pile up | Always check URL before pushing |
692
- | Ignoring `robots.txt` | Legal/ethical risk | Respect crawl rules; use public APIs when available |
693
- | JS-rendered sites with `requests` | Empty response | Use Playwright or look for the underlying API |
694
- | `maxOutputTokens` too low | Truncated JSON, parse error | Use 2048+ for batch responses |
695
-
696
- ---
697
-
698
- ## Free Tier Limits Reference
699
-
700
- | Service | Free Limit | Typical Usage |
701
- |---|---|---|
702
- | Gemini Flash Lite | 30 RPM, 1500 RPD | ~56 req/day at 3-hr intervals |
703
- | Gemini 2.0 Flash | 15 RPM, 1500 RPD | Good fallback |
704
- | Gemini 2.5 Flash | 10 RPM, 500 RPD | Use sparingly |
705
- | GitHub Actions | Unlimited (public repos) | ~20 min/day |
706
- | Notion API | Unlimited | ~200 writes/day |
707
- | Supabase | 500MB DB, 2GB transfer | Fine for most agents |
708
- | Google Sheets API | 300 req/min | Works for small agents |
709
-
710
- ---
711
-
712
- ## Requirements Template
713
-
714
- ```
715
- requests==2.31.0
716
- beautifulsoup4==4.12.3
717
- lxml==5.1.0
718
- python-dotenv==1.0.1
719
- pyyaml==6.0.2
720
- notion-client==2.2.1 # if using Notion
721
- # playwright==1.40.0 # uncomment for JS-rendered sites
722
- ```
723
-
724
- ---
725
-
726
- ## Quality Checklist
727
-
728
- Before marking the agent complete:
729
-
730
- - [ ] `config.yaml` controls all user-facing settings — no hardcoded values
731
- - [ ] `profile/context.md` holds user-specific context for AI matching
732
- - [ ] Deduplication by URL before every storage push
733
- - [ ] Gemini client has model fallback chain (4 models)
734
- - [ ] Batch size ≤ 5 items per API call
735
- - [ ] `maxOutputTokens` ≥ 2048
736
- - [ ] `.env` is in `.gitignore`
737
- - [ ] `.env.example` provided for onboarding
738
- - [ ] `setup.py` creates DB schema on first run
739
- - [ ] `enrich_existing.py` backfills AI scores on old rows
740
- - [ ] GitHub Actions workflow commits `feedback.json` after each run
741
- - [ ] README covers: setup in < 5 minutes, required secrets, customisation
742
-
743
- ---
744
-
745
- ## Real-World Examples
746
-
747
- ```
748
- "Build me an agent that monitors Hacker News for AI startup funding news"
749
- "Scrape product prices from 3 e-commerce sites and alert when they drop"
750
- "Track new GitHub repos tagged with 'llm' or 'agents' — summarise each one"
751
- "Collect Chief of Staff job listings from LinkedIn and Cutshort into Notion"
752
- "Monitor a subreddit for posts mentioning my company — classify sentiment"
753
- "Scrape new academic papers from arXiv on a topic I care about daily"
754
- "Track sports fixture results and keep a running table in Google Sheets"
755
- "Build a real estate listing watcher — alert on new properties under ₹1 Cr"
756
- ```
757
-
758
- ---
759
-
760
- ## Reference Implementation
761
-
762
- A complete working agent built with this exact architecture would scrape 4+ sources,
763
- batch Gemini calls, learn from Applied/Rejected decisions stored in Notion, and run
764
- 100% free on GitHub Actions. Follow Steps 1–9 above to build your own.
1
+ ---
2
+ name: data-scraper-agent
3
+ description: Build a fully automated AI-powered data collection agent for any public source — job boards, prices, news, GitHub, sports, anything. Scrapes on a schedule, enriches data with a free LLM (Gemini Flash), stores results in Notion/Sheets/Supabase, and learns from user feedback. Runs 100% free on GitHub Actions. Use when the user wants to monitor, collect, or track any public data automatically.
4
+ origin: community
5
+ ---
6
+
7
+ # Data Scraper Agent
8
+
9
+ Build a production-ready, AI-powered data collection agent for any public data source.
10
+ Runs on a schedule, enriches results with a free LLM, stores to a database, and improves over time.
11
+
12
+ **Stack: Python · Gemini Flash (free) · GitHub Actions (free) · Notion / Sheets / Supabase**
13
+
14
+ ## When to Activate
15
+
16
+ - User wants to scrape or monitor any public website or API
17
+ - User says "build a bot that checks...", "monitor X for me", "collect data from..."
18
+ - User wants to track jobs, prices, news, repos, sports scores, events, listings
19
+ - User asks how to automate data collection without paying for hosting
20
+ - User wants an agent that gets smarter over time based on their decisions
21
+
22
+ ## Core Concepts
23
+
24
+ ### The Three Layers
25
+
26
+ Every data scraper agent has three layers:
27
+
28
+ ```
29
+ COLLECT → ENRICH → STORE
30
+ │ │ │
31
+ Scraper AI (LLM) Database
32
+ runs on scores/ Notion /
33
+ schedule summarises Sheets /
34
+ & classifies Supabase
35
+ ```
36
+
37
+ ### Free Stack
38
+
39
+ | Layer | Tool | Why |
40
+ |---|---|---|
41
+ | **Scraping** | `requests` + `BeautifulSoup` | No cost, covers 80% of public sites |
42
+ | **JS-rendered sites** | `playwright` (free) | When HTML scraping fails |
43
+ | **AI enrichment** | Gemini Flash via REST API | 500 req/day, 1M tokens/day — free |
44
+ | **Storage** | Notion API | Free tier, great UI for review |
45
+ | **Schedule** | GitHub Actions cron | Free for public repos |
46
+ | **Learning** | JSON feedback file in repo | Zero infra, persists in git |
47
+
48
+ ### AI Model Fallback Chain
49
+
50
+ Build agents to auto-fallback across Gemini models on quota exhaustion:
51
+
52
+ ```
53
+ gemini-2.0-flash-lite (30 RPM) →
54
+ gemini-2.0-flash (15 RPM) →
55
+ gemini-2.5-flash (10 RPM) →
56
+ gemini-flash-lite-latest (fallback)
57
+ ```
58
+
59
+ ### Batch API Calls for Efficiency
60
+
61
+ Never call the LLM once per item. Always batch:
62
+
63
+ ```python
64
+ # BAD: 33 API calls for 33 items
65
+ for item in items:
66
+ result = call_ai(item) # 33 calls → hits rate limit
67
+
68
+ # GOOD: 7 API calls for 33 items (batch size 5)
69
+ for batch in chunks(items, size=5):
70
+ results = call_ai(batch) # 7 calls → stays within free tier
71
+ ```
72
+
73
+ ---
74
+
75
+ ## Workflow
76
+
77
+ ### Step 1: Understand the Goal
78
+
79
+ Ask the user:
80
+
81
+ 1. **What to collect:** "What data source? URL / API / RSS / public endpoint?"
82
+ 2. **What to extract:** "What fields matter? Title, price, URL, date, score?"
83
+ 3. **How to store:** "Where should results go? Notion, Google Sheets, Supabase, or local file?"
84
+ 4. **How to enrich:** "Do you want AI to score, summarise, classify, or match each item?"
85
+ 5. **Frequency:** "How often should it run? Every hour, daily, weekly?"
86
+
87
+ Common examples to prompt:
88
+ - Job boards → score relevance to resume
89
+ - Product prices → alert on drops
90
+ - GitHub repos → summarise new releases
91
+ - News feeds → classify by topic + sentiment
92
+ - Sports results → extract stats to tracker
93
+ - Events calendar → filter by interest
94
+
95
+ ---
96
+
97
+ ### Step 2: Design the Agent Architecture
98
+
99
+ Generate this directory structure for the user:
100
+
101
+ ```
102
+ my-agent/
103
+ ├── config.yaml # User customises this (keywords, filters, preferences)
104
+ ├── profile/
105
+ │ └── context.md # User context the AI uses (resume, interests, criteria)
106
+ ├── scraper/
107
+ │ ├── __init__.py
108
+ │ ├── main.py # Orchestrator: scrape → enrich → store
109
+ │ ├── filters.py # Rule-based pre-filter (fast, before AI)
110
+ │ └── sources/
111
+ │ ├── __init__.py
112
+ │ └── source_name.py # One file per data source
113
+ ├── ai/
114
+ │ ├── __init__.py
115
+ │ ├── client.py # Gemini REST client with model fallback
116
+ │ ├── pipeline.py # Batch AI analysis
117
+ │ ├── jd_fetcher.py # Fetch full content from URLs (optional)
118
+ │ └── memory.py # Learn from user feedback
119
+ ├── storage/
120
+ │ ├── __init__.py
121
+ │ └── notion_sync.py # Or sheets_sync.py / supabase_sync.py
122
+ ├── data/
123
+ │ └── feedback.json # User decision history (auto-updated)
124
+ ├── .env.example
125
+ ├── setup.py # One-time DB/schema creation
126
+ ├── enrich_existing.py # Backfill AI scores on old rows
127
+ ├── requirements.txt
128
+ └── .github/
129
+ └── workflows/
130
+ └── scraper.yml # GitHub Actions schedule
131
+ ```
132
+
133
+ ---
134
+
135
+ ### Step 3: Build the Scraper Source
136
+
137
+ Template for any data source:
138
+
139
+ ```python
140
+ # scraper/sources/my_source.py
141
+ """
142
+ [Source Name] — scrapes [what] from [where].
143
+ Method: [REST API / HTML scraping / RSS feed]
144
+ """
145
+ import requests
146
+ from bs4 import BeautifulSoup
147
+ from datetime import datetime, timezone
148
+ from scraper.filters import is_relevant
149
+
150
+ HEADERS = {
151
+ "User-Agent": "Mozilla/5.0 (compatible; research-bot/1.0)",
152
+ }
153
+
154
+
155
+ def fetch() -> list[dict]:
156
+ """
157
+ Returns a list of items with consistent schema.
158
+ Each item must have at minimum: name, url, date_found.
159
+ """
160
+ results = []
161
+
162
+ # ---- REST API source ----
163
+ resp = requests.get("https://api.example.com/items", headers=HEADERS, timeout=15)
164
+ if resp.status_code == 200:
165
+ for item in resp.json().get("results", []):
166
+ if not is_relevant(item.get("title", "")):
167
+ continue
168
+ results.append(_normalise(item))
169
+
170
+ return results
171
+
172
+
173
+ def _normalise(raw: dict) -> dict:
174
+ """Convert raw API/HTML data to the standard schema."""
175
+ return {
176
+ "name": raw.get("title", ""),
177
+ "url": raw.get("link", ""),
178
+ "source": "MySource",
179
+ "date_found": datetime.now(timezone.utc).date().isoformat(),
180
+ # add domain-specific fields here
181
+ }
182
+ ```
183
+
184
+ **HTML scraping pattern:**
185
+ ```python
186
+ soup = BeautifulSoup(resp.text, "lxml")
187
+ for card in soup.select("[class*='listing']"):
188
+ title = card.select_one("h2, h3").get_text(strip=True)
189
+ link = card.select_one("a")["href"]
190
+ if not link.startswith("http"):
191
+ link = f"https://example.com{link}"
192
+ ```
193
+
194
+ **RSS feed pattern:**
195
+ ```python
196
+ import xml.etree.ElementTree as ET
197
+ root = ET.fromstring(resp.text)
198
+ for item in root.findall(".//item"):
199
+ title = item.findtext("title", "")
200
+ link = item.findtext("link", "")
201
+ ```
202
+
203
+ ---
204
+
205
+ ### Step 4: Build the Gemini AI Client
206
+
207
+ ```python
208
+ # ai/client.py
209
+ import os, json, time, requests
210
+
211
+ _last_call = 0.0
212
+
213
+ MODEL_FALLBACK = [
214
+ "gemini-2.0-flash-lite",
215
+ "gemini-2.0-flash",
216
+ "gemini-2.5-flash",
217
+ "gemini-flash-lite-latest",
218
+ ]
219
+
220
+
221
+ def generate(prompt: str, model: str = "", rate_limit: float = 7.0) -> dict:
222
+ """Call Gemini with auto-fallback on 429. Returns parsed JSON or {}."""
223
+ global _last_call
224
+
225
+ api_key = os.environ.get("GEMINI_API_KEY", "")
226
+ if not api_key:
227
+ return {}
228
+
229
+ elapsed = time.time() - _last_call
230
+ if elapsed < rate_limit:
231
+ time.sleep(rate_limit - elapsed)
232
+
233
+ models = [model] + [m for m in MODEL_FALLBACK if m != model] if model else MODEL_FALLBACK
234
+ _last_call = time.time()
235
+
236
+ for m in models:
237
+ url = f"https://generativelanguage.googleapis.com/v1beta/models/{m}:generateContent?key={api_key}"
238
+ payload = {
239
+ "contents": [{"parts": [{"text": prompt}]}],
240
+ "generationConfig": {
241
+ "responseMimeType": "application/json",
242
+ "temperature": 0.3,
243
+ "maxOutputTokens": 2048,
244
+ },
245
+ }
246
+ try:
247
+ resp = requests.post(url, json=payload, timeout=30)
248
+ if resp.status_code == 200:
249
+ return _parse(resp)
250
+ if resp.status_code in (429, 404):
251
+ time.sleep(1)
252
+ continue
253
+ return {}
254
+ except requests.RequestException:
255
+ return {}
256
+
257
+ return {}
258
+
259
+
260
+ def _parse(resp) -> dict:
261
+ try:
262
+ text = (
263
+ resp.json()
264
+ .get("candidates", [{}])[0]
265
+ .get("content", {})
266
+ .get("parts", [{}])[0]
267
+ .get("text", "")
268
+ .strip()
269
+ )
270
+ if text.startswith("```"):
271
+ text = text.split("\n", 1)[-1].rsplit("```", 1)[0]
272
+ return json.loads(text)
273
+ except (json.JSONDecodeError, KeyError):
274
+ return {}
275
+ ```
276
+
277
+ ---
278
+
279
+ ### Step 5: Build the AI Pipeline (Batch)
280
+
281
+ ```python
282
+ # ai/pipeline.py
283
+ import json
284
+ import yaml
285
+ from pathlib import Path
286
+ from ai.client import generate
287
+
288
+ def analyse_batch(items: list[dict], context: str = "", preference_prompt: str = "") -> list[dict]:
289
+ """Analyse items in batches. Returns items enriched with AI fields."""
290
+ config = yaml.safe_load((Path(__file__).parent.parent / "config.yaml").read_text())
291
+ model = config.get("ai", {}).get("model", "gemini-2.5-flash")
292
+ rate_limit = config.get("ai", {}).get("rate_limit_seconds", 7.0)
293
+ min_score = config.get("ai", {}).get("min_score", 0)
294
+ batch_size = config.get("ai", {}).get("batch_size", 5)
295
+
296
+ batches = [items[i:i + batch_size] for i in range(0, len(items), batch_size)]
297
+ print(f" [AI] {len(items)} items → {len(batches)} API calls")
298
+
299
+ enriched = []
300
+ for i, batch in enumerate(batches):
301
+ print(f" [AI] Batch {i + 1}/{len(batches)}...")
302
+ prompt = _build_prompt(batch, context, preference_prompt, config)
303
+ result = generate(prompt, model=model, rate_limit=rate_limit)
304
+
305
+ analyses = result.get("analyses", [])
306
+ for j, item in enumerate(batch):
307
+ ai = analyses[j] if j < len(analyses) else {}
308
+ if ai:
309
+ score = max(0, min(100, int(ai.get("score", 0))))
310
+ if min_score and score < min_score:
311
+ continue
312
+ enriched.append({**item, "ai_score": score, "ai_summary": ai.get("summary", ""), "ai_notes": ai.get("notes", "")})
313
+ else:
314
+ enriched.append(item)
315
+
316
+ return enriched
317
+
318
+
319
+ def _build_prompt(batch, context, preference_prompt, config):
320
+ priorities = config.get("priorities", [])
321
+ items_text = "\n\n".join(
322
+ f"Item {i+1}: {json.dumps({k: v for k, v in item.items() if not k.startswith('_')})}"
323
+ for i, item in enumerate(batch)
324
+ )
325
+
326
+ return f"""Analyse these {len(batch)} items and return a JSON object.
327
+
328
+ # Items
329
+ {items_text}
330
+
331
+ # User Context
332
+ {context[:800] if context else "Not provided"}
333
+
334
+ # User Priorities
335
+ {chr(10).join(f"- {p}" for p in priorities)}
336
+
337
+ {preference_prompt}
338
+
339
+ # Instructions
340
+ Return: {{"analyses": [{{"score": <0-100>, "summary": "<2 sentences>", "notes": "<why this matches or doesn't>"}} for each item in order]}}
341
+ Be concise. Score 90+=excellent match, 70-89=good, 50-69=ok, <50=weak."""
342
+ ```
343
+
344
+ ---
345
+
346
+ ### Step 6: Build the Feedback Learning System
347
+
348
+ ```python
349
+ # ai/memory.py
350
+ """Learn from user decisions to improve future scoring."""
351
+ import json
352
+ from pathlib import Path
353
+
354
+ FEEDBACK_PATH = Path(__file__).parent.parent / "data" / "feedback.json"
355
+
356
+
357
+ def load_feedback() -> dict:
358
+ if FEEDBACK_PATH.exists():
359
+ try:
360
+ return json.loads(FEEDBACK_PATH.read_text())
361
+ except (json.JSONDecodeError, OSError):
362
+ pass
363
+ return {"positive": [], "negative": []}
364
+
365
+
366
+ def save_feedback(fb: dict):
367
+ FEEDBACK_PATH.parent.mkdir(parents=True, exist_ok=True)
368
+ FEEDBACK_PATH.write_text(json.dumps(fb, indent=2))
369
+
370
+
371
+ def build_preference_prompt(feedback: dict, max_examples: int = 15) -> str:
372
+ """Convert feedback history into a prompt bias section."""
373
+ lines = []
374
+ if feedback.get("positive"):
375
+ lines.append("# Items the user LIKED (positive signal):")
376
+ for e in feedback["positive"][-max_examples:]:
377
+ lines.append(f"- {e}")
378
+ if feedback.get("negative"):
379
+ lines.append("\n# Items the user SKIPPED/REJECTED (negative signal):")
380
+ for e in feedback["negative"][-max_examples:]:
381
+ lines.append(f"- {e}")
382
+ if lines:
383
+ lines.append("\nUse these patterns to bias scoring on new items.")
384
+ return "\n".join(lines)
385
+ ```
386
+
387
+ **Integration with your storage layer:** after each run, query your DB for items with positive/negative status and call `save_feedback()` with the extracted patterns.
388
+
389
+ ---
390
+
391
+ ### Step 7: Build Storage (Notion example)
392
+
393
+ ```python
394
+ # storage/notion_sync.py
395
+ import os
396
+ from notion_client import Client
397
+ from notion_client.errors import APIResponseError
398
+
399
+ _client = None
400
+
401
+ def get_client():
402
+ global _client
403
+ if _client is None:
404
+ _client = Client(auth=os.environ["NOTION_TOKEN"])
405
+ return _client
406
+
407
+ def get_existing_urls(db_id: str) -> set[str]:
408
+ """Fetch all URLs already stored — used for deduplication."""
409
+ client, seen, cursor = get_client(), set(), None
410
+ while True:
411
+ resp = client.databases.query(database_id=db_id, page_size=100, **{"start_cursor": cursor} if cursor else {})
412
+ for page in resp["results"]:
413
+ url = page["properties"].get("URL", {}).get("url", "")
414
+ if url: seen.add(url)
415
+ if not resp["has_more"]: break
416
+ cursor = resp["next_cursor"]
417
+ return seen
418
+
419
+ def push_item(db_id: str, item: dict) -> bool:
420
+ """Push one item to Notion. Returns True on success."""
421
+ props = {
422
+ "Name": {"title": [{"text": {"content": item.get("name", "")[:100]}}]},
423
+ "URL": {"url": item.get("url")},
424
+ "Source": {"select": {"name": item.get("source", "Unknown")}},
425
+ "Date Found": {"date": {"start": item.get("date_found")}},
426
+ "Status": {"select": {"name": "New"}},
427
+ }
428
+ # AI fields
429
+ if item.get("ai_score") is not None:
430
+ props["AI Score"] = {"number": item["ai_score"]}
431
+ if item.get("ai_summary"):
432
+ props["Summary"] = {"rich_text": [{"text": {"content": item["ai_summary"][:2000]}}]}
433
+ if item.get("ai_notes"):
434
+ props["Notes"] = {"rich_text": [{"text": {"content": item["ai_notes"][:2000]}}]}
435
+
436
+ try:
437
+ get_client().pages.create(parent={"database_id": db_id}, properties=props)
438
+ return True
439
+ except APIResponseError as e:
440
+ print(f"[notion] Push failed: {e}")
441
+ return False
442
+
443
+ def sync(db_id: str, items: list[dict]) -> tuple[int, int]:
444
+ existing = get_existing_urls(db_id)
445
+ added = skipped = 0
446
+ for item in items:
447
+ if item.get("url") in existing:
448
+ skipped += 1; continue
449
+ if push_item(db_id, item):
450
+ added += 1; existing.add(item["url"])
451
+ else:
452
+ skipped += 1
453
+ return added, skipped
454
+ ```
455
+
456
+ ---
457
+
458
+ ### Step 8: Orchestrate in main.py
459
+
460
+ ```python
461
+ # scraper/main.py
462
+ import os, sys, yaml
463
+ from pathlib import Path
464
+ from dotenv import load_dotenv
465
+
466
+ load_dotenv()
467
+
468
+ from scraper.sources import my_source # add your sources
469
+
470
+ # NOTE: This example uses Notion. If storage.provider is "sheets" or "supabase",
471
+ # replace this import with storage.sheets_sync or storage.supabase_sync and update
472
+ # the env var and sync() call accordingly.
473
+ from storage.notion_sync import sync
474
+
475
+ SOURCES = [
476
+ ("My Source", my_source.fetch),
477
+ ]
478
+
479
+ def ai_enabled():
480
+ return bool(os.environ.get("GEMINI_API_KEY"))
481
+
482
+ def main():
483
+ config = yaml.safe_load((Path(__file__).parent.parent / "config.yaml").read_text())
484
+ provider = config.get("storage", {}).get("provider", "notion")
485
+
486
+ # Resolve the storage target identifier from env based on provider
487
+ if provider == "notion":
488
+ db_id = os.environ.get("NOTION_DATABASE_ID")
489
+ if not db_id:
490
+ print("ERROR: NOTION_DATABASE_ID not set"); sys.exit(1)
491
+ else:
492
+ # Extend here for sheets (SHEET_ID) or supabase (SUPABASE_TABLE) etc.
493
+ print(f"ERROR: provider '{provider}' not yet wired in main.py"); sys.exit(1)
494
+
495
+ config = yaml.safe_load((Path(__file__).parent.parent / "config.yaml").read_text())
496
+ all_items = []
497
+
498
+ for name, fetch_fn in SOURCES:
499
+ try:
500
+ items = fetch_fn()
501
+ print(f"[{name}] {len(items)} items")
502
+ all_items.extend(items)
503
+ except Exception as e:
504
+ print(f"[{name}] FAILED: {e}")
505
+
506
+ # Deduplicate by URL
507
+ seen, deduped = set(), []
508
+ for item in all_items:
509
+ if (url := item.get("url", "")) and url not in seen:
510
+ seen.add(url); deduped.append(item)
511
+
512
+ print(f"Unique items: {len(deduped)}")
513
+
514
+ if ai_enabled() and deduped:
515
+ from ai.memory import load_feedback, build_preference_prompt
516
+ from ai.pipeline import analyse_batch
517
+
518
+ # load_feedback() reads data/feedback.json written by your feedback sync script.
519
+ # To keep it current, implement a separate feedback_sync.py that queries your
520
+ # storage provider for items with positive/negative statuses and calls save_feedback().
521
+ feedback = load_feedback()
522
+ preference = build_preference_prompt(feedback)
523
+ context_path = Path(__file__).parent.parent / "profile" / "context.md"
524
+ context = context_path.read_text() if context_path.exists() else ""
525
+ deduped = analyse_batch(deduped, context=context, preference_prompt=preference)
526
+ else:
527
+ print("[AI] Skipped — GEMINI_API_KEY not set")
528
+
529
+ added, skipped = sync(db_id, deduped)
530
+ print(f"Done — {added} new, {skipped} existing")
531
+
532
+ if __name__ == "__main__":
533
+ main()
534
+ ```
535
+
536
+ ---
537
+
538
+ ### Step 9: GitHub Actions Workflow
539
+
540
+ ```yaml
541
+ # .github/workflows/scraper.yml
542
+ name: Data Scraper Agent
543
+
544
+ on:
545
+ schedule:
546
+ - cron: "0 */3 * * *" # every 3 hours — adjust to your needs
547
+ workflow_dispatch: # allow manual trigger
548
+
549
+ permissions:
550
+ contents: write # required for the feedback-history commit step
551
+
552
+ jobs:
553
+ scrape:
554
+ runs-on: ubuntu-latest
555
+ timeout-minutes: 20
556
+
557
+ steps:
558
+ - uses: actions/checkout@v4
559
+
560
+ - uses: actions/setup-python@v5
561
+ with:
562
+ python-version: "3.11"
563
+ cache: "pip"
564
+
565
+ - run: pip install -r requirements.txt
566
+
567
+ # Uncomment if Playwright is enabled in requirements.txt
568
+ # - name: Install Playwright browsers
569
+ # run: python -m playwright install chromium --with-deps
570
+
571
+ - name: Run agent
572
+ env:
573
+ NOTION_TOKEN: ${{ secrets.NOTION_TOKEN }}
574
+ NOTION_DATABASE_ID: ${{ secrets.NOTION_DATABASE_ID }}
575
+ GEMINI_API_KEY: ${{ secrets.GEMINI_API_KEY }}
576
+ run: python -m scraper.main
577
+
578
+ - name: Commit feedback history
579
+ run: |
580
+ git config user.name "github-actions[bot]"
581
+ git config user.email "github-actions[bot]@users.noreply.github.com"
582
+ git add data/feedback.json || true
583
+ git diff --cached --quiet || git commit -m "chore: update feedback history"
584
+ git push
585
+ ```
586
+
587
+ ---
588
+
589
+ ### Step 10: config.yaml Template
590
+
591
+ ```yaml
592
+ # Customise this file — no code changes needed
593
+
594
+ # What to collect (pre-filter before AI)
595
+ filters:
596
+ required_keywords: [] # item must contain at least one
597
+ blocked_keywords: [] # item must not contain any
598
+
599
+ # Your priorities — AI uses these for scoring
600
+ priorities:
601
+ - "example priority 1"
602
+ - "example priority 2"
603
+
604
+ # Storage
605
+ storage:
606
+ provider: "notion" # notion | sheets | supabase | sqlite
607
+
608
+ # Feedback learning
609
+ feedback:
610
+ positive_statuses: ["Saved", "Applied", "Interested"]
611
+ negative_statuses: ["Skip", "Rejected", "Not relevant"]
612
+
613
+ # AI settings
614
+ ai:
615
+ enabled: true
616
+ model: "gemini-2.5-flash"
617
+ min_score: 0 # filter out items below this score
618
+ rate_limit_seconds: 7 # seconds between API calls
619
+ batch_size: 5 # items per API call
620
+ ```
621
+
622
+ ---
623
+
624
+ ## Common Scraping Patterns
625
+
626
+ ### Pattern 1: REST API (easiest)
627
+ ```python
628
+ resp = requests.get(url, params={"q": query}, headers=HEADERS, timeout=15)
629
+ items = resp.json().get("results", [])
630
+ ```
631
+
632
+ ### Pattern 2: HTML Scraping
633
+ ```python
634
+ soup = BeautifulSoup(resp.text, "lxml")
635
+ for card in soup.select(".listing-card"):
636
+ title = card.select_one("h2").get_text(strip=True)
637
+ href = card.select_one("a")["href"]
638
+ ```
639
+
640
+ ### Pattern 3: RSS Feed
641
+ ```python
642
+ import xml.etree.ElementTree as ET
643
+ root = ET.fromstring(resp.text)
644
+ for item in root.findall(".//item"):
645
+ title = item.findtext("title", "")
646
+ link = item.findtext("link", "")
647
+ pub_date = item.findtext("pubDate", "")
648
+ ```
649
+
650
+ ### Pattern 4: Paginated API
651
+ ```python
652
+ page = 1
653
+ while True:
654
+ resp = requests.get(url, params={"page": page, "limit": 50}, timeout=15)
655
+ data = resp.json()
656
+ items = data.get("results", [])
657
+ if not items:
658
+ break
659
+ for item in items:
660
+ results.append(_normalise(item))
661
+ if not data.get("has_more"):
662
+ break
663
+ page += 1
664
+ ```
665
+
666
+ ### Pattern 5: JS-Rendered Pages (Playwright)
667
+ ```python
668
+ from playwright.sync_api import sync_playwright
669
+
670
+ with sync_playwright() as p:
671
+ browser = p.chromium.launch()
672
+ page = browser.new_page()
673
+ page.goto(url)
674
+ page.wait_for_selector(".listing")
675
+ html = page.content()
676
+ browser.close()
677
+
678
+ soup = BeautifulSoup(html, "lxml")
679
+ ```
680
+
681
+ ---
682
+
683
+ ## Anti-Patterns to Avoid
684
+
685
+ | Anti-pattern | Problem | Fix |
686
+ |---|---|---|
687
+ | One LLM call per item | Hits rate limits instantly | Batch 5 items per call |
688
+ | Hardcoded keywords in code | Not reusable | Move all config to `config.yaml` |
689
+ | Scraping without rate limit | IP ban | Add `time.sleep(1)` between requests |
690
+ | Storing secrets in code | Security risk | Always use `.env` + GitHub Secrets |
691
+ | No deduplication | Duplicate rows pile up | Always check URL before pushing |
692
+ | Ignoring `robots.txt` | Legal/ethical risk | Respect crawl rules; use public APIs when available |
693
+ | JS-rendered sites with `requests` | Empty response | Use Playwright or look for the underlying API |
694
+ | `maxOutputTokens` too low | Truncated JSON, parse error | Use 2048+ for batch responses |
695
+
696
+ ---
697
+
698
+ ## Free Tier Limits Reference
699
+
700
+ | Service | Free Limit | Typical Usage |
701
+ |---|---|---|
702
+ | Gemini Flash Lite | 30 RPM, 1500 RPD | ~56 req/day at 3-hr intervals |
703
+ | Gemini 2.0 Flash | 15 RPM, 1500 RPD | Good fallback |
704
+ | Gemini 2.5 Flash | 10 RPM, 500 RPD | Use sparingly |
705
+ | GitHub Actions | Unlimited (public repos) | ~20 min/day |
706
+ | Notion API | Unlimited | ~200 writes/day |
707
+ | Supabase | 500MB DB, 2GB transfer | Fine for most agents |
708
+ | Google Sheets API | 300 req/min | Works for small agents |
709
+
710
+ ---
711
+
712
+ ## Requirements Template
713
+
714
+ ```
715
+ requests==2.31.0
716
+ beautifulsoup4==4.12.3
717
+ lxml==5.1.0
718
+ python-dotenv==1.0.1
719
+ pyyaml==6.0.2
720
+ notion-client==2.2.1 # if using Notion
721
+ # playwright==1.40.0 # uncomment for JS-rendered sites
722
+ ```
723
+
724
+ ---
725
+
726
+ ## Quality Checklist
727
+
728
+ Before marking the agent complete:
729
+
730
+ - [ ] `config.yaml` controls all user-facing settings — no hardcoded values
731
+ - [ ] `profile/context.md` holds user-specific context for AI matching
732
+ - [ ] Deduplication by URL before every storage push
733
+ - [ ] Gemini client has model fallback chain (4 models)
734
+ - [ ] Batch size ≤ 5 items per API call
735
+ - [ ] `maxOutputTokens` ≥ 2048
736
+ - [ ] `.env` is in `.gitignore`
737
+ - [ ] `.env.example` provided for onboarding
738
+ - [ ] `setup.py` creates DB schema on first run
739
+ - [ ] `enrich_existing.py` backfills AI scores on old rows
740
+ - [ ] GitHub Actions workflow commits `feedback.json` after each run
741
+ - [ ] README covers: setup in < 5 minutes, required secrets, customisation
742
+
743
+ ---
744
+
745
+ ## Real-World Examples
746
+
747
+ ```
748
+ "Build me an agent that monitors Hacker News for AI startup funding news"
749
+ "Scrape product prices from 3 e-commerce sites and alert when they drop"
750
+ "Track new GitHub repos tagged with 'llm' or 'agents' — summarise each one"
751
+ "Collect Chief of Staff job listings from LinkedIn and Cutshort into Notion"
752
+ "Monitor a subreddit for posts mentioning my company — classify sentiment"
753
+ "Scrape new academic papers from arXiv on a topic I care about daily"
754
+ "Track sports fixture results and keep a running table in Google Sheets"
755
+ "Build a real estate listing watcher — alert on new properties under ₹1 Cr"
756
+ ```
757
+
758
+ ---
759
+
760
+ ## Reference Implementation
761
+
762
+ A complete working agent built with this exact architecture would scrape 4+ sources,
763
+ batch Gemini calls, learn from Applied/Rejected decisions stored in Notion, and run
764
+ 100% free on GitHub Actions. Follow Steps 1–9 above to build your own.