ghimera 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (185) hide show
  1. ghimera-0.3.0/.gitignore +8 -0
  2. ghimera-0.3.0/CHANGELOG.md +134 -0
  3. ghimera-0.3.0/LICENSE +21 -0
  4. ghimera-0.3.0/PKG-INFO +314 -0
  5. ghimera-0.3.0/README.md +287 -0
  6. ghimera-0.3.0/docs/C0.md +166 -0
  7. ghimera-0.3.0/docs/C0_EVIDENCE.md +82 -0
  8. ghimera-0.3.0/docs/C1.md +94 -0
  9. ghimera-0.3.0/docs/C1_BROWSER.md +144 -0
  10. ghimera-0.3.0/docs/C1_BROWSER_EVIDENCE.md +98 -0
  11. ghimera-0.3.0/docs/C1_BROWSER_REDIRECT_EVIDENCE.md +124 -0
  12. ghimera-0.3.0/docs/C1_HTTP_EVIDENCE.md +71 -0
  13. ghimera-0.3.0/docs/C1_TOR_EVIDENCE.md +75 -0
  14. ghimera-0.3.0/docs/C2_DEDUP.md +51 -0
  15. ghimera-0.3.0/docs/C2_DEDUP_EVIDENCE.md +57 -0
  16. ghimera-0.3.0/docs/C2_DOCUMENTS.md +160 -0
  17. ghimera-0.3.0/docs/C2_DOCUMENTS_EVIDENCE.md +66 -0
  18. ghimera-0.3.0/docs/C2_EXTRACTION_RECOVERY_EVIDENCE.md +60 -0
  19. ghimera-0.3.0/docs/C2_HTML.md +156 -0
  20. ghimera-0.3.0/docs/C2_HTML_EVIDENCE.md +94 -0
  21. ghimera-0.3.0/docs/C2_LOCATOR_DRIFT_EVIDENCE.md +64 -0
  22. ghimera-0.3.0/docs/C2_PDF_MODELS_EVIDENCE.md +99 -0
  23. ghimera-0.3.0/docs/C3_EMBEDDING_SCORING.md +174 -0
  24. ghimera-0.3.0/docs/C3_EMBEDDING_SCORING_EVIDENCE.md +129 -0
  25. ghimera-0.3.0/docs/C3_GRAPH_EVIDENCE.md +66 -0
  26. ghimera-0.3.0/docs/C3_INTENT_SCORING_EVIDENCE.md +84 -0
  27. ghimera-0.3.0/docs/C3_LIVE_DISCOVERY.md +81 -0
  28. ghimera-0.3.0/docs/C3_LIVE_RESEARCH.md +83 -0
  29. ghimera-0.3.0/docs/C3_MODELS.md +140 -0
  30. ghimera-0.3.0/docs/C3_MODELS_EVIDENCE.md +48 -0
  31. ghimera-0.3.0/docs/C3_REAL_MODEL_EVIDENCE.md +197 -0
  32. ghimera-0.3.0/docs/C3_REFERENCES.md +70 -0
  33. ghimera-0.3.0/docs/C3_REFERENCES_EVIDENCE.md +60 -0
  34. ghimera-0.3.0/docs/C3_RESEARCH.md +138 -0
  35. ghimera-0.3.0/docs/C3_RESEARCH_EVIDENCE.md +70 -0
  36. ghimera-0.3.0/docs/C3_SEARCH_EVIDENCE.md +83 -0
  37. ghimera-0.3.0/docs/C3_SEARCH_HTML.md +94 -0
  38. ghimera-0.3.0/docs/COLLECTOR.md +144 -0
  39. ghimera-0.3.0/docs/COLLECTOR_COMMAND.md +103 -0
  40. ghimera-0.3.0/docs/COLLECTOR_COMMAND_EVIDENCE.md +101 -0
  41. ghimera-0.3.0/docs/COLLECTOR_EVIDENCE.md +116 -0
  42. ghimera-0.3.0/docs/ORGANIZATION_RESEARCH.md +102 -0
  43. ghimera-0.3.0/docs/RELEASE_030.md +55 -0
  44. ghimera-0.3.0/docs/RELEASE_0_2_0.md +66 -0
  45. ghimera-0.3.0/docs/RESEARCH_GRAPH.md +84 -0
  46. ghimera-0.3.0/docs/RUN_JOURNAL.md +79 -0
  47. ghimera-0.3.0/docs/RUN_JOURNAL_EVIDENCE.md +55 -0
  48. ghimera-0.3.0/docs/SOURCE_SESSIONS.md +84 -0
  49. ghimera-0.3.0/docs/SOURCE_SESSIONS_EVIDENCE.md +56 -0
  50. ghimera-0.3.0/docs/TOR.md +85 -0
  51. ghimera-0.3.0/examples/browser.toml +25 -0
  52. ghimera-0.3.0/examples/chimera-tor.toml +59 -0
  53. ghimera-0.3.0/examples/chimera.toml +54 -0
  54. ghimera-0.3.0/examples/collector-command.toml +13 -0
  55. ghimera-0.3.0/examples/collector.toml +274 -0
  56. ghimera-0.3.0/examples/credential-bindings.json +18 -0
  57. ghimera-0.3.0/examples/documents-native.toml +26 -0
  58. ghimera-0.3.0/examples/documents-standard.toml +110 -0
  59. ghimera-0.3.0/examples/embedding-references.fixture.json +14 -0
  60. ghimera-0.3.0/examples/extraction.toml +34 -0
  61. ghimera-0.3.0/examples/intent-research.toml +83 -0
  62. ghimera-0.3.0/examples/intent-scoring.toml +37 -0
  63. ghimera-0.3.0/examples/journal.toml +10 -0
  64. ghimera-0.3.0/examples/model-service.toml +30 -0
  65. ghimera-0.3.0/examples/pdf-expectations.json +7 -0
  66. ghimera-0.3.0/examples/research-graph.toml +75 -0
  67. ghimera-0.3.0/examples/research-request.json +4 -0
  68. ghimera-0.3.0/examples/scoring.toml +37 -0
  69. ghimera-0.3.0/examples/searxng-html.toml +11 -0
  70. ghimera-0.3.0/examples/searxng.toml +7 -0
  71. ghimera-0.3.0/examples/source-session.toml +8 -0
  72. ghimera-0.3.0/pyproject.toml +54 -0
  73. ghimera-0.3.0/scripts/gate.sh +17 -0
  74. ghimera-0.3.0/scripts/make_pdf_fixture.py +61 -0
  75. ghimera-0.3.0/src/chimera/__init__.py +11 -0
  76. ghimera-0.3.0/src/chimera/__main__.py +5 -0
  77. ghimera-0.3.0/src/ghimera/__init__.py +8 -0
  78. ghimera-0.3.0/src/ghimera/__main__.py +5 -0
  79. ghimera-0.3.0/src/ghimera/browser.py +363 -0
  80. ghimera-0.3.0/src/ghimera/browser_config.py +69 -0
  81. ghimera-0.3.0/src/ghimera/browser_types.py +174 -0
  82. ghimera-0.3.0/src/ghimera/browser_worker.py +348 -0
  83. ghimera-0.3.0/src/ghimera/budget.py +89 -0
  84. ghimera-0.3.0/src/ghimera/collector.py +157 -0
  85. ghimera-0.3.0/src/ghimera/command.py +198 -0
  86. ghimera-0.3.0/src/ghimera/config.py +173 -0
  87. ghimera-0.3.0/src/ghimera/content_dedup.py +205 -0
  88. ghimera-0.3.0/src/ghimera/dedup_config.py +35 -0
  89. ghimera-0.3.0/src/ghimera/dedup_types.py +40 -0
  90. ghimera-0.3.0/src/ghimera/document_acceptance.py +131 -0
  91. ghimera-0.3.0/src/ghimera/document_config.py +108 -0
  92. ghimera-0.3.0/src/ghimera/document_models.py +103 -0
  93. ghimera-0.3.0/src/ghimera/document_order.py +99 -0
  94. ghimera-0.3.0/src/ghimera/document_pipeline.py +179 -0
  95. ghimera-0.3.0/src/ghimera/document_types.py +56 -0
  96. ghimera-0.3.0/src/ghimera/document_worker.py +246 -0
  97. ghimera-0.3.0/src/ghimera/documents.py +137 -0
  98. ghimera-0.3.0/src/ghimera/doubles.py +126 -0
  99. ghimera-0.3.0/src/ghimera/embedding.py +149 -0
  100. ghimera-0.3.0/src/ghimera/embedding_types.py +151 -0
  101. ghimera-0.3.0/src/ghimera/evidence_context.py +175 -0
  102. ghimera-0.3.0/src/ghimera/extraction.py +316 -0
  103. ghimera-0.3.0/src/ghimera/extraction_attempts.py +140 -0
  104. ghimera-0.3.0/src/ghimera/extraction_config.py +81 -0
  105. ghimera-0.3.0/src/ghimera/extraction_types.py +93 -0
  106. ghimera-0.3.0/src/ghimera/fetch.py +461 -0
  107. ghimera-0.3.0/src/ghimera/graph.py +392 -0
  108. ghimera-0.3.0/src/ghimera/graph_types.py +190 -0
  109. ghimera-0.3.0/src/ghimera/html_worker.py +322 -0
  110. ghimera-0.3.0/src/ghimera/http.py +240 -0
  111. ghimera-0.3.0/src/ghimera/journal.py +350 -0
  112. ghimera-0.3.0/src/ghimera/journal_config.py +26 -0
  113. ghimera-0.3.0/src/ghimera/journal_types.py +86 -0
  114. ghimera-0.3.0/src/ghimera/ledger.py +43 -0
  115. ghimera-0.3.0/src/ghimera/locator_health.py +195 -0
  116. ghimera-0.3.0/src/ghimera/locator_types.py +47 -0
  117. ghimera-0.3.0/src/ghimera/loop.py +561 -0
  118. ghimera-0.3.0/src/ghimera/model_citations.py +107 -0
  119. ghimera-0.3.0/src/ghimera/model_client.py +476 -0
  120. ghimera-0.3.0/src/ghimera/model_config.py +131 -0
  121. ghimera-0.3.0/src/ghimera/model_http.py +146 -0
  122. ghimera-0.3.0/src/ghimera/model_types.py +47 -0
  123. ghimera-0.3.0/src/ghimera/models.py +660 -0
  124. ghimera-0.3.0/src/ghimera/passive_worker.py +125 -0
  125. ghimera-0.3.0/src/ghimera/politeness.py +117 -0
  126. ghimera-0.3.0/src/ghimera/ports.py +54 -0
  127. ghimera-0.3.0/src/ghimera/reference_config.py +54 -0
  128. ghimera-0.3.0/src/ghimera/reference_types.py +148 -0
  129. ghimera-0.3.0/src/ghimera/references.py +293 -0
  130. ghimera-0.3.0/src/ghimera/refusals.py +142 -0
  131. ghimera-0.3.0/src/ghimera/research.py +689 -0
  132. ghimera-0.3.0/src/ghimera/research_config.py +62 -0
  133. ghimera-0.3.0/src/ghimera/research_types.py +306 -0
  134. ghimera-0.3.0/src/ghimera/response.py +39 -0
  135. ghimera-0.3.0/src/ghimera/result_archive.py +167 -0
  136. ghimera-0.3.0/src/ghimera/scoring.py +60 -0
  137. ghimera-0.3.0/src/ghimera/scoring_config.py +42 -0
  138. ghimera-0.3.0/src/ghimera/scoring_types.py +75 -0
  139. ghimera-0.3.0/src/ghimera/scoring_validation.py +51 -0
  140. ghimera-0.3.0/src/ghimera/search.py +117 -0
  141. ghimera-0.3.0/src/ghimera/search_config.py +45 -0
  142. ghimera-0.3.0/src/ghimera/search_history.py +33 -0
  143. ghimera-0.3.0/src/ghimera/search_html_types.py +26 -0
  144. ghimera-0.3.0/src/ghimera/searx_html_worker.py +69 -0
  145. ghimera-0.3.0/src/ghimera/searxng.py +180 -0
  146. ghimera-0.3.0/src/ghimera/semantic_scoring.py +363 -0
  147. ghimera-0.3.0/src/ghimera/source_session_types.py +153 -0
  148. ghimera-0.3.0/src/ghimera/source_sessions.py +76 -0
  149. ghimera-0.3.0/src/ghimera/transport.py +410 -0
  150. ghimera-0.3.0/src/ghimera/transport_types.py +105 -0
  151. ghimera-0.3.0/tests/test_browser_redirects.py +293 -0
  152. ghimera-0.3.0/tests/test_browser_render.py +501 -0
  153. ghimera-0.3.0/tests/test_c0.py +198 -0
  154. ghimera-0.3.0/tests/test_cited_by_expansion.py +202 -0
  155. ghimera-0.3.0/tests/test_collector.py +320 -0
  156. ghimera-0.3.0/tests/test_collector_command.py +320 -0
  157. ghimera-0.3.0/tests/test_content_dedup.py +275 -0
  158. ghimera-0.3.0/tests/test_contract_mutations.py +152 -0
  159. ghimera-0.3.0/tests/test_document_acceptance.py +51 -0
  160. ghimera-0.3.0/tests/test_document_extraction.py +278 -0
  161. ghimera-0.3.0/tests/test_document_models.py +178 -0
  162. ghimera-0.3.0/tests/test_document_order.py +74 -0
  163. ghimera-0.3.0/tests/test_embedding_scoring.py +758 -0
  164. ghimera-0.3.0/tests/test_extraction_recovery.py +278 -0
  165. ghimera-0.3.0/tests/test_fetch_route_conformance.py +129 -0
  166. ghimera-0.3.0/tests/test_html_extraction.py +251 -0
  167. ghimera-0.3.0/tests/test_http_fetch.py +425 -0
  168. ghimera-0.3.0/tests/test_intent_research.py +350 -0
  169. ghimera-0.3.0/tests/test_locator_health.py +210 -0
  170. ghimera-0.3.0/tests/test_package_boundary.py +54 -0
  171. ghimera-0.3.0/tests/test_reference_expansion.py +220 -0
  172. ghimera-0.3.0/tests/test_release_metadata.py +52 -0
  173. ghimera-0.3.0/tests/test_research_graph.py +300 -0
  174. ghimera-0.3.0/tests/test_run_journal.py +238 -0
  175. ghimera-0.3.0/tests/test_scorer_conformance.py +70 -0
  176. ghimera-0.3.0/tests/test_search_archive.py +151 -0
  177. ghimera-0.3.0/tests/test_search_conformance.py +181 -0
  178. ghimera-0.3.0/tests/test_search_html.py +258 -0
  179. ghimera-0.3.0/tests/test_served_models.py +525 -0
  180. ghimera-0.3.0/tests/test_source_sessions.py +311 -0
  181. ghimera-0.3.0/tests/test_tor_transport.py +376 -0
  182. ghimera-0.3.0/typing/crawl4ai/__init__.pyi +1 -0
  183. ghimera-0.3.0/typing/crawl4ai/content_filter_strategy_lxml.pyi +2 -0
  184. ghimera-0.3.0/typing/crawl4ai/markdown_generation_strategy.pyi +9 -0
  185. ghimera-0.3.0/uv.lock +4559 -0
@@ -0,0 +1,8 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ .venv/
4
+ .pytest_cache/
5
+ .mypy_cache/
6
+ .ruff_cache/
7
+ dist/
8
+ gate-work/
@@ -0,0 +1,134 @@
1
+ # Changelog
2
+
3
+ ## 0.3.0 — 2026-10-06
4
+
5
+ - Consolidated project naming as `ghimera`: distribution, Python package,
6
+ command and GitHub repository. The primary configuration class is
7
+ `GhimeraConfig`. A narrow legacy `chimera` root facade and module command
8
+ reuse the same implementation; nested imports migrate to `ghimera.*`.
9
+ Existing `chimera.*` data schemas, prompt revisions and saved-result
10
+ identities are unchanged. The old `go-spider` releases remain unchanged.
11
+
12
+ - Concrete live intent research acceptance joined real discovery, laptop
13
+ HTTP/HTML extraction, a pinned semantic encoder, self-hosted model judgment,
14
+ answer/review, incremental graph and sealed archive readback. This English,
15
+ same-model-review diagnostic does not establish representative accuracy or
16
+ close the remaining browser/publisher/document/runtime acceptance gates.
17
+
18
+ - Explicit SearXNG ordinary-HTML search beside unchanged JSON mode. Version-2
19
+ search configuration selects the dialect; the Collector assembles the owning
20
+ adapter. The HTML adapter parses observed simple-theme results with pinned
21
+ Scrapling in the existing bounded passive worker, shares direct/Tor accounting,
22
+ and refuses unknown layouts or access barriers without format fallback.
23
+
24
+ - Completed intent results now retain successful search responses, exact queries,
25
+ parsed hits and transport in `chimera.research-result/2`. Readback binds each
26
+ response to its accounted fetch and refuses missing, substituted or duplicate
27
+ discovery evidence. Citing-source candidates must be actual observed hits.
28
+ Legacy `/1` results remain readable without claiming raw-search retention;
29
+ snippets still cannot serve as answer citations.
30
+
31
+ - Collection grading now explicitly evaluates retained evidence sufficiency,
32
+ not the presence of an answer draft, with a distinct recorded prompt revision.
33
+ A real self-hosted-model follow-up accepted sufficient native documentation
34
+ and rejected irrelevant, absent, and model-memory-only evidence. These bounded
35
+ controls do not establish representative accuracy or independent calibration.
36
+
37
+ - Explicit `citation_format = "template_ids"` for assessment and drafting:
38
+ the model selects supplied context IDs and the client restores exact native
39
+ quotations, offsets and hashes. Unknown/out-of-context IDs refuse without
40
+ similarity repair; full-citation mode remains the compatibility default.
41
+ Document judges' second looks now expand native context to the configured
42
+ character ceiling while retaining the first span. A real self-hosted-model
43
+ trial confirms these paths; its original grader failure, subsequent repair,
44
+ and independent-evaluation gaps remain recorded rather than being described
45
+ as complete acceptance.
46
+
47
+ - An explicit configuration-driven intent command (`python -m chimera`) with
48
+ bounded input reads, separately named environment credential bindings and
49
+ private no-overwrite complete-result archives. Originals, citations, graph,
50
+ ledger and model/extraction provenance survive revalidated checksum readback.
51
+ Partial outcomes and interrupted/unsealed archives never become answered runs.
52
+
53
+ - A configuration-driven `Collector` facade assembles actual search, HTTP,
54
+ HTML/document, optional browser, embedding and completion adapters. It supports
55
+ intent research and seeded collection with fresh per-run state, retains the
56
+ effective search recipe, and offers an explicitly bounded TOML read. Graph,
57
+ journal and source-session settings use their existing owning contracts.
58
+ Controlled-server composition acceptance is not real-model accuracy.
59
+
60
+ - Explicit intent-reference semantic scoring alongside unchanged pinned-shelf
61
+ scoring. The original intent's embedding spends the shared run budget once;
62
+ prepared vectors, call linkage, failure/cancellation and concurrent run
63
+ isolation are auditable through harvest and journal readers.
64
+
65
+ - Configured owner-private per-run JSONL observations, fsync-before-ack storage,
66
+ hash-chain replay checks, completion summaries and read-only run inspection.
67
+ Interrupted runs remain explicitly unsealed; no automatic refetch or resume.
68
+
69
+ - Configurable single generic HTML reparse over retained source bytes, without
70
+ a second fetch or renewed deadline. Success, refusal and cancellation attempts
71
+ retain typed source/configuration-bound provenance and round-trip ledger checks.
72
+
73
+ - Persistent exact-publisher/profile locator health with a configured miss bar,
74
+ generic extraction recovery, restart-safe/concurrent state, read-only doctor
75
+ output and source/config-bound harvest ledger findings. No publisher-redesign
76
+ accuracy claim follows from the regression fixtures.
77
+
78
+ - Explicit offline PDF layout/table/OCR recipe, pinned worker dependencies and
79
+ artifacts, configurable column-aware reading order and reproducible local PDF
80
+ acceptance with source-bound receipts. Representative-corpus and Marker
81
+ acceptance remain open.
82
+
83
+ - Authorized source sessions with caller-supplied cookies/headers, exact origin
84
+ and path scope, protected browser resource support, non-secret audit metadata,
85
+ and separation from discovery/model-control credentials.
86
+ - Configurable document-reference depth and citing-source discovery through the
87
+ injected search provider, using native parent context and the existing scorer.
88
+ - Source-bound reference decisions and native Docling URL/hyperlink locators;
89
+ saved-harvest validation of source hashes, query/provider binding and budgets.
90
+ - Shared reference/discovery host, parent, per-parent candidate and query limits
91
+ across follow-up rounds, with versioned configuration and an example.
92
+
93
+ ## 0.2.0 — 2026-10-06
94
+
95
+ Rebuild of go-spider as a typed, configured standalone library. Distribution
96
+ identity stays `go-spider`; the implementation package is `chimera`.
97
+
98
+ ### Added
99
+
100
+ - Goal collection and intent-driven planning/discovery/coverage/answer-review
101
+ orchestration with exact native-text citations and explicit partial results.
102
+ - Direct/Tor HTTP, native v3 onion support, bounded robots/redirect/retry/cache
103
+ policies and source-network separation from private model-control traffic.
104
+ - Isolated Patchright rendering with parent-owned HTTP resource/redirect handling.
105
+ - Native HTML fit-Markdown/adaptive locator extraction and language detection;
106
+ offline DOCX tables and native PDF text extraction with retained layout.
107
+ - Self-hosted completion and embedding clients with declared model binding,
108
+ bounded native evidence, token/encoding observations and recorded failure.
109
+ - Reference-vector relevance scoring, semantic/keyword link ranking, canonical
110
+ and near-duplicate grouping with independently retained source occurrences.
111
+ - Configurable research graph, immutable typed harvest/ledger/receipt contracts,
112
+ conformance tests, deliberate mutation witnesses and full package gate.
113
+ - Public README, package metadata, MIT licence text and migration guidance.
114
+
115
+ ### Breaking changes
116
+
117
+ - Requires Python 3.11 or newer; v0.1.0 required 3.10.
118
+ - `chimera` replaces the prototype `spider_core` imports. No compatibility shim
119
+ or legacy `spider` command is included.
120
+ - No cloud LLM client, automatic model-server launch, VPN manager or silent
121
+ external-model/direct-network fallback.
122
+ - Operational policy is explicit validated configuration and injected ports.
123
+
124
+ ### Not claimed complete
125
+
126
+ Public publisher/locator and real-model quality acceptance; full PDF/OCR/Marker;
127
+ Camoufox/nodriver; calibrated judge policy; one-hop references/cited-by expansion;
128
+ production runtime/egress acceptance. Publication
129
+ is not deployment or proof of completion of those planned features.
130
+
131
+ ## 0.1.0
132
+
133
+ Original prototype release with `spider_core` and the `spider` CLI. Existing
134
+ users may pin that version while migrating to the explicit v0.2.0 library API.
ghimera-0.3.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2025–2026 Josh Gompert
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
ghimera-0.3.0/PKG-INFO ADDED
@@ -0,0 +1,314 @@
1
+ Metadata-Version: 2.4
2
+ Name: ghimera
3
+ Version: 0.3.0
4
+ Summary: Intent-driven web research with bounded collection and source-bound evidence
5
+ Project-URL: Homepage, https://github.com/ginkorea/ghimera
6
+ Project-URL: Repository, https://github.com/ginkorea/ghimera
7
+ Project-URL: Issues, https://github.com/ginkorea/ghimera/issues
8
+ Project-URL: Changelog, https://github.com/ginkorea/ghimera/blob/main/CHANGELOG.md
9
+ Author: Josh Gompert
10
+ License-Expression: MIT
11
+ License-File: LICENSE
12
+ Requires-Python: >=3.11
13
+ Requires-Dist: curl-cffi==0.16.3
14
+ Requires-Dist: protego==0.7.0
15
+ Requires-Dist: pydantic==2.13.5
16
+ Provides-Extra: browser
17
+ Requires-Dist: patchright==1.63.0; extra == 'browser'
18
+ Provides-Extra: documents
19
+ Requires-Dist: docling-core==2.99.0; extra == 'documents'
20
+ Requires-Dist: docling-slim[convert-core,format-docx,format-pdf]==2.134.0; extra == 'documents'
21
+ Requires-Dist: lingua-language-detector==2.1.1; extra == 'documents'
22
+ Provides-Extra: html
23
+ Requires-Dist: crawl4ai==0.9.4; extra == 'html'
24
+ Requires-Dist: lingua-language-detector==2.1.1; extra == 'html'
25
+ Requires-Dist: scrapling==0.4.2; extra == 'html'
26
+ Description-Content-Type: text/markdown
27
+
28
+ # ghimera
29
+
30
+ Intent-driven web research: discover sources, collect native-language documents,
31
+ follow evidence gaps, and return a source-cited answer—or an explicit partial
32
+ result when the evidence or budget is insufficient.
33
+
34
+ **v0.3.0 consolidates the repository, distribution and import as `ghimera`.**
35
+ It succeeds the `go-spider` distribution and `chimera` implementation. It is not
36
+ backward-compatible with v0.1.0's `spider_core` API or `spider` CLI. Python
37
+ **3.11+** is required. Some planned browser/document adapters and public-corpus
38
+ acceptance are still in progress; see the limitations below.
39
+
40
+ ghimera is an independent library. Supply your own search provider,
41
+ self-hosted model services, extraction policies and graph profile.
42
+
43
+ ## What is implemented
44
+
45
+ - **Goal and intent loops.** Collect from configured seeds with `GoalLoop`, or
46
+ use `ResearchLoop` to plan questions, discover sources through an injected
47
+ search provider, assess gaps, and draft/review an evidence-cited answer.
48
+ - **Local models first.** Configured, already-served self-hosted models supply
49
+ planning, judging, answer generation, review and embeddings. Compatible HTTP
50
+ interfaces are supported; no external LLM fallback, model weights or model
51
+ server are included.
52
+ - **Bounded direct and Tor HTTP.** Native v3 onion collection and open-web
53
+ requests through Tor share the same route policy. Public-network validation,
54
+ pinned DNS for direct requests, per-hop redirect checks, robots policy,
55
+ concurrency/rate limits and retries are accounted for before returning data.
56
+ A failed Tor route never silently falls back to direct access.
57
+ - **Isolated JavaScript rendering.** Patchright runs in a network-isolated
58
+ Linux worker; the parent fetch boundary handles its permitted HTTP resources,
59
+ redirects and accounting. Browser binaries are explicitly configured and
60
+ verified, not downloaded on import. Camoufox/nodriver adapters remain planned.
61
+ - **Native extraction.** Configured HTML fit-Markdown, adaptive locator
62
+ profiles, language detection, DOCX tables and native PDF text preserve raw
63
+ bytes beside extracted native-language text. Full PDF/OCR and Marker
64
+ acceptance remain open.
65
+ - **Relevance and deduplication.** An injected self-hosted encoder scores
66
+ native text/windows and observed links against pinned reference vectors.
67
+ Keyword/semantic ranking, encoding budgets, canonical URL handling, SHA-256
68
+ and configured near-duplicate grouping retain source-qualified evidence.
69
+ Similarity is not a calibrated probability or a substitute for a verdict.
70
+ - **Graphs and audit records.** A configurable research graph starts with the
71
+ intent. Typed harvests, receipts and ledger rows retain configuration,
72
+ transport, model-call spend, omissions, verdicts, refusals and source hashes.
73
+ Operational discovery traces are distinct from evidence-supported claims.
74
+
75
+ Every candidate reaching acceptance receives an accept/reject/hold verdict. A hold gets a
76
+ second model pass. An intent is marked answered only after the coverage,
77
+ citations, review and configured confidence checks pass; budget exhaustion is
78
+ not silently presented as success.
79
+
80
+ ### Additions since go-spider 0.2.0
81
+
82
+ The current working branch also has explicit authorized source sessions,
83
+ configurable references/citing-source discovery, persistent publisher-locator
84
+ drift detection with generic recovery and a doctor, and offline model-based PDF
85
+ layout, tables, OCR and column-aware reading order, durable run journals, and
86
+ intent-based semantic scoring without a prebuilt reference-vector file, and a
87
+ configuration-driven `Collector` facade using actual adapters. These were not included
88
+ in the old `go-spider==0.2.0` wheel and are included in `ghimera==0.3.0`.
89
+ See [source sessions](docs/SOURCE_SESSIONS.md),
90
+ [references](docs/C3_REFERENCES.md), [locator health](docs/C2_HTML.md), and
91
+ [PDF configuration/acceptance](docs/C2_DOCUMENTS.md),
92
+ [run journals](docs/RUN_JOURNAL.md) and
93
+ [intent scoring](docs/C3_EMBEDDING_SCORING.md#intent-references-unreleased-source).
94
+ For the assembled intent-only API and full non-active template, see
95
+ [configured collector](docs/COLLECTOR.md) and `examples/collector.toml`.
96
+ The package also supports explicitly configured ordinary HTML search
97
+ alongside JSON, and complete research archives retain successful discovery
98
+ responses with query/fetch bindings. See [HTML search](docs/C3_SEARCH_HTML.md)
99
+ and [search evidence](docs/C3_SEARCH_EVIDENCE.md). Neither mode solves access
100
+ challenges, and a refusal is not a successful research result.
101
+ Representative-corpus accuracy and Marker acceptance remain open; passing a
102
+ controlled document check is not a universal quality claim.
103
+
104
+ ## Installation
105
+
106
+ Use a dedicated virtual environment:
107
+
108
+ ```bash
109
+ python3.11 -m venv .venv
110
+ . .venv/bin/activate
111
+ python -m pip install 'ghimera==0.3.0'
112
+ ```
113
+
114
+ Install the adapters you intend to configure:
115
+
116
+ ```bash
117
+ python -m pip install 'ghimera[html,documents,browser]==0.3.0'
118
+ ```
119
+
120
+ The base package contains the typed core, HTTP/Tor transport, research/search
121
+ and self-hosted model/embedding clients. Extras add pinned HTML, document and
122
+ Patchright dependencies. They do **not** install an inference server, browser
123
+ binary, Tor daemon or PDF/OCR model artifacts. The isolated browser adapter
124
+ requires Linux, a compatible explicitly supplied Chromium binary and Bubblewrap;
125
+ other operating systems have not been accepted for that adapter.
126
+
127
+ ## Configuration and API
128
+
129
+ Operational choices are typed, versioned configuration—not Python constants:
130
+ scope, budgets, endpoints, model identities/revisions, thresholds, private
131
+ worker directories, browser provenance and direct/Tor policy. Parse once with
132
+ `GhimeraConfig.from_toml(Path(...))`; inject the matching collaborators.
133
+
134
+ The [examples](https://github.com/ginkorea/ghimera/tree/v0.3.0/examples) are non-active templates. Replace invalid endpoints,
135
+ contact information, private paths and model identifiers; reference-vector
136
+ fixtures are **not** production relevance data. Adapter blocks belong in the
137
+ main configuration under their named keys, not as unrelated root settings.
138
+ Supply any model credential separately in memory, only to its authorized exact
139
+ endpoint; configuration is not credential or destination approval.
140
+
141
+ Download the starter configuration, or copy it from the repository:
142
+
143
+ ```bash
144
+ curl --fail --proto '=https' --tlsv1.2 \
145
+ https://raw.githubusercontent.com/ginkorea/ghimera/v0.3.0/examples/chimera.toml \
146
+ --output chimera.toml
147
+ ```
148
+
149
+ An entirely offline smoke example, using explicitly named test doubles:
150
+
151
+ ```python
152
+ import asyncio
153
+ from pathlib import Path
154
+
155
+ from ghimera import GhimeraConfig, Goal, GoalLoop, Harvest, Scope
156
+ from ghimera.doubles import FakeExtractor, FakeJudge, FakeRoute, KeywordScorer
157
+ from ghimera.fetch import FetchLadder
158
+
159
+
160
+ async def main() -> None:
161
+ config = GhimeraConfig.from_toml(Path("chimera.toml"))
162
+ collector = GoalLoop(
163
+ config=config,
164
+ fetcher=FetchLadder((FakeRoute(),)),
165
+ extractor=FakeExtractor(),
166
+ scorer=KeywordScorer(),
167
+ judge=FakeJudge(),
168
+ )
169
+ result = await collector.run(
170
+ Goal(text="ports", seeds=("https://example.org/start",)),
171
+ Scope(allowed_hosts=("example.org",), max_depth=1, content_types=("text/html",)),
172
+ )
173
+ # Reader revalidates retained sources, provenance and receipt accounting.
174
+ restored = Harvest.model_validate_json(result.model_dump_json())
175
+ print(restored.receipt.stop_reason)
176
+
177
+
178
+ asyncio.run(main())
179
+ ```
180
+
181
+ This example makes no network or real model calls and proves no research
182
+ accuracy. For actual collection bind `CurlRoute`, configured extraction and
183
+ scoring adapters, and the self-hosted judge through their ports. For intent-only
184
+ research, inject those into `ResearchLoop` alongside `GroundedSearch`,
185
+ `IntentPlanner`, `ResearchAnalyst` and `AnswerReviewer`, then call
186
+ `run(ResearchRequest(intent="your research question"))`. `SearxSearch` is the
187
+ implemented search adapter.
188
+
189
+ ### Configured intent API
190
+
191
+ The configured collector assembles real adapters, so applications need not manually
192
+ wire every port. Unlike the offline smoke above, this needs your configured
193
+ services and the adapted `examples/collector.toml` template:
194
+
195
+ ```python
196
+ import asyncio
197
+ from pathlib import Path
198
+
199
+ from ghimera import Collector
200
+
201
+
202
+ async def main() -> None:
203
+ collector = Collector.from_toml(Path("collector.toml"), max_config_bytes=100_000)
204
+ result = await collector.run("Your research question")
205
+ print(result.status)
206
+
207
+
208
+ asyncio.run(main())
209
+ ```
210
+
211
+ Use the [collector guide](docs/COLLECTOR.md) to configure actual private models,
212
+ source routing, extraction, optional graph/journal and separately supplied
213
+ credentials. This API is included in `ghimera==0.3.0`, not the old `go-spider==0.2.0` wheel. The
214
+ lower-level APIs remain supported for custom providers and composition.
215
+
216
+ The package also supplies a configured intent command that retains the
217
+ full result, original documents and citations in a private, checksum-sealed archive:
218
+
219
+ ```bash
220
+ python -m ghimera --job /absolute/path/collector-command.toml --max-job-bytes 100000
221
+ ```
222
+
223
+ See the [command guide](docs/COLLECTOR_COMMAND.md) and
224
+ `examples/collector-command.toml`. Existing output identities are never overwritten;
225
+ partial results remain partial. This command is not in the published 0.2.0 wheel.
226
+
227
+ The following guides cover the existing lower-level wiring:
228
+
229
+ | Area | Guide |
230
+ |---|---|
231
+ | Intent, discovery, coverage and answer review | [Intent research](https://github.com/ginkorea/ghimera/blob/v0.3.0/docs/C3_RESEARCH.md) |
232
+ | Model roles, credentials and native evidence context | [Self-hosted models](https://github.com/ginkorea/ghimera/blob/v0.3.0/docs/C3_MODELS.md) |
233
+ | Reference vectors, encoding and frontier ranking | [Embedding scoring](https://github.com/ginkorea/ghimera/blob/v0.3.0/docs/C3_EMBEDDING_SCORING.md) |
234
+ | Open web and native onion routing | [Tor policy](https://github.com/ginkorea/ghimera/blob/v0.3.0/docs/TOR.md) |
235
+ | Browser isolation, resources and redirects | [Browser rendering](https://github.com/ginkorea/ghimera/blob/v0.3.0/docs/C1_BROWSER.md) |
236
+ | HTML extraction and adaptive locators | [HTML extraction](https://github.com/ginkorea/ghimera/blob/v0.3.0/docs/C2_HTML.md) |
237
+ | DOCX/native PDF and offline artifacts | [Document extraction](https://github.com/ginkorea/ghimera/blob/v0.3.0/docs/C2_DOCUMENTS.md) |
238
+ | Canonical and near-duplicate source evidence | [Deduplication](https://github.com/ginkorea/ghimera/blob/v0.3.0/docs/C2_DEDUP.md) |
239
+ | Configurable research graphs | [Research graph](https://github.com/ginkorea/ghimera/blob/v0.3.0/docs/RESEARCH_GRAPH.md) |
240
+ | Architecture, contracts and completion tracker | [Specification](https://github.com/ginkorea/ghimera/blob/v0.3.0/docs/C0.md) |
241
+
242
+ ## Scope and limitations
243
+
244
+ Robots are honored by default. An override requires a recorded, reasoned,
245
+ exact-host configuration decision; it does not disable CAPTCHA, login, paywall
246
+ or challenge refusal. There is no challenge solver or authenticated-site bypass.
247
+ Tor routing is a transport capability, not a guarantee of anonymity or authority
248
+ to access a source.
249
+
250
+ Collection is not limited to anonymous access. Supply your own authorized
251
+ cookies or headers through explicitly configured source sessions, with exact
252
+ origin/path scope and no credential values in receipts. Browser resources use
253
+ the same parent-owned session boundary. See [authorized sessions](docs/SOURCE_SESSIONS.md).
254
+
255
+ Source/search requests run on the host executing the crawler. Model control is
256
+ a separate private-service boundary. Your application owns authorization,
257
+ deployment and storage; no external scheduler or registry is required to import
258
+ or use the library.
259
+
260
+ Configurable source expansion follows observed document URLs and discovers
261
+ candidate citing sources through your search provider. Depth, host policy and
262
+ budgets live in `[references]`, not in Python. Source hashes and native locators
263
+ are retained; a citing-source search hit is not proof that a citation exists.
264
+ See [reference expansion](docs/C3_REFERENCES.md).
265
+
266
+ For durable observations, configure `[journal]` and pass a unique `run_id`.
267
+ The collector persists JSONL events before acknowledgment and seals a completion
268
+ summary only after receipt reconciliation. Interrupted prefixes remain inspectable
269
+ without silently refetching sources. See [run journals](docs/RUN_JOURNAL.md).
270
+
271
+ Still required for the complete planned spider: the remaining browser adapters,
272
+ representative publisher/locator acceptance, full PDF/OCR and Marker validation,
273
+ real reference/cited-by adequacy, real served-model quality/admission and
274
+ calibrated decision policy and live runtime/egress acceptance.
275
+ The repository's detailed tracker retains those requirements; this release
276
+ does not erase them or describe fixture results as real-world model accuracy.
277
+
278
+ ## Development
279
+
280
+ ```bash
281
+ git clone https://github.com/ginkorea/ghimera.git
282
+ cd ghimera
283
+ uv sync --locked --extra html --extra documents --extra browser --python 3.11
284
+ # Explicit browser/isolation paths are required for the complete gate.
285
+ export CHIMERA_TEST_BROWSER=/absolute/path/to/compatible/chrome
286
+ export CHIMERA_TEST_ISOLATOR=/absolute/path/to/bwrap
287
+ bash scripts/gate.sh
288
+ uv build --no-sources
289
+ ```
290
+
291
+ The gate checks the actual interpreter/import path, lockfile, Ruff, strict mypy
292
+ and the entire test suite. Browser tests refuse absent acceptance prerequisites
293
+ rather than pretending they ran. Evidence records distinguish protocol fixtures,
294
+ installed-artifact checks, public-corpus acceptance and production activation.
295
+
296
+ ## Migration from go-spider
297
+
298
+ Install `ghimera==0.3.0` explicitly; this is a new distribution, not an in-place
299
+ rename of old PyPI releases. Use `from ghimera import Collector, GhimeraConfig`,
300
+ `ghimera.*` for submodules, and `ghimera` or `python -m ghimera` for the command.
301
+ The legacy root `from chimera import Collector, ChimeraConfig` and
302
+ `python -m chimera` forward to the same implementation. Old nested
303
+ `chimera.*` imports must migrate; there is no second implementation or import hook.
304
+ Existing versioned `chimera.*` configuration, graph, harvest, journal and result
305
+ schemas are retained so existing saved evidence does not change identity.
306
+
307
+ The v0.1.0 `spider_core` API and old `spider` command are not supplied; keep
308
+ `go-spider==0.1.0` while migrating those applications. The old cloud-client,
309
+ VPN-manager and implicit fallback design is not retained. Prototype source
310
+ remains in Git history.
311
+
312
+ See [CHANGELOG.md](https://github.com/ginkorea/ghimera/blob/v0.3.0/CHANGELOG.md) for release changes. Josh Gompert maintains
313
+ the project at [ginkorea/ghimera](https://github.com/ginkorea/ghimera).
314
+ Licensed under [MIT](https://github.com/ginkorea/ghimera/blob/v0.3.0/LICENSE), matching the existing PyPI licence declaration.