ghimera 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ghimera-0.3.0/.gitignore +8 -0
- ghimera-0.3.0/CHANGELOG.md +134 -0
- ghimera-0.3.0/LICENSE +21 -0
- ghimera-0.3.0/PKG-INFO +314 -0
- ghimera-0.3.0/README.md +287 -0
- ghimera-0.3.0/docs/C0.md +166 -0
- ghimera-0.3.0/docs/C0_EVIDENCE.md +82 -0
- ghimera-0.3.0/docs/C1.md +94 -0
- ghimera-0.3.0/docs/C1_BROWSER.md +144 -0
- ghimera-0.3.0/docs/C1_BROWSER_EVIDENCE.md +98 -0
- ghimera-0.3.0/docs/C1_BROWSER_REDIRECT_EVIDENCE.md +124 -0
- ghimera-0.3.0/docs/C1_HTTP_EVIDENCE.md +71 -0
- ghimera-0.3.0/docs/C1_TOR_EVIDENCE.md +75 -0
- ghimera-0.3.0/docs/C2_DEDUP.md +51 -0
- ghimera-0.3.0/docs/C2_DEDUP_EVIDENCE.md +57 -0
- ghimera-0.3.0/docs/C2_DOCUMENTS.md +160 -0
- ghimera-0.3.0/docs/C2_DOCUMENTS_EVIDENCE.md +66 -0
- ghimera-0.3.0/docs/C2_EXTRACTION_RECOVERY_EVIDENCE.md +60 -0
- ghimera-0.3.0/docs/C2_HTML.md +156 -0
- ghimera-0.3.0/docs/C2_HTML_EVIDENCE.md +94 -0
- ghimera-0.3.0/docs/C2_LOCATOR_DRIFT_EVIDENCE.md +64 -0
- ghimera-0.3.0/docs/C2_PDF_MODELS_EVIDENCE.md +99 -0
- ghimera-0.3.0/docs/C3_EMBEDDING_SCORING.md +174 -0
- ghimera-0.3.0/docs/C3_EMBEDDING_SCORING_EVIDENCE.md +129 -0
- ghimera-0.3.0/docs/C3_GRAPH_EVIDENCE.md +66 -0
- ghimera-0.3.0/docs/C3_INTENT_SCORING_EVIDENCE.md +84 -0
- ghimera-0.3.0/docs/C3_LIVE_DISCOVERY.md +81 -0
- ghimera-0.3.0/docs/C3_LIVE_RESEARCH.md +83 -0
- ghimera-0.3.0/docs/C3_MODELS.md +140 -0
- ghimera-0.3.0/docs/C3_MODELS_EVIDENCE.md +48 -0
- ghimera-0.3.0/docs/C3_REAL_MODEL_EVIDENCE.md +197 -0
- ghimera-0.3.0/docs/C3_REFERENCES.md +70 -0
- ghimera-0.3.0/docs/C3_REFERENCES_EVIDENCE.md +60 -0
- ghimera-0.3.0/docs/C3_RESEARCH.md +138 -0
- ghimera-0.3.0/docs/C3_RESEARCH_EVIDENCE.md +70 -0
- ghimera-0.3.0/docs/C3_SEARCH_EVIDENCE.md +83 -0
- ghimera-0.3.0/docs/C3_SEARCH_HTML.md +94 -0
- ghimera-0.3.0/docs/COLLECTOR.md +144 -0
- ghimera-0.3.0/docs/COLLECTOR_COMMAND.md +103 -0
- ghimera-0.3.0/docs/COLLECTOR_COMMAND_EVIDENCE.md +101 -0
- ghimera-0.3.0/docs/COLLECTOR_EVIDENCE.md +116 -0
- ghimera-0.3.0/docs/ORGANIZATION_RESEARCH.md +102 -0
- ghimera-0.3.0/docs/RELEASE_030.md +55 -0
- ghimera-0.3.0/docs/RELEASE_0_2_0.md +66 -0
- ghimera-0.3.0/docs/RESEARCH_GRAPH.md +84 -0
- ghimera-0.3.0/docs/RUN_JOURNAL.md +79 -0
- ghimera-0.3.0/docs/RUN_JOURNAL_EVIDENCE.md +55 -0
- ghimera-0.3.0/docs/SOURCE_SESSIONS.md +84 -0
- ghimera-0.3.0/docs/SOURCE_SESSIONS_EVIDENCE.md +56 -0
- ghimera-0.3.0/docs/TOR.md +85 -0
- ghimera-0.3.0/examples/browser.toml +25 -0
- ghimera-0.3.0/examples/chimera-tor.toml +59 -0
- ghimera-0.3.0/examples/chimera.toml +54 -0
- ghimera-0.3.0/examples/collector-command.toml +13 -0
- ghimera-0.3.0/examples/collector.toml +274 -0
- ghimera-0.3.0/examples/credential-bindings.json +18 -0
- ghimera-0.3.0/examples/documents-native.toml +26 -0
- ghimera-0.3.0/examples/documents-standard.toml +110 -0
- ghimera-0.3.0/examples/embedding-references.fixture.json +14 -0
- ghimera-0.3.0/examples/extraction.toml +34 -0
- ghimera-0.3.0/examples/intent-research.toml +83 -0
- ghimera-0.3.0/examples/intent-scoring.toml +37 -0
- ghimera-0.3.0/examples/journal.toml +10 -0
- ghimera-0.3.0/examples/model-service.toml +30 -0
- ghimera-0.3.0/examples/pdf-expectations.json +7 -0
- ghimera-0.3.0/examples/research-graph.toml +75 -0
- ghimera-0.3.0/examples/research-request.json +4 -0
- ghimera-0.3.0/examples/scoring.toml +37 -0
- ghimera-0.3.0/examples/searxng-html.toml +11 -0
- ghimera-0.3.0/examples/searxng.toml +7 -0
- ghimera-0.3.0/examples/source-session.toml +8 -0
- ghimera-0.3.0/pyproject.toml +54 -0
- ghimera-0.3.0/scripts/gate.sh +17 -0
- ghimera-0.3.0/scripts/make_pdf_fixture.py +61 -0
- ghimera-0.3.0/src/chimera/__init__.py +11 -0
- ghimera-0.3.0/src/chimera/__main__.py +5 -0
- ghimera-0.3.0/src/ghimera/__init__.py +8 -0
- ghimera-0.3.0/src/ghimera/__main__.py +5 -0
- ghimera-0.3.0/src/ghimera/browser.py +363 -0
- ghimera-0.3.0/src/ghimera/browser_config.py +69 -0
- ghimera-0.3.0/src/ghimera/browser_types.py +174 -0
- ghimera-0.3.0/src/ghimera/browser_worker.py +348 -0
- ghimera-0.3.0/src/ghimera/budget.py +89 -0
- ghimera-0.3.0/src/ghimera/collector.py +157 -0
- ghimera-0.3.0/src/ghimera/command.py +198 -0
- ghimera-0.3.0/src/ghimera/config.py +173 -0
- ghimera-0.3.0/src/ghimera/content_dedup.py +205 -0
- ghimera-0.3.0/src/ghimera/dedup_config.py +35 -0
- ghimera-0.3.0/src/ghimera/dedup_types.py +40 -0
- ghimera-0.3.0/src/ghimera/document_acceptance.py +131 -0
- ghimera-0.3.0/src/ghimera/document_config.py +108 -0
- ghimera-0.3.0/src/ghimera/document_models.py +103 -0
- ghimera-0.3.0/src/ghimera/document_order.py +99 -0
- ghimera-0.3.0/src/ghimera/document_pipeline.py +179 -0
- ghimera-0.3.0/src/ghimera/document_types.py +56 -0
- ghimera-0.3.0/src/ghimera/document_worker.py +246 -0
- ghimera-0.3.0/src/ghimera/documents.py +137 -0
- ghimera-0.3.0/src/ghimera/doubles.py +126 -0
- ghimera-0.3.0/src/ghimera/embedding.py +149 -0
- ghimera-0.3.0/src/ghimera/embedding_types.py +151 -0
- ghimera-0.3.0/src/ghimera/evidence_context.py +175 -0
- ghimera-0.3.0/src/ghimera/extraction.py +316 -0
- ghimera-0.3.0/src/ghimera/extraction_attempts.py +140 -0
- ghimera-0.3.0/src/ghimera/extraction_config.py +81 -0
- ghimera-0.3.0/src/ghimera/extraction_types.py +93 -0
- ghimera-0.3.0/src/ghimera/fetch.py +461 -0
- ghimera-0.3.0/src/ghimera/graph.py +392 -0
- ghimera-0.3.0/src/ghimera/graph_types.py +190 -0
- ghimera-0.3.0/src/ghimera/html_worker.py +322 -0
- ghimera-0.3.0/src/ghimera/http.py +240 -0
- ghimera-0.3.0/src/ghimera/journal.py +350 -0
- ghimera-0.3.0/src/ghimera/journal_config.py +26 -0
- ghimera-0.3.0/src/ghimera/journal_types.py +86 -0
- ghimera-0.3.0/src/ghimera/ledger.py +43 -0
- ghimera-0.3.0/src/ghimera/locator_health.py +195 -0
- ghimera-0.3.0/src/ghimera/locator_types.py +47 -0
- ghimera-0.3.0/src/ghimera/loop.py +561 -0
- ghimera-0.3.0/src/ghimera/model_citations.py +107 -0
- ghimera-0.3.0/src/ghimera/model_client.py +476 -0
- ghimera-0.3.0/src/ghimera/model_config.py +131 -0
- ghimera-0.3.0/src/ghimera/model_http.py +146 -0
- ghimera-0.3.0/src/ghimera/model_types.py +47 -0
- ghimera-0.3.0/src/ghimera/models.py +660 -0
- ghimera-0.3.0/src/ghimera/passive_worker.py +125 -0
- ghimera-0.3.0/src/ghimera/politeness.py +117 -0
- ghimera-0.3.0/src/ghimera/ports.py +54 -0
- ghimera-0.3.0/src/ghimera/reference_config.py +54 -0
- ghimera-0.3.0/src/ghimera/reference_types.py +148 -0
- ghimera-0.3.0/src/ghimera/references.py +293 -0
- ghimera-0.3.0/src/ghimera/refusals.py +142 -0
- ghimera-0.3.0/src/ghimera/research.py +689 -0
- ghimera-0.3.0/src/ghimera/research_config.py +62 -0
- ghimera-0.3.0/src/ghimera/research_types.py +306 -0
- ghimera-0.3.0/src/ghimera/response.py +39 -0
- ghimera-0.3.0/src/ghimera/result_archive.py +167 -0
- ghimera-0.3.0/src/ghimera/scoring.py +60 -0
- ghimera-0.3.0/src/ghimera/scoring_config.py +42 -0
- ghimera-0.3.0/src/ghimera/scoring_types.py +75 -0
- ghimera-0.3.0/src/ghimera/scoring_validation.py +51 -0
- ghimera-0.3.0/src/ghimera/search.py +117 -0
- ghimera-0.3.0/src/ghimera/search_config.py +45 -0
- ghimera-0.3.0/src/ghimera/search_history.py +33 -0
- ghimera-0.3.0/src/ghimera/search_html_types.py +26 -0
- ghimera-0.3.0/src/ghimera/searx_html_worker.py +69 -0
- ghimera-0.3.0/src/ghimera/searxng.py +180 -0
- ghimera-0.3.0/src/ghimera/semantic_scoring.py +363 -0
- ghimera-0.3.0/src/ghimera/source_session_types.py +153 -0
- ghimera-0.3.0/src/ghimera/source_sessions.py +76 -0
- ghimera-0.3.0/src/ghimera/transport.py +410 -0
- ghimera-0.3.0/src/ghimera/transport_types.py +105 -0
- ghimera-0.3.0/tests/test_browser_redirects.py +293 -0
- ghimera-0.3.0/tests/test_browser_render.py +501 -0
- ghimera-0.3.0/tests/test_c0.py +198 -0
- ghimera-0.3.0/tests/test_cited_by_expansion.py +202 -0
- ghimera-0.3.0/tests/test_collector.py +320 -0
- ghimera-0.3.0/tests/test_collector_command.py +320 -0
- ghimera-0.3.0/tests/test_content_dedup.py +275 -0
- ghimera-0.3.0/tests/test_contract_mutations.py +152 -0
- ghimera-0.3.0/tests/test_document_acceptance.py +51 -0
- ghimera-0.3.0/tests/test_document_extraction.py +278 -0
- ghimera-0.3.0/tests/test_document_models.py +178 -0
- ghimera-0.3.0/tests/test_document_order.py +74 -0
- ghimera-0.3.0/tests/test_embedding_scoring.py +758 -0
- ghimera-0.3.0/tests/test_extraction_recovery.py +278 -0
- ghimera-0.3.0/tests/test_fetch_route_conformance.py +129 -0
- ghimera-0.3.0/tests/test_html_extraction.py +251 -0
- ghimera-0.3.0/tests/test_http_fetch.py +425 -0
- ghimera-0.3.0/tests/test_intent_research.py +350 -0
- ghimera-0.3.0/tests/test_locator_health.py +210 -0
- ghimera-0.3.0/tests/test_package_boundary.py +54 -0
- ghimera-0.3.0/tests/test_reference_expansion.py +220 -0
- ghimera-0.3.0/tests/test_release_metadata.py +52 -0
- ghimera-0.3.0/tests/test_research_graph.py +300 -0
- ghimera-0.3.0/tests/test_run_journal.py +238 -0
- ghimera-0.3.0/tests/test_scorer_conformance.py +70 -0
- ghimera-0.3.0/tests/test_search_archive.py +151 -0
- ghimera-0.3.0/tests/test_search_conformance.py +181 -0
- ghimera-0.3.0/tests/test_search_html.py +258 -0
- ghimera-0.3.0/tests/test_served_models.py +525 -0
- ghimera-0.3.0/tests/test_source_sessions.py +311 -0
- ghimera-0.3.0/tests/test_tor_transport.py +376 -0
- ghimera-0.3.0/typing/crawl4ai/__init__.pyi +1 -0
- ghimera-0.3.0/typing/crawl4ai/content_filter_strategy_lxml.pyi +2 -0
- ghimera-0.3.0/typing/crawl4ai/markdown_generation_strategy.pyi +9 -0
- ghimera-0.3.0/uv.lock +4559 -0
ghimera-0.3.0/.gitignore
ADDED
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
## 0.3.0 — 2026-10-06
|
|
4
|
+
|
|
5
|
+
- Consolidated project naming as `ghimera`: distribution, Python package,
|
|
6
|
+
command and GitHub repository. The primary configuration class is
|
|
7
|
+
`GhimeraConfig`. A narrow legacy `chimera` root facade and module command
|
|
8
|
+
reuse the same implementation; nested imports migrate to `ghimera.*`.
|
|
9
|
+
Existing `chimera.*` data schemas, prompt revisions and saved-result
|
|
10
|
+
identities are unchanged. The old `go-spider` releases remain unchanged.
|
|
11
|
+
|
|
12
|
+
- Concrete live intent research acceptance joined real discovery, laptop
|
|
13
|
+
HTTP/HTML extraction, a pinned semantic encoder, self-hosted model judgment,
|
|
14
|
+
answer/review, incremental graph and sealed archive readback. This English,
|
|
15
|
+
same-model-review diagnostic does not establish representative accuracy or
|
|
16
|
+
close the remaining browser/publisher/document/runtime acceptance gates.
|
|
17
|
+
|
|
18
|
+
- Explicit SearXNG ordinary-HTML search beside unchanged JSON mode. Version-2
|
|
19
|
+
search configuration selects the dialect; the Collector assembles the owning
|
|
20
|
+
adapter. The HTML adapter parses observed simple-theme results with pinned
|
|
21
|
+
Scrapling in the existing bounded passive worker, shares direct/Tor accounting,
|
|
22
|
+
and refuses unknown layouts or access barriers without format fallback.
|
|
23
|
+
|
|
24
|
+
- Completed intent results now retain successful search responses, exact queries,
|
|
25
|
+
parsed hits and transport in `chimera.research-result/2`. Readback binds each
|
|
26
|
+
response to its accounted fetch and refuses missing, substituted or duplicate
|
|
27
|
+
discovery evidence. Citing-source candidates must be actual observed hits.
|
|
28
|
+
Legacy `/1` results remain readable without claiming raw-search retention;
|
|
29
|
+
snippets still cannot serve as answer citations.
|
|
30
|
+
|
|
31
|
+
- Collection grading now explicitly evaluates retained evidence sufficiency,
|
|
32
|
+
not the presence of an answer draft, with a distinct recorded prompt revision.
|
|
33
|
+
A real self-hosted-model follow-up accepted sufficient native documentation
|
|
34
|
+
and rejected irrelevant, absent, and model-memory-only evidence. These bounded
|
|
35
|
+
controls do not establish representative accuracy or independent calibration.
|
|
36
|
+
|
|
37
|
+
- Explicit `citation_format = "template_ids"` for assessment and drafting:
|
|
38
|
+
the model selects supplied context IDs and the client restores exact native
|
|
39
|
+
quotations, offsets and hashes. Unknown/out-of-context IDs refuse without
|
|
40
|
+
similarity repair; full-citation mode remains the compatibility default.
|
|
41
|
+
Document judges' second looks now expand native context to the configured
|
|
42
|
+
character ceiling while retaining the first span. A real self-hosted-model
|
|
43
|
+
trial confirms these paths; its original grader failure, subsequent repair,
|
|
44
|
+
and independent-evaluation gaps remain recorded rather than being described
|
|
45
|
+
as complete acceptance.
|
|
46
|
+
|
|
47
|
+
- An explicit configuration-driven intent command (`python -m chimera`) with
|
|
48
|
+
bounded input reads, separately named environment credential bindings and
|
|
49
|
+
private no-overwrite complete-result archives. Originals, citations, graph,
|
|
50
|
+
ledger and model/extraction provenance survive revalidated checksum readback.
|
|
51
|
+
Partial outcomes and interrupted/unsealed archives never become answered runs.
|
|
52
|
+
|
|
53
|
+
- A configuration-driven `Collector` facade assembles actual search, HTTP,
|
|
54
|
+
HTML/document, optional browser, embedding and completion adapters. It supports
|
|
55
|
+
intent research and seeded collection with fresh per-run state, retains the
|
|
56
|
+
effective search recipe, and offers an explicitly bounded TOML read. Graph,
|
|
57
|
+
journal and source-session settings use their existing owning contracts.
|
|
58
|
+
Controlled-server composition acceptance is not real-model accuracy.
|
|
59
|
+
|
|
60
|
+
- Explicit intent-reference semantic scoring alongside unchanged pinned-shelf
|
|
61
|
+
scoring. The original intent's embedding spends the shared run budget once;
|
|
62
|
+
prepared vectors, call linkage, failure/cancellation and concurrent run
|
|
63
|
+
isolation are auditable through harvest and journal readers.
|
|
64
|
+
|
|
65
|
+
- Configured owner-private per-run JSONL observations, fsync-before-ack storage,
|
|
66
|
+
hash-chain replay checks, completion summaries and read-only run inspection.
|
|
67
|
+
Interrupted runs remain explicitly unsealed; no automatic refetch or resume.
|
|
68
|
+
|
|
69
|
+
- Configurable single generic HTML reparse over retained source bytes, without
|
|
70
|
+
a second fetch or renewed deadline. Success, refusal and cancellation attempts
|
|
71
|
+
retain typed source/configuration-bound provenance and round-trip ledger checks.
|
|
72
|
+
|
|
73
|
+
- Persistent exact-publisher/profile locator health with a configured miss bar,
|
|
74
|
+
generic extraction recovery, restart-safe/concurrent state, read-only doctor
|
|
75
|
+
output and source/config-bound harvest ledger findings. No publisher-redesign
|
|
76
|
+
accuracy claim follows from the regression fixtures.
|
|
77
|
+
|
|
78
|
+
- Explicit offline PDF layout/table/OCR recipe, pinned worker dependencies and
|
|
79
|
+
artifacts, configurable column-aware reading order and reproducible local PDF
|
|
80
|
+
acceptance with source-bound receipts. Representative-corpus and Marker
|
|
81
|
+
acceptance remain open.
|
|
82
|
+
|
|
83
|
+
- Authorized source sessions with caller-supplied cookies/headers, exact origin
|
|
84
|
+
and path scope, protected browser resource support, non-secret audit metadata,
|
|
85
|
+
and separation from discovery/model-control credentials.
|
|
86
|
+
- Configurable document-reference depth and citing-source discovery through the
|
|
87
|
+
injected search provider, using native parent context and the existing scorer.
|
|
88
|
+
- Source-bound reference decisions and native Docling URL/hyperlink locators;
|
|
89
|
+
saved-harvest validation of source hashes, query/provider binding and budgets.
|
|
90
|
+
- Shared reference/discovery host, parent, per-parent candidate and query limits
|
|
91
|
+
across follow-up rounds, with versioned configuration and an example.
|
|
92
|
+
|
|
93
|
+
## 0.2.0 — 2026-10-06
|
|
94
|
+
|
|
95
|
+
Rebuild of go-spider as a typed, configured standalone library. Distribution
|
|
96
|
+
identity stays `go-spider`; the implementation package is `chimera`.
|
|
97
|
+
|
|
98
|
+
### Added
|
|
99
|
+
|
|
100
|
+
- Goal collection and intent-driven planning/discovery/coverage/answer-review
|
|
101
|
+
orchestration with exact native-text citations and explicit partial results.
|
|
102
|
+
- Direct/Tor HTTP, native v3 onion support, bounded robots/redirect/retry/cache
|
|
103
|
+
policies and source-network separation from private model-control traffic.
|
|
104
|
+
- Isolated Patchright rendering with parent-owned HTTP resource/redirect handling.
|
|
105
|
+
- Native HTML fit-Markdown/adaptive locator extraction and language detection;
|
|
106
|
+
offline DOCX tables and native PDF text extraction with retained layout.
|
|
107
|
+
- Self-hosted completion and embedding clients with declared model binding,
|
|
108
|
+
bounded native evidence, token/encoding observations and recorded failure.
|
|
109
|
+
- Reference-vector relevance scoring, semantic/keyword link ranking, canonical
|
|
110
|
+
and near-duplicate grouping with independently retained source occurrences.
|
|
111
|
+
- Configurable research graph, immutable typed harvest/ledger/receipt contracts,
|
|
112
|
+
conformance tests, deliberate mutation witnesses and full package gate.
|
|
113
|
+
- Public README, package metadata, MIT licence text and migration guidance.
|
|
114
|
+
|
|
115
|
+
### Breaking changes
|
|
116
|
+
|
|
117
|
+
- Requires Python 3.11 or newer; v0.1.0 required 3.10.
|
|
118
|
+
- `chimera` replaces the prototype `spider_core` imports. No compatibility shim
|
|
119
|
+
or legacy `spider` command is included.
|
|
120
|
+
- No cloud LLM client, automatic model-server launch, VPN manager or silent
|
|
121
|
+
external-model/direct-network fallback.
|
|
122
|
+
- Operational policy is explicit validated configuration and injected ports.
|
|
123
|
+
|
|
124
|
+
### Not claimed complete
|
|
125
|
+
|
|
126
|
+
Public publisher/locator and real-model quality acceptance; full PDF/OCR/Marker;
|
|
127
|
+
Camoufox/nodriver; calibrated judge policy; one-hop references/cited-by expansion;
|
|
128
|
+
production runtime/egress acceptance. Publication
|
|
129
|
+
is not deployment or proof of completion of those planned features.
|
|
130
|
+
|
|
131
|
+
## 0.1.0
|
|
132
|
+
|
|
133
|
+
Original prototype release with `spider_core` and the `spider` CLI. Existing
|
|
134
|
+
users may pin that version while migrating to the explicit v0.2.0 library API.
|
ghimera-0.3.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2025–2026 Josh Gompert
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
ghimera-0.3.0/PKG-INFO
ADDED
|
@@ -0,0 +1,314 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: ghimera
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: Intent-driven web research with bounded collection and source-bound evidence
|
|
5
|
+
Project-URL: Homepage, https://github.com/ginkorea/ghimera
|
|
6
|
+
Project-URL: Repository, https://github.com/ginkorea/ghimera
|
|
7
|
+
Project-URL: Issues, https://github.com/ginkorea/ghimera/issues
|
|
8
|
+
Project-URL: Changelog, https://github.com/ginkorea/ghimera/blob/main/CHANGELOG.md
|
|
9
|
+
Author: Josh Gompert
|
|
10
|
+
License-Expression: MIT
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Requires-Python: >=3.11
|
|
13
|
+
Requires-Dist: curl-cffi==0.16.3
|
|
14
|
+
Requires-Dist: protego==0.7.0
|
|
15
|
+
Requires-Dist: pydantic==2.13.5
|
|
16
|
+
Provides-Extra: browser
|
|
17
|
+
Requires-Dist: patchright==1.63.0; extra == 'browser'
|
|
18
|
+
Provides-Extra: documents
|
|
19
|
+
Requires-Dist: docling-core==2.99.0; extra == 'documents'
|
|
20
|
+
Requires-Dist: docling-slim[convert-core,format-docx,format-pdf]==2.134.0; extra == 'documents'
|
|
21
|
+
Requires-Dist: lingua-language-detector==2.1.1; extra == 'documents'
|
|
22
|
+
Provides-Extra: html
|
|
23
|
+
Requires-Dist: crawl4ai==0.9.4; extra == 'html'
|
|
24
|
+
Requires-Dist: lingua-language-detector==2.1.1; extra == 'html'
|
|
25
|
+
Requires-Dist: scrapling==0.4.2; extra == 'html'
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
|
|
28
|
+
# ghimera
|
|
29
|
+
|
|
30
|
+
Intent-driven web research: discover sources, collect native-language documents,
|
|
31
|
+
follow evidence gaps, and return a source-cited answer—or an explicit partial
|
|
32
|
+
result when the evidence or budget is insufficient.
|
|
33
|
+
|
|
34
|
+
**v0.3.0 consolidates the repository, distribution and import as `ghimera`.**
|
|
35
|
+
It succeeds the `go-spider` distribution and `chimera` implementation. It is not
|
|
36
|
+
backward-compatible with v0.1.0's `spider_core` API or `spider` CLI. Python
|
|
37
|
+
**3.11+** is required. Some planned browser/document adapters and public-corpus
|
|
38
|
+
acceptance are still in progress; see the limitations below.
|
|
39
|
+
|
|
40
|
+
ghimera is an independent library. Supply your own search provider,
|
|
41
|
+
self-hosted model services, extraction policies and graph profile.
|
|
42
|
+
|
|
43
|
+
## What is implemented
|
|
44
|
+
|
|
45
|
+
- **Goal and intent loops.** Collect from configured seeds with `GoalLoop`, or
|
|
46
|
+
use `ResearchLoop` to plan questions, discover sources through an injected
|
|
47
|
+
search provider, assess gaps, and draft/review an evidence-cited answer.
|
|
48
|
+
- **Local models first.** Configured, already-served self-hosted models supply
|
|
49
|
+
planning, judging, answer generation, review and embeddings. Compatible HTTP
|
|
50
|
+
interfaces are supported; no external LLM fallback, model weights or model
|
|
51
|
+
server are included.
|
|
52
|
+
- **Bounded direct and Tor HTTP.** Native v3 onion collection and open-web
|
|
53
|
+
requests through Tor share the same route policy. Public-network validation,
|
|
54
|
+
pinned DNS for direct requests, per-hop redirect checks, robots policy,
|
|
55
|
+
concurrency/rate limits and retries are accounted for before returning data.
|
|
56
|
+
A failed Tor route never silently falls back to direct access.
|
|
57
|
+
- **Isolated JavaScript rendering.** Patchright runs in a network-isolated
|
|
58
|
+
Linux worker; the parent fetch boundary handles its permitted HTTP resources,
|
|
59
|
+
redirects and accounting. Browser binaries are explicitly configured and
|
|
60
|
+
verified, not downloaded on import. Camoufox/nodriver adapters remain planned.
|
|
61
|
+
- **Native extraction.** Configured HTML fit-Markdown, adaptive locator
|
|
62
|
+
profiles, language detection, DOCX tables and native PDF text preserve raw
|
|
63
|
+
bytes beside extracted native-language text. Full PDF/OCR and Marker
|
|
64
|
+
acceptance remain open.
|
|
65
|
+
- **Relevance and deduplication.** An injected self-hosted encoder scores
|
|
66
|
+
native text/windows and observed links against pinned reference vectors.
|
|
67
|
+
Keyword/semantic ranking, encoding budgets, canonical URL handling, SHA-256
|
|
68
|
+
and configured near-duplicate grouping retain source-qualified evidence.
|
|
69
|
+
Similarity is not a calibrated probability or a substitute for a verdict.
|
|
70
|
+
- **Graphs and audit records.** A configurable research graph starts with the
|
|
71
|
+
intent. Typed harvests, receipts and ledger rows retain configuration,
|
|
72
|
+
transport, model-call spend, omissions, verdicts, refusals and source hashes.
|
|
73
|
+
Operational discovery traces are distinct from evidence-supported claims.
|
|
74
|
+
|
|
75
|
+
Every candidate reaching acceptance receives an accept/reject/hold verdict. A hold gets a
|
|
76
|
+
second model pass. An intent is marked answered only after the coverage,
|
|
77
|
+
citations, review and configured confidence checks pass; budget exhaustion is
|
|
78
|
+
not silently presented as success.
|
|
79
|
+
|
|
80
|
+
### Additions since go-spider 0.2.0
|
|
81
|
+
|
|
82
|
+
The current working branch also has explicit authorized source sessions,
|
|
83
|
+
configurable references/citing-source discovery, persistent publisher-locator
|
|
84
|
+
drift detection with generic recovery and a doctor, and offline model-based PDF
|
|
85
|
+
layout, tables, OCR and column-aware reading order, durable run journals, and
|
|
86
|
+
intent-based semantic scoring without a prebuilt reference-vector file, and a
|
|
87
|
+
configuration-driven `Collector` facade using actual adapters. These were not included
|
|
88
|
+
in the old `go-spider==0.2.0` wheel and are included in `ghimera==0.3.0`.
|
|
89
|
+
See [source sessions](docs/SOURCE_SESSIONS.md),
|
|
90
|
+
[references](docs/C3_REFERENCES.md), [locator health](docs/C2_HTML.md), and
|
|
91
|
+
[PDF configuration/acceptance](docs/C2_DOCUMENTS.md),
|
|
92
|
+
[run journals](docs/RUN_JOURNAL.md) and
|
|
93
|
+
[intent scoring](docs/C3_EMBEDDING_SCORING.md#intent-references-unreleased-source).
|
|
94
|
+
For the assembled intent-only API and full non-active template, see
|
|
95
|
+
[configured collector](docs/COLLECTOR.md) and `examples/collector.toml`.
|
|
96
|
+
The package also supports explicitly configured ordinary HTML search
|
|
97
|
+
alongside JSON, and complete research archives retain successful discovery
|
|
98
|
+
responses with query/fetch bindings. See [HTML search](docs/C3_SEARCH_HTML.md)
|
|
99
|
+
and [search evidence](docs/C3_SEARCH_EVIDENCE.md). Neither mode solves access
|
|
100
|
+
challenges, and a refusal is not a successful research result.
|
|
101
|
+
Representative-corpus accuracy and Marker acceptance remain open; passing a
|
|
102
|
+
controlled document check is not a universal quality claim.
|
|
103
|
+
|
|
104
|
+
## Installation
|
|
105
|
+
|
|
106
|
+
Use a dedicated virtual environment:
|
|
107
|
+
|
|
108
|
+
```bash
|
|
109
|
+
python3.11 -m venv .venv
|
|
110
|
+
. .venv/bin/activate
|
|
111
|
+
python -m pip install 'ghimera==0.3.0'
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
Install the adapters you intend to configure:
|
|
115
|
+
|
|
116
|
+
```bash
|
|
117
|
+
python -m pip install 'ghimera[html,documents,browser]==0.3.0'
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
The base package contains the typed core, HTTP/Tor transport, research/search
|
|
121
|
+
and self-hosted model/embedding clients. Extras add pinned HTML, document and
|
|
122
|
+
Patchright dependencies. They do **not** install an inference server, browser
|
|
123
|
+
binary, Tor daemon or PDF/OCR model artifacts. The isolated browser adapter
|
|
124
|
+
requires Linux, a compatible explicitly supplied Chromium binary and Bubblewrap;
|
|
125
|
+
other operating systems have not been accepted for that adapter.
|
|
126
|
+
|
|
127
|
+
## Configuration and API
|
|
128
|
+
|
|
129
|
+
Operational choices are typed, versioned configuration—not Python constants:
|
|
130
|
+
scope, budgets, endpoints, model identities/revisions, thresholds, private
|
|
131
|
+
worker directories, browser provenance and direct/Tor policy. Parse once with
|
|
132
|
+
`GhimeraConfig.from_toml(Path(...))`; inject the matching collaborators.
|
|
133
|
+
|
|
134
|
+
The [examples](https://github.com/ginkorea/ghimera/tree/v0.3.0/examples) are non-active templates. Replace invalid endpoints,
|
|
135
|
+
contact information, private paths and model identifiers; reference-vector
|
|
136
|
+
fixtures are **not** production relevance data. Adapter blocks belong in the
|
|
137
|
+
main configuration under their named keys, not as unrelated root settings.
|
|
138
|
+
Supply any model credential separately in memory, only to its authorized exact
|
|
139
|
+
endpoint; configuration is not credential or destination approval.
|
|
140
|
+
|
|
141
|
+
Download the starter configuration, or copy it from the repository:
|
|
142
|
+
|
|
143
|
+
```bash
|
|
144
|
+
curl --fail --proto '=https' --tlsv1.2 \
|
|
145
|
+
https://raw.githubusercontent.com/ginkorea/ghimera/v0.3.0/examples/chimera.toml \
|
|
146
|
+
--output chimera.toml
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
An entirely offline smoke example, using explicitly named test doubles:
|
|
150
|
+
|
|
151
|
+
```python
|
|
152
|
+
import asyncio
|
|
153
|
+
from pathlib import Path
|
|
154
|
+
|
|
155
|
+
from ghimera import GhimeraConfig, Goal, GoalLoop, Harvest, Scope
|
|
156
|
+
from ghimera.doubles import FakeExtractor, FakeJudge, FakeRoute, KeywordScorer
|
|
157
|
+
from ghimera.fetch import FetchLadder
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
async def main() -> None:
|
|
161
|
+
config = GhimeraConfig.from_toml(Path("chimera.toml"))
|
|
162
|
+
collector = GoalLoop(
|
|
163
|
+
config=config,
|
|
164
|
+
fetcher=FetchLadder((FakeRoute(),)),
|
|
165
|
+
extractor=FakeExtractor(),
|
|
166
|
+
scorer=KeywordScorer(),
|
|
167
|
+
judge=FakeJudge(),
|
|
168
|
+
)
|
|
169
|
+
result = await collector.run(
|
|
170
|
+
Goal(text="ports", seeds=("https://example.org/start",)),
|
|
171
|
+
Scope(allowed_hosts=("example.org",), max_depth=1, content_types=("text/html",)),
|
|
172
|
+
)
|
|
173
|
+
# Reader revalidates retained sources, provenance and receipt accounting.
|
|
174
|
+
restored = Harvest.model_validate_json(result.model_dump_json())
|
|
175
|
+
print(restored.receipt.stop_reason)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
asyncio.run(main())
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
This example makes no network or real model calls and proves no research
|
|
182
|
+
accuracy. For actual collection bind `CurlRoute`, configured extraction and
|
|
183
|
+
scoring adapters, and the self-hosted judge through their ports. For intent-only
|
|
184
|
+
research, inject those into `ResearchLoop` alongside `GroundedSearch`,
|
|
185
|
+
`IntentPlanner`, `ResearchAnalyst` and `AnswerReviewer`, then call
|
|
186
|
+
`run(ResearchRequest(intent="your research question"))`. `SearxSearch` is the
|
|
187
|
+
implemented search adapter.
|
|
188
|
+
|
|
189
|
+
### Configured intent API
|
|
190
|
+
|
|
191
|
+
The configured collector assembles real adapters, so applications need not manually
|
|
192
|
+
wire every port. Unlike the offline smoke above, this needs your configured
|
|
193
|
+
services and the adapted `examples/collector.toml` template:
|
|
194
|
+
|
|
195
|
+
```python
|
|
196
|
+
import asyncio
|
|
197
|
+
from pathlib import Path
|
|
198
|
+
|
|
199
|
+
from ghimera import Collector
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
async def main() -> None:
|
|
203
|
+
collector = Collector.from_toml(Path("collector.toml"), max_config_bytes=100_000)
|
|
204
|
+
result = await collector.run("Your research question")
|
|
205
|
+
print(result.status)
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
asyncio.run(main())
|
|
209
|
+
```
|
|
210
|
+
|
|
211
|
+
Use the [collector guide](docs/COLLECTOR.md) to configure actual private models,
|
|
212
|
+
source routing, extraction, optional graph/journal and separately supplied
|
|
213
|
+
credentials. This API is included in `ghimera==0.3.0`, not the old `go-spider==0.2.0` wheel. The
|
|
214
|
+
lower-level APIs remain supported for custom providers and composition.
|
|
215
|
+
|
|
216
|
+
The package also supplies a configured intent command that retains the
|
|
217
|
+
full result, original documents and citations in a private, checksum-sealed archive:
|
|
218
|
+
|
|
219
|
+
```bash
|
|
220
|
+
python -m ghimera --job /absolute/path/collector-command.toml --max-job-bytes 100000
|
|
221
|
+
```
|
|
222
|
+
|
|
223
|
+
See the [command guide](docs/COLLECTOR_COMMAND.md) and
|
|
224
|
+
`examples/collector-command.toml`. Existing output identities are never overwritten;
|
|
225
|
+
partial results remain partial. This command is not in the published 0.2.0 wheel.
|
|
226
|
+
|
|
227
|
+
The following guides cover the existing lower-level wiring:
|
|
228
|
+
|
|
229
|
+
| Area | Guide |
|
|
230
|
+
|---|---|
|
|
231
|
+
| Intent, discovery, coverage and answer review | [Intent research](https://github.com/ginkorea/ghimera/blob/v0.3.0/docs/C3_RESEARCH.md) |
|
|
232
|
+
| Model roles, credentials and native evidence context | [Self-hosted models](https://github.com/ginkorea/ghimera/blob/v0.3.0/docs/C3_MODELS.md) |
|
|
233
|
+
| Reference vectors, encoding and frontier ranking | [Embedding scoring](https://github.com/ginkorea/ghimera/blob/v0.3.0/docs/C3_EMBEDDING_SCORING.md) |
|
|
234
|
+
| Open web and native onion routing | [Tor policy](https://github.com/ginkorea/ghimera/blob/v0.3.0/docs/TOR.md) |
|
|
235
|
+
| Browser isolation, resources and redirects | [Browser rendering](https://github.com/ginkorea/ghimera/blob/v0.3.0/docs/C1_BROWSER.md) |
|
|
236
|
+
| HTML extraction and adaptive locators | [HTML extraction](https://github.com/ginkorea/ghimera/blob/v0.3.0/docs/C2_HTML.md) |
|
|
237
|
+
| DOCX/native PDF and offline artifacts | [Document extraction](https://github.com/ginkorea/ghimera/blob/v0.3.0/docs/C2_DOCUMENTS.md) |
|
|
238
|
+
| Canonical and near-duplicate source evidence | [Deduplication](https://github.com/ginkorea/ghimera/blob/v0.3.0/docs/C2_DEDUP.md) |
|
|
239
|
+
| Configurable research graphs | [Research graph](https://github.com/ginkorea/ghimera/blob/v0.3.0/docs/RESEARCH_GRAPH.md) |
|
|
240
|
+
| Architecture, contracts and completion tracker | [Specification](https://github.com/ginkorea/ghimera/blob/v0.3.0/docs/C0.md) |
|
|
241
|
+
|
|
242
|
+
## Scope and limitations
|
|
243
|
+
|
|
244
|
+
Robots are honored by default. An override requires a recorded, reasoned,
|
|
245
|
+
exact-host configuration decision; it does not disable CAPTCHA, login, paywall
|
|
246
|
+
or challenge refusal. There is no challenge solver or authenticated-site bypass.
|
|
247
|
+
Tor routing is a transport capability, not a guarantee of anonymity or authority
|
|
248
|
+
to access a source.
|
|
249
|
+
|
|
250
|
+
Collection is not limited to anonymous access. Supply your own authorized
|
|
251
|
+
cookies or headers through explicitly configured source sessions, with exact
|
|
252
|
+
origin/path scope and no credential values in receipts. Browser resources use
|
|
253
|
+
the same parent-owned session boundary. See [authorized sessions](docs/SOURCE_SESSIONS.md).
|
|
254
|
+
|
|
255
|
+
Source/search requests run on the host executing the crawler. Model control is
|
|
256
|
+
a separate private-service boundary. Your application owns authorization,
|
|
257
|
+
deployment and storage; no external scheduler or registry is required to import
|
|
258
|
+
or use the library.
|
|
259
|
+
|
|
260
|
+
Configurable source expansion follows observed document URLs and discovers
|
|
261
|
+
candidate citing sources through your search provider. Depth, host policy and
|
|
262
|
+
budgets live in `[references]`, not in Python. Source hashes and native locators
|
|
263
|
+
are retained; a citing-source search hit is not proof that a citation exists.
|
|
264
|
+
See [reference expansion](docs/C3_REFERENCES.md).
|
|
265
|
+
|
|
266
|
+
For durable observations, configure `[journal]` and pass a unique `run_id`.
|
|
267
|
+
The collector persists JSONL events before acknowledgment and seals a completion
|
|
268
|
+
summary only after receipt reconciliation. Interrupted prefixes remain inspectable
|
|
269
|
+
without silently refetching sources. See [run journals](docs/RUN_JOURNAL.md).
|
|
270
|
+
|
|
271
|
+
Still required for the complete planned spider: the remaining browser adapters,
|
|
272
|
+
representative publisher/locator acceptance, full PDF/OCR and Marker validation,
|
|
273
|
+
real reference/cited-by adequacy, real served-model quality/admission and
|
|
274
|
+
calibrated decision policy and live runtime/egress acceptance.
|
|
275
|
+
The repository's detailed tracker retains those requirements; this release
|
|
276
|
+
does not erase them or describe fixture results as real-world model accuracy.
|
|
277
|
+
|
|
278
|
+
## Development
|
|
279
|
+
|
|
280
|
+
```bash
|
|
281
|
+
git clone https://github.com/ginkorea/ghimera.git
|
|
282
|
+
cd ghimera
|
|
283
|
+
uv sync --locked --extra html --extra documents --extra browser --python 3.11
|
|
284
|
+
# Explicit browser/isolation paths are required for the complete gate.
|
|
285
|
+
export CHIMERA_TEST_BROWSER=/absolute/path/to/compatible/chrome
|
|
286
|
+
export CHIMERA_TEST_ISOLATOR=/absolute/path/to/bwrap
|
|
287
|
+
bash scripts/gate.sh
|
|
288
|
+
uv build --no-sources
|
|
289
|
+
```
|
|
290
|
+
|
|
291
|
+
The gate checks the actual interpreter/import path, lockfile, Ruff, strict mypy
|
|
292
|
+
and the entire test suite. Browser tests refuse absent acceptance prerequisites
|
|
293
|
+
rather than pretending they ran. Evidence records distinguish protocol fixtures,
|
|
294
|
+
installed-artifact checks, public-corpus acceptance and production activation.
|
|
295
|
+
|
|
296
|
+
## Migration from go-spider
|
|
297
|
+
|
|
298
|
+
Install `ghimera==0.3.0` explicitly; this is a new distribution, not an in-place
|
|
299
|
+
rename of old PyPI releases. Use `from ghimera import Collector, GhimeraConfig`,
|
|
300
|
+
`ghimera.*` for submodules, and `ghimera` or `python -m ghimera` for the command.
|
|
301
|
+
The legacy root `from chimera import Collector, ChimeraConfig` and
|
|
302
|
+
`python -m chimera` forward to the same implementation. Old nested
|
|
303
|
+
`chimera.*` imports must migrate; there is no second implementation or import hook.
|
|
304
|
+
Existing versioned `chimera.*` configuration, graph, harvest, journal and result
|
|
305
|
+
schemas are retained so existing saved evidence does not change identity.
|
|
306
|
+
|
|
307
|
+
The v0.1.0 `spider_core` API and old `spider` command are not supplied; keep
|
|
308
|
+
`go-spider==0.1.0` while migrating those applications. The old cloud-client,
|
|
309
|
+
VPN-manager and implicit fallback design is not retained. Prototype source
|
|
310
|
+
remains in Git history.
|
|
311
|
+
|
|
312
|
+
See [CHANGELOG.md](https://github.com/ginkorea/ghimera/blob/v0.3.0/CHANGELOG.md) for release changes. Josh Gompert maintains
|
|
313
|
+
the project at [ginkorea/ghimera](https://github.com/ginkorea/ghimera).
|
|
314
|
+
Licensed under [MIT](https://github.com/ginkorea/ghimera/blob/v0.3.0/LICENSE), matching the existing PyPI licence declaration.
|