geo-scope 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- geo_scope/__init__.py +37 -0
- geo_scope/benchmark/__init__.py +53 -0
- geo_scope/benchmark/builder.py +253 -0
- geo_scope/benchmark/calculator.py +479 -0
- geo_scope/benchmark/dataset_validator.py +232 -0
- geo_scope/benchmark/hasher.py +120 -0
- geo_scope/benchmark/models.py +184 -0
- geo_scope/benchmark/profile.py +82 -0
- geo_scope/benchmark/reproducer.py +196 -0
- geo_scope/benchmark/runner.py +668 -0
- geo_scope/benchmark/schemas.py +157 -0
- geo_scope/benchmark/validator.py +237 -0
- geo_scope/cli.py +1193 -0
- geo_scope/data/benchmark_history.json +451 -0
- geo_scope/engine/__init__.py +0 -0
- geo_scope/engine/algo_analyzer.py +361 -0
- geo_scope/engine/execution_mode.py +35 -0
- geo_scope/engine/export_manager.py +94 -0
- geo_scope/engine/feature_extractor.py +318 -0
- geo_scope/engine/history_tracker.py +106 -0
- geo_scope/engine/model_runner.py +445 -0
- geo_scope/engine/persistence.py +114 -0
- geo_scope/engine/query_generator.py +333 -0
- geo_scope/engine/query_loader.py +186 -0
- geo_scope/engine/report_generator.py +352 -0
- geo_scope/engine/response_import.py +37 -0
- geo_scope/engine/strategy_builder.py +60 -0
- geo_scope/entities/__init__.py +8 -0
- geo_scope/entities/models.py +37 -0
- geo_scope/entities/registry.py +181 -0
- geo_scope/mavi/__init__.py +27 -0
- geo_scope/mavi/engine.py +274 -0
- geo_scope/mavi/geoscope_evaluator.py +170 -0
- geo_scope/mavi/models.py +195 -0
- geo_scope/mavi/sage_evaluator.py +322 -0
- geo_scope/mcp_server.py +294 -0
- geo_scope/measurement/__init__.py +8 -0
- geo_scope/measurement/engine.py +557 -0
- geo_scope/measurement/replay.py +376 -0
- geo_scope/parser/__init__.py +11 -0
- geo_scope/parser/evaluator.py +217 -0
- geo_scope/parser/observation_parser.py +547 -0
- geo_scope/providers/__init__.py +26 -0
- geo_scope/providers/base.py +241 -0
- geo_scope/providers/claude_provider.py +65 -0
- geo_scope/providers/gemini_provider.py +63 -0
- geo_scope/providers/hamzad_provider.py +251 -0
- geo_scope/providers/keyless_wrapper_provider.py +40 -0
- geo_scope/providers/models.py +234 -0
- geo_scope/providers/ollama_provider.py +47 -0
- geo_scope/providers/openai_provider.py +66 -0
- geo_scope/providers/openrouter_provider.py +40 -0
- geo_scope/providers/perplexity_provider.py +74 -0
- geo_scope/providers/public_research_provider.py +24 -0
- geo_scope/providers/registry.py +166 -0
- geo_scope/providers/simulated.py +232 -0
- geo_scope/questions/__init__.py +19 -0
- geo_scope/questions/answerpath_connector.py +294 -0
- geo_scope/questions/discovery.py +228 -0
- geo_scope/questions/models.py +122 -0
- geo_scope/release_gate.py +293 -0
- geo_scope/server.py +320 -0
- geo_scope/static/featured_image_geo.png +0 -0
- geo_scope/static/index.html +1433 -0
- geo_scope-0.3.0.dist-info/METADATA +371 -0
- geo_scope-0.3.0.dist-info/RECORD +70 -0
- geo_scope-0.3.0.dist-info/WHEEL +5 -0
- geo_scope-0.3.0.dist-info/entry_points.txt +2 -0
- geo_scope-0.3.0.dist-info/licenses/LICENSE +21 -0
- geo_scope-0.3.0.dist-info/top_level.txt +1 -0
geo_scope/__init__.py
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""
|
|
2
|
+
GEO-Scope: Generative Engine Optimization (GEO) & Empirical AI Answer Visibility Framework
|
|
3
|
+
An open-source scientific framework for measuring entity visibility, recommendations, and citations across generative AI systems.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
__version__ = "0.3.0"
|
|
7
|
+
__author__ = "GEO-Scope Community & Contributors"
|
|
8
|
+
__license__ = "MIT"
|
|
9
|
+
|
|
10
|
+
from geo_scope.engine.query_generator import generate_prompt_dataset, INDUSTRY_PRESETS
|
|
11
|
+
from geo_scope.engine.feature_extractor import (
|
|
12
|
+
parse_model_response,
|
|
13
|
+
extract_citations_and_domains,
|
|
14
|
+
detect_brand_positions,
|
|
15
|
+
)
|
|
16
|
+
from geo_scope.engine.execution_mode import ExecutionMode
|
|
17
|
+
from geo_scope.engine.persistence import RawRunStore
|
|
18
|
+
from geo_scope.engine.model_runner import ModelRunner
|
|
19
|
+
from geo_scope.engine.strategy_builder import generate_geo_playbook
|
|
20
|
+
from geo_scope.mavi import MAVIEngine, MAVIReport, LayerWeights
|
|
21
|
+
|
|
22
|
+
__all__ = [
|
|
23
|
+
"__version__",
|
|
24
|
+
"generate_prompt_dataset",
|
|
25
|
+
"INDUSTRY_PRESETS",
|
|
26
|
+
"parse_model_response",
|
|
27
|
+
"extract_citations_and_domains",
|
|
28
|
+
"detect_brand_positions",
|
|
29
|
+
"ExecutionMode",
|
|
30
|
+
"RawRunStore",
|
|
31
|
+
"ModelRunner",
|
|
32
|
+
"AlgoAnalyzer",
|
|
33
|
+
"generate_geo_playbook",
|
|
34
|
+
"MAVIEngine",
|
|
35
|
+
"MAVIReport",
|
|
36
|
+
"LayerWeights",
|
|
37
|
+
]
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
"""
|
|
2
|
+
GEO-Scope Public Benchmark & Evidence Dataset Module.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
from geo_scope.benchmark.models import (
|
|
6
|
+
BenchmarkExecutionMode,
|
|
7
|
+
BenchmarkManifest,
|
|
8
|
+
PromptRecord,
|
|
9
|
+
ObservationRecord,
|
|
10
|
+
CitationEvidenceRecord,
|
|
11
|
+
BrandBenchmarkMetrics,
|
|
12
|
+
ProviderBenchmarkMetrics,
|
|
13
|
+
BenchmarkMetrics,
|
|
14
|
+
MetricEstimate,
|
|
15
|
+
StatisticalFactorAnalysis,
|
|
16
|
+
)
|
|
17
|
+
from geo_scope.benchmark.validator import ProviderValidator, ProviderValidationResult
|
|
18
|
+
from geo_scope.benchmark.hasher import (
|
|
19
|
+
compute_file_sha256,
|
|
20
|
+
compute_dataset_checksums,
|
|
21
|
+
compute_composite_hash,
|
|
22
|
+
write_checksums_file,
|
|
23
|
+
verify_dataset_checksums,
|
|
24
|
+
)
|
|
25
|
+
from geo_scope.benchmark.calculator import BenchmarkCalculator, calculate_bootstrap_ci
|
|
26
|
+
from geo_scope.benchmark.builder import BenchmarkBuilder
|
|
27
|
+
from geo_scope.benchmark.reproducer import BenchmarkReproducer
|
|
28
|
+
from geo_scope.benchmark.profile import BenchmarkProfile
|
|
29
|
+
from geo_scope.benchmark.runner import LiveBenchmarkRunner, estimate_benchmark_cost
|
|
30
|
+
|
|
31
|
+
__all__ = [
|
|
32
|
+
"BenchmarkManifest",
|
|
33
|
+
"PromptRecord",
|
|
34
|
+
"ObservationRecord",
|
|
35
|
+
"CitationEvidenceRecord",
|
|
36
|
+
"BrandBenchmarkMetrics",
|
|
37
|
+
"ProviderBenchmarkMetrics",
|
|
38
|
+
"BenchmarkMetrics",
|
|
39
|
+
"MetricEstimate",
|
|
40
|
+
"StatisticalFactorAnalysis",
|
|
41
|
+
"compute_file_sha256",
|
|
42
|
+
"compute_dataset_checksums",
|
|
43
|
+
"compute_composite_hash",
|
|
44
|
+
"write_checksums_file",
|
|
45
|
+
"verify_dataset_checksums",
|
|
46
|
+
"BenchmarkCalculator",
|
|
47
|
+
"calculate_bootstrap_ci",
|
|
48
|
+
"BenchmarkBuilder",
|
|
49
|
+
"BenchmarkReproducer",
|
|
50
|
+
"BenchmarkProfile",
|
|
51
|
+
"LiveBenchmarkRunner",
|
|
52
|
+
"estimate_benchmark_cost",
|
|
53
|
+
]
|
|
@@ -0,0 +1,253 @@
|
|
|
1
|
+
# Dataset package builder for GEO-Scope Public Benchmark v1.
|
|
2
|
+
import json
|
|
3
|
+
import os
|
|
4
|
+
import subprocess
|
|
5
|
+
from datetime import datetime, timezone
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Dict, List, Any, Optional
|
|
8
|
+
|
|
9
|
+
from geo_scope.benchmark.models import BenchmarkManifest, BenchmarkMetrics
|
|
10
|
+
from geo_scope.benchmark.calculator import BenchmarkCalculator
|
|
11
|
+
from geo_scope.benchmark.hasher import (
|
|
12
|
+
compute_dataset_checksums,
|
|
13
|
+
compute_composite_hash,
|
|
14
|
+
write_checksums_file,
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def get_git_commit(repo_dir: Optional[str | Path] = None) -> Optional[str]:
|
|
19
|
+
"""Attempt to get the current git commit hash."""
|
|
20
|
+
try:
|
|
21
|
+
cmd = ["git", "rev-parse", "HEAD"]
|
|
22
|
+
env = os.environ.copy()
|
|
23
|
+
env["DEVELOPER_DIR"] = "/Library/Developer/CommandLineTools"
|
|
24
|
+
res = subprocess.run(cmd, cwd=repo_dir, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env=env)
|
|
25
|
+
if res.returncode == 0:
|
|
26
|
+
return res.stdout.strip()
|
|
27
|
+
except Exception:
|
|
28
|
+
pass
|
|
29
|
+
return None
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class BenchmarkBuilder:
|
|
33
|
+
"""
|
|
34
|
+
Builds a complete, versioned, tamper-proof GEO-Scope benchmark dataset package.
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
def __init__(self, dataset_id: str = "geo-scope-benchmark-2026.1"):
|
|
38
|
+
self.dataset_id = dataset_id
|
|
39
|
+
self.calculator = BenchmarkCalculator()
|
|
40
|
+
|
|
41
|
+
def build_package(
|
|
42
|
+
self,
|
|
43
|
+
out_dir: str | Path,
|
|
44
|
+
prompts: List[Dict[str, Any]],
|
|
45
|
+
observations: List[Dict[str, Any]],
|
|
46
|
+
citations: List[Dict[str, Any]],
|
|
47
|
+
brands: List[Dict[str, Any]],
|
|
48
|
+
providers: List[Dict[str, Any]],
|
|
49
|
+
execution_mode: str = "synthetic",
|
|
50
|
+
benchmark_mode: str = "discovery",
|
|
51
|
+
research_status: str = "demo_only",
|
|
52
|
+
description: Optional[str] = None,
|
|
53
|
+
methodology_md: Optional[str] = None,
|
|
54
|
+
readme_md: Optional[str] = None,
|
|
55
|
+
provider_validation: Optional[Dict[str, Any]] = None,
|
|
56
|
+
model_provenance: Optional[Dict[str, Any]] = None,
|
|
57
|
+
) -> Path:
|
|
58
|
+
target_dir = Path(out_dir) / self.dataset_id
|
|
59
|
+
target_dir.mkdir(parents=True, exist_ok=True)
|
|
60
|
+
|
|
61
|
+
if description is None:
|
|
62
|
+
description = (
|
|
63
|
+
f"GEO-Scope Public Benchmark {self.dataset_id} containing multi-model "
|
|
64
|
+
f"observations across {len(providers)} providers and {len(prompts)} prompts."
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
# 1. Write data files
|
|
68
|
+
prompts_file = target_dir / "prompts.jsonl"
|
|
69
|
+
with open(prompts_file, "w", encoding="utf-8") as f:
|
|
70
|
+
for p in prompts:
|
|
71
|
+
f.write(json.dumps(p, ensure_ascii=False) + "\n")
|
|
72
|
+
|
|
73
|
+
# Write partitioned prompts directory
|
|
74
|
+
prompts_sub_dir = target_dir / "prompts"
|
|
75
|
+
prompts_sub_dir.mkdir(parents=True, exist_ok=True)
|
|
76
|
+
obs_prompts = [p for p in prompts if p.get("source_type") == "observed"]
|
|
77
|
+
gen_prompts = [p for p in prompts if p.get("source_type") != "observed"]
|
|
78
|
+
|
|
79
|
+
with open(prompts_sub_dir / "observed.jsonl", "w", encoding="utf-8") as f:
|
|
80
|
+
for p in obs_prompts:
|
|
81
|
+
f.write(json.dumps(p, ensure_ascii=False) + "\n")
|
|
82
|
+
|
|
83
|
+
with open(prompts_sub_dir / "generated.jsonl", "w", encoding="utf-8") as f:
|
|
84
|
+
for p in gen_prompts:
|
|
85
|
+
f.write(json.dumps(p, ensure_ascii=False) + "\n")
|
|
86
|
+
|
|
87
|
+
(target_dir / "entities.json").write_text(
|
|
88
|
+
json.dumps(brands, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
|
89
|
+
)
|
|
90
|
+
(target_dir / "brands.json").write_text(
|
|
91
|
+
json.dumps(brands, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
(target_dir / "providers.json").write_text(
|
|
95
|
+
json.dumps(providers, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
obs_file = target_dir / "observations.jsonl"
|
|
99
|
+
with open(obs_file, "w", encoding="utf-8") as f:
|
|
100
|
+
for o in observations:
|
|
101
|
+
f.write(json.dumps(o, ensure_ascii=False) + "\n")
|
|
102
|
+
|
|
103
|
+
cits_file = target_dir / "citations.jsonl"
|
|
104
|
+
with open(cits_file, "w", encoding="utf-8") as f:
|
|
105
|
+
for c in citations:
|
|
106
|
+
f.write(json.dumps(c, ensure_ascii=False) + "\n")
|
|
107
|
+
|
|
108
|
+
# 2. Compute and write metrics.json
|
|
109
|
+
metrics_obj = self.calculator.compute(
|
|
110
|
+
prompts=prompts,
|
|
111
|
+
observations=observations,
|
|
112
|
+
citations=citations,
|
|
113
|
+
brands=brands,
|
|
114
|
+
providers=providers,
|
|
115
|
+
dataset_id=self.dataset_id,
|
|
116
|
+
execution_mode=execution_mode,
|
|
117
|
+
benchmark_mode=benchmark_mode,
|
|
118
|
+
research_status=research_status,
|
|
119
|
+
)
|
|
120
|
+
(target_dir / "metrics.json").write_text(
|
|
121
|
+
json.dumps(metrics_obj.model_dump(), ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
|
122
|
+
)
|
|
123
|
+
|
|
124
|
+
# Write provenance.json
|
|
125
|
+
provenance_payload = {
|
|
126
|
+
"dataset_id": self.dataset_id,
|
|
127
|
+
"created_at": datetime.now(timezone.utc).isoformat(),
|
|
128
|
+
"execution_mode": execution_mode,
|
|
129
|
+
"benchmark_mode": benchmark_mode,
|
|
130
|
+
"question_provenance": {
|
|
131
|
+
"total_prompts": len(prompts),
|
|
132
|
+
"observed_count": len(obs_prompts),
|
|
133
|
+
"generated_count": len(gen_prompts),
|
|
134
|
+
"source_reference": "answerpath",
|
|
135
|
+
"observed_sources": list({p.get("source_reference", "observed") for p in obs_prompts}),
|
|
136
|
+
"generated_sources": list({p.get("source_reference", "generated") for p in gen_prompts}),
|
|
137
|
+
},
|
|
138
|
+
"model_provenance": model_provenance or {
|
|
139
|
+
"validated": True,
|
|
140
|
+
"fallbacks_recorded": True,
|
|
141
|
+
},
|
|
142
|
+
}
|
|
143
|
+
(target_dir / "provenance.json").write_text(
|
|
144
|
+
json.dumps(provenance_payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
# 3. Write documentation
|
|
148
|
+
if methodology_md is None:
|
|
149
|
+
methodology_md = f"""# GEO-Scope Benchmark Methodology ({self.dataset_id})
|
|
150
|
+
|
|
151
|
+
## Overview
|
|
152
|
+
This benchmark evaluates Generative Engine Optimization (GEO) performance, brand visibility, and citation presence across multi-model AI engines.
|
|
153
|
+
|
|
154
|
+
## Execution Mode & Status
|
|
155
|
+
- **Execution Mode**: `{execution_mode}`
|
|
156
|
+
- **Research Status**: `{research_status}`
|
|
157
|
+
- **Strict Separation**: Synthetic simulation runs are explicitly marked `demo_only` and must not be cited as real provider behavior.
|
|
158
|
+
|
|
159
|
+
## Question & Model Provenance
|
|
160
|
+
- **Observed Prompts**: {len(obs_prompts)} (Real user demand)
|
|
161
|
+
- **Generated Prompts**: {len(gen_prompts)} (Research exploration templates)
|
|
162
|
+
- **Model Provenance**: Explicit tracking of native vs fallback routes.
|
|
163
|
+
|
|
164
|
+
## Measured Metrics
|
|
165
|
+
1. **Share of Model (SoM)**: Percentage of total observed brand mentions attributed to the brand.
|
|
166
|
+
2. **Mention Rate**: Percentage of successful multi-model observations containing the brand (with 95% bootstrap CI).
|
|
167
|
+
3. **Top-1 Primary Rate**: Percentage of successful observations where the brand is the first/primary recommendation.
|
|
168
|
+
4. **Citation Rate**: Percentage of observations citing the brand or authoritative third-party source.
|
|
169
|
+
|
|
170
|
+
## Statistical Bounds & Language Guardrails
|
|
171
|
+
- All confidence intervals are non-parametric 95% percentile bootstrap estimates (1,000 resamples).
|
|
172
|
+
- Factor analyses report **observed empirical correlations** only, avoiding speculative "AI ranking algorithm" assertions.
|
|
173
|
+
"""
|
|
174
|
+
(target_dir / "methodology.md").write_text(methodology_md.strip() + "\n", encoding="utf-8")
|
|
175
|
+
|
|
176
|
+
if readme_md is None:
|
|
177
|
+
readme_md = f"""# {self.dataset_id}
|
|
178
|
+
|
|
179
|
+
GEO-Scope Public Benchmark & Evidence Dataset.
|
|
180
|
+
|
|
181
|
+
- Total Prompts: {len(prompts)}
|
|
182
|
+
- Observed User Questions: {len(obs_prompts)}
|
|
183
|
+
- Generated Research Prompts: {len(gen_prompts)}
|
|
184
|
+
- Total Observations: {len(observations)}
|
|
185
|
+
- Providers: {len(providers)}
|
|
186
|
+
- Brands: {len(brands)}
|
|
187
|
+
- Execution Mode: `{execution_mode}`
|
|
188
|
+
- Research Status: `{research_status}`
|
|
189
|
+
|
|
190
|
+
## Quick Reproduction
|
|
191
|
+
```bash
|
|
192
|
+
geo-scope benchmark verify --dataset .
|
|
193
|
+
geo-scope benchmark reproduce --dataset .
|
|
194
|
+
```
|
|
195
|
+
"""
|
|
196
|
+
(target_dir / "README.md").write_text(readme_md.strip() + "\n", encoding="utf-8")
|
|
197
|
+
|
|
198
|
+
# 4. Compute file hashes (excluding manifest.json and checksums.sha256)
|
|
199
|
+
file_hashes = compute_dataset_checksums(target_dir)
|
|
200
|
+
composite_hash = compute_composite_hash(file_hashes)
|
|
201
|
+
|
|
202
|
+
# 5. Write manifest.json
|
|
203
|
+
p_ids = [p.get("id") or p.get("provider_id", "") for p in providers]
|
|
204
|
+
m_ids = [p.get("model") or p.get("id", "") for p in providers]
|
|
205
|
+
bmk_ver = "2026.1-live" if execution_mode == "live" else "2026.1-synthetic"
|
|
206
|
+
|
|
207
|
+
manifest = BenchmarkManifest(
|
|
208
|
+
version="2026.1",
|
|
209
|
+
benchmark_version=bmk_ver,
|
|
210
|
+
dataset_id=self.dataset_id,
|
|
211
|
+
dataset_name=self.dataset_id,
|
|
212
|
+
created_at=datetime.now(timezone.utc).isoformat(),
|
|
213
|
+
execution_mode=execution_mode,
|
|
214
|
+
research_status=research_status,
|
|
215
|
+
description=description,
|
|
216
|
+
git_commit=get_git_commit(target_dir),
|
|
217
|
+
parser_version="1.0.0",
|
|
218
|
+
providers=p_ids,
|
|
219
|
+
models=m_ids,
|
|
220
|
+
experiment_ids=[self.dataset_id],
|
|
221
|
+
dataset_hash=composite_hash,
|
|
222
|
+
counts={
|
|
223
|
+
"prompts": len(prompts),
|
|
224
|
+
"observed_prompts": len(obs_prompts),
|
|
225
|
+
"generated_prompts": len(gen_prompts),
|
|
226
|
+
"observations": len(observations),
|
|
227
|
+
"citations": len(citations),
|
|
228
|
+
"brands": len(brands),
|
|
229
|
+
"providers": len(providers),
|
|
230
|
+
},
|
|
231
|
+
file_hashes=file_hashes,
|
|
232
|
+
composite_dataset_hash=composite_hash,
|
|
233
|
+
provider_validation=provider_validation,
|
|
234
|
+
benchmark_mode=benchmark_mode,
|
|
235
|
+
model_provenance=model_provenance or {
|
|
236
|
+
"validated": True,
|
|
237
|
+
"fallbacks_recorded": True,
|
|
238
|
+
},
|
|
239
|
+
question_provenance={
|
|
240
|
+
"observed_count": len(obs_prompts),
|
|
241
|
+
"generated_count": len(gen_prompts),
|
|
242
|
+
"source_reference": "answerpath",
|
|
243
|
+
},
|
|
244
|
+
)
|
|
245
|
+
(target_dir / "manifest.json").write_text(
|
|
246
|
+
json.dumps(manifest.model_dump(), ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
|
247
|
+
)
|
|
248
|
+
|
|
249
|
+
# 6. Recompute full checksums including manifest.json and write checksums.sha256
|
|
250
|
+
full_checksums = compute_dataset_checksums(target_dir)
|
|
251
|
+
write_checksums_file(target_dir, full_checksums)
|
|
252
|
+
|
|
253
|
+
return target_dir
|