geo-scope 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. geo_scope/__init__.py +37 -0
  2. geo_scope/benchmark/__init__.py +53 -0
  3. geo_scope/benchmark/builder.py +253 -0
  4. geo_scope/benchmark/calculator.py +479 -0
  5. geo_scope/benchmark/dataset_validator.py +232 -0
  6. geo_scope/benchmark/hasher.py +120 -0
  7. geo_scope/benchmark/models.py +184 -0
  8. geo_scope/benchmark/profile.py +82 -0
  9. geo_scope/benchmark/reproducer.py +196 -0
  10. geo_scope/benchmark/runner.py +668 -0
  11. geo_scope/benchmark/schemas.py +157 -0
  12. geo_scope/benchmark/validator.py +237 -0
  13. geo_scope/cli.py +1193 -0
  14. geo_scope/data/benchmark_history.json +451 -0
  15. geo_scope/engine/__init__.py +0 -0
  16. geo_scope/engine/algo_analyzer.py +361 -0
  17. geo_scope/engine/execution_mode.py +35 -0
  18. geo_scope/engine/export_manager.py +94 -0
  19. geo_scope/engine/feature_extractor.py +318 -0
  20. geo_scope/engine/history_tracker.py +106 -0
  21. geo_scope/engine/model_runner.py +445 -0
  22. geo_scope/engine/persistence.py +114 -0
  23. geo_scope/engine/query_generator.py +333 -0
  24. geo_scope/engine/query_loader.py +186 -0
  25. geo_scope/engine/report_generator.py +352 -0
  26. geo_scope/engine/response_import.py +37 -0
  27. geo_scope/engine/strategy_builder.py +60 -0
  28. geo_scope/entities/__init__.py +8 -0
  29. geo_scope/entities/models.py +37 -0
  30. geo_scope/entities/registry.py +181 -0
  31. geo_scope/mavi/__init__.py +27 -0
  32. geo_scope/mavi/engine.py +274 -0
  33. geo_scope/mavi/geoscope_evaluator.py +170 -0
  34. geo_scope/mavi/models.py +195 -0
  35. geo_scope/mavi/sage_evaluator.py +322 -0
  36. geo_scope/mcp_server.py +294 -0
  37. geo_scope/measurement/__init__.py +8 -0
  38. geo_scope/measurement/engine.py +557 -0
  39. geo_scope/measurement/replay.py +376 -0
  40. geo_scope/parser/__init__.py +11 -0
  41. geo_scope/parser/evaluator.py +217 -0
  42. geo_scope/parser/observation_parser.py +547 -0
  43. geo_scope/providers/__init__.py +26 -0
  44. geo_scope/providers/base.py +241 -0
  45. geo_scope/providers/claude_provider.py +65 -0
  46. geo_scope/providers/gemini_provider.py +63 -0
  47. geo_scope/providers/hamzad_provider.py +251 -0
  48. geo_scope/providers/keyless_wrapper_provider.py +40 -0
  49. geo_scope/providers/models.py +234 -0
  50. geo_scope/providers/ollama_provider.py +47 -0
  51. geo_scope/providers/openai_provider.py +66 -0
  52. geo_scope/providers/openrouter_provider.py +40 -0
  53. geo_scope/providers/perplexity_provider.py +74 -0
  54. geo_scope/providers/public_research_provider.py +24 -0
  55. geo_scope/providers/registry.py +166 -0
  56. geo_scope/providers/simulated.py +232 -0
  57. geo_scope/questions/__init__.py +19 -0
  58. geo_scope/questions/answerpath_connector.py +294 -0
  59. geo_scope/questions/discovery.py +228 -0
  60. geo_scope/questions/models.py +122 -0
  61. geo_scope/release_gate.py +293 -0
  62. geo_scope/server.py +320 -0
  63. geo_scope/static/featured_image_geo.png +0 -0
  64. geo_scope/static/index.html +1433 -0
  65. geo_scope-0.3.0.dist-info/METADATA +371 -0
  66. geo_scope-0.3.0.dist-info/RECORD +70 -0
  67. geo_scope-0.3.0.dist-info/WHEEL +5 -0
  68. geo_scope-0.3.0.dist-info/entry_points.txt +2 -0
  69. geo_scope-0.3.0.dist-info/licenses/LICENSE +21 -0
  70. geo_scope-0.3.0.dist-info/top_level.txt +1 -0
geo_scope/__init__.py ADDED
@@ -0,0 +1,37 @@
1
+ """
2
+ GEO-Scope: Generative Engine Optimization (GEO) & Empirical AI Answer Visibility Framework
3
+ An open-source scientific framework for measuring entity visibility, recommendations, and citations across generative AI systems.
4
+ """
5
+
6
+ __version__ = "0.3.0"
7
+ __author__ = "GEO-Scope Community & Contributors"
8
+ __license__ = "MIT"
9
+
10
+ from geo_scope.engine.query_generator import generate_prompt_dataset, INDUSTRY_PRESETS
11
+ from geo_scope.engine.feature_extractor import (
12
+ parse_model_response,
13
+ extract_citations_and_domains,
14
+ detect_brand_positions,
15
+ )
16
+ from geo_scope.engine.execution_mode import ExecutionMode
17
+ from geo_scope.engine.persistence import RawRunStore
18
+ from geo_scope.engine.model_runner import ModelRunner
19
+ from geo_scope.engine.strategy_builder import generate_geo_playbook
20
+ from geo_scope.mavi import MAVIEngine, MAVIReport, LayerWeights
21
+
22
+ __all__ = [
23
+ "__version__",
24
+ "generate_prompt_dataset",
25
+ "INDUSTRY_PRESETS",
26
+ "parse_model_response",
27
+ "extract_citations_and_domains",
28
+ "detect_brand_positions",
29
+ "ExecutionMode",
30
+ "RawRunStore",
31
+ "ModelRunner",
32
+ "AlgoAnalyzer",
33
+ "generate_geo_playbook",
34
+ "MAVIEngine",
35
+ "MAVIReport",
36
+ "LayerWeights",
37
+ ]
@@ -0,0 +1,53 @@
1
+ """
2
+ GEO-Scope Public Benchmark & Evidence Dataset Module.
3
+ """
4
+
5
+ from geo_scope.benchmark.models import (
6
+ BenchmarkExecutionMode,
7
+ BenchmarkManifest,
8
+ PromptRecord,
9
+ ObservationRecord,
10
+ CitationEvidenceRecord,
11
+ BrandBenchmarkMetrics,
12
+ ProviderBenchmarkMetrics,
13
+ BenchmarkMetrics,
14
+ MetricEstimate,
15
+ StatisticalFactorAnalysis,
16
+ )
17
+ from geo_scope.benchmark.validator import ProviderValidator, ProviderValidationResult
18
+ from geo_scope.benchmark.hasher import (
19
+ compute_file_sha256,
20
+ compute_dataset_checksums,
21
+ compute_composite_hash,
22
+ write_checksums_file,
23
+ verify_dataset_checksums,
24
+ )
25
+ from geo_scope.benchmark.calculator import BenchmarkCalculator, calculate_bootstrap_ci
26
+ from geo_scope.benchmark.builder import BenchmarkBuilder
27
+ from geo_scope.benchmark.reproducer import BenchmarkReproducer
28
+ from geo_scope.benchmark.profile import BenchmarkProfile
29
+ from geo_scope.benchmark.runner import LiveBenchmarkRunner, estimate_benchmark_cost
30
+
31
+ __all__ = [
32
+ "BenchmarkManifest",
33
+ "PromptRecord",
34
+ "ObservationRecord",
35
+ "CitationEvidenceRecord",
36
+ "BrandBenchmarkMetrics",
37
+ "ProviderBenchmarkMetrics",
38
+ "BenchmarkMetrics",
39
+ "MetricEstimate",
40
+ "StatisticalFactorAnalysis",
41
+ "compute_file_sha256",
42
+ "compute_dataset_checksums",
43
+ "compute_composite_hash",
44
+ "write_checksums_file",
45
+ "verify_dataset_checksums",
46
+ "BenchmarkCalculator",
47
+ "calculate_bootstrap_ci",
48
+ "BenchmarkBuilder",
49
+ "BenchmarkReproducer",
50
+ "BenchmarkProfile",
51
+ "LiveBenchmarkRunner",
52
+ "estimate_benchmark_cost",
53
+ ]
@@ -0,0 +1,253 @@
1
+ # Dataset package builder for GEO-Scope Public Benchmark v1.
2
+ import json
3
+ import os
4
+ import subprocess
5
+ from datetime import datetime, timezone
6
+ from pathlib import Path
7
+ from typing import Dict, List, Any, Optional
8
+
9
+ from geo_scope.benchmark.models import BenchmarkManifest, BenchmarkMetrics
10
+ from geo_scope.benchmark.calculator import BenchmarkCalculator
11
+ from geo_scope.benchmark.hasher import (
12
+ compute_dataset_checksums,
13
+ compute_composite_hash,
14
+ write_checksums_file,
15
+ )
16
+
17
+
18
+ def get_git_commit(repo_dir: Optional[str | Path] = None) -> Optional[str]:
19
+ """Attempt to get the current git commit hash."""
20
+ try:
21
+ cmd = ["git", "rev-parse", "HEAD"]
22
+ env = os.environ.copy()
23
+ env["DEVELOPER_DIR"] = "/Library/Developer/CommandLineTools"
24
+ res = subprocess.run(cmd, cwd=repo_dir, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, env=env)
25
+ if res.returncode == 0:
26
+ return res.stdout.strip()
27
+ except Exception:
28
+ pass
29
+ return None
30
+
31
+
32
+ class BenchmarkBuilder:
33
+ """
34
+ Builds a complete, versioned, tamper-proof GEO-Scope benchmark dataset package.
35
+ """
36
+
37
+ def __init__(self, dataset_id: str = "geo-scope-benchmark-2026.1"):
38
+ self.dataset_id = dataset_id
39
+ self.calculator = BenchmarkCalculator()
40
+
41
+ def build_package(
42
+ self,
43
+ out_dir: str | Path,
44
+ prompts: List[Dict[str, Any]],
45
+ observations: List[Dict[str, Any]],
46
+ citations: List[Dict[str, Any]],
47
+ brands: List[Dict[str, Any]],
48
+ providers: List[Dict[str, Any]],
49
+ execution_mode: str = "synthetic",
50
+ benchmark_mode: str = "discovery",
51
+ research_status: str = "demo_only",
52
+ description: Optional[str] = None,
53
+ methodology_md: Optional[str] = None,
54
+ readme_md: Optional[str] = None,
55
+ provider_validation: Optional[Dict[str, Any]] = None,
56
+ model_provenance: Optional[Dict[str, Any]] = None,
57
+ ) -> Path:
58
+ target_dir = Path(out_dir) / self.dataset_id
59
+ target_dir.mkdir(parents=True, exist_ok=True)
60
+
61
+ if description is None:
62
+ description = (
63
+ f"GEO-Scope Public Benchmark {self.dataset_id} containing multi-model "
64
+ f"observations across {len(providers)} providers and {len(prompts)} prompts."
65
+ )
66
+
67
+ # 1. Write data files
68
+ prompts_file = target_dir / "prompts.jsonl"
69
+ with open(prompts_file, "w", encoding="utf-8") as f:
70
+ for p in prompts:
71
+ f.write(json.dumps(p, ensure_ascii=False) + "\n")
72
+
73
+ # Write partitioned prompts directory
74
+ prompts_sub_dir = target_dir / "prompts"
75
+ prompts_sub_dir.mkdir(parents=True, exist_ok=True)
76
+ obs_prompts = [p for p in prompts if p.get("source_type") == "observed"]
77
+ gen_prompts = [p for p in prompts if p.get("source_type") != "observed"]
78
+
79
+ with open(prompts_sub_dir / "observed.jsonl", "w", encoding="utf-8") as f:
80
+ for p in obs_prompts:
81
+ f.write(json.dumps(p, ensure_ascii=False) + "\n")
82
+
83
+ with open(prompts_sub_dir / "generated.jsonl", "w", encoding="utf-8") as f:
84
+ for p in gen_prompts:
85
+ f.write(json.dumps(p, ensure_ascii=False) + "\n")
86
+
87
+ (target_dir / "entities.json").write_text(
88
+ json.dumps(brands, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
89
+ )
90
+ (target_dir / "brands.json").write_text(
91
+ json.dumps(brands, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
92
+ )
93
+
94
+ (target_dir / "providers.json").write_text(
95
+ json.dumps(providers, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
96
+ )
97
+
98
+ obs_file = target_dir / "observations.jsonl"
99
+ with open(obs_file, "w", encoding="utf-8") as f:
100
+ for o in observations:
101
+ f.write(json.dumps(o, ensure_ascii=False) + "\n")
102
+
103
+ cits_file = target_dir / "citations.jsonl"
104
+ with open(cits_file, "w", encoding="utf-8") as f:
105
+ for c in citations:
106
+ f.write(json.dumps(c, ensure_ascii=False) + "\n")
107
+
108
+ # 2. Compute and write metrics.json
109
+ metrics_obj = self.calculator.compute(
110
+ prompts=prompts,
111
+ observations=observations,
112
+ citations=citations,
113
+ brands=brands,
114
+ providers=providers,
115
+ dataset_id=self.dataset_id,
116
+ execution_mode=execution_mode,
117
+ benchmark_mode=benchmark_mode,
118
+ research_status=research_status,
119
+ )
120
+ (target_dir / "metrics.json").write_text(
121
+ json.dumps(metrics_obj.model_dump(), ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
122
+ )
123
+
124
+ # Write provenance.json
125
+ provenance_payload = {
126
+ "dataset_id": self.dataset_id,
127
+ "created_at": datetime.now(timezone.utc).isoformat(),
128
+ "execution_mode": execution_mode,
129
+ "benchmark_mode": benchmark_mode,
130
+ "question_provenance": {
131
+ "total_prompts": len(prompts),
132
+ "observed_count": len(obs_prompts),
133
+ "generated_count": len(gen_prompts),
134
+ "source_reference": "answerpath",
135
+ "observed_sources": list({p.get("source_reference", "observed") for p in obs_prompts}),
136
+ "generated_sources": list({p.get("source_reference", "generated") for p in gen_prompts}),
137
+ },
138
+ "model_provenance": model_provenance or {
139
+ "validated": True,
140
+ "fallbacks_recorded": True,
141
+ },
142
+ }
143
+ (target_dir / "provenance.json").write_text(
144
+ json.dumps(provenance_payload, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
145
+ )
146
+
147
+ # 3. Write documentation
148
+ if methodology_md is None:
149
+ methodology_md = f"""# GEO-Scope Benchmark Methodology ({self.dataset_id})
150
+
151
+ ## Overview
152
+ This benchmark evaluates Generative Engine Optimization (GEO) performance, brand visibility, and citation presence across multi-model AI engines.
153
+
154
+ ## Execution Mode & Status
155
+ - **Execution Mode**: `{execution_mode}`
156
+ - **Research Status**: `{research_status}`
157
+ - **Strict Separation**: Synthetic simulation runs are explicitly marked `demo_only` and must not be cited as real provider behavior.
158
+
159
+ ## Question & Model Provenance
160
+ - **Observed Prompts**: {len(obs_prompts)} (Real user demand)
161
+ - **Generated Prompts**: {len(gen_prompts)} (Research exploration templates)
162
+ - **Model Provenance**: Explicit tracking of native vs fallback routes.
163
+
164
+ ## Measured Metrics
165
+ 1. **Share of Model (SoM)**: Percentage of total observed brand mentions attributed to the brand.
166
+ 2. **Mention Rate**: Percentage of successful multi-model observations containing the brand (with 95% bootstrap CI).
167
+ 3. **Top-1 Primary Rate**: Percentage of successful observations where the brand is the first/primary recommendation.
168
+ 4. **Citation Rate**: Percentage of observations citing the brand or authoritative third-party source.
169
+
170
+ ## Statistical Bounds & Language Guardrails
171
+ - All confidence intervals are non-parametric 95% percentile bootstrap estimates (1,000 resamples).
172
+ - Factor analyses report **observed empirical correlations** only, avoiding speculative "AI ranking algorithm" assertions.
173
+ """
174
+ (target_dir / "methodology.md").write_text(methodology_md.strip() + "\n", encoding="utf-8")
175
+
176
+ if readme_md is None:
177
+ readme_md = f"""# {self.dataset_id}
178
+
179
+ GEO-Scope Public Benchmark & Evidence Dataset.
180
+
181
+ - Total Prompts: {len(prompts)}
182
+ - Observed User Questions: {len(obs_prompts)}
183
+ - Generated Research Prompts: {len(gen_prompts)}
184
+ - Total Observations: {len(observations)}
185
+ - Providers: {len(providers)}
186
+ - Brands: {len(brands)}
187
+ - Execution Mode: `{execution_mode}`
188
+ - Research Status: `{research_status}`
189
+
190
+ ## Quick Reproduction
191
+ ```bash
192
+ geo-scope benchmark verify --dataset .
193
+ geo-scope benchmark reproduce --dataset .
194
+ ```
195
+ """
196
+ (target_dir / "README.md").write_text(readme_md.strip() + "\n", encoding="utf-8")
197
+
198
+ # 4. Compute file hashes (excluding manifest.json and checksums.sha256)
199
+ file_hashes = compute_dataset_checksums(target_dir)
200
+ composite_hash = compute_composite_hash(file_hashes)
201
+
202
+ # 5. Write manifest.json
203
+ p_ids = [p.get("id") or p.get("provider_id", "") for p in providers]
204
+ m_ids = [p.get("model") or p.get("id", "") for p in providers]
205
+ bmk_ver = "2026.1-live" if execution_mode == "live" else "2026.1-synthetic"
206
+
207
+ manifest = BenchmarkManifest(
208
+ version="2026.1",
209
+ benchmark_version=bmk_ver,
210
+ dataset_id=self.dataset_id,
211
+ dataset_name=self.dataset_id,
212
+ created_at=datetime.now(timezone.utc).isoformat(),
213
+ execution_mode=execution_mode,
214
+ research_status=research_status,
215
+ description=description,
216
+ git_commit=get_git_commit(target_dir),
217
+ parser_version="1.0.0",
218
+ providers=p_ids,
219
+ models=m_ids,
220
+ experiment_ids=[self.dataset_id],
221
+ dataset_hash=composite_hash,
222
+ counts={
223
+ "prompts": len(prompts),
224
+ "observed_prompts": len(obs_prompts),
225
+ "generated_prompts": len(gen_prompts),
226
+ "observations": len(observations),
227
+ "citations": len(citations),
228
+ "brands": len(brands),
229
+ "providers": len(providers),
230
+ },
231
+ file_hashes=file_hashes,
232
+ composite_dataset_hash=composite_hash,
233
+ provider_validation=provider_validation,
234
+ benchmark_mode=benchmark_mode,
235
+ model_provenance=model_provenance or {
236
+ "validated": True,
237
+ "fallbacks_recorded": True,
238
+ },
239
+ question_provenance={
240
+ "observed_count": len(obs_prompts),
241
+ "generated_count": len(gen_prompts),
242
+ "source_reference": "answerpath",
243
+ },
244
+ )
245
+ (target_dir / "manifest.json").write_text(
246
+ json.dumps(manifest.model_dump(), ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
247
+ )
248
+
249
+ # 6. Recompute full checksums including manifest.json and write checksums.sha256
250
+ full_checksums = compute_dataset_checksums(target_dir)
251
+ write_checksums_file(target_dir, full_checksums)
252
+
253
+ return target_dir