standpoint 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- standpoint/__init__.py +1699 -0
- standpoint/__main__.py +4 -0
- standpoint/click_cli.py +59 -0
- standpoint/i18n.yaml +146 -0
- standpoint-0.2.0.dist-info/METADATA +309 -0
- standpoint-0.2.0.dist-info/RECORD +10 -0
- standpoint-0.2.0.dist-info/WHEEL +5 -0
- standpoint-0.2.0.dist-info/entry_points.txt +3 -0
- standpoint-0.2.0.dist-info/licenses/LICENSE +29 -0
- standpoint-0.2.0.dist-info/top_level.txt +1 -0
standpoint/__init__.py
ADDED
|
@@ -0,0 +1,1699 @@
|
|
|
1
|
+
"""Standpoint — know where each option actually stands.
|
|
2
|
+
|
|
3
|
+
Explainable 2D PCA positioning map from any comparison table.
|
|
4
|
+
|
|
5
|
+
Turn a table of *approaches x criteria* (CSV or Markdown, numeric ratings on any
|
|
6
|
+
scale) into a competitive positioning map, plus a written interpretation and a
|
|
7
|
+
full dump of the coefficients — a three-fold deliverable from one input file.
|
|
8
|
+
|
|
9
|
+
Pipeline
|
|
10
|
+
--------
|
|
11
|
+
1. parse : CSV or Markdown table -> numeric DataFrame (blanks -> minimum value
|
|
12
|
+
of the non-blank, non-NaN values in that column).
|
|
13
|
+
2. prepare : normalization (default = z-score standardization, i.e. correlation
|
|
14
|
+
PCA, because PCA is scale-sensitive and criteria carry different
|
|
15
|
+
variances). Missing cells are imputed with the column minimum.
|
|
16
|
+
3. pca_2d : PCA onto 2 components, keeping the canonical axes (loadings) so
|
|
17
|
+
every axis stays a readable linear combination of the criteria.
|
|
18
|
+
4. orient : rigidly rotate the 2D scatter so the reference row (the first row by
|
|
19
|
+
default) leads in the TOP-RIGHT, and reposition an all-max reference
|
|
20
|
+
to the best Pareto point; RECOMPUTE the canonical axes in the rotated
|
|
21
|
+
frame (new_components = R(alpha) @ components).
|
|
22
|
+
|
|
23
|
+
Then: automatic roles by principled projection, distinct OKLCH colours by map
|
|
24
|
+
position, local-LLM axis pole names from the loadings, and a de-cluttered
|
|
25
|
+
Vega-Lite figure. `export_all` writes PNG + SVG + Vega JSON + a Markdown analysis
|
|
26
|
+
+ a YAML of coordinates and coefficients.
|
|
27
|
+
|
|
28
|
+
Author
|
|
29
|
+
------
|
|
30
|
+
Warith Harchaoui — https://www.linkedin.com/in/warith-harchaoui
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
from __future__ import annotations
|
|
34
|
+
|
|
35
|
+
__author__ = "Warith Harchaoui"
|
|
36
|
+
__url__ = "https://www.linkedin.com/in/warith-harchaoui"
|
|
37
|
+
__version__ = "0.2.0"
|
|
38
|
+
|
|
39
|
+
import argparse
|
|
40
|
+
import json
|
|
41
|
+
import logging
|
|
42
|
+
import math
|
|
43
|
+
import os
|
|
44
|
+
import re
|
|
45
|
+
from dataclasses import dataclass
|
|
46
|
+
|
|
47
|
+
import numpy as np
|
|
48
|
+
import ollama
|
|
49
|
+
import pandas as pd
|
|
50
|
+
import vl_convert as vlc
|
|
51
|
+
import yaml
|
|
52
|
+
from langdetect import DetectorFactory
|
|
53
|
+
from langdetect import detect as _langdetect
|
|
54
|
+
from sklearn.decomposition import PCA
|
|
55
|
+
from sklearn.preprocessing import StandardScaler
|
|
56
|
+
|
|
57
|
+
DetectorFactory.seed = 0 # deterministic language detection
|
|
58
|
+
|
|
59
|
+
# Library diagnostics go through logging, never bare print (a library must not
|
|
60
|
+
# grab stdout). The CLI in `run()` is the one place that prints on purpose.
|
|
61
|
+
logger = logging.getLogger("standpoint")
|
|
62
|
+
|
|
63
|
+
# "Good Colors" Apple-base palette — https://harchaoui.org/warith/colors/.
|
|
64
|
+
# The four highlighted roles keep a fixed identity hue; the axis cross and labels
|
|
65
|
+
# use neutrals. Every other dot is coloured by its map position (`gradient_colors`).
|
|
66
|
+
PALETTE = {
|
|
67
|
+
"reference": "#FF3B30", # Red — the reference leader (best), sits top-right
|
|
68
|
+
"right": "#007AFF", # Blue — challenger that most defines the right pole
|
|
69
|
+
"worst": "#A52A2A", # Brown — weakest overall, sits bottom-left
|
|
70
|
+
"top": "#AF52DE", # Purple — challenger that most defines the top pole
|
|
71
|
+
"competitor": "#8E8E93", # Gray — placeholder; overridden by gradient_colors
|
|
72
|
+
"axis": "#C7C7CC", # light gray for the centred, dotted axis cross
|
|
73
|
+
"label": "#1C1C1E", # near-black label text
|
|
74
|
+
}
|
|
75
|
+
FONT = "Roboto, -apple-system, Helvetica, Arial, sans-serif"
|
|
76
|
+
|
|
77
|
+
# One qwen vision-LLM for everything: axis pole names, the written analysis, and
|
|
78
|
+
# the visual assessment of the rendered figure (see `vlm_assess`).
|
|
79
|
+
DEFAULT_MODEL = "qwen2.5vl:7b"
|
|
80
|
+
|
|
81
|
+
__all__ = [
|
|
82
|
+
"positioning",
|
|
83
|
+
"Positioning",
|
|
84
|
+
"parse_table",
|
|
85
|
+
"analyze",
|
|
86
|
+
"PCAResult",
|
|
87
|
+
"assign_roles",
|
|
88
|
+
"axis_poles",
|
|
89
|
+
"gradient_colors",
|
|
90
|
+
"to_vega",
|
|
91
|
+
"render_figures",
|
|
92
|
+
"png_on_white",
|
|
93
|
+
"export_all",
|
|
94
|
+
"analysis_markdown",
|
|
95
|
+
"results_yaml",
|
|
96
|
+
"validate_table",
|
|
97
|
+
"resolve_polarity",
|
|
98
|
+
"detect_language",
|
|
99
|
+
"i18n",
|
|
100
|
+
"vlm_assess",
|
|
101
|
+
"run",
|
|
102
|
+
"main",
|
|
103
|
+
]
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
# --------------------------------------------------------------------------- #
|
|
107
|
+
# 1. parse
|
|
108
|
+
# --------------------------------------------------------------------------- #
|
|
109
|
+
def _cell_to_number(cell: str) -> float:
|
|
110
|
+
"""Convert one table cell to a number (int or float); blanks -> NaN."""
|
|
111
|
+
cell = cell.replace("**", "").strip()
|
|
112
|
+
if cell.lower() in {"", "-", "—", "?", "n/a", "na", "null", "none"}:
|
|
113
|
+
return np.nan
|
|
114
|
+
try:
|
|
115
|
+
return float(cell.replace(",", "."))
|
|
116
|
+
except ValueError:
|
|
117
|
+
return np.nan
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _looks_like_markdown(text: str) -> bool:
|
|
121
|
+
"""True if any line starts with a pipe, i.e. the text is a Markdown table."""
|
|
122
|
+
return any(line.lstrip().startswith("|") for line in text.splitlines())
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _parse_markdown(text: str) -> pd.DataFrame:
|
|
126
|
+
"""Parse a GitHub-flavoured Markdown table into a numeric DataFrame.
|
|
127
|
+
|
|
128
|
+
The first pipe-delimited row is the header (its first cell names the index);
|
|
129
|
+
the separator row (only pipes/dashes/colons) is dropped, and every remaining
|
|
130
|
+
cell is coerced to a number via `_cell_to_number`.
|
|
131
|
+
"""
|
|
132
|
+
rows = [ln.strip() for ln in text.splitlines() if ln.strip().startswith("|")]
|
|
133
|
+
# A GitHub separator row is only pipes/dashes/colons/spaces.
|
|
134
|
+
rows = [r for r in rows if not re.fullmatch(r"[|\s:\-]+", r)]
|
|
135
|
+
|
|
136
|
+
def split(row: str) -> list[str]:
|
|
137
|
+
"""Split one table row into stripped cell strings, dropping edge pipes."""
|
|
138
|
+
return [c.strip() for c in row.strip().strip("|").split("|")]
|
|
139
|
+
|
|
140
|
+
header = split(rows[0])
|
|
141
|
+
records, index = [], []
|
|
142
|
+
for row in rows[1:]:
|
|
143
|
+
cells = split(row)
|
|
144
|
+
name = cells[0].replace("**", "").strip()
|
|
145
|
+
index.append(name)
|
|
146
|
+
records.append([_cell_to_number(c) for c in cells[1:]])
|
|
147
|
+
frame = pd.DataFrame(records, index=index, columns=header[1:])
|
|
148
|
+
frame.index.name = header[0] # keep the first-column name (e.g. "Language")
|
|
149
|
+
return frame
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def parse_table(source: str) -> pd.DataFrame:
|
|
153
|
+
"""Parse a markdown/CSV table (path or raw string) into a numeric DataFrame.
|
|
154
|
+
|
|
155
|
+
The first column becomes the row index (approach names); every other cell is
|
|
156
|
+
parsed as a number (int or float); blanks become NaN.
|
|
157
|
+
"""
|
|
158
|
+
text = source
|
|
159
|
+
is_path = "\n" not in source and len(source) < 4096
|
|
160
|
+
if is_path:
|
|
161
|
+
try:
|
|
162
|
+
with open(source, encoding="utf-8") as fh:
|
|
163
|
+
text = fh.read()
|
|
164
|
+
except (OSError, ValueError):
|
|
165
|
+
text, is_path = source, False # not a real path -> treat as raw text
|
|
166
|
+
|
|
167
|
+
if _looks_like_markdown(text):
|
|
168
|
+
return _parse_markdown(text)
|
|
169
|
+
|
|
170
|
+
df = pd.read_csv(source if is_path else pd.io.common.StringIO(text), index_col=0)
|
|
171
|
+
return df.map(lambda c: _cell_to_number(str(c)))
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
# --------------------------------------------------------------------------- #
|
|
175
|
+
# i18n — detect the table's language and localize the LLM prompts
|
|
176
|
+
# --------------------------------------------------------------------------- #
|
|
177
|
+
SUPPORTED_LANGS = ("en", "fr", "es")
|
|
178
|
+
_I18N_PATH = os.path.join(os.path.dirname(os.path.abspath(__file__)), "i18n.yaml")
|
|
179
|
+
_I18N_CACHE: dict | None = None
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def i18n(lang: str = "en") -> dict:
|
|
183
|
+
"""Prompt templates for `lang` (falls back to English), loaded from i18n.yaml."""
|
|
184
|
+
global _I18N_CACHE
|
|
185
|
+
if _I18N_CACHE is None:
|
|
186
|
+
with open(_I18N_PATH, encoding="utf-8") as fh:
|
|
187
|
+
_I18N_CACHE = yaml.safe_load(fh)
|
|
188
|
+
return _I18N_CACHE.get(lang, _I18N_CACHE["en"])
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def detect_language(texts: list[str]) -> str:
|
|
192
|
+
"""Detect the language (one of SUPPORTED_LANGS) from text; default English.
|
|
193
|
+
|
|
194
|
+
Used on the table's column names so the pole labels and written analysis come
|
|
195
|
+
out in the table's own language.
|
|
196
|
+
"""
|
|
197
|
+
sample = " ".join(t for t in texts if t).strip()
|
|
198
|
+
if not sample:
|
|
199
|
+
return "en"
|
|
200
|
+
try:
|
|
201
|
+
lang = _langdetect(sample)
|
|
202
|
+
except Exception:
|
|
203
|
+
return "en"
|
|
204
|
+
return lang if lang in SUPPORTED_LANGS else "en"
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
# --------------------------------------------------------------------------- #
|
|
208
|
+
# 2. prepare (normalization / preprocessing)
|
|
209
|
+
# --------------------------------------------------------------------------- #
|
|
210
|
+
def validate_table(df: pd.DataFrame) -> None:
|
|
211
|
+
"""Raise a clear ``ValueError`` if the table can't be positioned.
|
|
212
|
+
|
|
213
|
+
Needs at least 2 options (rows) and 2 numeric criteria (columns) with some
|
|
214
|
+
variation, and no fully-empty column — otherwise PCA is undefined or degenerate.
|
|
215
|
+
"""
|
|
216
|
+
if df.shape[0] < 2:
|
|
217
|
+
raise ValueError(f"need at least 2 options (rows); got {df.shape[0]}.")
|
|
218
|
+
if df.shape[1] < 2:
|
|
219
|
+
raise ValueError(f"need at least 2 criteria (columns); got {df.shape[1]}.")
|
|
220
|
+
all_nan = [c for c in df.columns if df[c].isna().all()]
|
|
221
|
+
if all_nan:
|
|
222
|
+
raise ValueError(f"criteria with no numeric values at all: {all_nan}.")
|
|
223
|
+
constant = [c for c in df.columns if df[c].nunique(dropna=True) <= 1]
|
|
224
|
+
if len(constant) == df.shape[1]:
|
|
225
|
+
raise ValueError("every criterion is constant; nothing to position.")
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def _resolve_reference(df: pd.DataFrame, reference: int | str) -> int:
|
|
229
|
+
"""Return the row index of the reference, with a helpful error if it's unknown."""
|
|
230
|
+
if isinstance(reference, str):
|
|
231
|
+
if reference not in df.index:
|
|
232
|
+
raise ValueError(f"reference {reference!r} is not one of the options.")
|
|
233
|
+
return int(df.index.get_loc(reference))
|
|
234
|
+
if not -df.shape[0] <= reference < df.shape[0]:
|
|
235
|
+
raise ValueError(f"reference index {reference} is out of range (0..{df.shape[0] - 1}).")
|
|
236
|
+
return int(reference % df.shape[0])
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def impute(df: pd.DataFrame) -> pd.DataFrame:
|
|
240
|
+
"""Fill missing cells with each column's minimum observed value.
|
|
241
|
+
|
|
242
|
+
A blank criterion is treated as the worst (minimum) value for that criterion,
|
|
243
|
+
rather than the mean — a missing rating should not flatter an approach.
|
|
244
|
+
"""
|
|
245
|
+
return df.fillna(df.min(numeric_only=True))
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
# A header marker declaring a criterion as lower-is-better, e.g. "Price (↓)",
|
|
249
|
+
# "Latency (lower)", "Errors (lower is better)". Stripped from the shown name.
|
|
250
|
+
_LOWER_MARK = re.compile(
|
|
251
|
+
r"\s*\(?\s*(↓|lower(?:\s+is\s+better)?|less\s+is\s+better)\s*\)?\s*$", re.I
|
|
252
|
+
)
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def resolve_polarity(
|
|
256
|
+
df: pd.DataFrame, lower_is_better: list[str] | None = None
|
|
257
|
+
) -> tuple[pd.DataFrame, frozenset[str]]:
|
|
258
|
+
"""Detect lower-is-better criteria and return a clean-named copy + their names.
|
|
259
|
+
|
|
260
|
+
A criterion is lower-is-better if its header carries a marker (``Price (↓)``,
|
|
261
|
+
``Latency (lower)``) or is named in `lower_is_better`. Markers are stripped from
|
|
262
|
+
the column name; the returned set uses the cleaned names.
|
|
263
|
+
"""
|
|
264
|
+
explicit = {c.strip() for c in (lower_is_better or [])}
|
|
265
|
+
rename, lower = {}, set()
|
|
266
|
+
for col in df.columns:
|
|
267
|
+
clean = _LOWER_MARK.sub("", str(col)).strip()
|
|
268
|
+
if clean != col: # had a marker
|
|
269
|
+
lower.add(clean)
|
|
270
|
+
if clean in explicit or col in explicit:
|
|
271
|
+
lower.add(clean)
|
|
272
|
+
rename[col] = clean
|
|
273
|
+
out = df.rename(columns=rename)
|
|
274
|
+
return out, frozenset(lower & set(out.columns))
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
def prepare(df: pd.DataFrame) -> tuple[np.ndarray, list[str]]:
|
|
278
|
+
"""Impute missing cells (column minimum), then z-score standardize -> correlation PCA.
|
|
279
|
+
|
|
280
|
+
Standardization (mean 0, sd 1 per criterion) is the right normalization here,
|
|
281
|
+
always: PCA is scale-sensitive, and criteria live on different scales and units,
|
|
282
|
+
so each must get an equal say. A criterion with a larger numeric spread would
|
|
283
|
+
otherwise dominate the components purely because of its units.
|
|
284
|
+
"""
|
|
285
|
+
x = StandardScaler().fit_transform(impute(df).to_numpy(dtype=float))
|
|
286
|
+
return x, list(df.columns)
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
# --------------------------------------------------------------------------- #
|
|
290
|
+
# 3./4. PCA + orientation
|
|
291
|
+
# --------------------------------------------------------------------------- #
|
|
292
|
+
def _rotation(alpha: float) -> np.ndarray:
|
|
293
|
+
"""The 2x2 counter-clockwise rotation matrix for an angle `alpha` (radians)."""
|
|
294
|
+
c, s = np.cos(alpha), np.sin(alpha)
|
|
295
|
+
return np.array([[c, -s], [s, c]])
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
@dataclass
|
|
299
|
+
class PCAResult:
|
|
300
|
+
names: list[str] # row labels
|
|
301
|
+
features: list[str] # attribute names
|
|
302
|
+
scores: np.ndarray # (n, 2) oriented coordinates
|
|
303
|
+
components: np.ndarray # (2, p) oriented canonical axes (loadings)
|
|
304
|
+
explained_variance_ratio: np.ndarray # from the original PCA fit
|
|
305
|
+
rotation_deg: float # alpha applied, in degrees
|
|
306
|
+
reference: str # row placed top-right
|
|
307
|
+
x_std: np.ndarray # (n, p) normalized feature matrix (PCA input)
|
|
308
|
+
lower: frozenset[str] = frozenset() # criteria where lower is better (negated)
|
|
309
|
+
|
|
310
|
+
def loadings(self) -> pd.DataFrame:
|
|
311
|
+
"""Criterion weights per oriented axis, as a features x (axis-1, axis-2) frame."""
|
|
312
|
+
return pd.DataFrame(self.components.T, index=self.features, columns=["axis-1", "axis-2"])
|
|
313
|
+
|
|
314
|
+
def coords(self) -> pd.DataFrame:
|
|
315
|
+
"""Oriented (axis-1, axis-2) coordinates, one row per option."""
|
|
316
|
+
return pd.DataFrame(self.scores, index=self.names, columns=["axis-1", "axis-2"])
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
def analyze(
|
|
320
|
+
df: pd.DataFrame,
|
|
321
|
+
reference: int | str = 0,
|
|
322
|
+
soften_reference: float = 1.0,
|
|
323
|
+
lower_is_better: list[str] | None = None,
|
|
324
|
+
) -> PCAResult:
|
|
325
|
+
"""Run the full pipeline: prepare -> PCA(2) -> rotate reference to top-right.
|
|
326
|
+
|
|
327
|
+
The reference row is rotated onto the +45 deg diagonal (equal, positive
|
|
328
|
+
coordinates = top-right corner). The canonical axes are then recomputed in
|
|
329
|
+
the rotated frame so their loadings describe the *displayed* axes.
|
|
330
|
+
|
|
331
|
+
`soften_reference` repositions an all-max reference (a straight-5-stars first
|
|
332
|
+
row otherwise lands as a far outlier) to the best **Pareto** point: max x and
|
|
333
|
+
max y of the competitors, times this factor (default 1.0 = exactly best-in-class
|
|
334
|
+
on each axis, so it weakly dominates everyone without being an outlier). Set to
|
|
335
|
+
0 or None to keep the raw PCA position.
|
|
336
|
+
|
|
337
|
+
`lower_is_better` names criteria where a lower value is better (price, latency).
|
|
338
|
+
They are negated before the PCA so the whole space is uniformly higher-is-better;
|
|
339
|
+
header markers like ``Price (↓)`` are picked up automatically too.
|
|
340
|
+
"""
|
|
341
|
+
df, lower = resolve_polarity(df, lower_is_better)
|
|
342
|
+
validate_table(df)
|
|
343
|
+
ref_idx = _resolve_reference(df, reference)
|
|
344
|
+
signed = df.copy()
|
|
345
|
+
if lower:
|
|
346
|
+
signed[list(lower)] = -signed[list(lower)] # flip so higher is better
|
|
347
|
+
x, features = prepare(signed)
|
|
348
|
+
|
|
349
|
+
pca = PCA(n_components=2)
|
|
350
|
+
scores = pca.fit_transform(x) # (n, 2) in original PC frame
|
|
351
|
+
components = pca.components_ # (2, p) rows = PC1, PC2
|
|
352
|
+
|
|
353
|
+
ref_vec = scores[ref_idx]
|
|
354
|
+
phi = np.arctan2(ref_vec[1], ref_vec[0]) # current angle of the reference
|
|
355
|
+
alpha = np.pi / 4 - phi # rotate it onto +45 deg
|
|
356
|
+
|
|
357
|
+
r = _rotation(alpha)
|
|
358
|
+
scores_rot = scores @ r.T # rotate every point
|
|
359
|
+
components_rot = r @ components # recompute canonical axes
|
|
360
|
+
|
|
361
|
+
if soften_reference:
|
|
362
|
+
# Place the reference at the best *Pareto* point: just beyond best-in-class
|
|
363
|
+
# on each axis, so it weakly dominates every competitor without being a far
|
|
364
|
+
# outlier. Realistic leader, top-right, on the frontier.
|
|
365
|
+
others = np.delete(scores_rot, ref_idx, axis=0)
|
|
366
|
+
ideal_x = max(float(others[:, 0].max()), 0.0) * soften_reference
|
|
367
|
+
ideal_y = max(float(others[:, 1].max()), 0.0) * soften_reference
|
|
368
|
+
if ideal_x > 0 and ideal_y > 0:
|
|
369
|
+
scores_rot[ref_idx] = [ideal_x, ideal_y]
|
|
370
|
+
|
|
371
|
+
# Centre the cloud on the origin (mid-range), so the axis cross sits in its
|
|
372
|
+
# middle with equal margins on every side. Because the reference is the max on
|
|
373
|
+
# both axes, this leaves it at the exact top-right corner.
|
|
374
|
+
scores_rot = scores_rot - (scores_rot.max(axis=0) + scores_rot.min(axis=0)) / 2
|
|
375
|
+
|
|
376
|
+
return PCAResult(
|
|
377
|
+
names=list(df.index),
|
|
378
|
+
features=features,
|
|
379
|
+
scores=scores_rot,
|
|
380
|
+
components=components_rot,
|
|
381
|
+
explained_variance_ratio=pca.explained_variance_ratio_,
|
|
382
|
+
rotation_deg=float(np.degrees(alpha)),
|
|
383
|
+
reference=str(df.index[ref_idx]),
|
|
384
|
+
x_std=x,
|
|
385
|
+
lower=lower,
|
|
386
|
+
)
|
|
387
|
+
|
|
388
|
+
|
|
389
|
+
# --------------------------------------------------------------------------- #
|
|
390
|
+
# roles (colour semantics)
|
|
391
|
+
# --------------------------------------------------------------------------- #
|
|
392
|
+
# Four highlighted roles, each a *domain-agnostic* pick from the map geometry
|
|
393
|
+
# (see `assign_roles`): the leader, the weakest, and the two challengers that
|
|
394
|
+
# reach furthest toward the top and right poles. Highest priority last (wins
|
|
395
|
+
# ties): competitor < right < top < worst < best.
|
|
396
|
+
ROLE_ORDER = ["competitor", "right", "top", "worst", "best"]
|
|
397
|
+
ROLE_STYLE = {
|
|
398
|
+
"best": {"color": PALETTE["reference"], "size": 170, "bold": True},
|
|
399
|
+
"worst": {"color": PALETTE["worst"], "size": 120, "bold": True},
|
|
400
|
+
"top": {"color": PALETTE["top"], "size": 120, "bold": True},
|
|
401
|
+
"right": {"color": PALETTE["right"], "size": 120, "bold": True},
|
|
402
|
+
"competitor": {"color": PALETTE["competitor"], "size": 70, "bold": False},
|
|
403
|
+
}
|
|
404
|
+
|
|
405
|
+
|
|
406
|
+
def _rgb_to_hex(rgb: tuple[float, float, float]) -> str:
|
|
407
|
+
"""Convert an (r, g, b) triple in [0, 1] to a clamped ``#RRGGBB`` hex string."""
|
|
408
|
+
r, g, b = (max(0, min(255, round(c * 255))) for c in rgb)
|
|
409
|
+
return f"#{r:02X}{g:02X}{b:02X}"
|
|
410
|
+
|
|
411
|
+
|
|
412
|
+
def _oklab_to_hex(lightness: float, a: float, b: float) -> str:
|
|
413
|
+
"""Convert an OKLab colour (Ottosson 2020) to a clamped sRGB hex string."""
|
|
414
|
+
l_ = lightness + 0.3963377774 * a + 0.2158037573 * b
|
|
415
|
+
m_ = lightness - 0.1055613458 * a - 0.0638541728 * b
|
|
416
|
+
s_ = lightness - 0.0894841775 * a - 1.2914855480 * b
|
|
417
|
+
lc, mc, sc = l_**3, m_**3, s_**3
|
|
418
|
+
rgb_lin = (
|
|
419
|
+
+4.0767416621 * lc - 3.3077115913 * mc + 0.2309699292 * sc,
|
|
420
|
+
-1.2684380046 * lc + 2.6097574011 * mc - 0.3413193965 * sc,
|
|
421
|
+
-0.0041960863 * lc - 0.7034186147 * mc + 1.7076147010 * sc,
|
|
422
|
+
)
|
|
423
|
+
|
|
424
|
+
def gamma(u: float) -> float:
|
|
425
|
+
"""Apply the sRGB transfer function to one clamped linear channel."""
|
|
426
|
+
u = max(0.0, min(1.0, u))
|
|
427
|
+
return 1.055 * u ** (1 / 2.4) - 0.055 if u > 0.0031308 else 12.92 * u
|
|
428
|
+
|
|
429
|
+
return _rgb_to_hex(tuple(gamma(c) for c in rgb_lin))
|
|
430
|
+
|
|
431
|
+
|
|
432
|
+
# Dot-colour tuning: competitors get vivid OKLCH hues spread EVENLY around the
|
|
433
|
+
# circle (ordered by map direction) so hues are balanced — no muddy midtones, no
|
|
434
|
+
# clumping toward pink — with a gentle per-name lightness spread for extra variety.
|
|
435
|
+
_DOT_CHROMA = 0.125
|
|
436
|
+
_L_LO, _L_HI = 0.62, 0.82
|
|
437
|
+
|
|
438
|
+
|
|
439
|
+
def gradient_colors(result: PCAResult, roles: list[str]) -> list[str]:
|
|
440
|
+
"""Distinct, clean per-approach colours.
|
|
441
|
+
|
|
442
|
+
Competitors are placed at EVENLY spaced hues around the OKLCH circle in order
|
|
443
|
+
of their direction on the map — balanced hues, every colour vivid (fixed
|
|
444
|
+
chroma, never a muddy centre), all distinct. Lightness gets a small per-name
|
|
445
|
+
spread for extra separation. Named roles keep their fixed identity hue.
|
|
446
|
+
"""
|
|
447
|
+
scores = result.scores
|
|
448
|
+
n = len(scores)
|
|
449
|
+
comps = [i for i in range(n) if roles[i] == "competitor"]
|
|
450
|
+
|
|
451
|
+
# Order competitors by map direction, then hand out evenly spaced hues.
|
|
452
|
+
angles = np.arctan2(scores[:, 1], scores[:, 0])
|
|
453
|
+
ordered = sorted(comps, key=lambda i: float(angles[i]))
|
|
454
|
+
m = max(1, len(ordered))
|
|
455
|
+
lightness_key = sorted(comps, key=lambda i: (sum(map(ord, result.names[i])), i))
|
|
456
|
+
l_of = {
|
|
457
|
+
i: _L_LO + (_L_HI - _L_LO) * (rank / max(1, len(comps) - 1))
|
|
458
|
+
for rank, i in enumerate(lightness_key)
|
|
459
|
+
}
|
|
460
|
+
|
|
461
|
+
colors = [""] * n
|
|
462
|
+
for rank, i in enumerate(ordered):
|
|
463
|
+
hue = 2 * math.pi * (rank / m) # evenly spaced around the wheel
|
|
464
|
+
colors[i] = _oklab_to_hex(l_of[i], _DOT_CHROMA * math.cos(hue), _DOT_CHROMA * math.sin(hue))
|
|
465
|
+
for i, role in enumerate(roles):
|
|
466
|
+
if role != "competitor":
|
|
467
|
+
colors[i] = ROLE_STYLE[role]["color"]
|
|
468
|
+
return colors
|
|
469
|
+
|
|
470
|
+
|
|
471
|
+
def legend_order(scores: np.ndarray) -> list[int]:
|
|
472
|
+
"""Indices in reading order that matches the map, starting at the extreme
|
|
473
|
+
top-right: banded rows top -> bottom, and within each row right -> left.
|
|
474
|
+
"""
|
|
475
|
+
n = len(scores)
|
|
476
|
+
if n == 0:
|
|
477
|
+
return []
|
|
478
|
+
bands = max(1, round(n**0.5))
|
|
479
|
+
per = math.ceil(n / bands)
|
|
480
|
+
top_to_bottom = sorted(range(n), key=lambda i: -float(scores[i][1]))
|
|
481
|
+
order: list[int] = []
|
|
482
|
+
for b in range(bands):
|
|
483
|
+
row = top_to_bottom[b * per : (b + 1) * per]
|
|
484
|
+
row.sort(key=lambda i: -float(scores[i][0])) # right -> left within the row
|
|
485
|
+
order.extend(row)
|
|
486
|
+
return order
|
|
487
|
+
|
|
488
|
+
|
|
489
|
+
def corner_extremes(scores: np.ndarray) -> dict[str, int]:
|
|
490
|
+
"""Index of the most extreme point toward each corner (tr, tl, br, bl)."""
|
|
491
|
+
sx, sy = scores[:, 0], scores[:, 1]
|
|
492
|
+
return {
|
|
493
|
+
"tr": int(np.argmax(sx + sy)),
|
|
494
|
+
"tl": int(np.argmax(sy - sx)),
|
|
495
|
+
"br": int(np.argmax(sx - sy)),
|
|
496
|
+
"bl": int(np.argmax(-sx - sy)),
|
|
497
|
+
}
|
|
498
|
+
|
|
499
|
+
|
|
500
|
+
# Candidate label placements around a dot, as (dir_x, dir_y): right, left, up, down,
|
|
501
|
+
# then the four diagonals — the first that doesn't collide wins.
|
|
502
|
+
_LABEL_DIRS = [(1, 0), (-1, 0), (0, 1), (0, -1), (1, 1), (-1, 1), (1, -1), (-1, -1)]
|
|
503
|
+
|
|
504
|
+
|
|
505
|
+
def _overlaps(a: tuple[float, float, float, float], b: tuple[float, float, float, float]) -> bool:
|
|
506
|
+
"""True if two axis-aligned boxes ``(x0, y0, x1, y1)`` intersect."""
|
|
507
|
+
return not (a[2] < b[0] or a[0] > b[2] or a[3] < b[1] or a[1] > b[3])
|
|
508
|
+
|
|
509
|
+
|
|
510
|
+
def label_placements(
|
|
511
|
+
result: PCAResult,
|
|
512
|
+
view_x: float,
|
|
513
|
+
view_y: float,
|
|
514
|
+
width_px: int = 900,
|
|
515
|
+
height_px: int = 760,
|
|
516
|
+
font_px: float = 11.0,
|
|
517
|
+
) -> dict[int, tuple[float, float]]:
|
|
518
|
+
"""Greedy de-clutter: choose which approaches to label and *where* to put each
|
|
519
|
+
label. For every dot (corner extremes first, then outermost), try eight
|
|
520
|
+
placements around it and keep the first that overlaps neither another label nor
|
|
521
|
+
any dot marker. Returns {index: (label_x, label_y)} for the labels that fit.
|
|
522
|
+
|
|
523
|
+
`view_x` / `view_y` are the half-extents of each axis's domain (they can differ),
|
|
524
|
+
so the pixel-to-data conversion is correct even when the map is not square.
|
|
525
|
+
"""
|
|
526
|
+
scores = result.scores
|
|
527
|
+
sx = 2 * view_x / width_px # data units per pixel, x
|
|
528
|
+
sy = 2 * view_y / height_px # data units per pixel, y
|
|
529
|
+
pad = 4 * sx
|
|
530
|
+
dot_rx, dot_ry = 7 * sx, 7 * sy
|
|
531
|
+
boxes = [(x - dot_rx, y - dot_ry, x + dot_rx, y + dot_ry) for x, y in scores]
|
|
532
|
+
|
|
533
|
+
corners = list(corner_extremes(scores).values())
|
|
534
|
+
others = sorted(
|
|
535
|
+
(i for i in range(len(result.names)) if i not in corners),
|
|
536
|
+
key=lambda i: -float(np.hypot(*scores[i])),
|
|
537
|
+
)
|
|
538
|
+
placements: dict[int, tuple[float, float]] = {}
|
|
539
|
+
for i in corners + others:
|
|
540
|
+
x, y = scores[i]
|
|
541
|
+
w = len(result.names[i]) * 0.58 * font_px * sx
|
|
542
|
+
h = 1.3 * font_px * sy
|
|
543
|
+
best = None # (distance, box, (lx, ly)) — pick the free side nearest the dot
|
|
544
|
+
for ox, oy in _LABEL_DIRS:
|
|
545
|
+
lx = x + ox * (dot_rx + pad + w / 2) # clear the dot marker, then pad
|
|
546
|
+
ly = y + oy * (dot_ry + pad + h / 2)
|
|
547
|
+
box = (lx - w / 2, ly - h / 2, lx + w / 2, ly + h / 2)
|
|
548
|
+
if any(_overlaps(box, b) for b in boxes):
|
|
549
|
+
continue
|
|
550
|
+
dist = math.hypot(lx - x, ly - y)
|
|
551
|
+
if best is None or dist < best[0]:
|
|
552
|
+
best = (dist, box, (float(lx), float(ly)))
|
|
553
|
+
if best is not None:
|
|
554
|
+
boxes.append(best[1])
|
|
555
|
+
placements[i] = best[2]
|
|
556
|
+
return placements
|
|
557
|
+
|
|
558
|
+
|
|
559
|
+
def _axis_champion(axis_values: np.ndarray, exclude: set[int]) -> int:
|
|
560
|
+
"""Index of the option reaching furthest (largest value) along one axis.
|
|
561
|
+
|
|
562
|
+
Parameters
|
|
563
|
+
----------
|
|
564
|
+
axis_values : np.ndarray
|
|
565
|
+
One column of the oriented scores, e.g. every option's axis-1 coordinate.
|
|
566
|
+
exclude : set[int]
|
|
567
|
+
Row indices to skip (typically the leader, and an already-claimed champion),
|
|
568
|
+
so the same option is never highlighted twice.
|
|
569
|
+
|
|
570
|
+
Returns
|
|
571
|
+
-------
|
|
572
|
+
int
|
|
573
|
+
Row index of the highest not-excluded value along `axis_values`.
|
|
574
|
+
"""
|
|
575
|
+
order = np.argsort(axis_values)[::-1] # highest coordinate first
|
|
576
|
+
return int(next(i for i in order if int(i) not in exclude))
|
|
577
|
+
|
|
578
|
+
|
|
579
|
+
def assign_roles(
|
|
580
|
+
result: PCAResult,
|
|
581
|
+
top: str | None = None,
|
|
582
|
+
right: str | None = None,
|
|
583
|
+
) -> list[str]:
|
|
584
|
+
"""Label four options by domain-agnostic map geometry; the rest are competitors.
|
|
585
|
+
|
|
586
|
+
Every pick is read straight off the oriented coordinates, so it means the same
|
|
587
|
+
thing for any table (no per-domain keyword list):
|
|
588
|
+
|
|
589
|
+
best the reference, sitting at the top-right corner by construction;
|
|
590
|
+
worst the weakest overall: the minimum projection onto the top-right hero
|
|
591
|
+
diagonal (equivalently the smallest axis-1 + axis-2);
|
|
592
|
+
top the challenger reaching furthest up the vertical axis — the peer that
|
|
593
|
+
most defines the map's *top* pole (the leader excluded);
|
|
594
|
+
right the challenger reaching furthest along the horizontal axis — the peer
|
|
595
|
+
that most defines the *right* pole (leader and top champion excluded).
|
|
596
|
+
|
|
597
|
+
Parameters
|
|
598
|
+
----------
|
|
599
|
+
result : PCAResult
|
|
600
|
+
The oriented positioning (`scores` and `reference`).
|
|
601
|
+
top, right : str, optional
|
|
602
|
+
Force a specific option into the top-pole / right-pole highlight by exact
|
|
603
|
+
name, bypassing the geometric pick.
|
|
604
|
+
|
|
605
|
+
Returns
|
|
606
|
+
-------
|
|
607
|
+
list[str]
|
|
608
|
+
One role per option, aligned with ``result.names``; collisions resolve by
|
|
609
|
+
``ROLE_ORDER`` (best beats worst beats the two champions).
|
|
610
|
+
"""
|
|
611
|
+
names = result.names
|
|
612
|
+
scores = result.scores
|
|
613
|
+
best_idx = names.index(result.reference)
|
|
614
|
+
|
|
615
|
+
# Hero axis = the +45 deg diagonal after orientation; project onto (1, 1)/sqrt(2).
|
|
616
|
+
hero_projection = scores @ (np.ones(2) / np.sqrt(2))
|
|
617
|
+
worst_idx = int(next(i for i in np.argsort(hero_projection) if i != best_idx))
|
|
618
|
+
|
|
619
|
+
# The leader is the max on both axes, so a champion is the *next* option out
|
|
620
|
+
# along each axis — the challenger that best embodies that winning pole.
|
|
621
|
+
top_idx = names.index(top) if top is not None else _axis_champion(scores[:, 1], {best_idx})
|
|
622
|
+
right_idx = (
|
|
623
|
+
names.index(right)
|
|
624
|
+
if right is not None
|
|
625
|
+
else _axis_champion(scores[:, 0], {best_idx, top_idx})
|
|
626
|
+
)
|
|
627
|
+
|
|
628
|
+
roles = ["competitor"] * len(names)
|
|
629
|
+
for role in ROLE_ORDER[1:]: # skip "competitor" (default); low -> high priority
|
|
630
|
+
idx = {"right": right_idx, "top": top_idx, "worst": worst_idx, "best": best_idx}[role]
|
|
631
|
+
roles[idx] = role
|
|
632
|
+
return roles
|
|
633
|
+
|
|
634
|
+
|
|
635
|
+
# --------------------------------------------------------------------------- #
|
|
636
|
+
# axis naming (local LLM interprets the loading weights + column names)
|
|
637
|
+
# --------------------------------------------------------------------------- #
|
|
638
|
+
# Expand common acronyms to real words — never show acronyms in the figure.
|
|
639
|
+
_ACRONYM_WORDS = {
|
|
640
|
+
"tco": "Cost",
|
|
641
|
+
"pii": "Privacy",
|
|
642
|
+
"gdpr": "Compliance",
|
|
643
|
+
"ux": "Experience",
|
|
644
|
+
"fr": "French",
|
|
645
|
+
"ev": "Vehicles",
|
|
646
|
+
"ai": "Intelligence",
|
|
647
|
+
"qa": "Quality",
|
|
648
|
+
"stt": "Speech",
|
|
649
|
+
"api": "Interface",
|
|
650
|
+
"diy": "Homemade",
|
|
651
|
+
}
|
|
652
|
+
|
|
653
|
+
|
|
654
|
+
def _deacronym(label: str) -> str:
|
|
655
|
+
"""Expand or drop acronym tokens in a label so the figure shows real words."""
|
|
656
|
+
out = []
|
|
657
|
+
for tok in label.split():
|
|
658
|
+
if tok.isupper() and len(tok) <= 5: # looks like an acronym
|
|
659
|
+
expanded = _ACRONYM_WORDS.get(tok.lower())
|
|
660
|
+
if expanded:
|
|
661
|
+
out.append(expanded)
|
|
662
|
+
# unknown acronym -> drop it
|
|
663
|
+
else:
|
|
664
|
+
out.append(tok)
|
|
665
|
+
return " ".join(out).strip()
|
|
666
|
+
|
|
667
|
+
|
|
668
|
+
def _one_word(feature: str) -> str:
|
|
669
|
+
"""A single real word from an attribute name — longest non-acronym token,
|
|
670
|
+
expanding known acronyms so the figure never shows abbreviations.
|
|
671
|
+
"""
|
|
672
|
+
toks = re.findall(r"[A-Za-z]+", feature)
|
|
673
|
+
words = [t for t in toks if len(t) > 1 and not t.isupper()] # drop acronyms
|
|
674
|
+
if words:
|
|
675
|
+
return max(words, key=len).capitalize()
|
|
676
|
+
for tok in toks: # only acronyms left
|
|
677
|
+
if tok.lower() in _ACRONYM_WORDS:
|
|
678
|
+
return _ACRONYM_WORDS[tok.lower()]
|
|
679
|
+
return _ACRONYM_WORDS.get(feature.strip().lower(), feature.strip().capitalize())
|
|
680
|
+
|
|
681
|
+
|
|
682
|
+
# Small stop-words ignored when comparing labels for shared content words.
|
|
683
|
+
_LABEL_STOP = {
|
|
684
|
+
"and",
|
|
685
|
+
"the",
|
|
686
|
+
"for",
|
|
687
|
+
"with",
|
|
688
|
+
"your",
|
|
689
|
+
"our",
|
|
690
|
+
"per",
|
|
691
|
+
"les",
|
|
692
|
+
"des",
|
|
693
|
+
"las",
|
|
694
|
+
"los",
|
|
695
|
+
"una",
|
|
696
|
+
"por",
|
|
697
|
+
"con",
|
|
698
|
+
"sur",
|
|
699
|
+
"del",
|
|
700
|
+
}
|
|
701
|
+
|
|
702
|
+
# A pole must be a positive quality; these markers signal a drawback (en/fr/es) and
|
|
703
|
+
# get the label rejected — e.g. "High Cost", "Slow", "Expensive" never appear.
|
|
704
|
+
_NEGATIVE_WORDS = {
|
|
705
|
+
"high",
|
|
706
|
+
"low",
|
|
707
|
+
"expensive",
|
|
708
|
+
"costly",
|
|
709
|
+
"slow",
|
|
710
|
+
"complex",
|
|
711
|
+
"complicated",
|
|
712
|
+
"poor",
|
|
713
|
+
"weak",
|
|
714
|
+
"insecure",
|
|
715
|
+
"unreliable",
|
|
716
|
+
"difficult",
|
|
717
|
+
"limited",
|
|
718
|
+
"hidden",
|
|
719
|
+
"risky",
|
|
720
|
+
"lack",
|
|
721
|
+
"worse",
|
|
722
|
+
"bad",
|
|
723
|
+
"élevé",
|
|
724
|
+
"eleve",
|
|
725
|
+
"cher",
|
|
726
|
+
"lent",
|
|
727
|
+
"complexe",
|
|
728
|
+
"coûteux",
|
|
729
|
+
"couteux",
|
|
730
|
+
"difficile",
|
|
731
|
+
"faible",
|
|
732
|
+
"alto",
|
|
733
|
+
"caro",
|
|
734
|
+
"lento",
|
|
735
|
+
"complejo",
|
|
736
|
+
"costoso",
|
|
737
|
+
"débil",
|
|
738
|
+
"debil",
|
|
739
|
+
"riesgo",
|
|
740
|
+
}
|
|
741
|
+
|
|
742
|
+
|
|
743
|
+
def _content_words(label: str) -> set[str]:
|
|
744
|
+
"""Significant lowercase words in a label (>= 3 letters, minus stop-words)."""
|
|
745
|
+
return {
|
|
746
|
+
t for t in re.findall(r"[a-zA-Z]+", label.lower()) if len(t) >= 3 and t not in _LABEL_STOP
|
|
747
|
+
}
|
|
748
|
+
|
|
749
|
+
|
|
750
|
+
def _clean_label(label: str) -> str:
|
|
751
|
+
"""Expand acronyms, split camelCase, and keep at most three words."""
|
|
752
|
+
label = _deacronym(label) if label else ""
|
|
753
|
+
label = re.sub(r"(?<=[a-z])(?=[A-Z])", " ", label).strip() # split camelCase
|
|
754
|
+
return " ".join(label.split()[:3])
|
|
755
|
+
|
|
756
|
+
|
|
757
|
+
def finalize_poles(raw: list[str], fallback: list[str]) -> list[str]:
|
|
758
|
+
"""Turn raw LLM pole labels into four clean, distinct, non-antonymous labels.
|
|
759
|
+
|
|
760
|
+
Enforces: real words (no acronyms), at most three words, no label repeated, and
|
|
761
|
+
no two labels sharing a content word — which rules out antonym pairs such as
|
|
762
|
+
'Cost Efficient' / 'High Cost'. A rejected label is replaced by its
|
|
763
|
+
loading-derived fallback (drawn from a different criterion).
|
|
764
|
+
"""
|
|
765
|
+
|
|
766
|
+
def bad(w: str) -> bool:
|
|
767
|
+
"""True if label `w` must be rejected: empty, duplicate, shares a content
|
|
768
|
+
word with an already-accepted label (rules out antonym pairs), or contains
|
|
769
|
+
a negative word (a pole must name a positive quality).
|
|
770
|
+
"""
|
|
771
|
+
cw = _content_words(w)
|
|
772
|
+
return (
|
|
773
|
+
not w or w.lower() in seen or bool(cw & used_words) or bool(cw & _NEGATIVE_WORDS)
|
|
774
|
+
) # never a drawback / negative
|
|
775
|
+
|
|
776
|
+
out: list[str] = []
|
|
777
|
+
seen: set[str] = set()
|
|
778
|
+
used_words: set[str] = set()
|
|
779
|
+
for i, (label, fb) in enumerate(zip(raw, fallback, strict=False)):
|
|
780
|
+
w = _clean_label(label)
|
|
781
|
+
if bad(w):
|
|
782
|
+
w = _clean_label(fb) # fall back to the loading word
|
|
783
|
+
if bad(w):
|
|
784
|
+
w = f"{w} {i}"
|
|
785
|
+
seen.add(w.lower())
|
|
786
|
+
used_words |= _content_words(w)
|
|
787
|
+
out.append(w)
|
|
788
|
+
return out
|
|
789
|
+
|
|
790
|
+
|
|
791
|
+
def _fallback_poles(components: np.ndarray, features: list[str]) -> list[str]:
|
|
792
|
+
"""Four distinct pole words [left, right, bottom, top] from the loadings.
|
|
793
|
+
|
|
794
|
+
left/right = low/high end of axis-1; bottom/top = low/high end of axis-2.
|
|
795
|
+
Each pole takes the most extreme not-yet-used attribute at that end.
|
|
796
|
+
"""
|
|
797
|
+
specs = [(0, 1), (0, -1), (1, 1), (1, -1)] # (axis, +1=ascending->low end first)
|
|
798
|
+
used: set[str] = set()
|
|
799
|
+
poles: list[str] = []
|
|
800
|
+
for axis, sign in specs:
|
|
801
|
+
order = np.argsort(components[axis])[::sign] # sign +1 -> low end first
|
|
802
|
+
word = next(
|
|
803
|
+
(w for i in order if (w := _one_word(features[i])).lower() not in used),
|
|
804
|
+
_one_word(features[order[0]]),
|
|
805
|
+
)
|
|
806
|
+
used.add(word.lower())
|
|
807
|
+
poles.append(word)
|
|
808
|
+
return poles
|
|
809
|
+
|
|
810
|
+
|
|
811
|
+
def _poles_to_names(poles: list[str]) -> list[str]:
|
|
812
|
+
"""[left, right, bottom, top] -> ['left ↔ right', 'bottom ↔ top']."""
|
|
813
|
+
left, right, bottom, top = poles
|
|
814
|
+
return [f"{left} ↔ {right}", f"{bottom} ↔ {top}"]
|
|
815
|
+
|
|
816
|
+
|
|
817
|
+
def axis_poles(
|
|
818
|
+
result: PCAResult, model: str = DEFAULT_MODEL, use_llm: bool = True, lang: str | None = None
|
|
819
|
+
) -> list[str]:
|
|
820
|
+
"""Four distinct pole labels [left, right, bottom, top] for the two axes.
|
|
821
|
+
|
|
822
|
+
Each PCA axis is a weighted mix of the criteria. The local LLM names each pole
|
|
823
|
+
(1-3 words) for what the approaches at that end are collectively strongest at,
|
|
824
|
+
from the signed loadings and the original column names — in the table's own
|
|
825
|
+
language (auto-detected from the column names; see `i18n.yaml`). Falls back to
|
|
826
|
+
loading-derived distinct words if the LLM is unavailable or misbehaves.
|
|
827
|
+
"""
|
|
828
|
+
feats = result.features
|
|
829
|
+
fallback_poles = _fallback_poles(result.components, feats)
|
|
830
|
+
if not use_llm:
|
|
831
|
+
return fallback_poles
|
|
832
|
+
if lang is None:
|
|
833
|
+
lang = detect_language(feats)
|
|
834
|
+
tpl = i18n(lang)
|
|
835
|
+
|
|
836
|
+
try:
|
|
837
|
+
# Every rating is higher-is-better, so a pole is best described by the
|
|
838
|
+
# criteria approaches THERE score high on (its sign of the loading).
|
|
839
|
+
def show(f: str) -> str:
|
|
840
|
+
"""Present a criterion to the model, flagging negated (lower-better) ones.
|
|
841
|
+
|
|
842
|
+
A lower-is-better criterion was negated for the PCA, so a high score
|
|
843
|
+
means a LOW raw value: show it as "low <name>" so the model names the
|
|
844
|
+
benefit ("Affordable") rather than the drawback ("Expensive").
|
|
845
|
+
"""
|
|
846
|
+
return f"low {f}" if f in result.lower else f
|
|
847
|
+
|
|
848
|
+
def pole_strengths(k: int, sign: int) -> str:
|
|
849
|
+
"""Criteria (with weights) that define one end of axis `k`.
|
|
850
|
+
|
|
851
|
+
`sign` selects the end: +1 for the positive-loading pole, -1 for the
|
|
852
|
+
negative one. Returns them strongest-first as a human-readable string,
|
|
853
|
+
or "—" when nothing loads meaningfully on that end.
|
|
854
|
+
"""
|
|
855
|
+
pairs = [
|
|
856
|
+
(f, w)
|
|
857
|
+
for f, w in zip(feats, result.components[k], strict=False)
|
|
858
|
+
if (w > 0) == (sign > 0) and abs(w) > 0.05
|
|
859
|
+
]
|
|
860
|
+
pairs.sort(key=lambda t: -abs(t[1]))
|
|
861
|
+
return ", ".join(f"{show(f)} (weight {abs(w):.2f})" for f, w in pairs) or "—"
|
|
862
|
+
|
|
863
|
+
# Glossary of any acronyms present in the columns, so the model translates
|
|
864
|
+
# them instead of echoing them (built from the actual column names).
|
|
865
|
+
present = {
|
|
866
|
+
a.upper(): w
|
|
867
|
+
for a, w in _ACRONYM_WORDS.items()
|
|
868
|
+
if any(a.upper() in f.upper() for f in feats)
|
|
869
|
+
}
|
|
870
|
+
glossary = (
|
|
871
|
+
(tpl["glossary_prefix"] + "; ".join(f"{k} = {v}" for k, v in present.items()) + ".\n\n")
|
|
872
|
+
if present
|
|
873
|
+
else ""
|
|
874
|
+
)
|
|
875
|
+
prompt = tpl["axis_prompt"].format(
|
|
876
|
+
glossary=glossary,
|
|
877
|
+
left=pole_strengths(0, -1),
|
|
878
|
+
right=pole_strengths(0, +1),
|
|
879
|
+
bottom=pole_strengths(1, -1),
|
|
880
|
+
top=pole_strengths(1, +1),
|
|
881
|
+
)
|
|
882
|
+
schema = {
|
|
883
|
+
"type": "object",
|
|
884
|
+
"properties": {k: {"type": "string"} for k in ("left", "right", "bottom", "top")},
|
|
885
|
+
"required": ["left", "right", "bottom", "top"],
|
|
886
|
+
}
|
|
887
|
+
resp = ollama.chat(
|
|
888
|
+
model=model,
|
|
889
|
+
format=schema,
|
|
890
|
+
options={"temperature": 0},
|
|
891
|
+
messages=[{"role": "user", "content": prompt}],
|
|
892
|
+
)
|
|
893
|
+
data = json.loads(resp["message"]["content"])
|
|
894
|
+
raw = [str(data.get(k, "")) for k in ("left", "right", "bottom", "top")]
|
|
895
|
+
# Clean, de-duplicate, and reject antonym/shared-word pairs.
|
|
896
|
+
return finalize_poles(raw, fallback_poles)
|
|
897
|
+
except Exception as exc: # ollama missing / model absent / bad JSON
|
|
898
|
+
logger.warning("axis naming: LLM unavailable (%s); using deterministic names", exc)
|
|
899
|
+
return fallback_poles
|
|
900
|
+
|
|
901
|
+
|
|
902
|
+
def noun_forms(
|
|
903
|
+
word: str, model: str = DEFAULT_MODEL, use_llm: bool = True, lang: str | None = None
|
|
904
|
+
) -> tuple[str, str]:
|
|
905
|
+
"""Singular and plural of `word` (the first-column name), in its own language.
|
|
906
|
+
|
|
907
|
+
Used for the figure title and legend heading, so a table of "Language" reads
|
|
908
|
+
"Languages in the Quadrant". The prompt lives in `i18n.yaml`. Falls back to a
|
|
909
|
+
naive `+s` plural without a model.
|
|
910
|
+
"""
|
|
911
|
+
word = (word or "Approach").strip() or "Approach"
|
|
912
|
+
if len(word) > 1 and word.lower().endswith("s"): # looks plural already
|
|
913
|
+
naive = (word[:-1].capitalize(), word.capitalize())
|
|
914
|
+
else:
|
|
915
|
+
naive = (word.capitalize(), word.capitalize() + "s")
|
|
916
|
+
if not use_llm:
|
|
917
|
+
return naive
|
|
918
|
+
if lang is None:
|
|
919
|
+
lang = detect_language([word])
|
|
920
|
+
try:
|
|
921
|
+
schema = {
|
|
922
|
+
"type": "object",
|
|
923
|
+
"properties": {"singular": {"type": "string"}, "plural": {"type": "string"}},
|
|
924
|
+
"required": ["singular", "plural"],
|
|
925
|
+
}
|
|
926
|
+
resp = ollama.chat(
|
|
927
|
+
model=model,
|
|
928
|
+
format=schema,
|
|
929
|
+
options={"temperature": 0},
|
|
930
|
+
messages=[{"role": "user", "content": i18n(lang)["noun_prompt"].format(word=word)}],
|
|
931
|
+
)
|
|
932
|
+
data = json.loads(resp["message"]["content"])
|
|
933
|
+
s = (str(data.get("singular") or "").strip() or naive[0]).capitalize()
|
|
934
|
+
p = (str(data.get("plural") or "").strip() or naive[1]).capitalize()
|
|
935
|
+
# Guard against the model swapping in a synonym (e.g. Voiture -> Véhicules):
|
|
936
|
+
# a valid form must share a prefix with the actual column word.
|
|
937
|
+
prefix = word.lower()[: max(3, len(word) - 2)]
|
|
938
|
+
if not s.lower().startswith(prefix):
|
|
939
|
+
s = naive[0]
|
|
940
|
+
if not p.lower().startswith(prefix):
|
|
941
|
+
p = naive[1]
|
|
942
|
+
return s, p
|
|
943
|
+
except Exception:
|
|
944
|
+
return naive
|
|
945
|
+
|
|
946
|
+
|
|
947
|
+
# --------------------------------------------------------------------------- #
|
|
948
|
+
# Vega-Lite
|
|
949
|
+
# --------------------------------------------------------------------------- #
|
|
950
|
+
def to_vega(
|
|
951
|
+
result: PCAResult,
|
|
952
|
+
roles: list[str] | None = None,
|
|
953
|
+
poles: list[str] | None = None,
|
|
954
|
+
colors: list[str] | None = None,
|
|
955
|
+
noun_plural: str = "Approaches",
|
|
956
|
+
title: str | None = None,
|
|
957
|
+
) -> dict:
|
|
958
|
+
"""Build a self-contained Vega-Lite v5 spec (inline data) for the map.
|
|
959
|
+
|
|
960
|
+
Layers, bottom to top: a centred cross of axes through the origin (the neutral
|
|
961
|
+
intersection), every approach coloured by its position (Apple-wheel HSV), the
|
|
962
|
+
four pole words at the axis ends, and labels for the four corner extremes. No
|
|
963
|
+
frame, spines, ticks, numeric scales, or arrows.
|
|
964
|
+
|
|
965
|
+
`title` is the fully-localized figure title (e.g. "Voitures dans le quadrant");
|
|
966
|
+
when omitted it defaults to the English "<plural> in the Quadrant" so direct
|
|
967
|
+
callers still get a sensible heading.
|
|
968
|
+
"""
|
|
969
|
+
ref = result.reference
|
|
970
|
+
names = result.names
|
|
971
|
+
if roles is None:
|
|
972
|
+
roles = ["best" if n == ref else "competitor" for n in names]
|
|
973
|
+
if poles is None:
|
|
974
|
+
poles = _fallback_poles(result.components, result.features)
|
|
975
|
+
left, right, bottom, top = poles
|
|
976
|
+
|
|
977
|
+
if colors is None:
|
|
978
|
+
colors = gradient_colors(result, roles)
|
|
979
|
+
n = len(names)
|
|
980
|
+
|
|
981
|
+
# Per-axis extents so each axis fills its own space: a low-variance axis (e.g.
|
|
982
|
+
# PC2) is not squashed flat against the cross. Each axis gets its own domain.
|
|
983
|
+
span_x = float(np.abs(result.scores[:, 0]).max()) or 1.0
|
|
984
|
+
span_y = float(np.abs(result.scores[:, 1]).max()) or 1.0
|
|
985
|
+
# Wide margin: the dots occupy the central ~65%, leaving the outer band clear
|
|
986
|
+
# for the pole phrases at the axis ends.
|
|
987
|
+
view_x, view_y = span_x * 1.55, span_y * 1.55
|
|
988
|
+
|
|
989
|
+
# Sizes adapt to the option count: bigger when few, smaller when many.
|
|
990
|
+
def _scaled(lo: int, hi: int, few: int = 8, many: int = 40) -> int:
|
|
991
|
+
"""Interpolate a size between `hi` (at `few` options) and `lo` (at `many`).
|
|
992
|
+
|
|
993
|
+
Keeps the map legible across table sizes: large glyphs on a sparse map,
|
|
994
|
+
smaller ones once the plot gets crowded. Clamped outside ``[few, many]``.
|
|
995
|
+
"""
|
|
996
|
+
t = (min(max(n, few), many) - few) / (many - few)
|
|
997
|
+
return round(hi + (lo - hi) * t)
|
|
998
|
+
|
|
999
|
+
label_font = _scaled(11, 17)
|
|
1000
|
+
pole_font = _scaled(13, 22)
|
|
1001
|
+
legend_font = _scaled(9, 13)
|
|
1002
|
+
dot_size = _scaled(90, 240)
|
|
1003
|
+
|
|
1004
|
+
placements = label_placements(result, view_x, view_y, font_px=label_font)
|
|
1005
|
+
|
|
1006
|
+
# Legend follows the map: rows top -> bottom, left -> right within each row.
|
|
1007
|
+
order = legend_order(result.scores)
|
|
1008
|
+
legend_names = [names[i] for i in order]
|
|
1009
|
+
legend_colors = [colors[i] for i in order]
|
|
1010
|
+
|
|
1011
|
+
points = [
|
|
1012
|
+
{
|
|
1013
|
+
"name": nm,
|
|
1014
|
+
"axis1": float(x),
|
|
1015
|
+
"axis2": float(y),
|
|
1016
|
+
"role": r,
|
|
1017
|
+
"color": c,
|
|
1018
|
+
"label": nm if i in placements else "",
|
|
1019
|
+
"labelx": placements.get(i, (x, y))[0],
|
|
1020
|
+
"labely": placements.get(i, (x, y))[1],
|
|
1021
|
+
}
|
|
1022
|
+
for i, ((x, y), nm, r, c) in enumerate(
|
|
1023
|
+
zip(result.scores, names, roles, colors, strict=False)
|
|
1024
|
+
)
|
|
1025
|
+
]
|
|
1026
|
+
|
|
1027
|
+
xdom = {"domain": [-view_x, view_x]}
|
|
1028
|
+
ydom = {"domain": [-view_y, view_y]}
|
|
1029
|
+
bare = {"domain": False, "ticks": False, "labels": False, "grid": False, "title": None}
|
|
1030
|
+
xenc = {"field": "axis1", "type": "quantitative", "scale": xdom, "axis": bare}
|
|
1031
|
+
yenc = {"field": "axis2", "type": "quantitative", "scale": ydom, "axis": bare}
|
|
1032
|
+
|
|
1033
|
+
def rule(x0: float, x1: float, y0: float, y1: float) -> dict:
|
|
1034
|
+
"""A Vega-Lite layer drawing one dotted axis segment in data coordinates."""
|
|
1035
|
+
return {
|
|
1036
|
+
"data": {"values": [{}]},
|
|
1037
|
+
"mark": {"type": "rule", "color": PALETTE["axis"], "size": 1.2, "strokeDash": [2, 4]},
|
|
1038
|
+
"encoding": {
|
|
1039
|
+
"x": {"datum": x0, "type": "quantitative", "scale": xdom, "axis": bare},
|
|
1040
|
+
"x2": {"datum": x1},
|
|
1041
|
+
"y": {"datum": y0, "type": "quantitative", "scale": ydom, "axis": bare},
|
|
1042
|
+
"y2": {"datum": y1},
|
|
1043
|
+
},
|
|
1044
|
+
}
|
|
1045
|
+
|
|
1046
|
+
def pole_label(x: float, y: float, text: str, align: str, baseline: str) -> dict:
|
|
1047
|
+
"""A Vega-Lite text layer placing one italic pole word at an axis end."""
|
|
1048
|
+
return {
|
|
1049
|
+
"data": {"values": [{"x": x, "y": y, "t": text}]},
|
|
1050
|
+
"mark": {
|
|
1051
|
+
"type": "text",
|
|
1052
|
+
"fontSize": pole_font,
|
|
1053
|
+
"fontStyle": "italic",
|
|
1054
|
+
"color": "#6E6E73",
|
|
1055
|
+
"align": align,
|
|
1056
|
+
"baseline": baseline,
|
|
1057
|
+
},
|
|
1058
|
+
"encoding": {
|
|
1059
|
+
"x": {"field": "x", "type": "quantitative", "scale": xdom, "axis": bare},
|
|
1060
|
+
"y": {"field": "y", "type": "quantitative", "scale": ydom, "axis": bare},
|
|
1061
|
+
"text": {"field": "t", "type": "nominal"},
|
|
1062
|
+
},
|
|
1063
|
+
}
|
|
1064
|
+
|
|
1065
|
+
edge_x, edge_y = view_x * 0.98, view_y * 0.98 # axes span the full view
|
|
1066
|
+
gap_x, gap_y = span_x * 0.04, span_y * 0.04 # keep pole words off the lines
|
|
1067
|
+
layers = [
|
|
1068
|
+
rule(-edge_x, edge_x, 0, 0), # horizontal axis
|
|
1069
|
+
rule(0, 0, -edge_y, edge_y), # vertical axis
|
|
1070
|
+
pole_label(edge_x, gap_y, right, "right", "bottom"),
|
|
1071
|
+
pole_label(-edge_x, gap_y, left, "left", "bottom"),
|
|
1072
|
+
pole_label(gap_x, edge_y, top, "left", "top"),
|
|
1073
|
+
pole_label(gap_x, -edge_y, bottom, "left", "bottom"),
|
|
1074
|
+
{ # every dot coloured by position; legend maps name -> colour
|
|
1075
|
+
"data": {"values": points},
|
|
1076
|
+
"mark": {
|
|
1077
|
+
"type": "point",
|
|
1078
|
+
"filled": True,
|
|
1079
|
+
"opacity": 0.95,
|
|
1080
|
+
"stroke": "white",
|
|
1081
|
+
"strokeWidth": 1,
|
|
1082
|
+
"size": dot_size,
|
|
1083
|
+
},
|
|
1084
|
+
"encoding": {
|
|
1085
|
+
"x": xenc,
|
|
1086
|
+
"y": yenc,
|
|
1087
|
+
"color": {
|
|
1088
|
+
"field": "name",
|
|
1089
|
+
"type": "nominal",
|
|
1090
|
+
"scale": {"domain": legend_names, "range": legend_colors},
|
|
1091
|
+
"legend": {
|
|
1092
|
+
"title": noun_plural,
|
|
1093
|
+
"symbolLimit": 0,
|
|
1094
|
+
"labelFontSize": legend_font,
|
|
1095
|
+
"symbolOpacity": 1,
|
|
1096
|
+
},
|
|
1097
|
+
},
|
|
1098
|
+
"tooltip": [
|
|
1099
|
+
{"field": "name", "type": "nominal"},
|
|
1100
|
+
{"field": "role", "type": "nominal"},
|
|
1101
|
+
{"field": "axis1", "type": "quantitative", "format": ".2f"},
|
|
1102
|
+
{"field": "axis2", "type": "quantitative", "format": ".2f"},
|
|
1103
|
+
],
|
|
1104
|
+
},
|
|
1105
|
+
},
|
|
1106
|
+
{ # labels — de-cluttered, placed on whichever side is free
|
|
1107
|
+
"data": {"values": points},
|
|
1108
|
+
"transform": [{"filter": "datum.label != ''"}],
|
|
1109
|
+
"mark": {
|
|
1110
|
+
"type": "text",
|
|
1111
|
+
"align": "center",
|
|
1112
|
+
"baseline": "middle",
|
|
1113
|
+
"fontSize": label_font,
|
|
1114
|
+
"color": PALETTE["label"],
|
|
1115
|
+
},
|
|
1116
|
+
"encoding": {
|
|
1117
|
+
"x": {"field": "labelx", "type": "quantitative"},
|
|
1118
|
+
"y": {"field": "labely", "type": "quantitative"},
|
|
1119
|
+
"text": {"field": "label", "type": "nominal"},
|
|
1120
|
+
},
|
|
1121
|
+
},
|
|
1122
|
+
]
|
|
1123
|
+
|
|
1124
|
+
# The plotting area is tall enough that the one-row-per-approach legend beside
|
|
1125
|
+
# it is never taller than the canvas (so it can't be clipped).
|
|
1126
|
+
height = max(720, 24 * n + 140)
|
|
1127
|
+
if title is None: # direct callers get the English default; localized via i18n
|
|
1128
|
+
title = f"{noun_plural} in the Quadrant"
|
|
1129
|
+
return {
|
|
1130
|
+
"$schema": "https://vega.github.io/schema/vega-lite/v5.json",
|
|
1131
|
+
"title": {"text": title, "font": FONT, "fontSize": 18},
|
|
1132
|
+
# Transparent background: Vega-Lite otherwise bakes an opaque white rectangle
|
|
1133
|
+
# into the PNG/SVG. Null lets the map drop cleanly onto any page or slide.
|
|
1134
|
+
"background": None,
|
|
1135
|
+
"width": 1000,
|
|
1136
|
+
"height": height,
|
|
1137
|
+
"autosize": {"type": "pad", "resize": True}, # grow to fit the legend
|
|
1138
|
+
"config": {
|
|
1139
|
+
"font": FONT,
|
|
1140
|
+
"padding": 12,
|
|
1141
|
+
"view": {"stroke": None}, # no box around the plotting area
|
|
1142
|
+
"axis": {
|
|
1143
|
+
"grid": False,
|
|
1144
|
+
"domain": False,
|
|
1145
|
+
"ticks": False,
|
|
1146
|
+
"labels": False,
|
|
1147
|
+
"labelFont": FONT,
|
|
1148
|
+
"titleFont": FONT,
|
|
1149
|
+
},
|
|
1150
|
+
"text": {"font": FONT},
|
|
1151
|
+
},
|
|
1152
|
+
"layer": layers,
|
|
1153
|
+
}
|
|
1154
|
+
|
|
1155
|
+
|
|
1156
|
+
# --------------------------------------------------------------------------- #
|
|
1157
|
+
# Three-fold export: figures (PNG + SVG + Vega JSON), markdown, YAML
|
|
1158
|
+
# --------------------------------------------------------------------------- #
|
|
1159
|
+
def render_figures(spec: dict, stem: str) -> list[str]:
|
|
1160
|
+
"""Rasterize/vectorize a Vega-Lite spec to transparent and white PNG + SVG.
|
|
1161
|
+
|
|
1162
|
+
Writes four files: the transparent `<stem>.png` / `<stem>.svg` (the default, for
|
|
1163
|
+
dropping onto any coloured page) and a white-background `<stem>.white.png` /
|
|
1164
|
+
`<stem>.white.svg` (for dark surfaces — e.g. GitHub dark mode — where the map's
|
|
1165
|
+
near-black labels would otherwise vanish on a transparent background). Returns
|
|
1166
|
+
the four paths in that order.
|
|
1167
|
+
"""
|
|
1168
|
+
written: list[str] = []
|
|
1169
|
+
# `spec` already carries background:null; the ".white" pass overrides it. Same
|
|
1170
|
+
# layout both times, so the only difference is the baked-in backdrop.
|
|
1171
|
+
for suffix, variant in ((".", spec), (".white.", {**spec, "background": "white"})):
|
|
1172
|
+
png_path, svg_path = f"{stem}{suffix}png", f"{stem}{suffix}svg"
|
|
1173
|
+
with open(png_path, "wb") as fh:
|
|
1174
|
+
fh.write(vlc.vegalite_to_png(vl_spec=variant, scale=2.0))
|
|
1175
|
+
with open(svg_path, "w", encoding="utf-8") as fh:
|
|
1176
|
+
fh.write(vlc.vegalite_to_svg(vl_spec=variant))
|
|
1177
|
+
written += [png_path, svg_path]
|
|
1178
|
+
return written
|
|
1179
|
+
|
|
1180
|
+
|
|
1181
|
+
def png_on_white(spec: dict) -> bytes:
|
|
1182
|
+
"""Render `spec` to PNG bytes on an opaque white background.
|
|
1183
|
+
|
|
1184
|
+
The exported figures are transparent, but the vision self-check sends the image
|
|
1185
|
+
to a model whose backend flattens transparency onto a dark canvas — which would
|
|
1186
|
+
hide the near-black labels and legend and make the check misfire. White is the
|
|
1187
|
+
figure's intended reading surface, so the check runs against a white-composited
|
|
1188
|
+
copy rather than the transparent file on disk.
|
|
1189
|
+
"""
|
|
1190
|
+
return vlc.vegalite_to_png(vl_spec={**spec, "background": "white"}, scale=2.0)
|
|
1191
|
+
|
|
1192
|
+
|
|
1193
|
+
def vlm_assess(image: str | bytes, model: str = DEFAULT_MODEL) -> dict:
|
|
1194
|
+
"""Ask the qwen vision-LLM to sanity-check a rendered positioning map.
|
|
1195
|
+
|
|
1196
|
+
`image` is a PNG path or raw PNG bytes (bytes let the caller assess a
|
|
1197
|
+
white-composited render without touching the transparent file on disk). Returns
|
|
1198
|
+
a verdict dict — whether the red leader dot sits top-right, whether the labels
|
|
1199
|
+
are readable, and whether the legend is fully visible — plus free-text notes.
|
|
1200
|
+
Empty dict if the model or a rendered image is unavailable.
|
|
1201
|
+
"""
|
|
1202
|
+
schema = {
|
|
1203
|
+
"type": "object",
|
|
1204
|
+
"properties": {
|
|
1205
|
+
"leader_top_right": {"type": "boolean"},
|
|
1206
|
+
"readable": {"type": "boolean"},
|
|
1207
|
+
"legend_visible": {"type": "boolean"},
|
|
1208
|
+
"notes": {"type": "string"},
|
|
1209
|
+
},
|
|
1210
|
+
"required": ["leader_top_right", "readable", "legend_visible", "notes"],
|
|
1211
|
+
}
|
|
1212
|
+
prompt = (
|
|
1213
|
+
"This image is a 2D competitor positioning map. The single RED dot is the "
|
|
1214
|
+
"leader and should sit in the TOP-RIGHT area. Assess three things: (1) is "
|
|
1215
|
+
"the red leader dot in the top-right? (2) are the point labels readable and "
|
|
1216
|
+
"not badly overlapping? (3) is the legend on the right fully visible, not "
|
|
1217
|
+
"cut off? Reply as JSON."
|
|
1218
|
+
)
|
|
1219
|
+
try:
|
|
1220
|
+
resp = ollama.chat(
|
|
1221
|
+
model=model,
|
|
1222
|
+
format=schema,
|
|
1223
|
+
options={"temperature": 0},
|
|
1224
|
+
messages=[{"role": "user", "content": prompt, "images": [image]}],
|
|
1225
|
+
)
|
|
1226
|
+
return json.loads(resp["message"]["content"])
|
|
1227
|
+
except Exception:
|
|
1228
|
+
return {}
|
|
1229
|
+
|
|
1230
|
+
|
|
1231
|
+
def _llm_text(prompt: str, model: str, use_llm: bool, fallback: str) -> str:
|
|
1232
|
+
"""Free-text completion from the local model; `fallback` if unavailable."""
|
|
1233
|
+
if not use_llm:
|
|
1234
|
+
return fallback
|
|
1235
|
+
try:
|
|
1236
|
+
resp = ollama.chat(
|
|
1237
|
+
model=model,
|
|
1238
|
+
options={"temperature": 0.3},
|
|
1239
|
+
messages=[{"role": "user", "content": prompt}],
|
|
1240
|
+
)
|
|
1241
|
+
return resp["message"]["content"].strip() or fallback
|
|
1242
|
+
except Exception:
|
|
1243
|
+
return fallback
|
|
1244
|
+
|
|
1245
|
+
|
|
1246
|
+
def analysis_markdown(
|
|
1247
|
+
result: PCAResult,
|
|
1248
|
+
roles: list[str],
|
|
1249
|
+
poles: list[str],
|
|
1250
|
+
model: str = DEFAULT_MODEL,
|
|
1251
|
+
use_llm: bool = True,
|
|
1252
|
+
lang: str | None = None,
|
|
1253
|
+
) -> str:
|
|
1254
|
+
"""A thoughtful, precise interpretation of the map as Markdown.
|
|
1255
|
+
|
|
1256
|
+
Combines data-derived facts (axis loadings, variance, roles, coordinates) with
|
|
1257
|
+
an LLM-written narrative in the table's own language (auto-detected). Falls
|
|
1258
|
+
back to a templated narrative when the model is unavailable.
|
|
1259
|
+
"""
|
|
1260
|
+
left, right, bottom, top = poles
|
|
1261
|
+
evr = result.explained_variance_ratio
|
|
1262
|
+
names = result.names
|
|
1263
|
+
role_of = dict(zip(names, roles, strict=False))
|
|
1264
|
+
coords = result.coords()
|
|
1265
|
+
if lang is None:
|
|
1266
|
+
lang = detect_language(result.features)
|
|
1267
|
+
|
|
1268
|
+
def loading_line(k: int) -> str:
|
|
1269
|
+
"""Axis `k`'s criteria and signed weights, highest-first, as one line."""
|
|
1270
|
+
pairs = sorted(
|
|
1271
|
+
zip(result.features, result.components[k], strict=False), key=lambda t: -t[1]
|
|
1272
|
+
)
|
|
1273
|
+
return " · ".join(f"{f} ({w:+.2f})" for f, w in pairs)
|
|
1274
|
+
|
|
1275
|
+
ranked = sorted(names, key=lambda n: -(coords.loc[n].sum()))
|
|
1276
|
+
role_rows = {
|
|
1277
|
+
r: next((n for n, rr in role_of.items() if rr == r), "—")
|
|
1278
|
+
for r in ("best", "worst", "top", "right")
|
|
1279
|
+
}
|
|
1280
|
+
|
|
1281
|
+
narrative = _llm_text(
|
|
1282
|
+
i18n(lang)["narrative_prompt"].format(
|
|
1283
|
+
left=left,
|
|
1284
|
+
right=right,
|
|
1285
|
+
bottom=bottom,
|
|
1286
|
+
top=top,
|
|
1287
|
+
reference=result.reference,
|
|
1288
|
+
best=role_rows["best"],
|
|
1289
|
+
worst=role_rows["worst"],
|
|
1290
|
+
champ_top=role_rows["top"],
|
|
1291
|
+
champ_right=role_rows["right"],
|
|
1292
|
+
leaderboard=", ".join(ranked[:8]),
|
|
1293
|
+
),
|
|
1294
|
+
model,
|
|
1295
|
+
use_llm,
|
|
1296
|
+
fallback=(
|
|
1297
|
+
f"The map's horizontal axis contrasts **{left}** (left) with **{right}** "
|
|
1298
|
+
f"(right); the vertical contrasts **{bottom}** (bottom) with **{top}** "
|
|
1299
|
+
f"(top), together capturing {evr.sum():.0%} of the variation between "
|
|
1300
|
+
f"approaches. **{result.reference}** anchors the top-right as the "
|
|
1301
|
+
f"reference leader, strongest on the {right.lower()} and {top.lower()} "
|
|
1302
|
+
f"directions. **{role_rows['worst']}** sits opposite as the weakest on "
|
|
1303
|
+
f"these dimensions, while among the challengers **{role_rows['top']}** "
|
|
1304
|
+
f"reaches furthest toward {top.lower()} and **{role_rows['right']}** "
|
|
1305
|
+
f"furthest toward {right.lower()}."
|
|
1306
|
+
),
|
|
1307
|
+
)
|
|
1308
|
+
|
|
1309
|
+
lines = [
|
|
1310
|
+
f"# {result.reference}",
|
|
1311
|
+
"",
|
|
1312
|
+
"## Interpretation",
|
|
1313
|
+
"",
|
|
1314
|
+
narrative,
|
|
1315
|
+
"",
|
|
1316
|
+
"## Axes",
|
|
1317
|
+
"",
|
|
1318
|
+
f"- **Horizontal — {left} ↔ {right}** ({evr[0]:.0%} of variance). "
|
|
1319
|
+
f"Columns by weight: {loading_line(0)}.",
|
|
1320
|
+
f"- **Vertical — {bottom} ↔ {top}** ({evr[1]:.0%} of variance). "
|
|
1321
|
+
f"Columns by weight: {loading_line(1)}.",
|
|
1322
|
+
f"- Together the two axes retain **{evr.sum():.0%}** of the total variation; "
|
|
1323
|
+
f"the reference was rotated {result.rotation_deg:+.1f}° to reach the top-right.",
|
|
1324
|
+
"",
|
|
1325
|
+
"## Highlighted approaches",
|
|
1326
|
+
"",
|
|
1327
|
+
f"- **Leader (reference):** {role_rows['best']}",
|
|
1328
|
+
f"- **Weakest overall:** {role_rows['worst']} (lowest projection on the leader diagonal)",
|
|
1329
|
+
f"- **Strongest toward {top}:** {role_rows['top']} (challenger furthest up "
|
|
1330
|
+
"the vertical axis)",
|
|
1331
|
+
f"- **Strongest toward {right}:** {role_rows['right']} (challenger furthest "
|
|
1332
|
+
"along the horizontal axis)",
|
|
1333
|
+
"",
|
|
1334
|
+
"## Leaderboard (by combined axis score)",
|
|
1335
|
+
"",
|
|
1336
|
+
]
|
|
1337
|
+
lines += [
|
|
1338
|
+
f"{i}. {n} ({coords.loc[n, 'axis-1']:+.2f}, {coords.loc[n, 'axis-2']:+.2f})"
|
|
1339
|
+
for i, n in enumerate(ranked, 1)
|
|
1340
|
+
]
|
|
1341
|
+
lines += ["", "*Coordinates are PCA units; see the companion YAML for full coefficients.*", ""]
|
|
1342
|
+
return "\n".join(lines)
|
|
1343
|
+
|
|
1344
|
+
|
|
1345
|
+
def results_yaml(
|
|
1346
|
+
df: pd.DataFrame,
|
|
1347
|
+
result: PCAResult,
|
|
1348
|
+
roles: list[str],
|
|
1349
|
+
poles: list[str],
|
|
1350
|
+
axis_names: list[str],
|
|
1351
|
+
colors: list[str],
|
|
1352
|
+
) -> str:
|
|
1353
|
+
"""Everything about the fit as YAML: metadata, axis loadings, and per-approach
|
|
1354
|
+
coordinates, roles, colours, and original attribute values."""
|
|
1355
|
+
evr = result.explained_variance_ratio
|
|
1356
|
+
left, right, bottom, top = poles
|
|
1357
|
+
feats = result.features
|
|
1358
|
+
raw = impute(df)
|
|
1359
|
+
|
|
1360
|
+
doc = {
|
|
1361
|
+
"meta": {
|
|
1362
|
+
"reference": result.reference,
|
|
1363
|
+
"rotation_deg": round(result.rotation_deg, 3),
|
|
1364
|
+
"explained_variance_ratio": [round(float(v), 4) for v in evr],
|
|
1365
|
+
"cumulative_variance": round(float(evr.sum()), 4),
|
|
1366
|
+
"n_approaches": len(result.names),
|
|
1367
|
+
"attributes": feats,
|
|
1368
|
+
"lower_is_better": sorted(result.lower),
|
|
1369
|
+
},
|
|
1370
|
+
"axes": {
|
|
1371
|
+
"axis_1": {
|
|
1372
|
+
"name": axis_names[0],
|
|
1373
|
+
"pole_left": left,
|
|
1374
|
+
"pole_right": right,
|
|
1375
|
+
"loadings": {
|
|
1376
|
+
f: round(float(w), 4) for f, w in zip(feats, result.components[0], strict=False)
|
|
1377
|
+
},
|
|
1378
|
+
},
|
|
1379
|
+
"axis_2": {
|
|
1380
|
+
"name": axis_names[1],
|
|
1381
|
+
"pole_bottom": bottom,
|
|
1382
|
+
"pole_top": top,
|
|
1383
|
+
"loadings": {
|
|
1384
|
+
f: round(float(w), 4) for f, w in zip(feats, result.components[1], strict=False)
|
|
1385
|
+
},
|
|
1386
|
+
},
|
|
1387
|
+
},
|
|
1388
|
+
"approaches": [
|
|
1389
|
+
{
|
|
1390
|
+
"name": n,
|
|
1391
|
+
"coordinates": {"axis_1": round(float(x), 4), "axis_2": round(float(y), 4)},
|
|
1392
|
+
"role": role,
|
|
1393
|
+
"color": color,
|
|
1394
|
+
"attributes": {f: round(float(raw.loc[n, f]), 3) for f in feats},
|
|
1395
|
+
}
|
|
1396
|
+
for n, (x, y), role, color in zip(
|
|
1397
|
+
result.names, result.scores, roles, colors, strict=False
|
|
1398
|
+
)
|
|
1399
|
+
],
|
|
1400
|
+
}
|
|
1401
|
+
return yaml.dump(doc, sort_keys=False, allow_unicode=True, width=100)
|
|
1402
|
+
|
|
1403
|
+
|
|
1404
|
+
def export_all(
|
|
1405
|
+
df: pd.DataFrame,
|
|
1406
|
+
result: PCAResult,
|
|
1407
|
+
roles: list[str],
|
|
1408
|
+
poles: list[str],
|
|
1409
|
+
axis_names: list[str],
|
|
1410
|
+
colors: list[str],
|
|
1411
|
+
stem: str,
|
|
1412
|
+
model: str = DEFAULT_MODEL,
|
|
1413
|
+
use_llm: bool = True,
|
|
1414
|
+
noun_plural: str = "Approaches",
|
|
1415
|
+
title: str | None = None,
|
|
1416
|
+
) -> list[str]:
|
|
1417
|
+
"""Write the full three-fold deliverable for one table: figures (PNG + SVG +
|
|
1418
|
+
Vega JSON), a Markdown interpretation, and a YAML of coordinates + coefficients.
|
|
1419
|
+
Returns the list of paths written.
|
|
1420
|
+
"""
|
|
1421
|
+
spec = to_vega(
|
|
1422
|
+
result, roles=roles, poles=poles, colors=colors, noun_plural=noun_plural, title=title
|
|
1423
|
+
)
|
|
1424
|
+
written = render_figures(spec, stem)
|
|
1425
|
+
for path, text in [
|
|
1426
|
+
(f"{stem}.vl.json", json.dumps(spec, indent=2, ensure_ascii=False)),
|
|
1427
|
+
(f"{stem}.md", analysis_markdown(result, roles, poles, model, use_llm)),
|
|
1428
|
+
(f"{stem}.yaml", results_yaml(df, result, roles, poles, axis_names, colors)),
|
|
1429
|
+
]:
|
|
1430
|
+
with open(path, "w", encoding="utf-8") as fh:
|
|
1431
|
+
fh.write(text)
|
|
1432
|
+
written.append(path)
|
|
1433
|
+
return written
|
|
1434
|
+
|
|
1435
|
+
|
|
1436
|
+
# --------------------------------------------------------------------------- #
|
|
1437
|
+
# Convenience API — the one-liner library face
|
|
1438
|
+
# --------------------------------------------------------------------------- #
|
|
1439
|
+
@dataclass
|
|
1440
|
+
class Positioning:
|
|
1441
|
+
"""Result of `positioning()` — the map plus everything computed for it."""
|
|
1442
|
+
|
|
1443
|
+
df: pd.DataFrame
|
|
1444
|
+
result: PCAResult
|
|
1445
|
+
roles: list[str]
|
|
1446
|
+
poles: list[str]
|
|
1447
|
+
axis_names: list[str]
|
|
1448
|
+
colors: list[str]
|
|
1449
|
+
noun_singular: str = "Approach"
|
|
1450
|
+
noun_plural: str = "Approaches"
|
|
1451
|
+
title: str = "Approaches in the Quadrant" # fully-localized figure title
|
|
1452
|
+
|
|
1453
|
+
@property
|
|
1454
|
+
def coords(self) -> pd.DataFrame:
|
|
1455
|
+
"""Oriented (axis-1, axis-2) coordinates per option."""
|
|
1456
|
+
return self.result.coords()
|
|
1457
|
+
|
|
1458
|
+
@property
|
|
1459
|
+
def loadings(self) -> pd.DataFrame:
|
|
1460
|
+
"""Axis loadings (criterion weights) per axis."""
|
|
1461
|
+
return self.result.loadings()
|
|
1462
|
+
|
|
1463
|
+
@property
|
|
1464
|
+
def axes(self) -> dict[str, str]:
|
|
1465
|
+
"""The two axis names, e.g. {'x': 'Cost ↔ Innovation', 'y': ...}."""
|
|
1466
|
+
return {"x": self.axis_names[0], "y": self.axis_names[1]}
|
|
1467
|
+
|
|
1468
|
+
@property
|
|
1469
|
+
def role_of(self) -> dict[str, str]:
|
|
1470
|
+
"""Map each option name to its role (best / worst / … / competitor)."""
|
|
1471
|
+
return dict(zip(self.result.names, self.roles, strict=False))
|
|
1472
|
+
|
|
1473
|
+
def to_vega(self) -> dict:
|
|
1474
|
+
"""The Vega-Lite spec for the map."""
|
|
1475
|
+
return to_vega(
|
|
1476
|
+
self.result,
|
|
1477
|
+
self.roles,
|
|
1478
|
+
self.poles,
|
|
1479
|
+
self.colors,
|
|
1480
|
+
noun_plural=self.noun_plural,
|
|
1481
|
+
title=self.title,
|
|
1482
|
+
)
|
|
1483
|
+
|
|
1484
|
+
def to_markdown(self, model: str = DEFAULT_MODEL, use_llm: bool = True) -> str:
|
|
1485
|
+
"""The written interpretation as Markdown."""
|
|
1486
|
+
return analysis_markdown(self.result, self.roles, self.poles, model, use_llm)
|
|
1487
|
+
|
|
1488
|
+
def to_yaml(self) -> str:
|
|
1489
|
+
"""All coordinates + coefficients as YAML."""
|
|
1490
|
+
return results_yaml(
|
|
1491
|
+
self.df, self.result, self.roles, self.poles, self.axis_names, self.colors
|
|
1492
|
+
)
|
|
1493
|
+
|
|
1494
|
+
def figure(self, stem: str) -> list[str]:
|
|
1495
|
+
"""Render the map to `<stem>.png` and `<stem>.svg`; returns the paths."""
|
|
1496
|
+
return render_figures(self.to_vega(), stem)
|
|
1497
|
+
|
|
1498
|
+
def export(
|
|
1499
|
+
self,
|
|
1500
|
+
outdir: str = ".",
|
|
1501
|
+
stem: str | None = None,
|
|
1502
|
+
model: str = DEFAULT_MODEL,
|
|
1503
|
+
use_llm: bool = True,
|
|
1504
|
+
) -> list[str]:
|
|
1505
|
+
"""Write the full three-fold deliverable into `outdir`; returns the paths."""
|
|
1506
|
+
os.makedirs(outdir, exist_ok=True)
|
|
1507
|
+
name = stem or re.sub(r"[^A-Za-z0-9]+", "_", self.result.reference).strip("_").lower()
|
|
1508
|
+
return export_all(
|
|
1509
|
+
self.df,
|
|
1510
|
+
self.result,
|
|
1511
|
+
self.roles,
|
|
1512
|
+
self.poles,
|
|
1513
|
+
self.axis_names,
|
|
1514
|
+
self.colors,
|
|
1515
|
+
os.path.join(outdir, name),
|
|
1516
|
+
model=model,
|
|
1517
|
+
use_llm=use_llm,
|
|
1518
|
+
noun_plural=self.noun_plural,
|
|
1519
|
+
title=self.title,
|
|
1520
|
+
)
|
|
1521
|
+
|
|
1522
|
+
|
|
1523
|
+
def positioning(
|
|
1524
|
+
data: pd.DataFrame | str,
|
|
1525
|
+
reference: int | str = 0,
|
|
1526
|
+
top: str | None = None,
|
|
1527
|
+
right: str | None = None,
|
|
1528
|
+
lower_is_better: list[str] | None = None,
|
|
1529
|
+
model: str = DEFAULT_MODEL,
|
|
1530
|
+
use_llm: bool = True,
|
|
1531
|
+
) -> Positioning:
|
|
1532
|
+
"""Position options from a table in one call.
|
|
1533
|
+
|
|
1534
|
+
`data` is a pandas DataFrame (options × numeric criteria) or a path / raw string
|
|
1535
|
+
of a CSV or Markdown table. `lower_is_better` names criteria where a lower value
|
|
1536
|
+
is better (also picked up from ``(↓)`` header markers). `top` / `right` force a
|
|
1537
|
+
named option into the top-pole / right-pole highlight (see `assign_roles`).
|
|
1538
|
+
Returns a `Positioning` with `.coords`, `.loadings`, `.axes`, `.to_vega()`,
|
|
1539
|
+
`.to_markdown()`, `.to_yaml()`, and `.export(outdir)`.
|
|
1540
|
+
|
|
1541
|
+
>>> pos = positioning("examples/programming_languages.csv")
|
|
1542
|
+
>>> pos.export("out")
|
|
1543
|
+
"""
|
|
1544
|
+
df = data if isinstance(data, pd.DataFrame) else parse_table(data)
|
|
1545
|
+
df, lower = resolve_polarity(df, lower_is_better) # clean names + lower set
|
|
1546
|
+
result = analyze(df, reference=reference, lower_is_better=list(lower))
|
|
1547
|
+
roles = assign_roles(result, top=top, right=right)
|
|
1548
|
+
lang = detect_language(result.features)
|
|
1549
|
+
poles = axis_poles(result, model=model, use_llm=use_llm, lang=lang)
|
|
1550
|
+
singular, plural = noun_forms(
|
|
1551
|
+
str(df.index.name or "Approach"), model=model, use_llm=use_llm, lang=lang
|
|
1552
|
+
)
|
|
1553
|
+
# Localize the whole title, not just the noun: a French table reads
|
|
1554
|
+
# "Voitures dans le quadrant", never "Voitures in the Quadrant".
|
|
1555
|
+
title = i18n(lang)["title_template"].format(plural=plural)
|
|
1556
|
+
return Positioning(
|
|
1557
|
+
df,
|
|
1558
|
+
result,
|
|
1559
|
+
roles,
|
|
1560
|
+
poles,
|
|
1561
|
+
_poles_to_names(poles),
|
|
1562
|
+
gradient_colors(result, roles),
|
|
1563
|
+
singular,
|
|
1564
|
+
plural,
|
|
1565
|
+
title,
|
|
1566
|
+
)
|
|
1567
|
+
|
|
1568
|
+
|
|
1569
|
+
# --------------------------------------------------------------------------- #
|
|
1570
|
+
# CLI
|
|
1571
|
+
# --------------------------------------------------------------------------- #
|
|
1572
|
+
def run(
|
|
1573
|
+
table: str,
|
|
1574
|
+
reference: str = "0",
|
|
1575
|
+
outdir: str = "out",
|
|
1576
|
+
stem: str | None = None,
|
|
1577
|
+
top: str | None = None,
|
|
1578
|
+
right: str | None = None,
|
|
1579
|
+
lower: str = "",
|
|
1580
|
+
model: str = DEFAULT_MODEL,
|
|
1581
|
+
no_llm: bool = False,
|
|
1582
|
+
check: bool = False,
|
|
1583
|
+
) -> list[str]:
|
|
1584
|
+
"""Shared CLI core: build the positioning, print a summary, write the files.
|
|
1585
|
+
|
|
1586
|
+
Used by both the argparse (`main`) and click (`main_click`) entry points.
|
|
1587
|
+
`top` / `right` force a named option into the top-pole / right-pole highlight.
|
|
1588
|
+
Returns the list of written paths.
|
|
1589
|
+
"""
|
|
1590
|
+
ref: int | str = int(reference) if reference.lstrip("-").isdigit() else reference
|
|
1591
|
+
lower_cols = [c.strip() for c in lower.split(",") if c.strip()]
|
|
1592
|
+
pos = positioning(
|
|
1593
|
+
parse_table(table),
|
|
1594
|
+
reference=ref,
|
|
1595
|
+
top=top,
|
|
1596
|
+
right=right,
|
|
1597
|
+
lower_is_better=lower_cols,
|
|
1598
|
+
model=model,
|
|
1599
|
+
use_llm=not no_llm,
|
|
1600
|
+
)
|
|
1601
|
+
result, evr = pos.result, pos.result.explained_variance_ratio
|
|
1602
|
+
|
|
1603
|
+
print(f"Parsed {pos.df.shape[0]} options x {pos.df.shape[1]} criteria")
|
|
1604
|
+
print(
|
|
1605
|
+
f"Reference '{result.reference}' rotated by {result.rotation_deg:+.1f} deg "
|
|
1606
|
+
"onto the top-right diagonal\n"
|
|
1607
|
+
)
|
|
1608
|
+
print(
|
|
1609
|
+
f"PCA explained variance: axis-1(PC1)={evr[0]:.1%} axis-2(PC2)={evr[1]:.1%} "
|
|
1610
|
+
f"(cumulative {evr.sum():.1%})\n"
|
|
1611
|
+
)
|
|
1612
|
+
print(f"Axis names: axis-1 = {pos.axis_names[0]!r} axis-2 = {pos.axis_names[1]!r}\n")
|
|
1613
|
+
# poles are [left, right, bottom, top]; name each highlight by its pole word.
|
|
1614
|
+
print("Highlighted options:")
|
|
1615
|
+
highlights = [
|
|
1616
|
+
("best", "leader (reference)"),
|
|
1617
|
+
("worst", "weakest overall"),
|
|
1618
|
+
("top", f"strongest toward {pos.poles[3]!r}"),
|
|
1619
|
+
("right", f"strongest toward {pos.poles[1]!r}"),
|
|
1620
|
+
]
|
|
1621
|
+
for role, label in highlights:
|
|
1622
|
+
who = next((n for n, r in pos.role_of.items() if r == role), "—")
|
|
1623
|
+
print(f" {label:34s}: {who}")
|
|
1624
|
+
print()
|
|
1625
|
+
print("Canonical axes in the oriented frame (loadings):")
|
|
1626
|
+
print(pos.loadings.round(3).to_string(), "\n")
|
|
1627
|
+
|
|
1628
|
+
written = pos.export(outdir, stem=stem, model=model, use_llm=not no_llm)
|
|
1629
|
+
print("Three-fold deliverable written:")
|
|
1630
|
+
for path in written:
|
|
1631
|
+
print(f" {path}")
|
|
1632
|
+
|
|
1633
|
+
if check:
|
|
1634
|
+
# Assess a white-composited render, not the transparent PNG on disk: the
|
|
1635
|
+
# vision model's backend would otherwise flatten transparency onto black and
|
|
1636
|
+
# wrongly report the dark legend as cut off (see `png_on_white`).
|
|
1637
|
+
verdict = vlm_assess(png_on_white(pos.to_vega()), model=model)
|
|
1638
|
+
if verdict:
|
|
1639
|
+
print("\nVision self-check:")
|
|
1640
|
+
for key in ("leader_top_right", "readable", "legend_visible"):
|
|
1641
|
+
print(f" {key:16s}: {verdict.get(key)}")
|
|
1642
|
+
if verdict.get("notes"):
|
|
1643
|
+
print(f" notes : {verdict['notes']}")
|
|
1644
|
+
else:
|
|
1645
|
+
print("\nVision self-check unavailable (model not reachable).")
|
|
1646
|
+
return written
|
|
1647
|
+
|
|
1648
|
+
|
|
1649
|
+
def main(argv: list[str] | None = None) -> None:
|
|
1650
|
+
"""argparse entry point (console command ``standpoint``)."""
|
|
1651
|
+
ap = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
|
1652
|
+
ap.add_argument("table", help="path to a markdown or CSV table")
|
|
1653
|
+
ap.add_argument(
|
|
1654
|
+
"-r",
|
|
1655
|
+
"--reference",
|
|
1656
|
+
default="0",
|
|
1657
|
+
help="row placed top-right: index (default 0) or exact name",
|
|
1658
|
+
)
|
|
1659
|
+
ap.add_argument(
|
|
1660
|
+
"-o",
|
|
1661
|
+
"--outdir",
|
|
1662
|
+
default="out",
|
|
1663
|
+
help="output directory for the three-fold deliverable (default out/)",
|
|
1664
|
+
)
|
|
1665
|
+
ap.add_argument("--stem", help="basename for outputs (default: derived from reference)")
|
|
1666
|
+
ap.add_argument(
|
|
1667
|
+
"--top",
|
|
1668
|
+
help="exact name of the option to highlight as strongest "
|
|
1669
|
+
"toward the top pole (default: picked from the map)",
|
|
1670
|
+
)
|
|
1671
|
+
ap.add_argument(
|
|
1672
|
+
"--right",
|
|
1673
|
+
help="exact name of the option to highlight as strongest "
|
|
1674
|
+
"toward the right pole (default: picked from the map)",
|
|
1675
|
+
)
|
|
1676
|
+
ap.add_argument(
|
|
1677
|
+
"--lower",
|
|
1678
|
+
default="",
|
|
1679
|
+
help="comma-separated criteria where lower is better (e.g. Price,Latency)",
|
|
1680
|
+
)
|
|
1681
|
+
ap.add_argument(
|
|
1682
|
+
"--model",
|
|
1683
|
+
default=DEFAULT_MODEL,
|
|
1684
|
+
help=f"Ollama model for axis naming (default {DEFAULT_MODEL})",
|
|
1685
|
+
)
|
|
1686
|
+
ap.add_argument(
|
|
1687
|
+
"--no-llm", action="store_true", help="skip the LLM; use deterministic axis names"
|
|
1688
|
+
)
|
|
1689
|
+
ap.add_argument(
|
|
1690
|
+
"--check",
|
|
1691
|
+
action="store_true",
|
|
1692
|
+
help="ask the vision model to sanity-check the rendered figure",
|
|
1693
|
+
)
|
|
1694
|
+
a = ap.parse_args(argv)
|
|
1695
|
+
run(a.table, a.reference, a.outdir, a.stem, a.top, a.right, a.lower, a.model, a.no_llm, a.check)
|
|
1696
|
+
|
|
1697
|
+
|
|
1698
|
+
if __name__ == "__main__":
|
|
1699
|
+
main()
|