standpoint 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
standpoint/__init__.py ADDED
@@ -0,0 +1,1699 @@
1
+ """Standpoint — know where each option actually stands.
2
+
3
+ Explainable 2D PCA positioning map from any comparison table.
4
+
5
+ Turn a table of *approaches x criteria* (CSV or Markdown, numeric ratings on any
6
+ scale) into a competitive positioning map, plus a written interpretation and a
7
+ full dump of the coefficients — a three-fold deliverable from one input file.
8
+
9
+ Pipeline
10
+ --------
11
+ 1. parse : CSV or Markdown table -> numeric DataFrame (blanks -> minimum value
12
+ of the non-blank, non-NaN values in that column).
13
+ 2. prepare : normalization (default = z-score standardization, i.e. correlation
14
+ PCA, because PCA is scale-sensitive and criteria carry different
15
+ variances). Missing cells are imputed with the column minimum.
16
+ 3. pca_2d : PCA onto 2 components, keeping the canonical axes (loadings) so
17
+ every axis stays a readable linear combination of the criteria.
18
+ 4. orient : rigidly rotate the 2D scatter so the reference row (the first row by
19
+ default) leads in the TOP-RIGHT, and reposition an all-max reference
20
+ to the best Pareto point; RECOMPUTE the canonical axes in the rotated
21
+ frame (new_components = R(alpha) @ components).
22
+
23
+ Then: automatic roles by principled projection, distinct OKLCH colours by map
24
+ position, local-LLM axis pole names from the loadings, and a de-cluttered
25
+ Vega-Lite figure. `export_all` writes PNG + SVG + Vega JSON + a Markdown analysis
26
+ + a YAML of coordinates and coefficients.
27
+
28
+ Author
29
+ ------
30
+ Warith Harchaoui — https://www.linkedin.com/in/warith-harchaoui
31
+ """
32
+
33
+ from __future__ import annotations
34
+
35
+ __author__ = "Warith Harchaoui"
36
+ __url__ = "https://www.linkedin.com/in/warith-harchaoui"
37
+ __version__ = "0.2.0"
38
+
39
+ import argparse
40
+ import json
41
+ import logging
42
+ import math
43
+ import os
44
+ import re
45
+ from dataclasses import dataclass
46
+
47
+ import numpy as np
48
+ import ollama
49
+ import pandas as pd
50
+ import vl_convert as vlc
51
+ import yaml
52
+ from langdetect import DetectorFactory
53
+ from langdetect import detect as _langdetect
54
+ from sklearn.decomposition import PCA
55
+ from sklearn.preprocessing import StandardScaler
56
+
57
+ DetectorFactory.seed = 0 # deterministic language detection
58
+
59
+ # Library diagnostics go through logging, never bare print (a library must not
60
+ # grab stdout). The CLI in `run()` is the one place that prints on purpose.
61
+ logger = logging.getLogger("standpoint")
62
+
63
+ # "Good Colors" Apple-base palette — https://harchaoui.org/warith/colors/.
64
+ # The four highlighted roles keep a fixed identity hue; the axis cross and labels
65
+ # use neutrals. Every other dot is coloured by its map position (`gradient_colors`).
66
+ PALETTE = {
67
+ "reference": "#FF3B30", # Red — the reference leader (best), sits top-right
68
+ "right": "#007AFF", # Blue — challenger that most defines the right pole
69
+ "worst": "#A52A2A", # Brown — weakest overall, sits bottom-left
70
+ "top": "#AF52DE", # Purple — challenger that most defines the top pole
71
+ "competitor": "#8E8E93", # Gray — placeholder; overridden by gradient_colors
72
+ "axis": "#C7C7CC", # light gray for the centred, dotted axis cross
73
+ "label": "#1C1C1E", # near-black label text
74
+ }
75
+ FONT = "Roboto, -apple-system, Helvetica, Arial, sans-serif"
76
+
77
+ # One qwen vision-LLM for everything: axis pole names, the written analysis, and
78
+ # the visual assessment of the rendered figure (see `vlm_assess`).
79
+ DEFAULT_MODEL = "qwen2.5vl:7b"
80
+
81
+ __all__ = [
82
+ "positioning",
83
+ "Positioning",
84
+ "parse_table",
85
+ "analyze",
86
+ "PCAResult",
87
+ "assign_roles",
88
+ "axis_poles",
89
+ "gradient_colors",
90
+ "to_vega",
91
+ "render_figures",
92
+ "png_on_white",
93
+ "export_all",
94
+ "analysis_markdown",
95
+ "results_yaml",
96
+ "validate_table",
97
+ "resolve_polarity",
98
+ "detect_language",
99
+ "i18n",
100
+ "vlm_assess",
101
+ "run",
102
+ "main",
103
+ ]
104
+
105
+
106
+ # --------------------------------------------------------------------------- #
107
+ # 1. parse
108
+ # --------------------------------------------------------------------------- #
109
+ def _cell_to_number(cell: str) -> float:
110
+ """Convert one table cell to a number (int or float); blanks -> NaN."""
111
+ cell = cell.replace("**", "").strip()
112
+ if cell.lower() in {"", "-", "—", "?", "n/a", "na", "null", "none"}:
113
+ return np.nan
114
+ try:
115
+ return float(cell.replace(",", "."))
116
+ except ValueError:
117
+ return np.nan
118
+
119
+
120
+ def _looks_like_markdown(text: str) -> bool:
121
+ """True if any line starts with a pipe, i.e. the text is a Markdown table."""
122
+ return any(line.lstrip().startswith("|") for line in text.splitlines())
123
+
124
+
125
+ def _parse_markdown(text: str) -> pd.DataFrame:
126
+ """Parse a GitHub-flavoured Markdown table into a numeric DataFrame.
127
+
128
+ The first pipe-delimited row is the header (its first cell names the index);
129
+ the separator row (only pipes/dashes/colons) is dropped, and every remaining
130
+ cell is coerced to a number via `_cell_to_number`.
131
+ """
132
+ rows = [ln.strip() for ln in text.splitlines() if ln.strip().startswith("|")]
133
+ # A GitHub separator row is only pipes/dashes/colons/spaces.
134
+ rows = [r for r in rows if not re.fullmatch(r"[|\s:\-]+", r)]
135
+
136
+ def split(row: str) -> list[str]:
137
+ """Split one table row into stripped cell strings, dropping edge pipes."""
138
+ return [c.strip() for c in row.strip().strip("|").split("|")]
139
+
140
+ header = split(rows[0])
141
+ records, index = [], []
142
+ for row in rows[1:]:
143
+ cells = split(row)
144
+ name = cells[0].replace("**", "").strip()
145
+ index.append(name)
146
+ records.append([_cell_to_number(c) for c in cells[1:]])
147
+ frame = pd.DataFrame(records, index=index, columns=header[1:])
148
+ frame.index.name = header[0] # keep the first-column name (e.g. "Language")
149
+ return frame
150
+
151
+
152
+ def parse_table(source: str) -> pd.DataFrame:
153
+ """Parse a markdown/CSV table (path or raw string) into a numeric DataFrame.
154
+
155
+ The first column becomes the row index (approach names); every other cell is
156
+ parsed as a number (int or float); blanks become NaN.
157
+ """
158
+ text = source
159
+ is_path = "\n" not in source and len(source) < 4096
160
+ if is_path:
161
+ try:
162
+ with open(source, encoding="utf-8") as fh:
163
+ text = fh.read()
164
+ except (OSError, ValueError):
165
+ text, is_path = source, False # not a real path -> treat as raw text
166
+
167
+ if _looks_like_markdown(text):
168
+ return _parse_markdown(text)
169
+
170
+ df = pd.read_csv(source if is_path else pd.io.common.StringIO(text), index_col=0)
171
+ return df.map(lambda c: _cell_to_number(str(c)))
172
+
173
+
174
+ # --------------------------------------------------------------------------- #
175
+ # i18n — detect the table's language and localize the LLM prompts
176
+ # --------------------------------------------------------------------------- #
177
+ SUPPORTED_LANGS = ("en", "fr", "es")
178
+ _I18N_PATH = os.path.join(os.path.dirname(os.path.abspath(__file__)), "i18n.yaml")
179
+ _I18N_CACHE: dict | None = None
180
+
181
+
182
+ def i18n(lang: str = "en") -> dict:
183
+ """Prompt templates for `lang` (falls back to English), loaded from i18n.yaml."""
184
+ global _I18N_CACHE
185
+ if _I18N_CACHE is None:
186
+ with open(_I18N_PATH, encoding="utf-8") as fh:
187
+ _I18N_CACHE = yaml.safe_load(fh)
188
+ return _I18N_CACHE.get(lang, _I18N_CACHE["en"])
189
+
190
+
191
+ def detect_language(texts: list[str]) -> str:
192
+ """Detect the language (one of SUPPORTED_LANGS) from text; default English.
193
+
194
+ Used on the table's column names so the pole labels and written analysis come
195
+ out in the table's own language.
196
+ """
197
+ sample = " ".join(t for t in texts if t).strip()
198
+ if not sample:
199
+ return "en"
200
+ try:
201
+ lang = _langdetect(sample)
202
+ except Exception:
203
+ return "en"
204
+ return lang if lang in SUPPORTED_LANGS else "en"
205
+
206
+
207
+ # --------------------------------------------------------------------------- #
208
+ # 2. prepare (normalization / preprocessing)
209
+ # --------------------------------------------------------------------------- #
210
+ def validate_table(df: pd.DataFrame) -> None:
211
+ """Raise a clear ``ValueError`` if the table can't be positioned.
212
+
213
+ Needs at least 2 options (rows) and 2 numeric criteria (columns) with some
214
+ variation, and no fully-empty column — otherwise PCA is undefined or degenerate.
215
+ """
216
+ if df.shape[0] < 2:
217
+ raise ValueError(f"need at least 2 options (rows); got {df.shape[0]}.")
218
+ if df.shape[1] < 2:
219
+ raise ValueError(f"need at least 2 criteria (columns); got {df.shape[1]}.")
220
+ all_nan = [c for c in df.columns if df[c].isna().all()]
221
+ if all_nan:
222
+ raise ValueError(f"criteria with no numeric values at all: {all_nan}.")
223
+ constant = [c for c in df.columns if df[c].nunique(dropna=True) <= 1]
224
+ if len(constant) == df.shape[1]:
225
+ raise ValueError("every criterion is constant; nothing to position.")
226
+
227
+
228
+ def _resolve_reference(df: pd.DataFrame, reference: int | str) -> int:
229
+ """Return the row index of the reference, with a helpful error if it's unknown."""
230
+ if isinstance(reference, str):
231
+ if reference not in df.index:
232
+ raise ValueError(f"reference {reference!r} is not one of the options.")
233
+ return int(df.index.get_loc(reference))
234
+ if not -df.shape[0] <= reference < df.shape[0]:
235
+ raise ValueError(f"reference index {reference} is out of range (0..{df.shape[0] - 1}).")
236
+ return int(reference % df.shape[0])
237
+
238
+
239
+ def impute(df: pd.DataFrame) -> pd.DataFrame:
240
+ """Fill missing cells with each column's minimum observed value.
241
+
242
+ A blank criterion is treated as the worst (minimum) value for that criterion,
243
+ rather than the mean — a missing rating should not flatter an approach.
244
+ """
245
+ return df.fillna(df.min(numeric_only=True))
246
+
247
+
248
+ # A header marker declaring a criterion as lower-is-better, e.g. "Price (↓)",
249
+ # "Latency (lower)", "Errors (lower is better)". Stripped from the shown name.
250
+ _LOWER_MARK = re.compile(
251
+ r"\s*\(?\s*(↓|lower(?:\s+is\s+better)?|less\s+is\s+better)\s*\)?\s*$", re.I
252
+ )
253
+
254
+
255
+ def resolve_polarity(
256
+ df: pd.DataFrame, lower_is_better: list[str] | None = None
257
+ ) -> tuple[pd.DataFrame, frozenset[str]]:
258
+ """Detect lower-is-better criteria and return a clean-named copy + their names.
259
+
260
+ A criterion is lower-is-better if its header carries a marker (``Price (↓)``,
261
+ ``Latency (lower)``) or is named in `lower_is_better`. Markers are stripped from
262
+ the column name; the returned set uses the cleaned names.
263
+ """
264
+ explicit = {c.strip() for c in (lower_is_better or [])}
265
+ rename, lower = {}, set()
266
+ for col in df.columns:
267
+ clean = _LOWER_MARK.sub("", str(col)).strip()
268
+ if clean != col: # had a marker
269
+ lower.add(clean)
270
+ if clean in explicit or col in explicit:
271
+ lower.add(clean)
272
+ rename[col] = clean
273
+ out = df.rename(columns=rename)
274
+ return out, frozenset(lower & set(out.columns))
275
+
276
+
277
+ def prepare(df: pd.DataFrame) -> tuple[np.ndarray, list[str]]:
278
+ """Impute missing cells (column minimum), then z-score standardize -> correlation PCA.
279
+
280
+ Standardization (mean 0, sd 1 per criterion) is the right normalization here,
281
+ always: PCA is scale-sensitive, and criteria live on different scales and units,
282
+ so each must get an equal say. A criterion with a larger numeric spread would
283
+ otherwise dominate the components purely because of its units.
284
+ """
285
+ x = StandardScaler().fit_transform(impute(df).to_numpy(dtype=float))
286
+ return x, list(df.columns)
287
+
288
+
289
+ # --------------------------------------------------------------------------- #
290
+ # 3./4. PCA + orientation
291
+ # --------------------------------------------------------------------------- #
292
+ def _rotation(alpha: float) -> np.ndarray:
293
+ """The 2x2 counter-clockwise rotation matrix for an angle `alpha` (radians)."""
294
+ c, s = np.cos(alpha), np.sin(alpha)
295
+ return np.array([[c, -s], [s, c]])
296
+
297
+
298
+ @dataclass
299
+ class PCAResult:
300
+ names: list[str] # row labels
301
+ features: list[str] # attribute names
302
+ scores: np.ndarray # (n, 2) oriented coordinates
303
+ components: np.ndarray # (2, p) oriented canonical axes (loadings)
304
+ explained_variance_ratio: np.ndarray # from the original PCA fit
305
+ rotation_deg: float # alpha applied, in degrees
306
+ reference: str # row placed top-right
307
+ x_std: np.ndarray # (n, p) normalized feature matrix (PCA input)
308
+ lower: frozenset[str] = frozenset() # criteria where lower is better (negated)
309
+
310
+ def loadings(self) -> pd.DataFrame:
311
+ """Criterion weights per oriented axis, as a features x (axis-1, axis-2) frame."""
312
+ return pd.DataFrame(self.components.T, index=self.features, columns=["axis-1", "axis-2"])
313
+
314
+ def coords(self) -> pd.DataFrame:
315
+ """Oriented (axis-1, axis-2) coordinates, one row per option."""
316
+ return pd.DataFrame(self.scores, index=self.names, columns=["axis-1", "axis-2"])
317
+
318
+
319
+ def analyze(
320
+ df: pd.DataFrame,
321
+ reference: int | str = 0,
322
+ soften_reference: float = 1.0,
323
+ lower_is_better: list[str] | None = None,
324
+ ) -> PCAResult:
325
+ """Run the full pipeline: prepare -> PCA(2) -> rotate reference to top-right.
326
+
327
+ The reference row is rotated onto the +45 deg diagonal (equal, positive
328
+ coordinates = top-right corner). The canonical axes are then recomputed in
329
+ the rotated frame so their loadings describe the *displayed* axes.
330
+
331
+ `soften_reference` repositions an all-max reference (a straight-5-stars first
332
+ row otherwise lands as a far outlier) to the best **Pareto** point: max x and
333
+ max y of the competitors, times this factor (default 1.0 = exactly best-in-class
334
+ on each axis, so it weakly dominates everyone without being an outlier). Set to
335
+ 0 or None to keep the raw PCA position.
336
+
337
+ `lower_is_better` names criteria where a lower value is better (price, latency).
338
+ They are negated before the PCA so the whole space is uniformly higher-is-better;
339
+ header markers like ``Price (↓)`` are picked up automatically too.
340
+ """
341
+ df, lower = resolve_polarity(df, lower_is_better)
342
+ validate_table(df)
343
+ ref_idx = _resolve_reference(df, reference)
344
+ signed = df.copy()
345
+ if lower:
346
+ signed[list(lower)] = -signed[list(lower)] # flip so higher is better
347
+ x, features = prepare(signed)
348
+
349
+ pca = PCA(n_components=2)
350
+ scores = pca.fit_transform(x) # (n, 2) in original PC frame
351
+ components = pca.components_ # (2, p) rows = PC1, PC2
352
+
353
+ ref_vec = scores[ref_idx]
354
+ phi = np.arctan2(ref_vec[1], ref_vec[0]) # current angle of the reference
355
+ alpha = np.pi / 4 - phi # rotate it onto +45 deg
356
+
357
+ r = _rotation(alpha)
358
+ scores_rot = scores @ r.T # rotate every point
359
+ components_rot = r @ components # recompute canonical axes
360
+
361
+ if soften_reference:
362
+ # Place the reference at the best *Pareto* point: just beyond best-in-class
363
+ # on each axis, so it weakly dominates every competitor without being a far
364
+ # outlier. Realistic leader, top-right, on the frontier.
365
+ others = np.delete(scores_rot, ref_idx, axis=0)
366
+ ideal_x = max(float(others[:, 0].max()), 0.0) * soften_reference
367
+ ideal_y = max(float(others[:, 1].max()), 0.0) * soften_reference
368
+ if ideal_x > 0 and ideal_y > 0:
369
+ scores_rot[ref_idx] = [ideal_x, ideal_y]
370
+
371
+ # Centre the cloud on the origin (mid-range), so the axis cross sits in its
372
+ # middle with equal margins on every side. Because the reference is the max on
373
+ # both axes, this leaves it at the exact top-right corner.
374
+ scores_rot = scores_rot - (scores_rot.max(axis=0) + scores_rot.min(axis=0)) / 2
375
+
376
+ return PCAResult(
377
+ names=list(df.index),
378
+ features=features,
379
+ scores=scores_rot,
380
+ components=components_rot,
381
+ explained_variance_ratio=pca.explained_variance_ratio_,
382
+ rotation_deg=float(np.degrees(alpha)),
383
+ reference=str(df.index[ref_idx]),
384
+ x_std=x,
385
+ lower=lower,
386
+ )
387
+
388
+
389
+ # --------------------------------------------------------------------------- #
390
+ # roles (colour semantics)
391
+ # --------------------------------------------------------------------------- #
392
+ # Four highlighted roles, each a *domain-agnostic* pick from the map geometry
393
+ # (see `assign_roles`): the leader, the weakest, and the two challengers that
394
+ # reach furthest toward the top and right poles. Highest priority last (wins
395
+ # ties): competitor < right < top < worst < best.
396
+ ROLE_ORDER = ["competitor", "right", "top", "worst", "best"]
397
+ ROLE_STYLE = {
398
+ "best": {"color": PALETTE["reference"], "size": 170, "bold": True},
399
+ "worst": {"color": PALETTE["worst"], "size": 120, "bold": True},
400
+ "top": {"color": PALETTE["top"], "size": 120, "bold": True},
401
+ "right": {"color": PALETTE["right"], "size": 120, "bold": True},
402
+ "competitor": {"color": PALETTE["competitor"], "size": 70, "bold": False},
403
+ }
404
+
405
+
406
+ def _rgb_to_hex(rgb: tuple[float, float, float]) -> str:
407
+ """Convert an (r, g, b) triple in [0, 1] to a clamped ``#RRGGBB`` hex string."""
408
+ r, g, b = (max(0, min(255, round(c * 255))) for c in rgb)
409
+ return f"#{r:02X}{g:02X}{b:02X}"
410
+
411
+
412
+ def _oklab_to_hex(lightness: float, a: float, b: float) -> str:
413
+ """Convert an OKLab colour (Ottosson 2020) to a clamped sRGB hex string."""
414
+ l_ = lightness + 0.3963377774 * a + 0.2158037573 * b
415
+ m_ = lightness - 0.1055613458 * a - 0.0638541728 * b
416
+ s_ = lightness - 0.0894841775 * a - 1.2914855480 * b
417
+ lc, mc, sc = l_**3, m_**3, s_**3
418
+ rgb_lin = (
419
+ +4.0767416621 * lc - 3.3077115913 * mc + 0.2309699292 * sc,
420
+ -1.2684380046 * lc + 2.6097574011 * mc - 0.3413193965 * sc,
421
+ -0.0041960863 * lc - 0.7034186147 * mc + 1.7076147010 * sc,
422
+ )
423
+
424
+ def gamma(u: float) -> float:
425
+ """Apply the sRGB transfer function to one clamped linear channel."""
426
+ u = max(0.0, min(1.0, u))
427
+ return 1.055 * u ** (1 / 2.4) - 0.055 if u > 0.0031308 else 12.92 * u
428
+
429
+ return _rgb_to_hex(tuple(gamma(c) for c in rgb_lin))
430
+
431
+
432
+ # Dot-colour tuning: competitors get vivid OKLCH hues spread EVENLY around the
433
+ # circle (ordered by map direction) so hues are balanced — no muddy midtones, no
434
+ # clumping toward pink — with a gentle per-name lightness spread for extra variety.
435
+ _DOT_CHROMA = 0.125
436
+ _L_LO, _L_HI = 0.62, 0.82
437
+
438
+
439
+ def gradient_colors(result: PCAResult, roles: list[str]) -> list[str]:
440
+ """Distinct, clean per-approach colours.
441
+
442
+ Competitors are placed at EVENLY spaced hues around the OKLCH circle in order
443
+ of their direction on the map — balanced hues, every colour vivid (fixed
444
+ chroma, never a muddy centre), all distinct. Lightness gets a small per-name
445
+ spread for extra separation. Named roles keep their fixed identity hue.
446
+ """
447
+ scores = result.scores
448
+ n = len(scores)
449
+ comps = [i for i in range(n) if roles[i] == "competitor"]
450
+
451
+ # Order competitors by map direction, then hand out evenly spaced hues.
452
+ angles = np.arctan2(scores[:, 1], scores[:, 0])
453
+ ordered = sorted(comps, key=lambda i: float(angles[i]))
454
+ m = max(1, len(ordered))
455
+ lightness_key = sorted(comps, key=lambda i: (sum(map(ord, result.names[i])), i))
456
+ l_of = {
457
+ i: _L_LO + (_L_HI - _L_LO) * (rank / max(1, len(comps) - 1))
458
+ for rank, i in enumerate(lightness_key)
459
+ }
460
+
461
+ colors = [""] * n
462
+ for rank, i in enumerate(ordered):
463
+ hue = 2 * math.pi * (rank / m) # evenly spaced around the wheel
464
+ colors[i] = _oklab_to_hex(l_of[i], _DOT_CHROMA * math.cos(hue), _DOT_CHROMA * math.sin(hue))
465
+ for i, role in enumerate(roles):
466
+ if role != "competitor":
467
+ colors[i] = ROLE_STYLE[role]["color"]
468
+ return colors
469
+
470
+
471
+ def legend_order(scores: np.ndarray) -> list[int]:
472
+ """Indices in reading order that matches the map, starting at the extreme
473
+ top-right: banded rows top -> bottom, and within each row right -> left.
474
+ """
475
+ n = len(scores)
476
+ if n == 0:
477
+ return []
478
+ bands = max(1, round(n**0.5))
479
+ per = math.ceil(n / bands)
480
+ top_to_bottom = sorted(range(n), key=lambda i: -float(scores[i][1]))
481
+ order: list[int] = []
482
+ for b in range(bands):
483
+ row = top_to_bottom[b * per : (b + 1) * per]
484
+ row.sort(key=lambda i: -float(scores[i][0])) # right -> left within the row
485
+ order.extend(row)
486
+ return order
487
+
488
+
489
+ def corner_extremes(scores: np.ndarray) -> dict[str, int]:
490
+ """Index of the most extreme point toward each corner (tr, tl, br, bl)."""
491
+ sx, sy = scores[:, 0], scores[:, 1]
492
+ return {
493
+ "tr": int(np.argmax(sx + sy)),
494
+ "tl": int(np.argmax(sy - sx)),
495
+ "br": int(np.argmax(sx - sy)),
496
+ "bl": int(np.argmax(-sx - sy)),
497
+ }
498
+
499
+
500
+ # Candidate label placements around a dot, as (dir_x, dir_y): right, left, up, down,
501
+ # then the four diagonals — the first that doesn't collide wins.
502
+ _LABEL_DIRS = [(1, 0), (-1, 0), (0, 1), (0, -1), (1, 1), (-1, 1), (1, -1), (-1, -1)]
503
+
504
+
505
+ def _overlaps(a: tuple[float, float, float, float], b: tuple[float, float, float, float]) -> bool:
506
+ """True if two axis-aligned boxes ``(x0, y0, x1, y1)`` intersect."""
507
+ return not (a[2] < b[0] or a[0] > b[2] or a[3] < b[1] or a[1] > b[3])
508
+
509
+
510
+ def label_placements(
511
+ result: PCAResult,
512
+ view_x: float,
513
+ view_y: float,
514
+ width_px: int = 900,
515
+ height_px: int = 760,
516
+ font_px: float = 11.0,
517
+ ) -> dict[int, tuple[float, float]]:
518
+ """Greedy de-clutter: choose which approaches to label and *where* to put each
519
+ label. For every dot (corner extremes first, then outermost), try eight
520
+ placements around it and keep the first that overlaps neither another label nor
521
+ any dot marker. Returns {index: (label_x, label_y)} for the labels that fit.
522
+
523
+ `view_x` / `view_y` are the half-extents of each axis's domain (they can differ),
524
+ so the pixel-to-data conversion is correct even when the map is not square.
525
+ """
526
+ scores = result.scores
527
+ sx = 2 * view_x / width_px # data units per pixel, x
528
+ sy = 2 * view_y / height_px # data units per pixel, y
529
+ pad = 4 * sx
530
+ dot_rx, dot_ry = 7 * sx, 7 * sy
531
+ boxes = [(x - dot_rx, y - dot_ry, x + dot_rx, y + dot_ry) for x, y in scores]
532
+
533
+ corners = list(corner_extremes(scores).values())
534
+ others = sorted(
535
+ (i for i in range(len(result.names)) if i not in corners),
536
+ key=lambda i: -float(np.hypot(*scores[i])),
537
+ )
538
+ placements: dict[int, tuple[float, float]] = {}
539
+ for i in corners + others:
540
+ x, y = scores[i]
541
+ w = len(result.names[i]) * 0.58 * font_px * sx
542
+ h = 1.3 * font_px * sy
543
+ best = None # (distance, box, (lx, ly)) — pick the free side nearest the dot
544
+ for ox, oy in _LABEL_DIRS:
545
+ lx = x + ox * (dot_rx + pad + w / 2) # clear the dot marker, then pad
546
+ ly = y + oy * (dot_ry + pad + h / 2)
547
+ box = (lx - w / 2, ly - h / 2, lx + w / 2, ly + h / 2)
548
+ if any(_overlaps(box, b) for b in boxes):
549
+ continue
550
+ dist = math.hypot(lx - x, ly - y)
551
+ if best is None or dist < best[0]:
552
+ best = (dist, box, (float(lx), float(ly)))
553
+ if best is not None:
554
+ boxes.append(best[1])
555
+ placements[i] = best[2]
556
+ return placements
557
+
558
+
559
+ def _axis_champion(axis_values: np.ndarray, exclude: set[int]) -> int:
560
+ """Index of the option reaching furthest (largest value) along one axis.
561
+
562
+ Parameters
563
+ ----------
564
+ axis_values : np.ndarray
565
+ One column of the oriented scores, e.g. every option's axis-1 coordinate.
566
+ exclude : set[int]
567
+ Row indices to skip (typically the leader, and an already-claimed champion),
568
+ so the same option is never highlighted twice.
569
+
570
+ Returns
571
+ -------
572
+ int
573
+ Row index of the highest not-excluded value along `axis_values`.
574
+ """
575
+ order = np.argsort(axis_values)[::-1] # highest coordinate first
576
+ return int(next(i for i in order if int(i) not in exclude))
577
+
578
+
579
+ def assign_roles(
580
+ result: PCAResult,
581
+ top: str | None = None,
582
+ right: str | None = None,
583
+ ) -> list[str]:
584
+ """Label four options by domain-agnostic map geometry; the rest are competitors.
585
+
586
+ Every pick is read straight off the oriented coordinates, so it means the same
587
+ thing for any table (no per-domain keyword list):
588
+
589
+ best the reference, sitting at the top-right corner by construction;
590
+ worst the weakest overall: the minimum projection onto the top-right hero
591
+ diagonal (equivalently the smallest axis-1 + axis-2);
592
+ top the challenger reaching furthest up the vertical axis — the peer that
593
+ most defines the map's *top* pole (the leader excluded);
594
+ right the challenger reaching furthest along the horizontal axis — the peer
595
+ that most defines the *right* pole (leader and top champion excluded).
596
+
597
+ Parameters
598
+ ----------
599
+ result : PCAResult
600
+ The oriented positioning (`scores` and `reference`).
601
+ top, right : str, optional
602
+ Force a specific option into the top-pole / right-pole highlight by exact
603
+ name, bypassing the geometric pick.
604
+
605
+ Returns
606
+ -------
607
+ list[str]
608
+ One role per option, aligned with ``result.names``; collisions resolve by
609
+ ``ROLE_ORDER`` (best beats worst beats the two champions).
610
+ """
611
+ names = result.names
612
+ scores = result.scores
613
+ best_idx = names.index(result.reference)
614
+
615
+ # Hero axis = the +45 deg diagonal after orientation; project onto (1, 1)/sqrt(2).
616
+ hero_projection = scores @ (np.ones(2) / np.sqrt(2))
617
+ worst_idx = int(next(i for i in np.argsort(hero_projection) if i != best_idx))
618
+
619
+ # The leader is the max on both axes, so a champion is the *next* option out
620
+ # along each axis — the challenger that best embodies that winning pole.
621
+ top_idx = names.index(top) if top is not None else _axis_champion(scores[:, 1], {best_idx})
622
+ right_idx = (
623
+ names.index(right)
624
+ if right is not None
625
+ else _axis_champion(scores[:, 0], {best_idx, top_idx})
626
+ )
627
+
628
+ roles = ["competitor"] * len(names)
629
+ for role in ROLE_ORDER[1:]: # skip "competitor" (default); low -> high priority
630
+ idx = {"right": right_idx, "top": top_idx, "worst": worst_idx, "best": best_idx}[role]
631
+ roles[idx] = role
632
+ return roles
633
+
634
+
635
+ # --------------------------------------------------------------------------- #
636
+ # axis naming (local LLM interprets the loading weights + column names)
637
+ # --------------------------------------------------------------------------- #
638
+ # Expand common acronyms to real words — never show acronyms in the figure.
639
+ _ACRONYM_WORDS = {
640
+ "tco": "Cost",
641
+ "pii": "Privacy",
642
+ "gdpr": "Compliance",
643
+ "ux": "Experience",
644
+ "fr": "French",
645
+ "ev": "Vehicles",
646
+ "ai": "Intelligence",
647
+ "qa": "Quality",
648
+ "stt": "Speech",
649
+ "api": "Interface",
650
+ "diy": "Homemade",
651
+ }
652
+
653
+
654
+ def _deacronym(label: str) -> str:
655
+ """Expand or drop acronym tokens in a label so the figure shows real words."""
656
+ out = []
657
+ for tok in label.split():
658
+ if tok.isupper() and len(tok) <= 5: # looks like an acronym
659
+ expanded = _ACRONYM_WORDS.get(tok.lower())
660
+ if expanded:
661
+ out.append(expanded)
662
+ # unknown acronym -> drop it
663
+ else:
664
+ out.append(tok)
665
+ return " ".join(out).strip()
666
+
667
+
668
+ def _one_word(feature: str) -> str:
669
+ """A single real word from an attribute name — longest non-acronym token,
670
+ expanding known acronyms so the figure never shows abbreviations.
671
+ """
672
+ toks = re.findall(r"[A-Za-z]+", feature)
673
+ words = [t for t in toks if len(t) > 1 and not t.isupper()] # drop acronyms
674
+ if words:
675
+ return max(words, key=len).capitalize()
676
+ for tok in toks: # only acronyms left
677
+ if tok.lower() in _ACRONYM_WORDS:
678
+ return _ACRONYM_WORDS[tok.lower()]
679
+ return _ACRONYM_WORDS.get(feature.strip().lower(), feature.strip().capitalize())
680
+
681
+
682
+ # Small stop-words ignored when comparing labels for shared content words.
683
+ _LABEL_STOP = {
684
+ "and",
685
+ "the",
686
+ "for",
687
+ "with",
688
+ "your",
689
+ "our",
690
+ "per",
691
+ "les",
692
+ "des",
693
+ "las",
694
+ "los",
695
+ "una",
696
+ "por",
697
+ "con",
698
+ "sur",
699
+ "del",
700
+ }
701
+
702
+ # A pole must be a positive quality; these markers signal a drawback (en/fr/es) and
703
+ # get the label rejected — e.g. "High Cost", "Slow", "Expensive" never appear.
704
+ _NEGATIVE_WORDS = {
705
+ "high",
706
+ "low",
707
+ "expensive",
708
+ "costly",
709
+ "slow",
710
+ "complex",
711
+ "complicated",
712
+ "poor",
713
+ "weak",
714
+ "insecure",
715
+ "unreliable",
716
+ "difficult",
717
+ "limited",
718
+ "hidden",
719
+ "risky",
720
+ "lack",
721
+ "worse",
722
+ "bad",
723
+ "élevé",
724
+ "eleve",
725
+ "cher",
726
+ "lent",
727
+ "complexe",
728
+ "coûteux",
729
+ "couteux",
730
+ "difficile",
731
+ "faible",
732
+ "alto",
733
+ "caro",
734
+ "lento",
735
+ "complejo",
736
+ "costoso",
737
+ "débil",
738
+ "debil",
739
+ "riesgo",
740
+ }
741
+
742
+
743
+ def _content_words(label: str) -> set[str]:
744
+ """Significant lowercase words in a label (>= 3 letters, minus stop-words)."""
745
+ return {
746
+ t for t in re.findall(r"[a-zA-Z]+", label.lower()) if len(t) >= 3 and t not in _LABEL_STOP
747
+ }
748
+
749
+
750
+ def _clean_label(label: str) -> str:
751
+ """Expand acronyms, split camelCase, and keep at most three words."""
752
+ label = _deacronym(label) if label else ""
753
+ label = re.sub(r"(?<=[a-z])(?=[A-Z])", " ", label).strip() # split camelCase
754
+ return " ".join(label.split()[:3])
755
+
756
+
757
+ def finalize_poles(raw: list[str], fallback: list[str]) -> list[str]:
758
+ """Turn raw LLM pole labels into four clean, distinct, non-antonymous labels.
759
+
760
+ Enforces: real words (no acronyms), at most three words, no label repeated, and
761
+ no two labels sharing a content word — which rules out antonym pairs such as
762
+ 'Cost Efficient' / 'High Cost'. A rejected label is replaced by its
763
+ loading-derived fallback (drawn from a different criterion).
764
+ """
765
+
766
+ def bad(w: str) -> bool:
767
+ """True if label `w` must be rejected: empty, duplicate, shares a content
768
+ word with an already-accepted label (rules out antonym pairs), or contains
769
+ a negative word (a pole must name a positive quality).
770
+ """
771
+ cw = _content_words(w)
772
+ return (
773
+ not w or w.lower() in seen or bool(cw & used_words) or bool(cw & _NEGATIVE_WORDS)
774
+ ) # never a drawback / negative
775
+
776
+ out: list[str] = []
777
+ seen: set[str] = set()
778
+ used_words: set[str] = set()
779
+ for i, (label, fb) in enumerate(zip(raw, fallback, strict=False)):
780
+ w = _clean_label(label)
781
+ if bad(w):
782
+ w = _clean_label(fb) # fall back to the loading word
783
+ if bad(w):
784
+ w = f"{w} {i}"
785
+ seen.add(w.lower())
786
+ used_words |= _content_words(w)
787
+ out.append(w)
788
+ return out
789
+
790
+
791
+ def _fallback_poles(components: np.ndarray, features: list[str]) -> list[str]:
792
+ """Four distinct pole words [left, right, bottom, top] from the loadings.
793
+
794
+ left/right = low/high end of axis-1; bottom/top = low/high end of axis-2.
795
+ Each pole takes the most extreme not-yet-used attribute at that end.
796
+ """
797
+ specs = [(0, 1), (0, -1), (1, 1), (1, -1)] # (axis, +1=ascending->low end first)
798
+ used: set[str] = set()
799
+ poles: list[str] = []
800
+ for axis, sign in specs:
801
+ order = np.argsort(components[axis])[::sign] # sign +1 -> low end first
802
+ word = next(
803
+ (w for i in order if (w := _one_word(features[i])).lower() not in used),
804
+ _one_word(features[order[0]]),
805
+ )
806
+ used.add(word.lower())
807
+ poles.append(word)
808
+ return poles
809
+
810
+
811
+ def _poles_to_names(poles: list[str]) -> list[str]:
812
+ """[left, right, bottom, top] -> ['left ↔ right', 'bottom ↔ top']."""
813
+ left, right, bottom, top = poles
814
+ return [f"{left} ↔ {right}", f"{bottom} ↔ {top}"]
815
+
816
+
817
+ def axis_poles(
818
+ result: PCAResult, model: str = DEFAULT_MODEL, use_llm: bool = True, lang: str | None = None
819
+ ) -> list[str]:
820
+ """Four distinct pole labels [left, right, bottom, top] for the two axes.
821
+
822
+ Each PCA axis is a weighted mix of the criteria. The local LLM names each pole
823
+ (1-3 words) for what the approaches at that end are collectively strongest at,
824
+ from the signed loadings and the original column names — in the table's own
825
+ language (auto-detected from the column names; see `i18n.yaml`). Falls back to
826
+ loading-derived distinct words if the LLM is unavailable or misbehaves.
827
+ """
828
+ feats = result.features
829
+ fallback_poles = _fallback_poles(result.components, feats)
830
+ if not use_llm:
831
+ return fallback_poles
832
+ if lang is None:
833
+ lang = detect_language(feats)
834
+ tpl = i18n(lang)
835
+
836
+ try:
837
+ # Every rating is higher-is-better, so a pole is best described by the
838
+ # criteria approaches THERE score high on (its sign of the loading).
839
+ def show(f: str) -> str:
840
+ """Present a criterion to the model, flagging negated (lower-better) ones.
841
+
842
+ A lower-is-better criterion was negated for the PCA, so a high score
843
+ means a LOW raw value: show it as "low <name>" so the model names the
844
+ benefit ("Affordable") rather than the drawback ("Expensive").
845
+ """
846
+ return f"low {f}" if f in result.lower else f
847
+
848
+ def pole_strengths(k: int, sign: int) -> str:
849
+ """Criteria (with weights) that define one end of axis `k`.
850
+
851
+ `sign` selects the end: +1 for the positive-loading pole, -1 for the
852
+ negative one. Returns them strongest-first as a human-readable string,
853
+ or "—" when nothing loads meaningfully on that end.
854
+ """
855
+ pairs = [
856
+ (f, w)
857
+ for f, w in zip(feats, result.components[k], strict=False)
858
+ if (w > 0) == (sign > 0) and abs(w) > 0.05
859
+ ]
860
+ pairs.sort(key=lambda t: -abs(t[1]))
861
+ return ", ".join(f"{show(f)} (weight {abs(w):.2f})" for f, w in pairs) or "—"
862
+
863
+ # Glossary of any acronyms present in the columns, so the model translates
864
+ # them instead of echoing them (built from the actual column names).
865
+ present = {
866
+ a.upper(): w
867
+ for a, w in _ACRONYM_WORDS.items()
868
+ if any(a.upper() in f.upper() for f in feats)
869
+ }
870
+ glossary = (
871
+ (tpl["glossary_prefix"] + "; ".join(f"{k} = {v}" for k, v in present.items()) + ".\n\n")
872
+ if present
873
+ else ""
874
+ )
875
+ prompt = tpl["axis_prompt"].format(
876
+ glossary=glossary,
877
+ left=pole_strengths(0, -1),
878
+ right=pole_strengths(0, +1),
879
+ bottom=pole_strengths(1, -1),
880
+ top=pole_strengths(1, +1),
881
+ )
882
+ schema = {
883
+ "type": "object",
884
+ "properties": {k: {"type": "string"} for k in ("left", "right", "bottom", "top")},
885
+ "required": ["left", "right", "bottom", "top"],
886
+ }
887
+ resp = ollama.chat(
888
+ model=model,
889
+ format=schema,
890
+ options={"temperature": 0},
891
+ messages=[{"role": "user", "content": prompt}],
892
+ )
893
+ data = json.loads(resp["message"]["content"])
894
+ raw = [str(data.get(k, "")) for k in ("left", "right", "bottom", "top")]
895
+ # Clean, de-duplicate, and reject antonym/shared-word pairs.
896
+ return finalize_poles(raw, fallback_poles)
897
+ except Exception as exc: # ollama missing / model absent / bad JSON
898
+ logger.warning("axis naming: LLM unavailable (%s); using deterministic names", exc)
899
+ return fallback_poles
900
+
901
+
902
+ def noun_forms(
903
+ word: str, model: str = DEFAULT_MODEL, use_llm: bool = True, lang: str | None = None
904
+ ) -> tuple[str, str]:
905
+ """Singular and plural of `word` (the first-column name), in its own language.
906
+
907
+ Used for the figure title and legend heading, so a table of "Language" reads
908
+ "Languages in the Quadrant". The prompt lives in `i18n.yaml`. Falls back to a
909
+ naive `+s` plural without a model.
910
+ """
911
+ word = (word or "Approach").strip() or "Approach"
912
+ if len(word) > 1 and word.lower().endswith("s"): # looks plural already
913
+ naive = (word[:-1].capitalize(), word.capitalize())
914
+ else:
915
+ naive = (word.capitalize(), word.capitalize() + "s")
916
+ if not use_llm:
917
+ return naive
918
+ if lang is None:
919
+ lang = detect_language([word])
920
+ try:
921
+ schema = {
922
+ "type": "object",
923
+ "properties": {"singular": {"type": "string"}, "plural": {"type": "string"}},
924
+ "required": ["singular", "plural"],
925
+ }
926
+ resp = ollama.chat(
927
+ model=model,
928
+ format=schema,
929
+ options={"temperature": 0},
930
+ messages=[{"role": "user", "content": i18n(lang)["noun_prompt"].format(word=word)}],
931
+ )
932
+ data = json.loads(resp["message"]["content"])
933
+ s = (str(data.get("singular") or "").strip() or naive[0]).capitalize()
934
+ p = (str(data.get("plural") or "").strip() or naive[1]).capitalize()
935
+ # Guard against the model swapping in a synonym (e.g. Voiture -> Véhicules):
936
+ # a valid form must share a prefix with the actual column word.
937
+ prefix = word.lower()[: max(3, len(word) - 2)]
938
+ if not s.lower().startswith(prefix):
939
+ s = naive[0]
940
+ if not p.lower().startswith(prefix):
941
+ p = naive[1]
942
+ return s, p
943
+ except Exception:
944
+ return naive
945
+
946
+
947
+ # --------------------------------------------------------------------------- #
948
+ # Vega-Lite
949
+ # --------------------------------------------------------------------------- #
950
+ def to_vega(
951
+ result: PCAResult,
952
+ roles: list[str] | None = None,
953
+ poles: list[str] | None = None,
954
+ colors: list[str] | None = None,
955
+ noun_plural: str = "Approaches",
956
+ title: str | None = None,
957
+ ) -> dict:
958
+ """Build a self-contained Vega-Lite v5 spec (inline data) for the map.
959
+
960
+ Layers, bottom to top: a centred cross of axes through the origin (the neutral
961
+ intersection), every approach coloured by its position (Apple-wheel HSV), the
962
+ four pole words at the axis ends, and labels for the four corner extremes. No
963
+ frame, spines, ticks, numeric scales, or arrows.
964
+
965
+ `title` is the fully-localized figure title (e.g. "Voitures dans le quadrant");
966
+ when omitted it defaults to the English "<plural> in the Quadrant" so direct
967
+ callers still get a sensible heading.
968
+ """
969
+ ref = result.reference
970
+ names = result.names
971
+ if roles is None:
972
+ roles = ["best" if n == ref else "competitor" for n in names]
973
+ if poles is None:
974
+ poles = _fallback_poles(result.components, result.features)
975
+ left, right, bottom, top = poles
976
+
977
+ if colors is None:
978
+ colors = gradient_colors(result, roles)
979
+ n = len(names)
980
+
981
+ # Per-axis extents so each axis fills its own space: a low-variance axis (e.g.
982
+ # PC2) is not squashed flat against the cross. Each axis gets its own domain.
983
+ span_x = float(np.abs(result.scores[:, 0]).max()) or 1.0
984
+ span_y = float(np.abs(result.scores[:, 1]).max()) or 1.0
985
+ # Wide margin: the dots occupy the central ~65%, leaving the outer band clear
986
+ # for the pole phrases at the axis ends.
987
+ view_x, view_y = span_x * 1.55, span_y * 1.55
988
+
989
+ # Sizes adapt to the option count: bigger when few, smaller when many.
990
+ def _scaled(lo: int, hi: int, few: int = 8, many: int = 40) -> int:
991
+ """Interpolate a size between `hi` (at `few` options) and `lo` (at `many`).
992
+
993
+ Keeps the map legible across table sizes: large glyphs on a sparse map,
994
+ smaller ones once the plot gets crowded. Clamped outside ``[few, many]``.
995
+ """
996
+ t = (min(max(n, few), many) - few) / (many - few)
997
+ return round(hi + (lo - hi) * t)
998
+
999
+ label_font = _scaled(11, 17)
1000
+ pole_font = _scaled(13, 22)
1001
+ legend_font = _scaled(9, 13)
1002
+ dot_size = _scaled(90, 240)
1003
+
1004
+ placements = label_placements(result, view_x, view_y, font_px=label_font)
1005
+
1006
+ # Legend follows the map: rows top -> bottom, left -> right within each row.
1007
+ order = legend_order(result.scores)
1008
+ legend_names = [names[i] for i in order]
1009
+ legend_colors = [colors[i] for i in order]
1010
+
1011
+ points = [
1012
+ {
1013
+ "name": nm,
1014
+ "axis1": float(x),
1015
+ "axis2": float(y),
1016
+ "role": r,
1017
+ "color": c,
1018
+ "label": nm if i in placements else "",
1019
+ "labelx": placements.get(i, (x, y))[0],
1020
+ "labely": placements.get(i, (x, y))[1],
1021
+ }
1022
+ for i, ((x, y), nm, r, c) in enumerate(
1023
+ zip(result.scores, names, roles, colors, strict=False)
1024
+ )
1025
+ ]
1026
+
1027
+ xdom = {"domain": [-view_x, view_x]}
1028
+ ydom = {"domain": [-view_y, view_y]}
1029
+ bare = {"domain": False, "ticks": False, "labels": False, "grid": False, "title": None}
1030
+ xenc = {"field": "axis1", "type": "quantitative", "scale": xdom, "axis": bare}
1031
+ yenc = {"field": "axis2", "type": "quantitative", "scale": ydom, "axis": bare}
1032
+
1033
+ def rule(x0: float, x1: float, y0: float, y1: float) -> dict:
1034
+ """A Vega-Lite layer drawing one dotted axis segment in data coordinates."""
1035
+ return {
1036
+ "data": {"values": [{}]},
1037
+ "mark": {"type": "rule", "color": PALETTE["axis"], "size": 1.2, "strokeDash": [2, 4]},
1038
+ "encoding": {
1039
+ "x": {"datum": x0, "type": "quantitative", "scale": xdom, "axis": bare},
1040
+ "x2": {"datum": x1},
1041
+ "y": {"datum": y0, "type": "quantitative", "scale": ydom, "axis": bare},
1042
+ "y2": {"datum": y1},
1043
+ },
1044
+ }
1045
+
1046
+ def pole_label(x: float, y: float, text: str, align: str, baseline: str) -> dict:
1047
+ """A Vega-Lite text layer placing one italic pole word at an axis end."""
1048
+ return {
1049
+ "data": {"values": [{"x": x, "y": y, "t": text}]},
1050
+ "mark": {
1051
+ "type": "text",
1052
+ "fontSize": pole_font,
1053
+ "fontStyle": "italic",
1054
+ "color": "#6E6E73",
1055
+ "align": align,
1056
+ "baseline": baseline,
1057
+ },
1058
+ "encoding": {
1059
+ "x": {"field": "x", "type": "quantitative", "scale": xdom, "axis": bare},
1060
+ "y": {"field": "y", "type": "quantitative", "scale": ydom, "axis": bare},
1061
+ "text": {"field": "t", "type": "nominal"},
1062
+ },
1063
+ }
1064
+
1065
+ edge_x, edge_y = view_x * 0.98, view_y * 0.98 # axes span the full view
1066
+ gap_x, gap_y = span_x * 0.04, span_y * 0.04 # keep pole words off the lines
1067
+ layers = [
1068
+ rule(-edge_x, edge_x, 0, 0), # horizontal axis
1069
+ rule(0, 0, -edge_y, edge_y), # vertical axis
1070
+ pole_label(edge_x, gap_y, right, "right", "bottom"),
1071
+ pole_label(-edge_x, gap_y, left, "left", "bottom"),
1072
+ pole_label(gap_x, edge_y, top, "left", "top"),
1073
+ pole_label(gap_x, -edge_y, bottom, "left", "bottom"),
1074
+ { # every dot coloured by position; legend maps name -> colour
1075
+ "data": {"values": points},
1076
+ "mark": {
1077
+ "type": "point",
1078
+ "filled": True,
1079
+ "opacity": 0.95,
1080
+ "stroke": "white",
1081
+ "strokeWidth": 1,
1082
+ "size": dot_size,
1083
+ },
1084
+ "encoding": {
1085
+ "x": xenc,
1086
+ "y": yenc,
1087
+ "color": {
1088
+ "field": "name",
1089
+ "type": "nominal",
1090
+ "scale": {"domain": legend_names, "range": legend_colors},
1091
+ "legend": {
1092
+ "title": noun_plural,
1093
+ "symbolLimit": 0,
1094
+ "labelFontSize": legend_font,
1095
+ "symbolOpacity": 1,
1096
+ },
1097
+ },
1098
+ "tooltip": [
1099
+ {"field": "name", "type": "nominal"},
1100
+ {"field": "role", "type": "nominal"},
1101
+ {"field": "axis1", "type": "quantitative", "format": ".2f"},
1102
+ {"field": "axis2", "type": "quantitative", "format": ".2f"},
1103
+ ],
1104
+ },
1105
+ },
1106
+ { # labels — de-cluttered, placed on whichever side is free
1107
+ "data": {"values": points},
1108
+ "transform": [{"filter": "datum.label != ''"}],
1109
+ "mark": {
1110
+ "type": "text",
1111
+ "align": "center",
1112
+ "baseline": "middle",
1113
+ "fontSize": label_font,
1114
+ "color": PALETTE["label"],
1115
+ },
1116
+ "encoding": {
1117
+ "x": {"field": "labelx", "type": "quantitative"},
1118
+ "y": {"field": "labely", "type": "quantitative"},
1119
+ "text": {"field": "label", "type": "nominal"},
1120
+ },
1121
+ },
1122
+ ]
1123
+
1124
+ # The plotting area is tall enough that the one-row-per-approach legend beside
1125
+ # it is never taller than the canvas (so it can't be clipped).
1126
+ height = max(720, 24 * n + 140)
1127
+ if title is None: # direct callers get the English default; localized via i18n
1128
+ title = f"{noun_plural} in the Quadrant"
1129
+ return {
1130
+ "$schema": "https://vega.github.io/schema/vega-lite/v5.json",
1131
+ "title": {"text": title, "font": FONT, "fontSize": 18},
1132
+ # Transparent background: Vega-Lite otherwise bakes an opaque white rectangle
1133
+ # into the PNG/SVG. Null lets the map drop cleanly onto any page or slide.
1134
+ "background": None,
1135
+ "width": 1000,
1136
+ "height": height,
1137
+ "autosize": {"type": "pad", "resize": True}, # grow to fit the legend
1138
+ "config": {
1139
+ "font": FONT,
1140
+ "padding": 12,
1141
+ "view": {"stroke": None}, # no box around the plotting area
1142
+ "axis": {
1143
+ "grid": False,
1144
+ "domain": False,
1145
+ "ticks": False,
1146
+ "labels": False,
1147
+ "labelFont": FONT,
1148
+ "titleFont": FONT,
1149
+ },
1150
+ "text": {"font": FONT},
1151
+ },
1152
+ "layer": layers,
1153
+ }
1154
+
1155
+
1156
+ # --------------------------------------------------------------------------- #
1157
+ # Three-fold export: figures (PNG + SVG + Vega JSON), markdown, YAML
1158
+ # --------------------------------------------------------------------------- #
1159
+ def render_figures(spec: dict, stem: str) -> list[str]:
1160
+ """Rasterize/vectorize a Vega-Lite spec to transparent and white PNG + SVG.
1161
+
1162
+ Writes four files: the transparent `<stem>.png` / `<stem>.svg` (the default, for
1163
+ dropping onto any coloured page) and a white-background `<stem>.white.png` /
1164
+ `<stem>.white.svg` (for dark surfaces — e.g. GitHub dark mode — where the map's
1165
+ near-black labels would otherwise vanish on a transparent background). Returns
1166
+ the four paths in that order.
1167
+ """
1168
+ written: list[str] = []
1169
+ # `spec` already carries background:null; the ".white" pass overrides it. Same
1170
+ # layout both times, so the only difference is the baked-in backdrop.
1171
+ for suffix, variant in ((".", spec), (".white.", {**spec, "background": "white"})):
1172
+ png_path, svg_path = f"{stem}{suffix}png", f"{stem}{suffix}svg"
1173
+ with open(png_path, "wb") as fh:
1174
+ fh.write(vlc.vegalite_to_png(vl_spec=variant, scale=2.0))
1175
+ with open(svg_path, "w", encoding="utf-8") as fh:
1176
+ fh.write(vlc.vegalite_to_svg(vl_spec=variant))
1177
+ written += [png_path, svg_path]
1178
+ return written
1179
+
1180
+
1181
+ def png_on_white(spec: dict) -> bytes:
1182
+ """Render `spec` to PNG bytes on an opaque white background.
1183
+
1184
+ The exported figures are transparent, but the vision self-check sends the image
1185
+ to a model whose backend flattens transparency onto a dark canvas — which would
1186
+ hide the near-black labels and legend and make the check misfire. White is the
1187
+ figure's intended reading surface, so the check runs against a white-composited
1188
+ copy rather than the transparent file on disk.
1189
+ """
1190
+ return vlc.vegalite_to_png(vl_spec={**spec, "background": "white"}, scale=2.0)
1191
+
1192
+
1193
+ def vlm_assess(image: str | bytes, model: str = DEFAULT_MODEL) -> dict:
1194
+ """Ask the qwen vision-LLM to sanity-check a rendered positioning map.
1195
+
1196
+ `image` is a PNG path or raw PNG bytes (bytes let the caller assess a
1197
+ white-composited render without touching the transparent file on disk). Returns
1198
+ a verdict dict — whether the red leader dot sits top-right, whether the labels
1199
+ are readable, and whether the legend is fully visible — plus free-text notes.
1200
+ Empty dict if the model or a rendered image is unavailable.
1201
+ """
1202
+ schema = {
1203
+ "type": "object",
1204
+ "properties": {
1205
+ "leader_top_right": {"type": "boolean"},
1206
+ "readable": {"type": "boolean"},
1207
+ "legend_visible": {"type": "boolean"},
1208
+ "notes": {"type": "string"},
1209
+ },
1210
+ "required": ["leader_top_right", "readable", "legend_visible", "notes"],
1211
+ }
1212
+ prompt = (
1213
+ "This image is a 2D competitor positioning map. The single RED dot is the "
1214
+ "leader and should sit in the TOP-RIGHT area. Assess three things: (1) is "
1215
+ "the red leader dot in the top-right? (2) are the point labels readable and "
1216
+ "not badly overlapping? (3) is the legend on the right fully visible, not "
1217
+ "cut off? Reply as JSON."
1218
+ )
1219
+ try:
1220
+ resp = ollama.chat(
1221
+ model=model,
1222
+ format=schema,
1223
+ options={"temperature": 0},
1224
+ messages=[{"role": "user", "content": prompt, "images": [image]}],
1225
+ )
1226
+ return json.loads(resp["message"]["content"])
1227
+ except Exception:
1228
+ return {}
1229
+
1230
+
1231
+ def _llm_text(prompt: str, model: str, use_llm: bool, fallback: str) -> str:
1232
+ """Free-text completion from the local model; `fallback` if unavailable."""
1233
+ if not use_llm:
1234
+ return fallback
1235
+ try:
1236
+ resp = ollama.chat(
1237
+ model=model,
1238
+ options={"temperature": 0.3},
1239
+ messages=[{"role": "user", "content": prompt}],
1240
+ )
1241
+ return resp["message"]["content"].strip() or fallback
1242
+ except Exception:
1243
+ return fallback
1244
+
1245
+
1246
+ def analysis_markdown(
1247
+ result: PCAResult,
1248
+ roles: list[str],
1249
+ poles: list[str],
1250
+ model: str = DEFAULT_MODEL,
1251
+ use_llm: bool = True,
1252
+ lang: str | None = None,
1253
+ ) -> str:
1254
+ """A thoughtful, precise interpretation of the map as Markdown.
1255
+
1256
+ Combines data-derived facts (axis loadings, variance, roles, coordinates) with
1257
+ an LLM-written narrative in the table's own language (auto-detected). Falls
1258
+ back to a templated narrative when the model is unavailable.
1259
+ """
1260
+ left, right, bottom, top = poles
1261
+ evr = result.explained_variance_ratio
1262
+ names = result.names
1263
+ role_of = dict(zip(names, roles, strict=False))
1264
+ coords = result.coords()
1265
+ if lang is None:
1266
+ lang = detect_language(result.features)
1267
+
1268
+ def loading_line(k: int) -> str:
1269
+ """Axis `k`'s criteria and signed weights, highest-first, as one line."""
1270
+ pairs = sorted(
1271
+ zip(result.features, result.components[k], strict=False), key=lambda t: -t[1]
1272
+ )
1273
+ return " · ".join(f"{f} ({w:+.2f})" for f, w in pairs)
1274
+
1275
+ ranked = sorted(names, key=lambda n: -(coords.loc[n].sum()))
1276
+ role_rows = {
1277
+ r: next((n for n, rr in role_of.items() if rr == r), "—")
1278
+ for r in ("best", "worst", "top", "right")
1279
+ }
1280
+
1281
+ narrative = _llm_text(
1282
+ i18n(lang)["narrative_prompt"].format(
1283
+ left=left,
1284
+ right=right,
1285
+ bottom=bottom,
1286
+ top=top,
1287
+ reference=result.reference,
1288
+ best=role_rows["best"],
1289
+ worst=role_rows["worst"],
1290
+ champ_top=role_rows["top"],
1291
+ champ_right=role_rows["right"],
1292
+ leaderboard=", ".join(ranked[:8]),
1293
+ ),
1294
+ model,
1295
+ use_llm,
1296
+ fallback=(
1297
+ f"The map's horizontal axis contrasts **{left}** (left) with **{right}** "
1298
+ f"(right); the vertical contrasts **{bottom}** (bottom) with **{top}** "
1299
+ f"(top), together capturing {evr.sum():.0%} of the variation between "
1300
+ f"approaches. **{result.reference}** anchors the top-right as the "
1301
+ f"reference leader, strongest on the {right.lower()} and {top.lower()} "
1302
+ f"directions. **{role_rows['worst']}** sits opposite as the weakest on "
1303
+ f"these dimensions, while among the challengers **{role_rows['top']}** "
1304
+ f"reaches furthest toward {top.lower()} and **{role_rows['right']}** "
1305
+ f"furthest toward {right.lower()}."
1306
+ ),
1307
+ )
1308
+
1309
+ lines = [
1310
+ f"# {result.reference}",
1311
+ "",
1312
+ "## Interpretation",
1313
+ "",
1314
+ narrative,
1315
+ "",
1316
+ "## Axes",
1317
+ "",
1318
+ f"- **Horizontal — {left} ↔ {right}** ({evr[0]:.0%} of variance). "
1319
+ f"Columns by weight: {loading_line(0)}.",
1320
+ f"- **Vertical — {bottom} ↔ {top}** ({evr[1]:.0%} of variance). "
1321
+ f"Columns by weight: {loading_line(1)}.",
1322
+ f"- Together the two axes retain **{evr.sum():.0%}** of the total variation; "
1323
+ f"the reference was rotated {result.rotation_deg:+.1f}° to reach the top-right.",
1324
+ "",
1325
+ "## Highlighted approaches",
1326
+ "",
1327
+ f"- **Leader (reference):** {role_rows['best']}",
1328
+ f"- **Weakest overall:** {role_rows['worst']} (lowest projection on the leader diagonal)",
1329
+ f"- **Strongest toward {top}:** {role_rows['top']} (challenger furthest up "
1330
+ "the vertical axis)",
1331
+ f"- **Strongest toward {right}:** {role_rows['right']} (challenger furthest "
1332
+ "along the horizontal axis)",
1333
+ "",
1334
+ "## Leaderboard (by combined axis score)",
1335
+ "",
1336
+ ]
1337
+ lines += [
1338
+ f"{i}. {n} ({coords.loc[n, 'axis-1']:+.2f}, {coords.loc[n, 'axis-2']:+.2f})"
1339
+ for i, n in enumerate(ranked, 1)
1340
+ ]
1341
+ lines += ["", "*Coordinates are PCA units; see the companion YAML for full coefficients.*", ""]
1342
+ return "\n".join(lines)
1343
+
1344
+
1345
+ def results_yaml(
1346
+ df: pd.DataFrame,
1347
+ result: PCAResult,
1348
+ roles: list[str],
1349
+ poles: list[str],
1350
+ axis_names: list[str],
1351
+ colors: list[str],
1352
+ ) -> str:
1353
+ """Everything about the fit as YAML: metadata, axis loadings, and per-approach
1354
+ coordinates, roles, colours, and original attribute values."""
1355
+ evr = result.explained_variance_ratio
1356
+ left, right, bottom, top = poles
1357
+ feats = result.features
1358
+ raw = impute(df)
1359
+
1360
+ doc = {
1361
+ "meta": {
1362
+ "reference": result.reference,
1363
+ "rotation_deg": round(result.rotation_deg, 3),
1364
+ "explained_variance_ratio": [round(float(v), 4) for v in evr],
1365
+ "cumulative_variance": round(float(evr.sum()), 4),
1366
+ "n_approaches": len(result.names),
1367
+ "attributes": feats,
1368
+ "lower_is_better": sorted(result.lower),
1369
+ },
1370
+ "axes": {
1371
+ "axis_1": {
1372
+ "name": axis_names[0],
1373
+ "pole_left": left,
1374
+ "pole_right": right,
1375
+ "loadings": {
1376
+ f: round(float(w), 4) for f, w in zip(feats, result.components[0], strict=False)
1377
+ },
1378
+ },
1379
+ "axis_2": {
1380
+ "name": axis_names[1],
1381
+ "pole_bottom": bottom,
1382
+ "pole_top": top,
1383
+ "loadings": {
1384
+ f: round(float(w), 4) for f, w in zip(feats, result.components[1], strict=False)
1385
+ },
1386
+ },
1387
+ },
1388
+ "approaches": [
1389
+ {
1390
+ "name": n,
1391
+ "coordinates": {"axis_1": round(float(x), 4), "axis_2": round(float(y), 4)},
1392
+ "role": role,
1393
+ "color": color,
1394
+ "attributes": {f: round(float(raw.loc[n, f]), 3) for f in feats},
1395
+ }
1396
+ for n, (x, y), role, color in zip(
1397
+ result.names, result.scores, roles, colors, strict=False
1398
+ )
1399
+ ],
1400
+ }
1401
+ return yaml.dump(doc, sort_keys=False, allow_unicode=True, width=100)
1402
+
1403
+
1404
+ def export_all(
1405
+ df: pd.DataFrame,
1406
+ result: PCAResult,
1407
+ roles: list[str],
1408
+ poles: list[str],
1409
+ axis_names: list[str],
1410
+ colors: list[str],
1411
+ stem: str,
1412
+ model: str = DEFAULT_MODEL,
1413
+ use_llm: bool = True,
1414
+ noun_plural: str = "Approaches",
1415
+ title: str | None = None,
1416
+ ) -> list[str]:
1417
+ """Write the full three-fold deliverable for one table: figures (PNG + SVG +
1418
+ Vega JSON), a Markdown interpretation, and a YAML of coordinates + coefficients.
1419
+ Returns the list of paths written.
1420
+ """
1421
+ spec = to_vega(
1422
+ result, roles=roles, poles=poles, colors=colors, noun_plural=noun_plural, title=title
1423
+ )
1424
+ written = render_figures(spec, stem)
1425
+ for path, text in [
1426
+ (f"{stem}.vl.json", json.dumps(spec, indent=2, ensure_ascii=False)),
1427
+ (f"{stem}.md", analysis_markdown(result, roles, poles, model, use_llm)),
1428
+ (f"{stem}.yaml", results_yaml(df, result, roles, poles, axis_names, colors)),
1429
+ ]:
1430
+ with open(path, "w", encoding="utf-8") as fh:
1431
+ fh.write(text)
1432
+ written.append(path)
1433
+ return written
1434
+
1435
+
1436
+ # --------------------------------------------------------------------------- #
1437
+ # Convenience API — the one-liner library face
1438
+ # --------------------------------------------------------------------------- #
1439
+ @dataclass
1440
+ class Positioning:
1441
+ """Result of `positioning()` — the map plus everything computed for it."""
1442
+
1443
+ df: pd.DataFrame
1444
+ result: PCAResult
1445
+ roles: list[str]
1446
+ poles: list[str]
1447
+ axis_names: list[str]
1448
+ colors: list[str]
1449
+ noun_singular: str = "Approach"
1450
+ noun_plural: str = "Approaches"
1451
+ title: str = "Approaches in the Quadrant" # fully-localized figure title
1452
+
1453
+ @property
1454
+ def coords(self) -> pd.DataFrame:
1455
+ """Oriented (axis-1, axis-2) coordinates per option."""
1456
+ return self.result.coords()
1457
+
1458
+ @property
1459
+ def loadings(self) -> pd.DataFrame:
1460
+ """Axis loadings (criterion weights) per axis."""
1461
+ return self.result.loadings()
1462
+
1463
+ @property
1464
+ def axes(self) -> dict[str, str]:
1465
+ """The two axis names, e.g. {'x': 'Cost ↔ Innovation', 'y': ...}."""
1466
+ return {"x": self.axis_names[0], "y": self.axis_names[1]}
1467
+
1468
+ @property
1469
+ def role_of(self) -> dict[str, str]:
1470
+ """Map each option name to its role (best / worst / … / competitor)."""
1471
+ return dict(zip(self.result.names, self.roles, strict=False))
1472
+
1473
+ def to_vega(self) -> dict:
1474
+ """The Vega-Lite spec for the map."""
1475
+ return to_vega(
1476
+ self.result,
1477
+ self.roles,
1478
+ self.poles,
1479
+ self.colors,
1480
+ noun_plural=self.noun_plural,
1481
+ title=self.title,
1482
+ )
1483
+
1484
+ def to_markdown(self, model: str = DEFAULT_MODEL, use_llm: bool = True) -> str:
1485
+ """The written interpretation as Markdown."""
1486
+ return analysis_markdown(self.result, self.roles, self.poles, model, use_llm)
1487
+
1488
+ def to_yaml(self) -> str:
1489
+ """All coordinates + coefficients as YAML."""
1490
+ return results_yaml(
1491
+ self.df, self.result, self.roles, self.poles, self.axis_names, self.colors
1492
+ )
1493
+
1494
+ def figure(self, stem: str) -> list[str]:
1495
+ """Render the map to `<stem>.png` and `<stem>.svg`; returns the paths."""
1496
+ return render_figures(self.to_vega(), stem)
1497
+
1498
+ def export(
1499
+ self,
1500
+ outdir: str = ".",
1501
+ stem: str | None = None,
1502
+ model: str = DEFAULT_MODEL,
1503
+ use_llm: bool = True,
1504
+ ) -> list[str]:
1505
+ """Write the full three-fold deliverable into `outdir`; returns the paths."""
1506
+ os.makedirs(outdir, exist_ok=True)
1507
+ name = stem or re.sub(r"[^A-Za-z0-9]+", "_", self.result.reference).strip("_").lower()
1508
+ return export_all(
1509
+ self.df,
1510
+ self.result,
1511
+ self.roles,
1512
+ self.poles,
1513
+ self.axis_names,
1514
+ self.colors,
1515
+ os.path.join(outdir, name),
1516
+ model=model,
1517
+ use_llm=use_llm,
1518
+ noun_plural=self.noun_plural,
1519
+ title=self.title,
1520
+ )
1521
+
1522
+
1523
+ def positioning(
1524
+ data: pd.DataFrame | str,
1525
+ reference: int | str = 0,
1526
+ top: str | None = None,
1527
+ right: str | None = None,
1528
+ lower_is_better: list[str] | None = None,
1529
+ model: str = DEFAULT_MODEL,
1530
+ use_llm: bool = True,
1531
+ ) -> Positioning:
1532
+ """Position options from a table in one call.
1533
+
1534
+ `data` is a pandas DataFrame (options × numeric criteria) or a path / raw string
1535
+ of a CSV or Markdown table. `lower_is_better` names criteria where a lower value
1536
+ is better (also picked up from ``(↓)`` header markers). `top` / `right` force a
1537
+ named option into the top-pole / right-pole highlight (see `assign_roles`).
1538
+ Returns a `Positioning` with `.coords`, `.loadings`, `.axes`, `.to_vega()`,
1539
+ `.to_markdown()`, `.to_yaml()`, and `.export(outdir)`.
1540
+
1541
+ >>> pos = positioning("examples/programming_languages.csv")
1542
+ >>> pos.export("out")
1543
+ """
1544
+ df = data if isinstance(data, pd.DataFrame) else parse_table(data)
1545
+ df, lower = resolve_polarity(df, lower_is_better) # clean names + lower set
1546
+ result = analyze(df, reference=reference, lower_is_better=list(lower))
1547
+ roles = assign_roles(result, top=top, right=right)
1548
+ lang = detect_language(result.features)
1549
+ poles = axis_poles(result, model=model, use_llm=use_llm, lang=lang)
1550
+ singular, plural = noun_forms(
1551
+ str(df.index.name or "Approach"), model=model, use_llm=use_llm, lang=lang
1552
+ )
1553
+ # Localize the whole title, not just the noun: a French table reads
1554
+ # "Voitures dans le quadrant", never "Voitures in the Quadrant".
1555
+ title = i18n(lang)["title_template"].format(plural=plural)
1556
+ return Positioning(
1557
+ df,
1558
+ result,
1559
+ roles,
1560
+ poles,
1561
+ _poles_to_names(poles),
1562
+ gradient_colors(result, roles),
1563
+ singular,
1564
+ plural,
1565
+ title,
1566
+ )
1567
+
1568
+
1569
+ # --------------------------------------------------------------------------- #
1570
+ # CLI
1571
+ # --------------------------------------------------------------------------- #
1572
+ def run(
1573
+ table: str,
1574
+ reference: str = "0",
1575
+ outdir: str = "out",
1576
+ stem: str | None = None,
1577
+ top: str | None = None,
1578
+ right: str | None = None,
1579
+ lower: str = "",
1580
+ model: str = DEFAULT_MODEL,
1581
+ no_llm: bool = False,
1582
+ check: bool = False,
1583
+ ) -> list[str]:
1584
+ """Shared CLI core: build the positioning, print a summary, write the files.
1585
+
1586
+ Used by both the argparse (`main`) and click (`main_click`) entry points.
1587
+ `top` / `right` force a named option into the top-pole / right-pole highlight.
1588
+ Returns the list of written paths.
1589
+ """
1590
+ ref: int | str = int(reference) if reference.lstrip("-").isdigit() else reference
1591
+ lower_cols = [c.strip() for c in lower.split(",") if c.strip()]
1592
+ pos = positioning(
1593
+ parse_table(table),
1594
+ reference=ref,
1595
+ top=top,
1596
+ right=right,
1597
+ lower_is_better=lower_cols,
1598
+ model=model,
1599
+ use_llm=not no_llm,
1600
+ )
1601
+ result, evr = pos.result, pos.result.explained_variance_ratio
1602
+
1603
+ print(f"Parsed {pos.df.shape[0]} options x {pos.df.shape[1]} criteria")
1604
+ print(
1605
+ f"Reference '{result.reference}' rotated by {result.rotation_deg:+.1f} deg "
1606
+ "onto the top-right diagonal\n"
1607
+ )
1608
+ print(
1609
+ f"PCA explained variance: axis-1(PC1)={evr[0]:.1%} axis-2(PC2)={evr[1]:.1%} "
1610
+ f"(cumulative {evr.sum():.1%})\n"
1611
+ )
1612
+ print(f"Axis names: axis-1 = {pos.axis_names[0]!r} axis-2 = {pos.axis_names[1]!r}\n")
1613
+ # poles are [left, right, bottom, top]; name each highlight by its pole word.
1614
+ print("Highlighted options:")
1615
+ highlights = [
1616
+ ("best", "leader (reference)"),
1617
+ ("worst", "weakest overall"),
1618
+ ("top", f"strongest toward {pos.poles[3]!r}"),
1619
+ ("right", f"strongest toward {pos.poles[1]!r}"),
1620
+ ]
1621
+ for role, label in highlights:
1622
+ who = next((n for n, r in pos.role_of.items() if r == role), "—")
1623
+ print(f" {label:34s}: {who}")
1624
+ print()
1625
+ print("Canonical axes in the oriented frame (loadings):")
1626
+ print(pos.loadings.round(3).to_string(), "\n")
1627
+
1628
+ written = pos.export(outdir, stem=stem, model=model, use_llm=not no_llm)
1629
+ print("Three-fold deliverable written:")
1630
+ for path in written:
1631
+ print(f" {path}")
1632
+
1633
+ if check:
1634
+ # Assess a white-composited render, not the transparent PNG on disk: the
1635
+ # vision model's backend would otherwise flatten transparency onto black and
1636
+ # wrongly report the dark legend as cut off (see `png_on_white`).
1637
+ verdict = vlm_assess(png_on_white(pos.to_vega()), model=model)
1638
+ if verdict:
1639
+ print("\nVision self-check:")
1640
+ for key in ("leader_top_right", "readable", "legend_visible"):
1641
+ print(f" {key:16s}: {verdict.get(key)}")
1642
+ if verdict.get("notes"):
1643
+ print(f" notes : {verdict['notes']}")
1644
+ else:
1645
+ print("\nVision self-check unavailable (model not reachable).")
1646
+ return written
1647
+
1648
+
1649
+ def main(argv: list[str] | None = None) -> None:
1650
+ """argparse entry point (console command ``standpoint``)."""
1651
+ ap = argparse.ArgumentParser(description=__doc__.splitlines()[0])
1652
+ ap.add_argument("table", help="path to a markdown or CSV table")
1653
+ ap.add_argument(
1654
+ "-r",
1655
+ "--reference",
1656
+ default="0",
1657
+ help="row placed top-right: index (default 0) or exact name",
1658
+ )
1659
+ ap.add_argument(
1660
+ "-o",
1661
+ "--outdir",
1662
+ default="out",
1663
+ help="output directory for the three-fold deliverable (default out/)",
1664
+ )
1665
+ ap.add_argument("--stem", help="basename for outputs (default: derived from reference)")
1666
+ ap.add_argument(
1667
+ "--top",
1668
+ help="exact name of the option to highlight as strongest "
1669
+ "toward the top pole (default: picked from the map)",
1670
+ )
1671
+ ap.add_argument(
1672
+ "--right",
1673
+ help="exact name of the option to highlight as strongest "
1674
+ "toward the right pole (default: picked from the map)",
1675
+ )
1676
+ ap.add_argument(
1677
+ "--lower",
1678
+ default="",
1679
+ help="comma-separated criteria where lower is better (e.g. Price,Latency)",
1680
+ )
1681
+ ap.add_argument(
1682
+ "--model",
1683
+ default=DEFAULT_MODEL,
1684
+ help=f"Ollama model for axis naming (default {DEFAULT_MODEL})",
1685
+ )
1686
+ ap.add_argument(
1687
+ "--no-llm", action="store_true", help="skip the LLM; use deterministic axis names"
1688
+ )
1689
+ ap.add_argument(
1690
+ "--check",
1691
+ action="store_true",
1692
+ help="ask the vision model to sanity-check the rendered figure",
1693
+ )
1694
+ a = ap.parse_args(argv)
1695
+ run(a.table, a.reference, a.outdir, a.stem, a.top, a.right, a.lower, a.model, a.no_llm, a.check)
1696
+
1697
+
1698
+ if __name__ == "__main__":
1699
+ main()