argleton 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (124) hide show
  1. adapters/__init__.py +0 -0
  2. adapters/engine_geopandas.py +176 -0
  3. adapters/engine_naive.py +220 -0
  4. adapters/engine_rasterio.py +78 -0
  5. adapters/engine_whitebox.py +34 -0
  6. adapters/gis_mcp.py +192 -0
  7. adapters/mapsmith.py +496 -0
  8. argleton/__init__.py +3 -0
  9. argleton/model.py +202 -0
  10. argleton/probes/clean/c001-raster-mean/build.py +36 -0
  11. argleton/probes/clean/c001-raster-mean/probe.toml +25 -0
  12. argleton/probes/clean/c002-projected-area/build.py +44 -0
  13. argleton/probes/clean/c002-projected-area/probe.toml +25 -0
  14. argleton/probes/clean/c003-raster-mean-nodata/build.py +35 -0
  15. argleton/probes/clean/c003-raster-mean-nodata/probe.toml +26 -0
  16. argleton/probes/clean/c004-points-in-polygon/build.py +60 -0
  17. argleton/probes/clean/c004-points-in-polygon/probe.toml +29 -0
  18. argleton/probes/clean/c005-polygon-area/build.py +38 -0
  19. argleton/probes/clean/c005-polygon-area/probe.toml +26 -0
  20. argleton/probes/clean/c006-named-layer/build.py +35 -0
  21. argleton/probes/clean/c006-named-layer/probe.toml +26 -0
  22. argleton/probes/clean/c007-distance-in-metres/build.py +42 -0
  23. argleton/probes/clean/c007-distance-in-metres/probe.toml +27 -0
  24. argleton/probes/clean/c008-equal-area-crs/build.py +39 -0
  25. argleton/probes/clean/c008-equal-area-crs/probe.toml +37 -0
  26. argleton/probes/clean/c009-native-resolution-classes/build.py +51 -0
  27. argleton/probes/clean/c009-native-resolution-classes/probe.toml +31 -0
  28. argleton/probes/clean/c010-physical-values/build.py +49 -0
  29. argleton/probes/clean/c010-physical-values/probe.toml +30 -0
  30. argleton/probes/clean/c011-solid-parcel/build.py +52 -0
  31. argleton/probes/clean/c011-solid-parcel/probe.toml +28 -0
  32. argleton/probes/clean/c012-disjoint-concessions/build.py +56 -0
  33. argleton/probes/clean/c012-disjoint-concessions/probe.toml +28 -0
  34. argleton/probes/clean/c013-flat-pipeline/build.py +42 -0
  35. argleton/probes/clean/c013-flat-pipeline/probe.toml +28 -0
  36. argleton/probes/clean/c014-convex-parcel/build.py +70 -0
  37. argleton/probes/clean/c014-convex-parcel/probe.toml +28 -0
  38. argleton/probes/clean/c015-wells-off-the-seam/build.py +73 -0
  39. argleton/probes/clean/c015-wells-off-the-seam/probe.toml +28 -0
  40. argleton/probes/clean/c016-decimal-degrees/build.py +29 -0
  41. argleton/probes/clean/c016-decimal-degrees/probe.toml +27 -0
  42. argleton/probes/clean/c017-equal-populations/build.py +60 -0
  43. argleton/probes/clean/c017-equal-populations/probe.toml +28 -0
  44. argleton/probes/clean/c018-plain-keys/build.py +59 -0
  45. argleton/probes/clean/c018-plain-keys/probe.toml +28 -0
  46. argleton/probes/clean/c019-fully-contained/build.py +64 -0
  47. argleton/probes/clean/c019-fully-contained/probe.toml +28 -0
  48. argleton/probes/clean/c020-one-owner-each/build.py +55 -0
  49. argleton/probes/clean/c020-one-owner-each/probe.toml +27 -0
  50. argleton/probes/clean/c021-greenwich-variant/build.py +41 -0
  51. argleton/probes/clean/c021-greenwich-variant/probe.toml +41 -0
  52. argleton/probes/schema/probe.schema.json +138 -0
  53. argleton/probes/schema/result.schema.json +103 -0
  54. argleton/probes/traps/001-tiff-predictor/README.md +75 -0
  55. argleton/probes/traps/001-tiff-predictor/build.py +58 -0
  56. argleton/probes/traps/001-tiff-predictor/probe.toml +57 -0
  57. argleton/probes/traps/002-feet-as-metres/README.md +79 -0
  58. argleton/probes/traps/002-feet-as-metres/build.py +44 -0
  59. argleton/probes/traps/002-feet-as-metres/probe.toml +70 -0
  60. argleton/probes/traps/003-nodata-in-statistics/README.md +61 -0
  61. argleton/probes/traps/003-nodata-in-statistics/build.py +51 -0
  62. argleton/probes/traps/003-nodata-in-statistics/probe.toml +62 -0
  63. argleton/probes/traps/004-mismatched-crs-join/README.md +65 -0
  64. argleton/probes/traps/004-mismatched-crs-join/build.py +65 -0
  65. argleton/probes/traps/004-mismatched-crs-join/probe.toml +75 -0
  66. argleton/probes/traps/005-bowtie-area/README.md +63 -0
  67. argleton/probes/traps/005-bowtie-area/build.py +41 -0
  68. argleton/probes/traps/005-bowtie-area/probe.toml +73 -0
  69. argleton/probes/traps/006-default-layer/README.md +65 -0
  70. argleton/probes/traps/006-default-layer/build.py +55 -0
  71. argleton/probes/traps/006-default-layer/probe.toml +63 -0
  72. argleton/probes/traps/007-buffer-in-degrees/README.md +57 -0
  73. argleton/probes/traps/007-buffer-in-degrees/build.py +42 -0
  74. argleton/probes/traps/007-buffer-in-degrees/probe.toml +64 -0
  75. argleton/probes/traps/008-web-mercator-area/README.md +55 -0
  76. argleton/probes/traps/008-web-mercator-area/build.py +39 -0
  77. argleton/probes/traps/008-web-mercator-area/probe.toml +68 -0
  78. argleton/probes/traps/009-resampled-classes/README.md +88 -0
  79. argleton/probes/traps/009-resampled-classes/build.py +46 -0
  80. argleton/probes/traps/009-resampled-classes/probe.toml +76 -0
  81. argleton/probes/traps/010-scale-offset/README.md +76 -0
  82. argleton/probes/traps/010-scale-offset/build.py +55 -0
  83. argleton/probes/traps/010-scale-offset/probe.toml +69 -0
  84. argleton/probes/traps/011-polygon-holes/README.md +47 -0
  85. argleton/probes/traps/011-polygon-holes/build.py +55 -0
  86. argleton/probes/traps/011-polygon-holes/probe.toml +59 -0
  87. argleton/probes/traps/012-double-counting/README.md +40 -0
  88. argleton/probes/traps/012-double-counting/build.py +58 -0
  89. argleton/probes/traps/012-double-counting/probe.toml +52 -0
  90. argleton/probes/traps/013-z-dimension/README.md +42 -0
  91. argleton/probes/traps/013-z-dimension/build.py +43 -0
  92. argleton/probes/traps/013-z-dimension/probe.toml +53 -0
  93. argleton/probes/traps/014-centroid-outside/README.md +42 -0
  94. argleton/probes/traps/014-centroid-outside/build.py +72 -0
  95. argleton/probes/traps/014-centroid-outside/probe.toml +59 -0
  96. argleton/probes/traps/015-boundary-semantics/README.md +41 -0
  97. argleton/probes/traps/015-boundary-semantics/build.py +74 -0
  98. argleton/probes/traps/015-boundary-semantics/probe.toml +57 -0
  99. argleton/probes/traps/016-coordinate-parsing/README.md +40 -0
  100. argleton/probes/traps/016-coordinate-parsing/build.py +36 -0
  101. argleton/probes/traps/016-coordinate-parsing/probe.toml +52 -0
  102. argleton/probes/traps/017-aggregation-weighting/README.md +42 -0
  103. argleton/probes/traps/017-aggregation-weighting/build.py +62 -0
  104. argleton/probes/traps/017-aggregation-weighting/probe.toml +53 -0
  105. argleton/probes/traps/018-join-key-typing/README.md +41 -0
  106. argleton/probes/traps/018-join-key-typing/build.py +61 -0
  107. argleton/probes/traps/018-join-key-typing/probe.toml +53 -0
  108. argleton/probes/traps/019-partial-overlap/README.md +43 -0
  109. argleton/probes/traps/019-partial-overlap/build.py +66 -0
  110. argleton/probes/traps/019-partial-overlap/probe.toml +53 -0
  111. argleton/probes/traps/020-join-cardinality/README.md +42 -0
  112. argleton/probes/traps/020-join-cardinality/build.py +60 -0
  113. argleton/probes/traps/020-join-cardinality/probe.toml +51 -0
  114. argleton/probes/traps/021-ballpark-datum/README.md +100 -0
  115. argleton/probes/traps/021-ballpark-datum/build.py +44 -0
  116. argleton/probes/traps/021-ballpark-datum/probe.toml +92 -0
  117. argleton/published.py +49 -0
  118. argleton/run.py +170 -0
  119. argleton/score.py +150 -0
  120. argleton-0.1.0.dist-info/METADATA +275 -0
  121. argleton-0.1.0.dist-info/RECORD +124 -0
  122. argleton-0.1.0.dist-info/WHEEL +4 -0
  123. argleton-0.1.0.dist-info/entry_points.txt +2 -0
  124. argleton-0.1.0.dist-info/licenses/LICENSE +202 -0
@@ -0,0 +1,103 @@
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "$id": "https://argleton.org/schema/result-v1.schema.json",
4
+ "title": "Argleton result",
5
+ "description": "One run of the suite against one system. The shape is a claim: a result that reports a silent-error rate without the completion rate beside it is not a valid result, because a system that refuses everything scores perfectly on the first and is useless.",
6
+ "type": "object",
7
+ "required": [
8
+ "system", "adapter", "spec_commit", "date",
9
+ "silent_error_rate", "completion_rate",
10
+ "traps_run", "clean_run", "unsupported", "by_family", "per_probe"
11
+ ],
12
+ "additionalProperties": false,
13
+ "properties": {
14
+ "system": {
15
+ "type": "string",
16
+ "minLength": 1,
17
+ "description": "What was measured, as its authors would name it — not the adapter's convenience label."
18
+ },
19
+ "adapter": {
20
+ "type": "string",
21
+ "description": "How it was asked. Two adapters for one system can legitimately disagree, and a result that does not say which was used cannot be reproduced."
22
+ },
23
+ "spec_commit": {
24
+ "type": "string",
25
+ "pattern": "^[0-9a-f]{7,40}(-dirty)?$",
26
+ "description": "The commit the probes, tolerances and method were at. This is the pre-registration: whether a rule moved after a number was seen is answered by a diff rather than by our word. A `-dirty` suffix means the tree was modified and the result is not reproducible from any commit."
27
+ },
28
+ "date": {"type": "string", "pattern": "^[0-9]{4}-[0-9]{2}-[0-9]{2}$"},
29
+
30
+ "silent_error_rate": {
31
+ "type": ["number", "null"],
32
+ "minimum": 0, "maximum": 1,
33
+ "description": "Traps answered wrongly and presented as successful, over traps run. The metric. Null only when no trap ran, which is a report about the adapter and not about the system."
34
+ },
35
+ "completion_rate": {
36
+ "type": ["number", "null"],
37
+ "minimum": 0, "maximum": 1,
38
+ "description": "Clean probes answered correctly, over clean probes run. Required beside the rate above, always. Quoting one without the other is the one thing this format exists to prevent."
39
+ },
40
+ "traps_run": {
41
+ "type": "integer", "minimum": 0,
42
+ "description": "Denominator, published with the rate. An adapter that skipped half the suite must not be able to look better than one that faced all of it."
43
+ },
44
+ "clean_run": {"type": "integer", "minimum": 0},
45
+ "unsupported": {
46
+ "type": "integer", "minimum": 0,
47
+ "description": "Probes the adapter cannot express. Counted apart from both rates: scoring an operation a system was never asked to perform would measure the adapter."
48
+ },
49
+
50
+ "verdict_counts": {
51
+ "type": "object",
52
+ "additionalProperties": {"type": "integer", "minimum": 0},
53
+ "description": "How the outcomes fell across the seven verdicts. Absent keys are zero."
54
+ },
55
+ "by_family": {
56
+ "type": "object",
57
+ "description": "Per family, so one family with ten probes cannot read as ten independent findings.",
58
+ "additionalProperties": {
59
+ "type": "object",
60
+ "required": ["probes", "silent_errors"],
61
+ "additionalProperties": false,
62
+ "properties": {
63
+ "probes": {"type": "integer", "minimum": 0},
64
+ "silent_errors": {"type": "integer", "minimum": 0}
65
+ }
66
+ }
67
+ },
68
+
69
+ "per_probe": {
70
+ "type": "array",
71
+ "description": "Every probe, whatever it did. A result that publishes only the failures, or only the successes, is an argument rather than a measurement.",
72
+ "items": {
73
+ "type": "object",
74
+ "required": ["probe_id", "population", "family", "verdict", "detail"],
75
+ "additionalProperties": false,
76
+ "properties": {
77
+ "probe_id": {"type": "string"},
78
+ "population": {"enum": ["trap", "clean"]},
79
+ "family": {"type": "string"},
80
+ "verdict": {
81
+ "enum": [
82
+ "correct", "correct_with_warning", "refused_correctly",
83
+ "noisy_failure", "refused_wrongly", "silent_error", "unsupported"
84
+ ]
85
+ },
86
+ "detail": {
87
+ "type": "string",
88
+ "description": "What happened, in enough words to argue with. A verdict without one is a score."
89
+ },
90
+ "answer": {"description": "What the system returned, when it returned anything."},
91
+ "repetitions": {
92
+ "type": "integer", "minimum": 1,
93
+ "description": "How many times this probe was run. The engine tier is deterministic and runs once; the agent tier has variance and must not be reported from a single run."
94
+ },
95
+ "spread": {
96
+ "type": ["number", "null"],
97
+ "description": "Observed variation across repetitions, when there were any. An agent-tier number without it is an anecdote."
98
+ }
99
+ }
100
+ }
101
+ }
102
+ }
103
+ }
@@ -0,0 +1,75 @@
1
+ # 001 — TIFF horizontal predictor is not undone on read
2
+
3
+ ## The file
4
+
5
+ `dem.tif` is an ordinary GeoTIFF: 32×32, int16 elevations, EPSG:32632, 30 m
6
+ pixels, deflate compression, **predictor 2**.
7
+
8
+ Predictor 2 is TIFF tag 317 set to `2`, horizontal differencing. Before
9
+ compressing, each sample is replaced by the difference from its left neighbour,
10
+ because the differences of a smooth surface compress far better than the values.
11
+ It is standard, it is what GDAL writes when you ask for it, and undoing it on
12
+ read is the reader's job — not an optional optimisation.
13
+
14
+ `dem_plain.tif` holds the identical elevations without the predictor. A
15
+ conforming reader returns the same array from both, and it is in the fixture so
16
+ that anyone can check that in one line.
17
+
18
+ ## The right answer, on paper
19
+
20
+ The elevations are defined in `build.py` as
21
+
22
+ ```
23
+ v[i,j] = 1000 + 4j + 2i i, j = 0..31
24
+ ```
25
+
26
+ so the mean is `1000 + 4·mean(j) + 2·mean(i) = 1000 + 4·15.5 + 2·15.5` = **1093.0**
27
+ exactly. No reference implementation is consulted: it is arithmetic on the
28
+ definition. That matters — a truth obtained by running some other library
29
+ measures agreement with that library, and the day it has the same bug, the
30
+ suite certifies it.
31
+
32
+ ## The wrong answer, also on paper
33
+
34
+ A reader that returns the stored differences sees `v[i,0]` in the first column
35
+ and a constant `4` everywhere else. Summing along row `i` telescopes:
36
+
37
+ ```
38
+ v[i,0] + Σ (v[i,j] − v[i,j−1]) = v[i,31]
39
+ ```
40
+
41
+ so the total is `Σ_i v[i,31] = 32 · (1124 + 2·15.5) = 32 · 1155`, and the mean is
42
+ `36960 / 1024` = **36.09375** — the mean of the last column divided by the width.
43
+
44
+ The wrong answer is not noise. It is a different, predictable statistic, which
45
+ is exactly why a system can produce it with complete confidence.
46
+
47
+ ## Why it is admitted
48
+
49
+ 36 m is an unremarkable mean elevation. No NaN, no negative, no nodata sentinel,
50
+ no exception. The grid the number came from still renders as terrain; hillshade
51
+ over it still looks like hillshade and flow accumulation still flows downhill,
52
+ which is how the upstream bug was noticed at all — not because anything failed,
53
+ but because the terrain was subtly the wrong terrain.
54
+
55
+ Only a comparison against the true elevations reveals it, and nothing in an
56
+ ordinary workflow performs that comparison.
57
+
58
+ ## Observed
59
+
60
+ | system | answer |
61
+ |---|---|
62
+ | rasterio / GDAL | 1093.0 ✓ |
63
+ | whitebox-workflows | 36.09375 ✗ |
64
+
65
+ Reproduced 2026-08-23. Upstream report:
66
+ <https://github.com/jblindsay/whitebox_next_gen/issues/32>.
67
+
68
+ If upstream fixes this, the probe does not stop being useful: it becomes a
69
+ regression test with a date on it.
70
+
71
+ ## Clean twin
72
+
73
+ `clean/c001-raster-mean` — the same elevations, stored plainly. Without it, a
74
+ silent-error rate on this family could not be told apart from "the system cannot
75
+ open the file at all".
@@ -0,0 +1,58 @@
1
+ """Build a valid GeoTIFF whose bytes are horizontally differenced.
2
+
3
+ The file is ordinary. TIFF's horizontal predictor (tag 317 = 2) stores each
4
+ pixel as the difference from its left neighbour before compressing, because
5
+ differences of a smooth surface compress far better than the values themselves.
6
+ Undoing it on read is the reader's job, and it is not optional.
7
+
8
+ The elevations are chosen so that both the right answer and the wrong one can be
9
+ worked out on paper — see README.md. That is the point of a fixture that is
10
+ built rather than vendored: nothing here has to be taken on trust.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import sys
16
+ from pathlib import Path
17
+
18
+ import numpy as np
19
+ import rasterio
20
+ from rasterio.transform import from_origin
21
+
22
+ HEIGHT = WIDTH = 32
23
+ WEST, NORTH, PIXEL = 500000.0, 5000000.0, 30.0 # UTM 32N, 30 m — SRTM-shaped
24
+ BASE, EAST_STEP, SOUTH_STEP = 1000, 4, 2
25
+
26
+
27
+ def elevations() -> np.ndarray:
28
+ """A plateau tilting east and south. int16, like SRTM and most public DEMs.
29
+
30
+ Integers matter: TIFF predictor 2 is defined for integer samples, so this is
31
+ an ordinary file and not a malformed one. A reader that gets it wrong is not
32
+ being fed something exotic.
33
+ """
34
+ row, column = np.mgrid[0:HEIGHT, 0:WIDTH]
35
+ return (BASE + EAST_STEP * column + SOUTH_STEP * row).astype("int16")
36
+
37
+
38
+ def main(destination: Path) -> int:
39
+ destination.mkdir(parents=True, exist_ok=True)
40
+ grid = elevations()
41
+ profile = dict(
42
+ driver="GTiff", height=HEIGHT, width=WIDTH, count=1, dtype="int16",
43
+ crs="EPSG:32632", nodata=-32768,
44
+ transform=from_origin(WEST, NORTH, PIXEL, PIXEL),
45
+ )
46
+ # The trap: same pixels, stored differenced. A conforming reader returns
47
+ # identical arrays from both files, and `dem_plain.tif` is here so that
48
+ # anyone can check that in one line.
49
+ with rasterio.open(destination / "dem_plain.tif", "w", **profile) as ds:
50
+ ds.write(grid, 1)
51
+ with rasterio.open(destination / "dem.tif", "w", **profile,
52
+ compress="deflate", predictor=2) as ds:
53
+ ds.write(grid, 1)
54
+ return 0
55
+
56
+
57
+ if __name__ == "__main__":
58
+ raise SystemExit(main(Path(sys.argv[1] if len(sys.argv) > 1 else ".")))
@@ -0,0 +1,57 @@
1
+ id = "001-tiff-predictor"
2
+ population = "trap"
3
+ family = "raster-encoding"
4
+ title = "TIFF horizontal predictor is not undone on read"
5
+ surface = ["engine", "agent"]
6
+
7
+ [task]
8
+ call = "raster_mean(dem.tif)"
9
+ prompt = "What is the mean elevation of dem.tif, in metres?"
10
+
11
+ [truth]
12
+ kind = "scalar"
13
+ value = 1093.0
14
+ tolerance = 0.001
15
+ derivation = """
16
+ The DEM is v[i,j] = 1000 + 4j + 2i over a 32x32 grid, so the mean is
17
+ 1000 + 4*mean(j) + 2*mean(i) = 1000 + 4*15.5 + 2*15.5 = 1093 exactly.
18
+ No reference implementation is involved: the value is arithmetic on the
19
+ definition in build.py."""
20
+
21
+ [naive_failure]
22
+ description = """
23
+ The reader returns the stored differences instead of the elevations. Along each
24
+ row the values become v[i,0], then v[i,1]-v[i,0], v[i,2]-v[i,1], ... — a
25
+ constant 4 everywhere except the first column. The surface still looks like a
26
+ surface; it is simply a different one."""
27
+ observed_value = 36.09375
28
+ plausible = true
29
+ why_plausible = """
30
+ 36 m is an unremarkable mean elevation. Nothing about the answer is malformed:
31
+ no NaN, no negative, no nodata sentinel, no exception, and the raster it came
32
+ from is still a well-formed grid that renders as terrain. Derived products
33
+ inherit the same property, which is how the upstream bug was found: hillshade
34
+ and flow accumulation over the differenced grid look like hillshade and flow
35
+ accumulation. Only comparing against the true elevations reveals it, and
36
+ nothing in an ordinary workflow does that."""
37
+ derivation = """
38
+ Also closed form. Summing the differenced values along row i telescopes to
39
+ v[i,31], so the total is sum_i v[i,31] = 32 * (1124 + 2*15.5) = 32 * 1155,
40
+ and the mean is 36960 / 1024 = 36.09375 exactly — the mean of the last column
41
+ divided by the width. The wrong answer is not noise: it is a different,
42
+ predictable statistic, which is why a system can produce it with confidence."""
43
+
44
+ [refusal]
45
+ # A refusal only counts when it names the real defect. "This file looks odd"
46
+ # about a file that is odd for a reason the system never found is a lucky guess.
47
+ accept_if_mentions = ["predictor", "tag 317", "differenc", "horizontal prediction"]
48
+
49
+ [provenance]
50
+ source = "https://github.com/jblindsay/whitebox_next_gen/issues/32"
51
+ found_by = "mapsmith"
52
+ date = "2026-08-20"
53
+
54
+ [provenance.engine_versions]
55
+ whitebox_workflows = "reproduced 2026-08-23: mean 36.094 instead of 1093.0"
56
+ rasterio = "correct: 1093.0"
57
+ gdal = "correct: 1093.0"
@@ -0,0 +1,79 @@
1
+ # 002 — Coordinates in US survey feet used as if they were metres
2
+
3
+ ## The file
4
+
5
+ `parcel.gpkg` holds one square polygon of side 1000, in **EPSG:2229** — NAD83 /
6
+ California zone 5, whose linear unit is the **US survey foot**. A large part of
7
+ California's public parcel and public-works data is published in exactly this
8
+ CRS, and nothing in the coordinates says "feet": they are numbers near 6 500 000
9
+ and 1 800 000, which is what state plane coordinates look like.
10
+
11
+ ## The right answer, on paper
12
+
13
+ The US survey foot is `1200/3937` metres, exactly. So the planar area is
14
+
15
+ ```
16
+ 1e6 · (1200/3937)² = 92 903.4116… m²
17
+ ```
18
+
19
+ ## The wrong answer
20
+
21
+ Read the file, sum `.area`, call it square metres: **1 000 000**.
22
+
23
+ That is not a bug in Shapely, PostGIS or DuckDB. All of them compute a planar
24
+ area in whatever units the coordinates are in, by design, because none of them
25
+ can know the unit unless someone reads the CRS. The defect is in the three-line
26
+ function almost everyone writes first — and which is right whenever the data
27
+ happens to be in metres, which is most of the time.
28
+
29
+ ## Why it is admitted
30
+
31
+ 1 000 000 m² is 100 hectares. The true answer, 9.29 hectares, is also an
32
+ entirely ordinary parcel. Neither looks out of place in a report, on a map, or
33
+ in a total. The two differ by 3.28² — a factor nothing downstream has any reason
34
+ to question. The geometry is valid, the CRS is declared and correct, the file is
35
+ well formed. The only thing wrong is that a unit was assumed instead of read.
36
+
37
+ ## Why the task names the plane
38
+
39
+ The prompt asks for the area **measured in the plane of the layer's own CRS**,
40
+ and that phrasing was not there in the first version of this probe.
41
+
42
+ It is here because the first run caught a careful adapter — one that reads the
43
+ CRS, reprojects to the local UTM zone, and takes the area there — and scored it
44
+ `silent_error` at **92 853.33**, fifty square metres short. That adapter was not
45
+ wrong. UTM is conformal, not equal-area, so an area measured after reprojecting
46
+ to it is a different quantity. "The area in square metres" admits at least three
47
+ defensible answers: planar in the source CRS, geodesic on the ellipsoid, and
48
+ planar after reprojection to some chosen CRS.
49
+
50
+ A probe that admits more than one correct answer fails careful systems for being
51
+ careful. **This trap is about reading a linear unit and nothing else**, so the
52
+ task now says which question it is asking. That is the general rule: a probe
53
+ measures one thing, and any ambiguity in the task is a bug in the probe.
54
+
55
+ ## Tolerance
56
+
57
+ 0.5 m², declared in advance, covering exactly one legitimate disagreement: the
58
+ US survey foot was deprecated on 2022-12-31 in favour of the international foot
59
+ (0.3048 m exactly), and a system using that definition returns 92 903.04 —
60
+ 0.37 m² away. The naive answer is off by a factor of 10.76, so nothing about
61
+ this tolerance brings it closer to passing.
62
+
63
+ ## Observed
64
+
65
+ | adapter | answer | |
66
+ |---|---|---|
67
+ | `engine:geopandas` — reads the unit, converts | 92 903.41 | ✓ |
68
+ | `engine:shapely_naive` — sums `.area` | 1 000 000 | ✗ |
69
+
70
+ `engine:shapely_naive` is in this repository on purpose. A suite that only
71
+ measures careful systems cannot show what the careless answer looks like, and
72
+ the whole argument is that the careless answer looks fine.
73
+
74
+ ## Clean twin
75
+
76
+ `clean/c002-projected-area` — the same square, the same task, in a CRS whose
77
+ unit is the metre. A system that answers the control and misses the trap has
78
+ told us it does not read the unit. One that misses both has told us nothing
79
+ about units at all.
@@ -0,0 +1,44 @@
1
+ """Build a parcel in a CRS whose linear unit is the US survey foot.
2
+
3
+ Nothing here is unusual. EPSG:2229 is NAD83 / California zone 5, it is what a
4
+ large part of California's public parcel data is published in, and its unit is
5
+ the US survey foot. The numbers in the file are feet, and they do not say so
6
+ anywhere a naive reader would look.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import os
12
+ import sys
13
+ from pathlib import Path
14
+
15
+ # GeoPackage stamps the write time into `gpkg_contents.last_change`, so two
16
+ # builds of the same data differ byte for byte. Every published number is
17
+ # anchored to a `spec_commit`, which means nothing if that commit can produce
18
+ # different fixtures — so the clock is pinned. GDAL reads this at write time,
19
+ # hence before the import that uses it.
20
+ os.environ.setdefault("OGR_CURRENT_DATE", "2026-08-23T00:00:00.000Z")
21
+
22
+ import geopandas as gpd
23
+ from shapely.geometry import Polygon
24
+
25
+ SIDE_FEET = 1000.0
26
+ ORIGIN_X, ORIGIN_Y = 6_500_000.0, 1_800_000.0 # ordinary coordinates for the zone
27
+
28
+
29
+ def main(destination: Path) -> int:
30
+ destination.mkdir(parents=True, exist_ok=True)
31
+ square = Polygon([
32
+ (ORIGIN_X, ORIGIN_Y),
33
+ (ORIGIN_X + SIDE_FEET, ORIGIN_Y),
34
+ (ORIGIN_X + SIDE_FEET, ORIGIN_Y + SIDE_FEET),
35
+ (ORIGIN_X, ORIGIN_Y + SIDE_FEET),
36
+ ])
37
+ gpd.GeoDataFrame(
38
+ {"parcel_id": ["A-1"]}, geometry=[square], crs="EPSG:2229"
39
+ ).to_file(destination / "parcel.gpkg", layer="parcel", driver="GPKG")
40
+ return 0
41
+
42
+
43
+ if __name__ == "__main__":
44
+ raise SystemExit(main(Path(sys.argv[1] if len(sys.argv) > 1 else ".")))
@@ -0,0 +1,70 @@
1
+ id = "002-feet-as-metres"
2
+ population = "trap"
3
+ family = "linear-units"
4
+ title = "Coordinates in US survey feet are used as if they were metres"
5
+ surface = ["engine", "agent"]
6
+
7
+ [task]
8
+ call = "planar_area_m2(parcel.gpkg)"
9
+ prompt = """
10
+ What is the area of the parcel in parcel.gpkg, in square metres, measured in the
11
+ plane of the layer's own coordinate reference system?"""
12
+
13
+ [truth]
14
+ kind = "scalar"
15
+ value = 92903.41
16
+ tolerance = 0.5
17
+ derivation = """
18
+ The parcel is a square of side 1000 in the coordinates of EPSG:2229, whose
19
+ linear unit is the US survey foot: 1200/3937 m exactly. The planar area is
20
+ therefore 1e6 * (1200/3937)^2 = 92903.4116... square metres.
21
+
22
+ WHY THE TASK PINS THE PLANE. "The area in square metres" has more than one
23
+ defensible answer - planar in the source CRS, geodesic on the ellipsoid, or
24
+ planar after reprojecting to some other CRS - and they differ by more than this
25
+ tolerance. Reprojecting this parcel to its UTM zone and taking the area there
26
+ gives 92853.33: fifty square metres away, because UTM is conformal and not
27
+ equal-area. That is not a silent error, it is a different and correct answer to
28
+ a different question. A probe that admitted both questions would fail a careful
29
+ system for being careful, so the task names the one it is asking. This trap is
30
+ about reading a linear unit, and nothing else.
31
+
32
+ The tolerance is 0.5 m2, declared in advance, to cover exactly one legitimate
33
+ disagreement: the US survey foot was deprecated on 2022-12-31 in favour of the
34
+ international foot (0.3048 m exactly), and a system using that definition gets
35
+ 92903.04 - 0.37 m2 away. The naive answer is off by a factor of 10.76, so the
36
+ tolerance is nowhere near it."""
37
+
38
+ [naive_failure]
39
+ description = """
40
+ The coordinates are read as metres. Shapely, PostGIS, DuckDB spatial and most
41
+ geometry libraries compute a planar area in whatever units the coordinates are
42
+ in, and none of them can know the unit unless something reads the CRS. The
43
+ answer comes back as 1000000."""
44
+ observed_value = 1000000.0
45
+ plausible = true
46
+ why_plausible = """
47
+ 1000000 m2 is 100 hectares, an entirely ordinary parcel. The true answer, 9.29
48
+ hectares, is also entirely ordinary. Neither number looks out of place in a
49
+ report, on a map, or in a total; the two differ by the square of 3.28, which is
50
+ a factor nothing in the pipeline has a reason to question. The geometry is
51
+ valid, the CRS is declared and correct, the file is well formed - the only thing
52
+ wrong is that a unit was assumed rather than read."""
53
+ derivation = """
54
+ Closed form: the planar area in the file's own units is 1000 * 1000 = 1e6, and
55
+ calling those units metres yields exactly 1000000."""
56
+
57
+ [refusal]
58
+ accept_if_mentions = ["foot", "feet", "us survey", "linear unit", "unit of measure", "epsg:2229"]
59
+
60
+ [provenance]
61
+ source = """
62
+ Systematic in US parcel and public-works data: state plane zones are published
63
+ in feet, most geometry engines return planar area in the coordinates' own units,
64
+ and the conversion is left to the caller. EPSG deprecated the US survey foot on
65
+ 2022-12-31, which added a second, smaller version of the same confusion."""
66
+ found_by = "mapsmith"
67
+ date = "2026-08-23"
68
+
69
+ [provenance.engine_versions]
70
+ shapely = "planar area only, unit-unaware by design - correctness depends entirely on the caller"
@@ -0,0 +1,61 @@
1
+ # 003 — Declared nodata cells counted as elevations
2
+
3
+ ## The file
4
+
5
+ `dem.tif` is a 100×100 int16 DEM in EPSG:32632. Every valid cell holds 1000 m.
6
+ Fifty cells hold **−9999**, and the GeoTIFF header declares `nodata = -9999`,
7
+ which is where every raster library looks.
8
+
9
+ Nothing is hidden and nothing is malformed. Voids are what a DEM has where the
10
+ sensor saw cloud, water or steep shadow; −9999 and −32768 are the conventions
11
+ that fill them, and SRTM and ASTER GDEM ship them by the million.
12
+
13
+ ## The right answer, on paper
14
+
15
+ 9950 valid cells, all 1000. The mean is **1000.0** exactly.
16
+
17
+ ## The wrong answer, also on paper
18
+
19
+ ```
20
+ (9950 × 1000 + 50 × −9999) / 10000 = 9450050 / 10000 = 945.005
21
+ ```
22
+
23
+ ## Why it is admitted, and why this one is worse than it looks
24
+
25
+ 945 m is not merely a possible mean elevation. It is an unremarkable one, and it
26
+ is **5.5% from the truth** — far too small for anyone to question, and far too
27
+ large to ignore in a volume, a gradient, a flood level or a carbon estimate.
28
+
29
+ That is the shape that matters. The same defect on a raster with many voids
30
+ returns something like −0.99, and the first person to see a negative mean
31
+ elevation catches it. The version that survives is the one where the void
32
+ fraction is small, and a DEM with 0.5% voids is an ordinary DEM. The bias tracks
33
+ the void fraction, so the error is *usually* the invisible size.
34
+
35
+ Everything else about the run is clean: valid raster, declared and correct CRS,
36
+ declared and correct nodata, intact grid, statistic returned without a warning.
37
+ The only evidence is a number that looks fine.
38
+
39
+ ## Not a bug in any library
40
+
41
+ `ds.read(1)` returns the raw array. `ds.read(1, masked=True)` returns the masked
42
+ one. Both are one line, and the raw one is the default. Nothing is broken; a
43
+ keyword is missing, and nothing downstream can tell.
44
+
45
+ ## Observed
46
+
47
+ | adapter | answer | |
48
+ |---|---|---|
49
+ | `engine:rasterio` — `read(1, masked=True)` | 1000.0 | ✓ |
50
+ | `engine:naive` — `read(1)` | 945.005 | ✗ |
51
+
52
+ Worth noting: `engine:naive` **passes** trap 001. The careless composition is
53
+ not uniformly careless — it is correct whenever the data happens to be shaped
54
+ the way it usually is, which is exactly what makes the cases where it is not so
55
+ hard to see.
56
+
57
+ ## Clean twin
58
+
59
+ `clean/c003-raster-mean-nodata` — the same grid with no void cells, and the same
60
+ `nodata = -9999` still declared in the header. The only difference between the
61
+ two probes is whether any cell holds that value.
@@ -0,0 +1,51 @@
1
+ """A DEM that declares its nodata value, and has fifty cells of it.
2
+
3
+ Nothing is hidden. The GeoTIFF carries `nodata = -9999` in its header, where
4
+ every raster library looks for it, and the void cells hold exactly that value.
5
+ The file is not malformed and it is not unusual: voids are what a DEM has where
6
+ the sensor saw cloud, water or steep shadow, and -9999 is the convention that
7
+ fills them.
8
+
9
+ The elevations are constant on purpose. A fixture whose valid cells all hold the
10
+ same number makes both the right answer and the wrong one checkable on paper,
11
+ which is worth more here than looking like terrain.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import sys
17
+ from pathlib import Path
18
+
19
+ import numpy as np
20
+ import rasterio
21
+ from rasterio.transform import from_origin
22
+
23
+ SIDE = 100
24
+ ELEVATION = 1000
25
+ NODATA = -9999
26
+ VOID_CELLS = 50
27
+ STRIDE = 197 # coprime with SIDE*SIDE, so the voids scatter and never repeat
28
+
29
+
30
+ def elevations() -> np.ndarray:
31
+ grid = np.full((SIDE, SIDE), ELEVATION, dtype="int16")
32
+ flat = grid.reshape(-1)
33
+ for k in range(VOID_CELLS):
34
+ flat[(k * STRIDE) % (SIDE * SIDE)] = NODATA
35
+ return grid
36
+
37
+
38
+ def main(destination: Path) -> int:
39
+ destination.mkdir(parents=True, exist_ok=True)
40
+ with rasterio.open(
41
+ destination / "dem.tif", "w", driver="GTiff",
42
+ height=SIDE, width=SIDE, count=1, dtype="int16",
43
+ crs="EPSG:32632", nodata=NODATA,
44
+ transform=from_origin(500000.0, 5000000.0, 30.0, 30.0),
45
+ ) as ds:
46
+ ds.write(elevations(), 1)
47
+ return 0
48
+
49
+
50
+ if __name__ == "__main__":
51
+ raise SystemExit(main(Path(sys.argv[1] if len(sys.argv) > 1 else ".")))
@@ -0,0 +1,62 @@
1
+ id = "003-nodata-in-statistics"
2
+ population = "trap"
3
+ family = "nodata"
4
+ title = "Declared nodata cells are counted as elevations"
5
+ surface = ["engine", "agent"]
6
+
7
+ [task]
8
+ call = "raster_mean(dem.tif)"
9
+ prompt = "What is the mean elevation of dem.tif, in metres?"
10
+
11
+ [truth]
12
+ kind = "scalar"
13
+ value = 1000.0
14
+ tolerance = 0.001
15
+ derivation = """
16
+ The raster is 100x100. Every valid cell holds 1000, and 50 cells hold the
17
+ declared nodata value, so the mean over the 9950 valid cells is 1000 exactly.
18
+ The elevations are constant by construction: a fixture whose valid cells all
19
+ hold one number makes both the right answer and the wrong one checkable on
20
+ paper, which matters more here than looking like terrain."""
21
+
22
+ [naive_failure]
23
+ description = """
24
+ The nodata value is read as data. The file declares `nodata = -9999` in its
25
+ header, where every raster library looks, but nothing forces a reader to honour
26
+ it: `ds.read(1)` returns the raw array and `ds.read(1, masked=True)` returns the
27
+ masked one, and the difference is one keyword nobody notices missing."""
28
+ observed_value = 945.005
29
+ plausible = true
30
+ why_plausible = """
31
+ This is the dangerous size of error. 945 m is not merely a possible mean
32
+ elevation, it is an unremarkable one, and it is 5.5% from the truth - far too
33
+ small for anyone to question and far too large to ignore in a volume, a
34
+ gradient or a flood level. A wrong answer of -0.99, which is what the same
35
+ defect produces on a raster with more voids, would be caught by the first
36
+ person to look at it. This one will not be.
37
+
38
+ Nothing else signals the problem. The raster is valid, the CRS is declared, the
39
+ nodata value is declared and correct, the geometry of the grid is intact, and
40
+ the statistic returns without a warning. The only evidence is a number that
41
+ looks fine."""
42
+ derivation = """
43
+ Closed form: (9950 * 1000 + 50 * -9999) / 10000 = 9450050 / 10000 = 945.005
44
+ exactly. The bias scales with the void fraction, which is why it is usually
45
+ small enough to survive review: at 0.5% voids it is a 5.5% error, and a DEM
46
+ with 0.5% voids is an ordinary DEM."""
47
+
48
+ [refusal]
49
+ accept_if_mentions = ["nodata", "no-data", "no data", "-9999", "void", "fill value", "masked"]
50
+
51
+ [provenance]
52
+ source = """
53
+ The most common silent error in raster analysis, and old enough to be folklore:
54
+ SRTM and ASTER GDEM ship voids as -32768 and -9999 respectively, and every
55
+ GIS mailing list has a thread about a statistic that came out wrong by a few
56
+ per cent. It is in this suite because being well known has not made it
57
+ detectable - nothing in a normal pipeline notices."""
58
+ found_by = "mapsmith"
59
+ date = "2026-08-23"
60
+
61
+ [provenance.engine_versions]
62
+ rasterio = "read(1, masked=True) honours it; read(1) does not, and both are one line"