@kanzo-tech/graph 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +15 -7
- package/dist/bounded.d.ts +119 -69
- package/dist/bounded.d.ts.map +1 -1
- package/dist/bounded.js +13 -34
- package/dist/bounded.js.map +1 -1
- package/dist/duck-source.d.ts +60 -72
- package/dist/duck-source.d.ts.map +1 -1
- package/dist/duck-source.js +185 -255
- package/dist/duck-source.js.map +1 -1
- package/dist/graph-canvas.d.ts +1 -1
- package/dist/graph-canvas.js.map +1 -1
- package/dist/graph-looks.d.ts +63 -18
- package/dist/graph-looks.d.ts.map +1 -1
- package/dist/graph-looks.js +37 -26
- package/dist/graph-looks.js.map +1 -1
- package/dist/graph-model.d.ts +2 -2
- package/dist/graph-model.d.ts.map +1 -1
- package/dist/graph-model.js +19 -14
- package/dist/graph-model.js.map +1 -1
- package/dist/graph-sim.d.ts +15 -1
- package/dist/graph-sim.d.ts.map +1 -1
- package/dist/graph-sim.js +17 -13
- package/dist/graph-sim.js.map +1 -1
- package/dist/index.d.ts +14 -14
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +30 -48
- package/dist/index.js.map +1 -1
- package/dist/obligations.d.ts +1 -3
- package/dist/obligations.d.ts.map +1 -1
- package/dist/shape-glyph.d.ts +30 -0
- package/dist/shape-glyph.d.ts.map +1 -0
- package/dist/shape-glyph.js +15 -0
- package/dist/shape-glyph.js.map +1 -0
- package/dist/slice-client.d.ts.map +1 -1
- package/dist/slice-client.js +36 -36
- package/dist/slice-client.js.map +1 -1
- package/dist/use-graph-overlays.d.ts +21 -30
- package/dist/use-graph-overlays.d.ts.map +1 -1
- package/dist/use-graph-overlays.js +39 -37
- package/dist/use-graph-overlays.js.map +1 -1
- package/dist/use-graph.d.ts +38 -12
- package/dist/use-graph.d.ts.map +1 -1
- package/dist/use-graph.js +62 -61
- package/dist/use-graph.js.map +1 -1
- package/dist/use-query-loop.d.ts +8 -4
- package/dist/use-query-loop.d.ts.map +1 -1
- package/dist/use-query-loop.js +93 -96
- package/dist/use-query-loop.js.map +1 -1
- package/dist/use-renderer.d.ts.map +1 -1
- package/dist/use-renderer.js +69 -66
- package/dist/use-renderer.js.map +1 -1
- package/package.json +11 -5
- package/dist/memory-source.d.ts +0 -32
- package/dist/memory-source.d.ts.map +0 -1
- package/dist/memory-source.js +0 -134
- package/dist/memory-source.js.map +0 -1
package/dist/duck-source.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"duck-source.js","sources":["../src/duck-source.ts"],"sourcesContent":["\"use client\";\n\nimport { clausePoints, column, fillColumn, numbers } from \"@kanzo-tech/mosaic\";\nimport type { Coordinator, FilterExpr, Selection } from \"@kanzo-tech/mosaic\";\nimport { BOUNDED_DEFAULTS, SUPERSEDED, type BoundedSource, type Slice, type SliceRequest, type Viewport } from \"./bounded\";\nimport { SliceRead } from \"./slice-client\";\nimport { denseOf, typeOf, vertexId, type VertexId } from \"./resident\";\n\n/**\n * A `BoundedSource` over two ordinary relations in DuckDB.\n *\n * On a subpath because Mosaic is an optional peer and this is the half that needs it: a host drawing\n * arrays it already holds takes `memorySource` and pays for no database. Splitting them is what lets\n * that promise be true rather than merely stated.\n *\n * **Neutral about storage, and that is the point.** The neutrality let bounded be measured against\n * unbounded before anything committed to a layout on disk — and it is the reason this file survived\n * a decision on the other side of the seam. The verb it was written to sit beside never landed:\n * `viewport` was dropped and GraphAr with it, because the camera is addressed rather than queried.\n * What replaces it is a tile fetched by a computed URL, which is another source.\n *\n * **Every query in this file goes through a `SliceRead`, and there is no other path.** `onceQuery`\n * was the other one — a throwaway client per query, on this subpath, re-exported for the two\n * showcases that also read a relation directly. It is gone: a read the page's filters cannot reach\n * is a picture that disagrees with the page, and `slice-client.ts` carries what that cost.\n */\n\n/**\n * A DuckDB-backed source, and the one thing it can do that the render contract knows nothing about.\n *\n * `BoundedSource` says what a renderer needs: answer a bounded question. Publishing a selection is\n * the other direction of the same seam and it is Mosaic's, not the renderer's — so it lives on the\n * concrete type rather than on the contract, beside `watch`, which is on the contract because the\n * query loop is what has to act on it.\n */\nexport interface DuckSource extends BoundedSource {\n /**\n * The reader's own selection, as a clause the rest of the page filters by. `null` retracts it.\n *\n * **The graph is exempt from its own clause, and that is the whole difference from the greyout it\n * replaces.** While the canvas *faded* excluded rows it could take its own clause too — the row was\n * still drawn and still selectable, and the fade was the brush. A canvas that now draws what\n * survives would answer a lasso by deleting everything the reader did not lasso, which is not a\n * selection, it is a filter nobody asked for.\n */\n publish(vertices: readonly VertexId[] | null): void;\n}\n\nexport interface DuckSourceOptions {\n coordinator: Coordinator;\n /**\n * The crossfilter this graph draws inside.\n *\n * Given, the page's predicate rides in the slice query and the canvas draws **what survives**.\n * Omitted, the source is a reader of a relation and nothing else — which is what a graph with no\n * charts beside it is.\n */\n filterBy?: Selection;\n /** The node relation. */\n nodes: string;\n /** The edge relation, as pairs of node ids. */\n edges: string;\n /**\n * Which vertex type this relation is.\n *\n * Required, and with no default, because the source is the only thing that knows: a `dense_id`\n * numbers within one type, so the identity a slice carries is only completed here. A corpus of one\n * type is type `0` and has to say so — a defaulted `0` would let a second relation ship the same\n * identities as the first with nothing raised.\n */\n typeIndex: number;\n /**\n * A **dense** integer id — `0..n-1`, no gaps.\n *\n * Dense because `links` refers to positions rather than to ids, so a consumer never pays for an\n * id→index map. GraphAr's `dense_id` is this column by another name.\n */\n idField?: string;\n /**\n * The identity column — the subject IRI. **Omitted, a slice carries addresses only.**\n *\n * A `dense_id` says where a vertex is; the IRI says which vertex it is, and only the second\n * survives the layout being redone. The corpus writes it as `subject`, non-null and unique within\n * a type, which is why that is the name here — but it stays opt-in rather than defaulted, because\n * reading it costs 1.87× the drawing tile and most points are painted rather than named.\n *\n * Set it when something outlives a session: a bookmark, a link out, a selection that has to mean\n * the same thing after the next `fossil run`.\n */\n subjectField?: string;\n xField?: string;\n yField?: string;\n /**\n * **What colours and what sizes are not here**, and their absence is the shape rather than an\n * omission: they are `fill` and `r` on the request, because a channel is what the caller wants\n * drawn now and this object is where the bytes are. Given here too, recolouring meant building a\n * second source — and two places deciding one colour is the state that move ended.\n */\n sourceField?: string;\n targetField?: string;\n}\n\ninterface Columns {\n id: string;\n /** `undefined` when the host did not ask to be able to name a vertex. */\n subject: string | undefined;\n x: string;\n y: string;\n /**\n * The categorical column, when a channel named one — **and `undefined` is not a missing value.**\n *\n * It used to default to `community`, in both sources, which is this side writing what the corpus\n * owns: a relation that has no such column answered `Referenced column \"community\" not found`, and\n * one that has a differently named cluster column was silently coloured by the wrong thing. Neither\n * failure is the caller's, and both were invented here.\n *\n * Unbound, every point is one colour — which is Plot's own answer to a mark with no `fill` channel,\n * and an honest picture rather than a guess.\n */\n category: string | undefined;\n size: string | undefined;\n source: string;\n target: string;\n}\n\n/**\n * The three reads a source makes, and which of them the page can filter.\n *\n * `points` and `links` carry the crossfilter; `meta` deliberately does not. How big the corpus is,\n * where it sits and what its tile footers say are facts about the corpus rather than about the\n * page's current question — and a `total()` that shrank with the filters would make the view's own\n * \"20,000 of 1,000,000\" a fraction of itself, which is the one number a bounded renderer owes its\n * reader honestly.\n */\ninterface Reads {\n points: SliceRead;\n links: SliceRead;\n meta: SliceRead;\n}\n\nfunction openReads(coordinator: Coordinator, filterBy?: Selection): Reads {\n return {\n points: new SliceRead(coordinator, filterBy),\n links: new SliceRead(coordinator, filterBy),\n meta: new SliceRead(coordinator),\n };\n}\n\n/**\n * The metadata reads, queued behind each other.\n *\n * One client answers one question at a time — a second `ask` supersedes the first — and `total()`,\n * `extent()` and the tile probing are issued by different effects with no ordering between them. A\n * queue rather than a client each, because they *already* run one at a time: DuckDB-WASM answers\n * over one connection, measured, so three clients would buy three registrations and no concurrency.\n */\nfunction metaAsker(read: SliceRead): (sql: string) => Promise<unknown> {\n let queue: Promise<unknown> = Promise.resolve();\n return (sql) => {\n const next = queue.then(() => read.ask(() => sql));\n queue = next.catch(() => undefined);\n return next;\n };\n}\n\n/**\n * The page's predicate, as SQL text.\n *\n * The reads here are CTEs over window functions rather than builder queries — `row_number()` over the\n * visible set is what makes a slice's links speak in buffer positions — so the predicate has to be\n * interpolated rather than handed to `Query.where`. Mosaic's expression nodes stringify to the same\n * SQL the builder would emit, which is what makes that safe rather than a re-implementation.\n */\nfunction predicateSql(filter: FilterExpr | undefined): string {\n if (filter == null) return \"\";\n const list = Array.isArray(filter) ? filter : [filter];\n const clauses = list.filter((node) => node != null).map((node) => String(node));\n return clauses.length > 0 ? clauses.map((c) => `(${c})`).join(\" AND \") : \"\";\n}\n\n/** Two predicates, conjoined, where an absent one contributes nothing rather than `AND TRUE`. */\nfunction both(left: string, right: string): string {\n if (!left) return right || \"TRUE\";\n if (!right) return left;\n return `(${left}) AND (${right})`;\n}\n\nexport function duckBoundedSource(options: DuckSourceOptions): DuckSource {\n const { coordinator, edges, filterBy, nodes, typeIndex } = options;\n /**\n * Everything this relation *is*, and nothing about what to draw.\n *\n * `category` and `size` are absent here and filled in per request from `fill` and `r` — the same\n * split `openCorpus` makes with its own `fixed` block, for the same reason.\n */\n const columns: Omit<Columns, \"category\" | \"size\"> = {\n id: options.idField ?? \"id\",\n subject: options.subjectField,\n x: options.xField ?? \"x\",\n y: options.yField ?? \"y\",\n source: options.sourceField ?? \"source\",\n target: options.targetField ?? \"target\",\n };\n\n const reads = openReads(coordinator, filterBy);\n const meta = metaAsker(reads.meta);\n const watching = watcher(reads);\n\n return {\n ...watching.api,\n\n publish(vertices) {\n publishSelection(reads, filterBy, columns.id, vertices);\n },\n\n async total() {\n const rows = await meta(`SELECT count(*) AS n FROM ${nodes}`);\n return Number(numbers(rows, \"n\")[0] ?? 0);\n },\n\n /**\n * Four aggregates, so the canvas can frame what is actually there.\n *\n * One scan of two columns, once — against a first paint that otherwise opens on the renderer's\n * default box and finds the corpus occupying a corner of it.\n */\n async extent() {\n const rows = await meta(\n `SELECT min(${columns.x}) AS x0, max(${columns.x}) AS x1,\n min(${columns.y}) AS y0, max(${columns.y}) AS y1\n FROM ${nodes}`,\n );\n const at = (field: string) => Number(numbers(rows, field)[0] ?? 0);\n return { xMin: at(\"x0\"), yMin: at(\"y0\"), xMax: at(\"x1\"), yMax: at(\"y1\") };\n },\n\n /**\n * Regions only, and it says so by **not having** `explore`.\n *\n * This source is two relations and a spatial predicate — it has no adjacency index, so a\n * neighbourhood query would mean recursive joins over the whole edge table, which is the\n * unbounded pattern wearing a bounded interface. It used to say that with a predicate and a\n * throw; now the absence is the statement, and asking is a compile error.\n */\n async slice(request: SliceRequest): Promise<Slice> {\n const { fill, limit, perPixel, pinned, r, view } = request;\n const asked: Columns = { ...columns, category: fill, size: r };\n // No `held`: this source reads a relation rather than addressing bytes, so there is nothing\n // \"already in hand\" to draw a far end from — see `anchorCte`.\n return watching.run(\n region(nodes, edges, asked, view, limit, typeIndex, pinned, perPixel, undefined),\n );\n },\n };\n}\n\n/**\n * The re-indexing happens in SQL, and that is the whole trick.\n *\n * `row_number() - 1` over the visible set gives every returned point a position in the arrays about\n * to be built, so the edge query can join to it twice and hand back links that already speak in\n * those positions. No id→index map is constructed in JavaScript — which is the 148 ms `load()` spent\n * at 200,000 nodes, gone by construction rather than by optimisation.\n *\n * The `LIMIT` sits inside the CTE, so the numbering is over what survives it. Numbering first and\n * limiting after would hand out indices into an array that was never built.\n */\n/**\n * A rectangle as a SQL predicate, and an **unbounded** rectangle as no predicate at all.\n *\n * `shouldSlice` answers `false` for a graph that fits, and the loop then asks for everything — a\n * viewport whose edges are `±Infinity`. Interpolated, that reads `x BETWEEN -Infinity AND Infinity`,\n * and SQL has no infinity literal: DuckDB parses `Infinity` as a **column name** and fails with\n * `Referenced column \"Infinity\" not found`. So an open edge contributes no clause, and a rectangle\n * open on every side is `TRUE` — which is also the right plan, because a query that wants every row\n * has nothing to prune.\n */\nfunction bboxSql(c: Columns, view: Viewport): string {\n const bounds: [string, number, string][] = [\n [c.x, view.xMin, \">=\"],\n [c.x, view.xMax, \"<=\"],\n [c.y, view.yMin, \">=\"],\n [c.y, view.yMax, \"<=\"],\n ];\n const clauses = bounds\n .filter(([, value]) => Number.isFinite(value))\n .map(([column, value, op]) => `${column} ${op} ${value}`);\n return clauses.length > 0 ? clauses.join(\" AND \") : \"TRUE\";\n}\n\n/**\n * How far apart the sampled ids are — one every `ceil(matched / limit)`.\n *\n * **In SQL rather than in JavaScript because the number it divides is only known inside the query.**\n * `matched` is a window aggregate over the rows the `WHERE` kept, so a caller wanting to compute\n * this outside would have to count first and slice second — two round trips down a connection that\n * answers one at a time, which is the shape `BENCHMARKS.md` records as a hung tab rather than a slow\n * one. As a column reference it costs the pass that was being made anyway.\n *\n * `greatest(1, …)` because an empty window makes the divisor zero, and a modulo by zero is an error\n * rather than an empty answer. At `matched <= limit` it is exactly 1 and `id % 1 = 0` keeps every\n * row: a window that fits is not sampled, it is returned.\n */\nconst strideSql = (limit: number) => `greatest(1, CAST(ceil(matched / ${limit}.0) AS BIGINT))`;\n\n/**\n * The visible set: what the rectangle matched, and the sample of it that gets drawn.\n *\n * **Two CTEs, and the second one is the whole of the far view.** `pool` is every row the predicate\n * kept, carrying `count(*) OVER ()` — a window function is evaluated over everything the `WHERE`\n * kept and `LIMIT` applies after it, so that column is the number that *matched* rather than the\n * number returned. It is what deleted the third query: `SELECT count(*) FROM … WHERE <the same\n * predicate>` was a second scan to learn a number the first scan already had to compute.\n *\n * `vis` then keeps one row in `stride`, **striding over the id rather than taking the front of the\n * ordering**, and that is the difference between a picture of the window and a picture of one corner\n * of it. A corpus numbers `dense_id` along the Morton curve, so every `s`-th id is a spatially\n * stratified sample; the same `LIMIT` with no stride returns a contiguous run of the curve, which is\n * a sub-region. Measured against the truth at screen resolution, L1@8px over blocks of eight pixels:\n * a stride sample of 20,000 scores 0.167 / 0.240 / 0.269 at 200k / 1M / 5M against a uniform null of\n * 1.044 / 0.829 / 0.731 — see `/docs/design/graph`.\n *\n * @param matched Whether to project the pre-sample count out to the caller.\n *\n * It rides on the points read only. The links read builds the same CTEs to join against and never\n * looks at the column — but it does compute it, because the stride is a function of it and both\n * reads have to select the *same* rows or a slice would draw edges to vertices it did not return.\n */\nfunction visibleCte(\n nodes: string,\n c: Columns,\n where: string,\n limit: number,\n matched = true,\n): string {\n const size = c.size ? `, ${c.size} AS size` : \"\";\n // Selected in the CTE rather than joined back afterwards: the numbering is over what survives the\n // LIMIT, and a second pass keyed on `local` would be a second scan to fetch a column the first one\n // was already standing on.\n const subject = c.subject ? `, ${c.subject} AS subject` : \"\";\n // No categorical binding, no ranking: a literal zero is the ordinal every point wears, and the\n // scale hands that one colour. Ranking a column nobody named is how a default column gets invented.\n // Ranked over the sample rather than over the window, so the ordinals are contiguous across what\n // is actually drawn — which is what the colour scale is handed.\n const category = c.category\n ? `(dense_rank() OVER (ORDER BY cat) - 1)::INTEGER AS category`\n : \"0::INTEGER AS category\";\n return `WITH pool AS (\n SELECT ${c.id} AS id, ${c.x} AS x, ${c.y} AS y${size}${subject}${\n c.category ? `, ${c.category} AS cat` : \"\"\n },\n count(*) OVER () AS matched\n FROM ${nodes}\n WHERE ${where}\n ), vis AS (\n SELECT id, x, y${c.size ? \", size\" : \"\"}${c.subject ? \", subject\" : \"\"}${\n matched ? \", matched\" : \"\"\n },\n ${category},\n (row_number() OVER (ORDER BY id) - 1)::INTEGER AS local\n FROM pool\n WHERE id % ${strideSql(limit)} = 0\n LIMIT ${limit}\n )`;\n}\n\n/**\n * The shortest edge worth a row, as a predicate over the two endpoints — **squared, and on purpose.**\n *\n * A distance is compared against a threshold, and squaring both sides removes a `sqrt` per row from\n * a predicate evaluated once per candidate edge. It changes no answer: both sides are non-negative.\n *\n * `undefined` when the caller said nothing about resolution, and then there is no predicate at all\n * rather than a permissive one — a request with no canvas behind it (`EVERYTHING`) has no pixels to\n * measure three of.\n */\nfunction longEnough(a: string, b: string, perPixel: number | undefined): string {\n if (perPixel === undefined || !Number.isFinite(perPixel) || perPixel <= 0) return \"\";\n const floor = BOUNDED_DEFAULTS.minLinkPixels * perPixel;\n return `(${a}.x - ${b}.x) * (${a}.x - ${b}.x) + (${a}.y - ${b}.y) * (${a}.y - ${b}.y) >= ${floor * floor}`;\n}\n\n/**\n * The far ends, and the edges that reach them — **out of bytes the reader already fetched.**\n *\n * An edge with one end outside the rectangle is dropped today, and that loses 19.31% / 31.94% /\n * 28.92% of the edges incident to a window at 200k / 1M / 5M; 7,930 of 20,000 vertices carry at least\n * one at five million. What was missing was never the edge row — a window reads the `by_source` tiles\n * of every vertex it draws, so the row is in hand — it was **a position to draw the far end at**.\n *\n * **And a tile answers that for free.** A tile is 4,096 rows of a Morton-ordered relation and its\n * bounding box is far wider than the rows the rectangle keeps, so the vertices just outside the\n * window are usually in a tile the window already fetched. Measured over five windows of\n * `docs/public/bench/1000000`, 2026-08-19: relaxing the join from *inside the rectangle* to *inside\n * the tiles that were read* takes the drawn edges of five windows from 616,885 to 781,562 of 906,337\n * incident — **56.9% of everything the reader was losing, at no request, no query and no byte.**\n *\n * **The tile boundary is also a distance filter, and that is what settles the drawing.** Over every\n * far end a window loses, the median sits 1.64 semi-widths out and the worst 47.1 — which is why a\n * stub clipped to the viewport is a lie: nothing distinguishes 1.1 from 47. The far ends a held tile\n * can answer are the near ones: median 1.11–1.85 semi-widths, 90th percentile 1.29–2.60, worst\n * **6.74**, against 7.09–16.91 for the full reachable set. So the honest picture and the free one are\n * the same picture, and there is no trade to make: the anchors are drawn where the vertices are, and\n * the long tail stays undrawn because its bytes are not here — not because we decided.\n *\n * @param held The relation whose rows the reader is holding — the tiles a corpus fetched for this\n * window. **`undefined` for a source over an ordinary relation, and that is not a downgrade, it is\n * the truth:** nothing is \"already in hand\" there, so `held` would be the whole node table and the\n * join would scan the corpus twice per camera move. Only an addressed source knows what it holds.\n */\nfunction anchorCte(\n held: string,\n edges: string,\n c: Columns,\n spatial: string,\n filter: string,\n perPixel: number | undefined,\n): string {\n const outside = both(`NOT (${spatial})`, filter);\n const long = longEnough(\"a\", \"b\", perPixel);\n return `, out AS (\n SELECT ${c.id} AS id, ${c.x} AS x, ${c.y} AS y FROM ${held} WHERE ${outside}\n ), reach AS (\n SELECT id, x, y FROM vis UNION ALL SELECT id, x, y FROM out\n ), span AS (\n SELECT sv.local AS src, tv.local AS dst, a.id AS src_id, b.id AS dst_id\n FROM ${edges} e\n JOIN reach a ON e.${c.source} = a.id\n JOIN reach b ON e.${c.target} = b.id\n LEFT JOIN vis sv ON sv.id = a.id\n LEFT JOIN vis tv ON tv.id = b.id\n WHERE (sv.id IS NOT NULL OR tv.id IS NOT NULL)${long ? ` AND ${long}` : \"\"}\n ), anchor AS (\n SELECT o.id, o.x, o.y,\n ((SELECT count(*) FROM vis) + row_number() OVER (ORDER BY o.id) - 1)::INTEGER AS local\n FROM out o\n WHERE o.id IN (SELECT src_id FROM span WHERE src IS NULL\n UNION SELECT dst_id FROM span WHERE dst IS NULL)\n )`;\n}\n\n/**\n * A question, in the two halves it is asked in and the one place they are put back together.\n *\n * The reads run through the coordinator, so the *same* pair of SQL builders serves both directions:\n * the camera pulling an answer, and the page's filters pushing one. `assemble` is therefore a\n * function of the two results and of nothing else — no closure over which of the two paths asked,\n * because a slice that came back because somebody brushed a histogram is the same slice.\n */\ninterface Plan {\n points: (filter: FilterExpr) => string;\n links: (filter: FilterExpr) => string;\n assemble: (points: unknown, links: unknown) => Slice;\n}\n\n/**\n * The one plan there is: a rectangle, its sample, and the edges both of whose ends survived it.\n *\n * It was called `detail` because it was one of two, and the other one — a `GROUP BY` over a\n * categorical column, one super-node per group — is gone. There is no mode to be in.\n */\nfunction region(\n nodes: string,\n edges: string,\n c: Columns,\n view: Viewport,\n limit: number,\n typeIndex: number,\n pinned: VertexId[] | undefined,\n perPixel: number | undefined,\n held: string | undefined,\n): Plan {\n const bbox = bboxSql(c, view);\n // A dragged node is drawn where the reader dropped it and indexed where it always was, so the\n // rectangle cannot find it. Riding along in the predicate is what keeps it on screen — and it\n // stays a predicate rather than a second query so the numbering still covers everything returned.\n //\n // Only this relation's own vertices: a pinned set spans the whole canvas, and asking one node\n // table for another type's dense ids returns the wrong rows rather than none.\n const mine = (pinned ?? []).filter((v) => typeOf(v) === typeIndex).map(denseOf);\n const pins = mine.length > 0 ? ` OR ${c.id} IN (${mine.join(\",\")})` : \"\";\n const spatial = `(${bbox})${pins}`;\n /**\n * The page's predicate outside the pin, not inside it.\n *\n * A pin says *where to look*; the filters say *what exists*. Written the other way round —\n * `bbox AND filter OR pinned` — a pinned node would survive a filter that excludes it, and the\n * canvas would draw a vertex the rest of the page has agreed is not there.\n */\n const where = (filter: FilterExpr) => both(spatial, predicateSql(filter));\n const size = c.size ? \", size\" : \"\";\n const subject = c.subject ? \", subject\" : \"\";\n const near = longEnough(\"s\", \"t\", perPixel);\n const anchors = (filter: FilterExpr) =>\n held ? anchorCte(held, edges, c, spatial, predicateSql(filter), perPixel) : \"\";\n\n return {\n /**\n * The marks, then the anchors, in one answer — because they are one buffer.\n *\n * `ORDER BY local` is load-bearing rather than tidy: `local` runs `0..marks-1` over the sample\n * and continues past it over the anchors, so ordering by it puts every mark before every anchor\n * and makes `marks` a prefix length. `matched` is then still read off row zero.\n */\n points: (filter) =>\n `${visibleCte(nodes, c, where(filter), limit)}${anchors(filter)}\n SELECT local, id, x, y, category, matched, TRUE AS mark${size}${subject} FROM vis${\n held\n ? `\n UNION ALL\n SELECT local, id, x, y, 0::INTEGER, NULL::BIGINT, FALSE AS mark${\n c.size ? \", CAST(NULL AS DOUBLE)\" : \"\"\n }${c.subject ? \", CAST(NULL AS VARCHAR)\" : \"\"} FROM anchor`\n : \"\"\n }\n ORDER BY local`,\n /**\n * One end drawn and both ends positioned — or, where nothing is held, both ends drawn.\n *\n * The second form is what a source over an ordinary relation gets, and it is the old query plus\n * the length predicate. It still drops an edge that leaves the window, and the reason is now\n * exact rather than a limitation of the reader: no bytes beyond the rectangle were fetched, so\n * there is no position to draw the far end at. A corpus fetches tiles and therefore has some.\n */\n links: (filter) =>\n held\n ? `${visibleCte(nodes, c, where(filter), limit, false)}${anchors(filter)}\n SELECT coalesce(sp.src, sa.local) AS src, coalesce(sp.dst, da.local) AS dst\n FROM span sp\n LEFT JOIN anchor sa ON sa.id = sp.src_id\n LEFT JOIN anchor da ON da.id = sp.dst_id`\n : `${visibleCte(nodes, c, where(filter), limit, false)}\n SELECT s.local AS src, t.local AS dst\n FROM ${edges} e\n JOIN vis s ON e.${c.source} = s.id\n JOIN vis t ON e.${c.target} = t.id${\n near ? `\n WHERE ${near}` : \"\"\n }`,\n assemble: (points, links) => ({\n /**\n * How many matched, separately from how many came back.\n *\n * Without it the view cannot tell a reader \"there is more here than I am showing you\", and a\n * truncated slice looks exactly like a complete one — which is the failure this whole branch\n * has been about. Read off the first row rather than asked for: `matched` is constant down the\n * column, and an empty answer has no row and no matches, which agree.\n */\n n: Number(numbers(points, \"matched\")[0] ?? 0),\n ...arrays(points, links, typeIndex, c.size ? \"size\" : undefined, c.subject !== undefined),\n }),\n };\n}\n\n/**\n * The half of a source that runs a [`Plan`], and the half that answers when nobody asked.\n *\n * Both sources need exactly this and neither should own a second copy of it — which is why it is a\n * function over the reads rather than two blocks of the same bookkeeping. What it holds is the\n * *standing* plan: the last question the camera put, kept so an answer arriving because the page\n * filtered something can be put back together the same way.\n */\nfunction watcher(reads: Reads) {\n let standing: Plan | null = null;\n let listener: ((slice: Slice) => void) | null = null;\n /** The half-answers of a push, waiting for their sibling. */\n const landed = new Map<\"points\" | \"links\", unknown>();\n\n const arrived = (half: \"points\" | \"links\") => (data: unknown) => {\n if (!standing || !listener) return;\n landed.set(half, data);\n if (landed.size < 2) return;\n const points = landed.get(\"points\");\n const links = landed.get(\"links\");\n landed.clear();\n listener(standing.assemble(points, links));\n };\n\n return {\n api: {\n /**\n * Say when the answer changes for a reason the camera cannot see.\n *\n * The reason it hands over a whole `Slice` rather than a nudge to ask again: the coordinator\n * has *already* re-run both reads with the new predicate by the time we hear about it. Asking\n * again would run the same two queries a second time to learn what is in hand.\n */\n watch(answered: (slice: Slice) => void): () => void {\n listener = answered;\n reads.points.onAnswer = arrived(\"points\");\n reads.links.onAnswer = arrived(\"links\");\n return () => {\n listener = null;\n landed.clear();\n standing = null;\n reads.points.release();\n reads.links.release();\n reads.meta.release();\n };\n },\n },\n async run(plan: Plan): Promise<Slice> {\n standing = plan;\n // Cleared because these two are halves of the *previous* question: keeping one would pair a\n // stale rectangle's points with the new rectangle's links the next time the page filters.\n landed.clear();\n const [points, links] = await Promise.all([\n reads.points.ask(plan.points),\n reads.links.ask(plan.links),\n ]);\n return plan.assemble(points, links);\n },\n };\n}\n\n/**\n * The reader's own gesture, as a clause — and the graph exempted from it.\n *\n * `clausePoints` defaults `clients` to the clause's source when that source is itself a client, which\n * is the exemption a crossfilter is built on. Here the source is a plain object and there are two\n * clients to exempt, so the set is written out: a lasso must filter the page's charts and leave the\n * canvas showing what the reader lassoed *in context*, rather than deleting everything else.\n */\nfunction publishSelection(\n reads: Reads,\n filterBy: Selection | undefined,\n idField: string,\n vertices: readonly VertexId[] | null,\n): void {\n if (!filterBy) return;\n filterBy.update(\n clausePoints([idField], vertices?.map((vertex) => [denseOf(vertex)]), {\n source: reads.points,\n clients: new Set([reads.points, reads.links]),\n }),\n );\n}\n\n/**\n * Arrow columns to the parallel typed arrays the renderer takes — sized once, filled in place.\n *\n * `fillColumn` rather than `numbers`: Arrow already hands back a typed buffer, and the obvious route\n * through `Array.from(...).map(Number)` allocates two full-length boxed arrays on the way to a third\n * that was the actual destination. Three copies to move nothing. Here the point buffers are sized\n * against the query's own `LIMIT` and each column is written straight into its stride, so `x` and\n * `y` interleave with no seam between them and no intermediate at all.\n */\nfunction arrays(\n points: unknown,\n links: unknown,\n typeIndex: number,\n sizeField?: string,\n withSubjects = false,\n): Omit<Slice, \"n\"> {\n // Sized from the answer rather than from the request's `limit`, which it used to be: an answer\n // carries the window's marks *and* the anchors the edges leaving it end at, so `limit` is no longer\n // an upper bound on the rows. Under-sizing here would drop the anchors silently and leave every\n // link that pointed at one indexing past the buffer.\n const rows = countOf(points, \"x\");\n const positions = new Float32Array(rows * 2);\n const n = fillColumn(points, \"x\", positions, 0, 2);\n fillColumn(points, \"y\", positions, 1, 2);\n\n // Where the marks stop. `mark` is `TRUE` down the sample and `FALSE` down the anchors, and the\n // query orders by `local`, so this is a prefix length rather than a count — which is what lets\n // `residentOf` and `buffers` treat \"is this a mark\" as an index comparison.\n const marked = new Int32Array(rows);\n fillColumn(points, \"mark\", marked);\n let marks = 0;\n while (marks < n && marked[marks] !== 0) marks++;\n\n // The dense ids land in a scratch and are widened into identities as they are copied across.\n //\n // In place, into the destination, would be better and is not available: `fillColumn` writes\n // `Number(…)`, and a `BigUint64Array` element takes a `bigint` only — assigning a `number` to one\n // throws rather than coercing. That refusal is the same guarantee this whole change is for, so the\n // extra `n`-long buffer is the price of the boundary being enforced by the runtime and not by us.\n const dense = new Float64Array(n);\n fillColumn(points, \"id\", dense);\n const vertices = new BigUint64Array(n);\n for (let i = 0; i < n; i++) vertices[i] = vertexId(typeIndex, dense[i] as number);\n const categories = new Uint16Array(n);\n fillColumn(points, \"category\", categories);\n\n // `column` rather than `fillColumn`: an IRI is a string, so there is no typed buffer to write\n // into and no interleaving to express. It is the one thing a slice carries that never reaches the\n // GPU, which is why asking for it is a decision rather than a default.\n const subjects = withSubjects ? (column(points, \"subject\") as string[]) : undefined;\n\n let sizes: Float32Array | undefined;\n if (sizeField) {\n sizes = new Float32Array(n);\n fillColumn(points, sizeField, sizes);\n }\n\n // The edge count is not bounded by the point limit, so it is asked for rather than assumed.\n const edgeCount = countOf(links, \"src\");\n const edges = new Float32Array(edgeCount * 2);\n const wrote = fillColumn(links, \"src\", edges, 0, 2);\n fillColumn(links, \"dst\", edges, 1, 2);\n\n return {\n marks,\n vertices,\n subjects,\n positions: positions.subarray(0, n * 2),\n links: edges.subarray(0, wrote * 2),\n categories,\n sizes,\n };\n}\n\n/** A row count for a result that does not advertise one, without materialising the rows. */\nfunction countOf(rows: unknown, field: string): number {\n const advertised = (rows as { numRows?: number } | null)?.numRows;\n if (typeof advertised === \"number\") return advertised;\n const child = (rows as { getChild?: (f: string) => { length: number } | null })?.getChild?.(field);\n if (child) return child.length;\n return Array.from(rows as Iterable<unknown>).length;\n}\n\n/**\n * A corpus that fossil wrote, read by address.\n *\n * **The five things a call site used to know, and now does not.** Drawing a corpus meant deriving\n * the chunk URLs from a `chunk_size` copied by hand, knowing how a tile is named, knowing what the\n * edge directory is called, knowing GraphAr's column names, and knowing that a glob cannot work over\n * a plain HTTP origin because there is no listing. Five conventions and about forty lines, none of\n * it the business of something that wants to draw a graph. `/docs/design/graph` carries the\n * argument; the copied `chunk_size` carries the evidence, because it went stale and read\n * a fraction of a corpus in silence for as long as it did.\n *\n * The consumer knows one thing: **where the corpus is.**\n *\n * ```ts\n * const { source } = await openCorpus({ coordinator, dest: \"/bench/1000000\" });\n * ```\n *\n * **Addressed, not queried.** The manifest and the per-tile boxes are read once and kept; after that\n * a camera move is arithmetic over boxes and a list of URLs. There is deliberately no request on the\n * path between the camera moving and a URL being computable — the moment there is one, this has\n * become the `viewport` verb fossil deleted.\n *\n * **What it is not.** It takes no column names and no type index. Those come from the manifest or\n * they do not come: a corpus reader that also accepts `idField` is `duckBoundedSource` with extra\n * steps, and there is already one of those for the case this is not — an arbitrary relation with\n * `x`/`y` that nobody wrote as a corpus.\n */\n\nexport interface OpenCorpusOptions {\n coordinator: Coordinator;\n /**\n * The crossfilter this graph draws inside — the same `Selection` the page's charts filter by.\n *\n * Given, the predicate rides in the slice query and the canvas draws what survives.\n */\n filterBy?: Selection;\n /** Where the corpus lives, without a trailing slash — the directory holding `graph.graph.yml`. */\n dest: string;\n /**\n * Which vertex type to draw, when a corpus carries more than one.\n *\n * Defaults to the first the manifest names. A corpus of one type never passes it; a corpus of\n * several has to, because *which graph do you mean* is not a question a reader can answer.\n */\n vertexType?: string;\n /**\n * Read the identity column as well. **Off by default, and the same trade as everywhere else:**\n * `subject` costs about twice the drawing tile, so the drawing path carries addresses and a host\n * asks for names when something has to be *named* rather than painted.\n */\n subjects?: boolean;\n}\n\n/** One tile's bounding box, from the footer. `null` for a tile whose statistics are missing. */\ninterface TileBox {\n tile: number;\n x0: number;\n x1: number;\n y0: number;\n y1: number;\n}\n\n/**\n * The manifests, flat enough to read with a line scan.\n *\n * A YAML library would be a dependency for six keys, and fossil's own checker makes the same call —\n * sixty lines that refuse what they cannot parse rather than guessing at it. This does the same: a\n * key it cannot find is an error naming the file, not a default that draws an empty graph.\n */\nfunction scalar(yaml: string, key: string): string | undefined {\n const line = yaml.split(\"\\n\").find((row) => row.startsWith(`${key}:`));\n return line?.slice(key.length + 1).trim().replace(/^['\"]|['\"]$/g, \"\");\n}\n\nfunction listItems(yaml: string, key: string): string[] {\n const rows = yaml.split(\"\\n\");\n const at = rows.findIndex((row) => row.startsWith(`${key}:`));\n if (at < 0) return [];\n const items: string[] = [];\n for (const row of rows.slice(at + 1)) {\n if (!row.startsWith(\"- \")) break;\n items.push(row.slice(2).trim());\n }\n return items;\n}\n\nasync function manifest(dest: string, path: string): Promise<string> {\n const response = await fetch(`${dest}/${path}`);\n if (!response.ok) throw new Error(`corpus: ${path} is not readable (${response.status})`);\n return response.text();\n}\n\n/**\n * An opened corpus: the half that **draws** and the half that **answers**.\n *\n * A host needs both over the same bytes and they are not the same access. The canvas reads tiles by\n * address — a handful of files per camera move, chosen from the footer's boxes, with no query. A\n * chart, a crossfilter clause or a verb reads the *relation*: every row, by column name, in SQL.\n * Hiding the URLs behind `source` is right for the first and leaves the second with nothing to\n * query, so opening a corpus registers views for it.\n *\n * This is the other side's own shape. `fossil-mcp` describes itself as opening a dataset,\n * *registering views over the Parquet the corpus already holds*, and dispatching a verb — the same\n * two halves, named the same way, one call apart.\n */\nexport interface OpenedCorpus {\n /**\n * For the canvas: `<GraphCanvas source={…}>`. Reads tiles, never the whole relation.\n *\n * A `DuckSource`, and `CorpusSource` is gone with the reason it existed: `extent()` was declared\n * there because a source over an unlaid-out relation has no answer to it, and *optional on the\n * base contract* says that better than a second interface — a relation with `x`/`y` has an extent\n * too, and it was the one host that could not frame its opening view.\n */\n source: DuckSource;\n /**\n * The vertex relation, registered and ready to query by name.\n *\n * Every column the manifest declares, including the corpus' own properties — so a clause a chart\n * publishes over `kind` or `region` lands here with no translation, which is what makes one\n * crossfilter serve the canvas and the charts.\n */\n nodes: string;\n /** The source-ordered edge relation, or `undefined` when the corpus declares no edge for this type. */\n edges: string | undefined;\n}\n\nexport async function openCorpus(options: OpenCorpusOptions): Promise<OpenedCorpus> {\n const { coordinator, dest, filterBy, subjects = false, vertexType } = options;\n\n const reads = openReads(coordinator, filterBy);\n const meta = metaAsker(reads.meta);\n const watching = watcher(reads);\n\n const root = await manifest(dest, \"graph.graph.yml\");\n const vertexPaths = listItems(root, \"vertices\");\n const edgePaths = listItems(root, \"edges\");\n\n const vertices = await Promise.all(vertexPaths.map((p) => manifest(dest, p)));\n const wanted =\n vertexType === undefined\n ? vertices[0]\n : vertices.find((y) => scalar(y, \"type\") === vertexType);\n if (!wanted) {\n throw new Error(\n `corpus: no vertex type ${vertexType ?? \"(none declared)\"} in ${dest}/graph.graph.yml`,\n );\n }\n\n const type = scalar(wanted, \"type\");\n const prefix = scalar(wanted, \"prefix\");\n const chunkSize = Number(scalar(wanted, \"chunk_size\"));\n if (!type || !prefix || !Number.isFinite(chunkSize) || chunkSize <= 0) {\n throw new Error(`corpus: ${dest} declares no usable type, prefix and chunk_size`);\n }\n\n // The edge relation whose source is this vertex type. Its tiles are keyed by the same range as the\n // vertices — `src_chunk_size` equals the source type's `chunk_size`, and a different number there\n // would address nothing — which is what lets one tile set serve both relations.\n const edges = await Promise.all(edgePaths.map((p) => manifest(dest, p)));\n const edge = edges.find((y) => scalar(y, \"src_type\") === type);\n const edgePrefix = edge ? scalar(edge, \"prefix\") : undefined;\n\n /**\n * The boxes, and the one query this source makes that is not a slice.\n *\n * Read on first use rather than in the factory: a host that constructs a source and never draws\n * should not pay for it, and the cost is a footer read over every tile. Kept forever after —\n * tiles are precomputed and their boxes cannot move without the corpus being rewritten.\n *\n * **Held as the promise rather than as the answer**, which the move to one shared metadata client\n * forced and which was a latent defect before it: `total()` and `extent()` are called by different\n * effects with nothing ordering them, so two loads used to run concurrently and probe the whole\n * tile range twice.\n */\n let loading: Promise<TileBox[]> | null = null;\n let total: number | undefined;\n\n const tileUrl = (k: number) => `'${dest}/${prefix}chunk${k}.parquet'`;\n const edgeTileUrl = (k: number) => `'${dest}/${edgePrefix}by_source/tile${k}.parquet'`;\n\n /**\n * The tiles a window needs, held as bytes so that **panning back is free**.\n *\n * This is the one item on `BENCHMARKS.md`'s fix list that no amount of query tuning substitutes\n * for, and the measurement that puts it there is blunt: two of six drag steps at ten million\n * transfer zero new bytes and still cost 247 requests each, because every visit re-reads the same\n * footers and column chunks over HTTP. A tile fetched once and registered as a file is read from\n * memory forever after — no request, no range negotiation, no metadata round trip.\n *\n * **A tile address is what makes this possible at all**, and it is why the cache lives here rather\n * than in the render loop: a rectangle is a continuous key nothing can memoise, and a tile index is\n * a discrete one. `/docs/design/graph` argued the camera is addressed;\n * this is the first thing that spends the address on something.\n *\n * **The trade is honest and it is not free.** A registered tile is the *whole* tile, where DuckDB\n * over HTTP reads only the column chunks a query projects — so the first visit costs more bytes and\n * every later one costs none. Which way that nets out depends on tile size, which is the corpus's\n * to choose and not ours: at 122,880 rows a tile is about a megabyte, and at the 4,096 the request\n * arithmetic asks for it is about forty kilobytes.\n *\n * Discovered rather than required: a coordinator whose connector is not DuckDB-WASM has no\n * filesystem to register into, and reads by URL exactly as before. `@duckdb/duckdb-wasm` is\n * deliberately not a dependency of this package, so the capability is named structurally.\n */\n interface Registrar {\n registerFileBuffer(name: string, buffer: Uint8Array): Promise<void>;\n dropFile(name: string): Promise<void>;\n }\n const registrar = async (): Promise<Registrar | null> => {\n const connector = coordinator.databaseConnector?.() as\n | { getDuckDB?: () => Promise<Registrar> }\n | null\n | undefined;\n if (!connector?.getDuckDB) return null;\n try {\n return await connector.getDuckDB();\n } catch {\n return null;\n }\n };\n\n /** Registered tile name → how many bytes it is holding. Insertion order is the eviction order. */\n const resident = new Map<string, number>();\n let held = 0;\n\n /**\n * Sixty-four megabytes of tiles, evicted oldest-first.\n *\n * A budget rather than a count, because a tile's size is the corpus's decision and a count would\n * mean something different for every one. Oldest-first rather than least-recently-used: a reader\n * pans, and a pan revisits what it just left, so recency and insertion order agree where it\n * matters — and an LRU's bookkeeping is a second structure to keep correct for a difference nobody\n * has measured.\n */\n const BUDGET = 64 * 1024 * 1024;\n\n /**\n * A tile is worth holding when it is **cheap to fetch whole** — 256 KB, and the number is measured.\n *\n * Registering a tile means downloading all of it, where DuckDB over HTTP reads only the column\n * chunks a query projects. On the million-node corpus a vertex tile is 74 KB and the edge tiles are\n * far larger, and caching both took a cold window from about 200 ms to **17,979 ms** while a repeat\n * of the same window fell to **11 ms**. Holding everything is a thousandfold win on revisit paid\n * for with an eighteen-second first paint, which is not a trade anybody would take.\n *\n * So the rule is a property of the tile rather than a flag: under the bar it is cached, over it the\n * URL is handed to DuckDB and the range reads happen as before. It also means the corpus decides —\n * `BENCHMARKS.md` asks for 4,096-row tiles on the request arithmetic alone, and at that size every\n * tile falls under this bar and the whole read path becomes cacheable without a line changing here.\n */\n const WORTH_HOLDING = 256 * 1024;\n\n /**\n * The names DuckDB should read these tiles from — registered buffers where possible, URLs where\n * not.\n *\n * Fetched concurrently, because a window is a handful of tiles and they are independent; a\n * sequential loop here would make the first paint the sum of its tiles rather than the slowest.\n */\n async function readable(kind: \"vertex\" | \"edge\", tiles: number[]): Promise<string[]> {\n const url = kind === \"vertex\" ? tileUrl : edgeTileUrl;\n const db = await registrar();\n if (!db) return tiles.map(url);\n const names = await Promise.all(\n tiles.map(async (k) => {\n const name = `corpus_${type}_${kind}_${k}.parquet`;\n if (resident.has(name)) return name;\n const address = url(k).slice(1, -1);\n // The size first, which is one metadata request against a body that may be megabytes — and\n // the same request the tile count already probes with, so the shape is not new here.\n const probe = await fetch(address, { method: \"HEAD\" });\n const size = Number(probe.headers.get(\"content-length\"));\n if (!probe.ok || !Number.isFinite(size) || size > WORTH_HOLDING) return url(k);\n const response = await fetch(address);\n // A tile that will not load is not a reason to fail the whole window: fall back to the URL\n // and let DuckDB report whatever it finds there, which is the error a reader can act on.\n if (!response.ok) return url(k);\n const bytes = new Uint8Array(await response.arrayBuffer());\n await db.registerFileBuffer(name, bytes);\n resident.set(name, bytes.byteLength);\n held += bytes.byteLength;\n return name;\n }),\n );\n while (held > BUDGET && resident.size > 0) {\n const [oldest, size] = resident.entries().next().value as [string, number];\n // Never evict a tile this very window is about to read, or the query reads a dropped file.\n if (names.includes(oldest)) break;\n resident.delete(oldest);\n held -= size;\n await db.dropFile(oldest);\n }\n return names.map((n) => (n.startsWith(\"'\") ? n : `'${n}'`));\n }\n\n function load(): Promise<TileBox[]> {\n loading ??= (async () => {\n /**\n * How many tiles there are, without listing anything.\n *\n * `ceil(V / chunk_size)` is the published arithmetic and it needs `V`, which **no manifest\n * carries** — the vertex YAML declares the type, the prefix and the chunk size and stops. So\n * the count has to come from the tiles themselves, and the obvious route is the one that does\n * not work: `read_parquet('…/chunk*.parquet')` expands a glob, expanding a glob lists a\n * directory, and a plain HTTP origin has no listing. It succeeds against a local path and\n * against a bucket, which is what makes the mistake easy to keep.\n *\n * So the last tile is found by probing — double until a `HEAD` misses, then bisect. That is\n * about a dozen requests once for a corpus of any size, and every one of them is a request a\n * plain origin can answer.\n */\n const exists = async (k: number) =>\n (await fetch(`${dest}/${prefix}chunk${k}.parquet`, { method: \"HEAD\" })).ok;\n if (!(await exists(0))) throw new Error(`corpus: ${dest}/${prefix} holds no chunk0`);\n let low = 0;\n let high = 1;\n while (await exists(high)) {\n low = high;\n high *= 2;\n }\n while (high - low > 1) {\n const mid = Math.floor((low + high) / 2);\n if (await exists(mid)) low = mid;\n else high = mid;\n }\n const tiles = low + 1;\n // The last tile is short unless the count divides evenly, and its footer says by how much —\n // metadata only, so this reads no column.\n const tail = await meta(\n `SELECT num_rows AS n FROM parquet_file_metadata('${tileUrl(low).slice(1, -1)}')`,\n );\n total = low * chunkSize + Number(numbers(tail, \"n\")[0] ?? 0);\n const urls = Array.from({ length: tiles }, (_, k) => tileUrl(k)).join(\", \");\n /**\n * `min_value`/`max_value`, never `min`/`max`.\n *\n * Parquet's original statistics fields are defined by *signed* byte comparison, which is\n * meaningless for an unsigned column — a writer that gets this right leaves them empty. A\n * reader that only knows the deprecated pair concludes the footer carries no box for the column\n * the whole address is built on. `coalesce` keeps the float columns working either way.\n */\n const stats = await meta(\n `SELECT CAST(regexp_extract(file_name, 'chunk(\\\\d+)', 1) AS INTEGER) AS tile,\n min(CASE WHEN path_in_schema = 'x' THEN coalesce(stats_min_value, stats_min)::DOUBLE END) AS x0,\n max(CASE WHEN path_in_schema = 'x' THEN coalesce(stats_max_value, stats_max)::DOUBLE END) AS x1,\n min(CASE WHEN path_in_schema = 'y' THEN coalesce(stats_min_value, stats_min)::DOUBLE END) AS y0,\n max(CASE WHEN path_in_schema = 'y' THEN coalesce(stats_max_value, stats_max)::DOUBLE END) AS y1\n FROM parquet_metadata([${urls}])\n WHERE path_in_schema IN ('x', 'y') GROUP BY 1 ORDER BY 1`,\n );\n const tile = numbers(stats, \"tile\");\n const [x0, x1, y0, y1] = [\"x0\", \"x1\", \"y0\", \"y1\"].map((f) => numbers(stats, f));\n return tile.map((t, i) => ({\n tile: t as number,\n x0: x0?.[i] as number,\n x1: x1?.[i] as number,\n y0: y0?.[i] as number,\n y1: y1?.[i] as number,\n }));\n })();\n return loading;\n }\n\n /** The tiles a rectangle touches. Pure — this is the whole of the addressing, and it makes no call. */\n function intersecting(all: TileBox[], view: Viewport): number[] {\n return all\n .filter((b) => b.x1 >= view.xMin && b.x0 <= view.xMax && b.y1 >= view.yMin && b.y0 <= view.yMax)\n .map((b) => b.tile);\n }\n\n /**\n * Everything the corpus fixes, and nothing it does not.\n *\n * `dense_id`, `subject`, `x` and `y` are facts of the format — a corpus has them under those\n * names or it is not one. What colours and what sizes are channels, so they arrive with the\n * request and are filled in per slice.\n */\n const fixed = {\n id: \"dense_id\",\n subject: subjects ? \"subject\" : undefined,\n x: \"x\",\n y: \"y\",\n source: \"src_dense\",\n target: \"dst_dense\",\n } as const;\n\n /**\n * The relation half, registered once at open.\n *\n * A view rather than a table: `CREATE TABLE AS` would pull the corpus into memory, which is the\n * working set the whole bounded path exists to refuse. A view leaves the bytes where they are and\n * lets each query fetch the ranges it needs.\n *\n * Over **every** tile, deliberately — this is the surface that answers *what does it mean*, and a\n * count, a histogram or a crossfilter clause is a question about the corpus rather than about the\n * window. The addressed reading is `slice`, beside it, and the two are different access to the\n * same bytes rather than two versions of one.\n */\n const nodesView = `corpus_${type}`;\n const edgesView = edgePrefix ? `corpus_${type}_edges` : undefined;\n const allTiles = async () => {\n const boxes = await load();\n return boxes.map((b) => tileUrl(b.tile)).join(\", \");\n };\n await coordinator.exec(\n `CREATE OR REPLACE VIEW ${nodesView} AS SELECT * FROM read_parquet([${await allTiles()}])`,\n );\n if (edgesView) {\n await coordinator.exec(\n `CREATE OR REPLACE VIEW ${edgesView} AS\n SELECT * FROM read_parquet('${dest}/${edgePrefix}by_source.parquet')`,\n );\n }\n\n const source: DuckSource = {\n ...watching.api,\n\n publish(vertices) {\n publishSelection(reads, filterBy, fixed.id, vertices);\n },\n\n async total() {\n await load();\n return total ?? 0;\n },\n\n async extent() {\n const all = await load();\n return {\n xMin: Math.min(...all.map((b) => b.x0)),\n yMin: Math.min(...all.map((b) => b.y0)),\n xMax: Math.max(...all.map((b) => b.x1)),\n yMax: Math.max(...all.map((b) => b.y1)),\n };\n },\n\n // Regions only, said the same way `duckBoundedSource` says it: no `explore`. A neighbourhood\n // needs adjacency this source does not index, and fossil's `expand` is what answers it —\n // `ExploringSource` is the shape waiting for whoever writes that one.\n async slice(request: SliceRequest): Promise<Slice> {\n const { fill, limit, perPixel, pinned, r, signal, view } = request;\n const columns: Columns = { ...fixed, category: fill, size: r };\n const all = await load();\n /**\n * The tiles the rectangle touches — **and the far view is not a special case of this.**\n *\n * It used to be: past a zoom threshold the selection was replaced by every tile, because the\n * aggregate branch was going to read the whole relation anyway. A window that covers the\n * extent already intersects every box, so the branch was arithmetic restating itself, and it\n * is the reason `Viewport` carried a `zoom` at all.\n */\n const selected = intersecting(all, view);\n // Nothing selected is a legitimate answer — the camera is over empty space — and asking\n // `read_parquet([])` is a syntax error rather than an empty result. Checked before the tiles\n // are fetched, so an empty window costs no bytes at all.\n if (selected.length === 0) {\n return {\n n: 0,\n marks: 0,\n vertices: new BigUint64Array(0),\n positions: new Float32Array(0),\n links: new Float32Array(0),\n categories: new Uint16Array(0),\n };\n }\n\n // Both halves at once: the vertex tiles and the edge tiles a window touches are independent\n // reads, and the window is not drawable until both have landed.\n const [vertexTiles, edgeTiles] = await Promise.all([\n readable(\"vertex\", selected),\n readable(\"edge\", edgePrefix ? selected : []),\n ]);\n /**\n * The camera moved while the tiles were arriving, so this question is already the wrong one.\n *\n * Checked here rather than left to the caller because of what comes next: the reads hold one\n * standing question each, so a request that resumes after the loop moved on would *supersede*\n * the newer one and reject it — the stale question winning the race against the live one.\n */\n if (signal?.aborted) throw SUPERSEDED;\n const nodes = `read_parquet([${vertexTiles.join(\", \")}])`;\n const relation = `read_parquet([${edgeTiles.join(\", \")}])`;\n\n // `nodes` twice, and the repetition is the statement: the relation this window reads *is* the\n // bytes the reader is holding, so the tiles that answer \"what is in the rectangle\" also answer\n // \"where is the far end of an edge that leaves it\".\n return watching.run(\n region(nodes, relation, columns, view, limit, 0, pinned, perPixel, nodes),\n );\n },\n };\n\n return { source, nodes: nodesView, edges: edgesView };\n}\n"],"names":[],"mappings":";;;;;AA4IA;AACE;AAAO;AACsC;AACD;AACX;AAEnC;AAUA;AACE;AACA;AACE;AACA;AAAmB;AACZ;AAEX;AAUA;AACE;AAEA;AACA;AACF;AAGA;AACE;AAGF;AAEO;AACL;AAOoD;AAC3B;AACN;AACI;AACA;AACU;AACA;AAOjC;AAAO;AACO;AAGV;AAAsD;AACxD;AAGE;AACA;AAAwC;AAC1C;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AASE;AAAmB;AAC+B;AACA;AACnC;AAGf;AAAsE;AACxE;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAWE;AAIA;AAAgB;AACiE;AAAA;AAEnF;AAEJ;AAuBA;AAOE;AAN2C;AACpB;AACA;AACA;AACA;AAKvB;AACF;AAeA;AAyBA;AAOE;AAYA;AAAO;AAGL;AAAA;AAEY;AACC;AAAA;AAIb;AACiB;AAAA;AAAA;AAGY;AAChB;AAEjB;AAYA;AACE;AACA;AACA;AACF;AA8BA;AAQE;AAEA;AAAO;AACsE;AAAA;AAAA;AAAA;AAAA;AAK/D;AACgB;AACA;AAAA;AAAA;AAG8C;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAQ9E;AAsBA;AAWE;AAwBA;AAAO;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAS4D;AAGxD;AAAA;AAMN;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAYyE;AAAA;AAAA;AAAA;AAKlB;AAAA;AAE1C;AACc;AAEjB;AAET;AAC0B;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AASgB;AAC4C;AAAA;AAG9F;AAUA;AACE;AAGA;AAKE;AACA;AAEA;AACyC;AAG3C;AAAO;AACA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AASD;AAIE;AAKW;AACb;AACF;AAAA;AAGA;AAIA;AAA0C;AACZ;AACF;AAE5B;AAAkC;AACpC;AAEJ;AAUA;AAME;AACS;AAC+D;AACtD;AAC8B;AAC7C;AAEL;AAWA;AAWE;AAGA;AAKA;AACA;AACA;AACA;AAQA;AACA;AACA;AACA;AACA;AACA;AAKA;AAEA;AACA;AAMA;AAGA;AAEO;AACL;AACA;AACA;AACsC;AACJ;AAClC;AACA;AAEJ;AAGA;;AACE;AACA;AACA;AACA;AAEF;AAuEA;AACE;AAAwB;AACxB;AACF;AAEA;AACE;AAAwB;AAExB;AACA;AACA;AACE;AACA;AAA8B;AAEhC;AACF;AAEA;AACE;AACA;AACA;AACF;AAqCA;AACE;AAeA;AACE;AAAU;AAC4D;AAIxE;AAGA;AACE;AAOF;AAeA;AAGA;;AAgCE;AAIA;AACA;AACE;AAAuB;AAEvB;AAAO;AACT;AAKF;AAWA;AAyBA;AACE;AAEA;AACA;AAA4B;AAExB;AACA;AACA;AAKA;AACA;AAGA;AACA;AACA;AAGO;AACR;AAEH;AACE;AAEA;AACA;AAEwB;AAE1B;AAA0D;AAG5D;AACE;AAeE;AAEA;AACA;AAEA;AACE;AAGF;AACE;AACA;AACY;AAEd;AAGmB;AAC4D;AAE/E;AACA;AASoB;AAClB;AAAA;AAAA;AAAA;AAAA;AAK8B;AAAA;AAKhC;AAA2B;AACnB;AACG;AACA;AACA;AACA;AACT;AAEG;AAIT;AACE;AAEoB;AAUtB;AAAc;AACR;AAC4B;AAC7B;AACA;AACK;AACA;AAqBV;AAAkB;AACsE;AAGpE;AACmB;AACgB;AAI5B;AACb;AAGV;AAAoD;AACtD;AAGE;AACgB;AAClB;AAGE;AACA;AAAO;AACiC;AACA;AACA;AACA;AAAA;AAE1C;AAAA;AAAA;AAAA;AAME;AAeA;AACE;AAAO;AACF;AACI;AACuB;AACD;AACJ;AACI;AAMjC;AAAmD;AACtB;AACgB;AAS7C;AACA;AAMA;AAAgB;AAC0D;AAAA;AAE5E;AAIJ;;;;;"}
|
|
1
|
+
{"version":3,"file":"duck-source.js","sources":["../src/duck-source.ts"],"sourcesContent":["\"use client\";\n\nimport {\n PAYLOAD_ADDRESS,\n PAYLOAD_COORDINATES,\n PAYLOAD_IDENTITY,\n open as openFossilCorpus,\n} from \"@fossil-lang/corpus\";\nimport type { OpenOptions } from \"@fossil-lang/corpus\";\nimport { clausePoints, column, fillColumn, numbers } from \"@kanzo-tech/mosaic\";\nimport type { Coordinator, FilterExpr, Selection } from \"@kanzo-tech/mosaic\";\nimport type { BoundedSource, Slice, SliceRequest, Viewport } from \"./bounded\";\nimport { SliceRead } from \"./slice-client\";\nimport { denseOf, typeOf, vertexId, type VertexId } from \"./resident\";\n\n/**\n * The source: a corpus fossil wrote, read through DuckDB.\n *\n * On a subpath because Mosaic and fossil's reader are optional peers and this is the half that needs\n * them. The root barrel ships no source at all, which is the whole of what that split now means: a\n * host that installs neither gets the rendering surface and draws nothing.\n *\n * **There used to be two here and the second was neutral about storage.** `duckBoundedSource` took\n * two ordinary relations and the column names that made sense of them, and that neutrality earned\n * its keep once — it let bounded be measured against unbounded before anything committed to a layout\n * on disk. What it cost afterwards is the reason it is gone: a corpus already declares those names\n * in its manifest, so the second source was this side writing down what the other side owns, and the\n * two answered the same question in two dialects of the same SQL.\n *\n * **Every query in this file goes through a `SliceRead`, and there is no other path.** `onceQuery`\n * was the other one — a throwaway client per query, on this subpath, re-exported for the two\n * showcases that also read a relation directly. It is gone: a read the page's filters cannot reach\n * is a picture that disagrees with the page, and `slice-client.ts` carries what that cost.\n */\n\n/**\n * A DuckDB-backed source, and the one thing it can do that the render contract knows nothing about.\n *\n * `BoundedSource` says what a renderer needs: answer a bounded question. Publishing a selection is\n * the other direction of the same seam and it is Mosaic's, not the renderer's — so it lives on the\n * concrete type rather than on the contract, beside `watch`, which is on the contract because the\n * query loop is what has to act on it.\n */\nexport interface DuckSource extends BoundedSource {\n /**\n * The reader's own selection, as a clause the rest of the page filters by. `null` retracts it.\n *\n * **The graph is exempt from its own clause, and that is the whole difference from the greyout it\n * replaces.** While the canvas *faded* excluded rows it could take its own clause too — the row was\n * still drawn and still selectable, and the fade was the brush. A canvas that now draws what\n * survives would answer a lasso by deleting everything the reader did not lasso, which is not a\n * selection, it is a filter nobody asked for.\n */\n publish(vertices: readonly VertexId[] | null): void;\n}\n\n/**\n * The type index every vertex this source returns wears — **zero, because there is one of them.**\n *\n * A `dense_id` numbers within one vertex type, so an identity is the pair and the second half has to\n * come from somewhere. It used to be an option: the general source took a `typeIndex` because only\n * the caller knew which relation it had handed over, and a defaulted `0` would have let a second\n * relation ship the first one's identities with nothing raised. `openCorpus` draws **one** vertex\n * type — `vertexType` picks which, and it is numbered zero either way — so the option had one legal\n * argument and it is written here instead, once, where the reason fits beside it.\n *\n * **What would reverse it:** a canvas drawing two vertex types at once, which is what a multi-type\n * corpus asks for. Then the number comes back — off the manifest's own type ordering rather than off\n * a caller, because by then the corpus is what knows.\n */\nconst VERTEX_TYPE = 0;\n\ninterface Columns {\n id: string;\n /** `undefined` when the host did not ask to be able to name a vertex. */\n subject: string | undefined;\n x: string;\n y: string;\n /**\n * The categorical column, when a channel named one — **and `undefined` is not a missing value.**\n *\n * It used to default to `community`, in both sources, which is this side writing what the corpus\n * owns: a relation that has no such column answered `Referenced column \"community\" not found`, and\n * one that has a differently named cluster column was silently coloured by the wrong thing. Neither\n * failure is the caller's, and both were invented here.\n *\n * Unbound, every point is one colour — which is Plot's own answer to a mark with no `fill` channel,\n * and an honest picture rather than a guess.\n */\n category: string | undefined;\n size: string | undefined;\n source: string;\n target: string;\n}\n\n/**\n * The three reads a source makes, and which of them the page can filter.\n *\n * `points` and `links` carry the crossfilter; `meta` deliberately does not. How big the corpus is,\n * where it sits and what its tile footers say are facts about the corpus rather than about the\n * page's current question — and a `total()` that shrank with the filters would make the view's own\n * \"20,000 of 1,000,000\" a fraction of itself, which is the one number a bounded renderer owes its\n * reader honestly.\n */\ninterface Reads {\n points: SliceRead;\n links: SliceRead;\n meta: SliceRead;\n}\n\nfunction openReads(coordinator: Coordinator, filterBy?: Selection): Reads {\n return {\n points: new SliceRead(coordinator, filterBy),\n links: new SliceRead(coordinator, filterBy),\n meta: new SliceRead(coordinator),\n };\n}\n\n/**\n * The metadata reads, queued behind each other.\n *\n * One client answers one question at a time — a second `ask` supersedes the first — and `total()`,\n * `extent()` and the tile probing are issued by different effects with no ordering between them. A\n * queue rather than a client each, because they *already* run one at a time: DuckDB-WASM answers\n * over one connection, measured, so three clients would buy three registrations and no concurrency.\n */\nfunction metaAsker(read: SliceRead): (sql: string) => Promise<unknown> {\n let queue: Promise<unknown> = Promise.resolve();\n return (sql) => {\n const next = queue.then(() => read.ask(() => sql));\n queue = next.catch(() => undefined);\n return next;\n };\n}\n\n/**\n * The page's predicate, as SQL text.\n *\n * The reads here are CTEs over window functions rather than builder queries — `row_number()` over the\n * visible set is what makes a slice's links speak in buffer positions — so the predicate has to be\n * interpolated rather than handed to `Query.where`. Mosaic's expression nodes stringify to the same\n * SQL the builder would emit, which is what makes that safe rather than a re-implementation.\n */\nfunction predicateSql(filter: FilterExpr | undefined): string {\n if (filter == null) return \"\";\n const list = Array.isArray(filter) ? filter : [filter];\n const clauses = list.filter((node) => node != null).map((node) => String(node));\n return clauses.length > 0 ? clauses.map((c) => `(${c})`).join(\" AND \") : \"\";\n}\n\n/** Two predicates, conjoined, where an absent one contributes nothing rather than `AND TRUE`. */\nfunction both(left: string, right: string): string {\n if (!left) return right || \"TRUE\";\n if (!right) return left;\n return `(${left}) AND (${right})`;\n}\n\n/**\n * The re-indexing happens in SQL, and that is the whole trick.\n *\n * `row_number() - 1` over the visible set gives every returned point a position in the arrays about\n * to be built, so the edge query can join to it twice and hand back links that already speak in\n * those positions. No id→index map is constructed in JavaScript — which is the 148 ms `load()` spent\n * at 200,000 nodes, gone by construction rather than by optimisation.\n *\n * The `LIMIT` sits inside the CTE, so the numbering is over what survives it. Numbering first and\n * limiting after would hand out indices into an array that was never built.\n */\n/**\n * A rectangle as a SQL predicate, and an **unbounded** rectangle as no predicate at all.\n *\n * `shouldSlice` answers `false` for a graph that fits, and the loop then asks for everything — a\n * viewport whose edges are `±Infinity`. Interpolated, that reads `x BETWEEN -Infinity AND Infinity`,\n * and SQL has no infinity literal: DuckDB parses `Infinity` as a **column name** and fails with\n * `Referenced column \"Infinity\" not found`. So an open edge contributes no clause, and a rectangle\n * open on every side is `TRUE` — which is also the right plan, because a query that wants every row\n * has nothing to prune.\n */\nfunction bboxSql(c: Columns, view: Viewport): string {\n const bounds: [string, number, string][] = [\n [c.x, view.xMin, \">=\"],\n [c.x, view.xMax, \"<=\"],\n [c.y, view.yMin, \">=\"],\n [c.y, view.yMax, \"<=\"],\n ];\n const clauses = bounds\n .filter(([, value]) => Number.isFinite(value))\n .map(([column, value, op]) => `${column} ${op} ${value}`);\n return clauses.length > 0 ? clauses.join(\" AND \") : \"TRUE\";\n}\n\n/**\n * How far apart the sampled ids are — one every `ceil(matched / limit)`.\n *\n * **In SQL rather than in JavaScript because the number it divides is only known inside the query.**\n * `matched` is a window aggregate over the rows the `WHERE` kept, so a caller wanting to compute\n * this outside would have to count first and slice second — two round trips down a connection that\n * answers one at a time, which is the shape `BENCHMARKS.md` records as a hung tab rather than a slow\n * one. As a column reference it costs the pass that was being made anyway.\n *\n * `greatest(1, …)` because an empty window makes the divisor zero, and a modulo by zero is an error\n * rather than an empty answer. At `matched <= limit` it is exactly 1 and `id % 1 = 0` keeps every\n * row: a window that fits is not sampled, it is returned.\n */\nconst strideSql = (limit: number) => `greatest(1, CAST(ceil(matched / ${limit}.0) AS BIGINT))`;\n\n/**\n * The visible set: what the rectangle matched, and the sample of it that gets drawn.\n *\n * **Two CTEs, and the second one is the whole of the far view.** `pool` is every row the predicate\n * kept, carrying `count(*) OVER ()` — a window function is evaluated over everything the `WHERE`\n * kept and `LIMIT` applies after it, so that column is the number that *matched* rather than the\n * number returned. It is what deleted the third query: `SELECT count(*) FROM … WHERE <the same\n * predicate>` was a second scan to learn a number the first scan already had to compute.\n *\n * `vis` then keeps one row in `stride`, **striding over the id rather than taking the front of the\n * ordering**, and that is the difference between a picture of the window and a picture of one corner\n * of it. A corpus numbers `dense_id` along the Morton curve, so every `s`-th id is a spatially\n * stratified sample; the same `LIMIT` with no stride returns a contiguous run of the curve, which is\n * a sub-region. Measured against the truth at screen resolution, L1@8px over blocks of eight pixels:\n * a stride sample of 20,000 scores 0.167 / 0.240 / 0.269 at 200k / 1M / 5M against a uniform null of\n * 1.044 / 0.829 / 0.731 — see `/docs/design/graph`.\n *\n * @param matched Whether to project the pre-sample count out to the caller.\n *\n * It rides on the points read only. The links read builds the same CTEs to join against and never\n * looks at the column — but it does compute it, because the stride is a function of it and both\n * reads have to select the *same* rows or a slice would draw edges to vertices it did not return.\n */\nfunction visibleCte(\n nodes: string,\n c: Columns,\n where: string,\n limit: number,\n matched = true,\n): string {\n const size = c.size ? `, ${c.size} AS size` : \"\";\n // Selected in the CTE rather than joined back afterwards: the numbering is over what survives the\n // LIMIT, and a second pass keyed on `local` would be a second scan to fetch a column the first one\n // was already standing on.\n const subject = c.subject ? `, ${c.subject} AS subject` : \"\";\n // No categorical binding, no ranking: a literal zero is the ordinal every point wears, and the\n // scale hands that one colour. Ranking a column nobody named is how a default column gets invented.\n // Ranked over the sample rather than over the window, so the ordinals are contiguous across what\n // is actually drawn — which is what the colour scale is handed.\n const category = c.category\n ? `(dense_rank() OVER (ORDER BY cat) - 1)::INTEGER AS category`\n : \"0::INTEGER AS category\";\n return `WITH pool AS (\n SELECT ${c.id} AS id, ${c.x} AS x, ${c.y} AS y${size}${subject}${\n c.category ? `, ${c.category} AS cat` : \"\"\n },\n count(*) OVER () AS matched\n FROM ${nodes}\n WHERE ${where}\n ), vis AS (\n SELECT id, x, y${c.size ? \", size\" : \"\"}${c.subject ? \", subject\" : \"\"}${\n matched ? \", matched\" : \"\"\n },\n ${category},\n (row_number() OVER (ORDER BY id) - 1)::INTEGER AS local\n FROM pool\n WHERE id % ${strideSql(limit)} = 0\n LIMIT ${limit}\n )`;\n}\n\n/**\n * The shortest edge worth a row, as a predicate over the two endpoints — **squared, and on purpose.**\n *\n * A distance is compared against a threshold, and squaring both sides removes a `sqrt` per row from\n * a predicate evaluated once per candidate edge. It changes no answer: both sides are non-negative.\n *\n * `undefined` when the caller said nothing about resolution, and then there is no predicate at all\n * rather than a permissive one — a request with no canvas behind it (`EVERYTHING`) has no pixels to\n * measure three of.\n */\nfunction longEnough(\n a: string,\n b: string,\n perPixel: number | undefined,\n minLinkPixels: number,\n): string {\n if (perPixel === undefined || !Number.isFinite(perPixel) || perPixel <= 0) return \"\";\n const floor = minLinkPixels * perPixel;\n return `(${a}.x - ${b}.x) * (${a}.x - ${b}.x) + (${a}.y - ${b}.y) * (${a}.y - ${b}.y) >= ${floor * floor}`;\n}\n\n/**\n * The far ends, and the edges that reach them — **out of bytes the reader already fetched.**\n *\n * An edge with one end outside the rectangle is dropped today, and that loses 19.31% / 31.94% /\n * 28.92% of the edges incident to a window at 200k / 1M / 5M; 7,930 of 20,000 vertices carry at least\n * one at five million. What was missing was never the edge row — a window reads the `by_source` tiles\n * of every vertex it draws, so the row is in hand — it was **a position to draw the far end at**.\n *\n * **And a tile answers that for free.** A tile is 4,096 rows of a Morton-ordered relation and its\n * bounding box is far wider than the rows the rectangle keeps, so the vertices just outside the\n * window are usually in a tile the window already fetched. Measured over five windows of\n * `docs/public/bench/1000000`, 2026-08-19: relaxing the join from *inside the rectangle* to *inside\n * the tiles that were read* takes the drawn edges of five windows from 616,885 to 781,562 of 906,337\n * incident — **56.9% of everything the reader was losing, at no request, no query and no byte.**\n *\n * **The tile boundary is also a distance filter, and that is what settles the drawing.** Over every\n * far end a window loses, the median sits 1.64 semi-widths out and the worst 47.1 — which is why a\n * stub clipped to the viewport is a lie: nothing distinguishes 1.1 from 47. The far ends a held tile\n * can answer are the near ones: median 1.11–1.85 semi-widths, 90th percentile 1.29–2.60, worst\n * **6.74**, against 7.09–16.91 for the full reachable set. So the honest picture and the free one are\n * the same picture, and there is no trade to make: the anchors are drawn where the vertices are, and\n * the long tail stays undrawn because its bytes are not here — not because we decided.\n *\n * @param held The relation whose rows the reader is holding — the tiles a corpus fetched for this\n * window, which is the same relation the marks were read from. **Only an addressed source can name\n * one**, and that is why there is no longer a branch here for a source that cannot: over an\n * ordinary relation `held` would be the whole node table, and the join would scan the corpus twice\n * per camera move — the unbounded pattern wearing a bounded interface.\n */\nfunction anchorCte(\n held: string,\n edges: string,\n c: Columns,\n spatial: string,\n filter: string,\n perPixel: number | undefined,\n minLinkPixels: number,\n): string {\n const outside = both(`NOT (${spatial})`, filter);\n const long = longEnough(\"a\", \"b\", perPixel, minLinkPixels);\n return `, out AS (\n SELECT ${c.id} AS id, ${c.x} AS x, ${c.y} AS y FROM ${held} WHERE ${outside}\n ), reach AS (\n SELECT id, x, y FROM vis UNION ALL SELECT id, x, y FROM out\n ), span AS (\n SELECT sv.local AS src, tv.local AS dst, a.id AS src_id, b.id AS dst_id\n FROM ${edges} e\n JOIN reach a ON e.${c.source} = a.id\n JOIN reach b ON e.${c.target} = b.id\n LEFT JOIN vis sv ON sv.id = a.id\n LEFT JOIN vis tv ON tv.id = b.id\n WHERE (sv.id IS NOT NULL OR tv.id IS NOT NULL)${long ? ` AND ${long}` : \"\"}\n ), anchor AS (\n SELECT o.id, o.x, o.y,\n ((SELECT count(*) FROM vis) + row_number() OVER (ORDER BY o.id) - 1)::INTEGER AS local\n FROM out o\n WHERE o.id IN (SELECT src_id FROM span WHERE src IS NULL\n UNION SELECT dst_id FROM span WHERE dst IS NULL)\n )`;\n}\n\n/**\n * A question, in the two halves it is asked in and the one place they are put back together.\n *\n * The reads run through the coordinator, so the *same* pair of SQL builders serves both directions:\n * the camera pulling an answer, and the page's filters pushing one. `assemble` is therefore a\n * function of the two results and of nothing else — no closure over which of the two paths asked,\n * because a slice that came back because somebody brushed a histogram is the same slice.\n */\ninterface Plan {\n points: (filter: FilterExpr) => string;\n links: (filter: FilterExpr) => string;\n assemble: (points: unknown, links: unknown) => Slice;\n}\n\n/**\n * The one plan there is: a rectangle, its sample, and the edges both of whose ends survived it.\n *\n * It was called `detail` because it was one of two, and the other one — a `GROUP BY` over a\n * categorical column, one super-node per group — is gone. There is no mode to be in.\n */\nfunction region(\n nodes: string,\n edges: string,\n c: Columns,\n view: Viewport,\n limit: number,\n pinned: VertexId[] | undefined,\n perPixel: number | undefined,\n minLinkPixels: number,\n): Plan {\n const bbox = bboxSql(c, view);\n // A dragged node is drawn where the reader dropped it and indexed where it always was, so the\n // rectangle cannot find it. Riding along in the predicate is what keeps it on screen — and it\n // stays a predicate rather than a second query so the numbering still covers everything returned.\n //\n // Only this relation's own vertices: a pinned set spans the whole canvas, and asking one node\n // table for another type's dense ids returns the wrong rows rather than none. `VERTEX_TYPE` is the\n // whole of what \"this relation\" means here — see the constant.\n const mine = (pinned ?? []).filter((v) => typeOf(v) === VERTEX_TYPE).map(denseOf);\n const pins = mine.length > 0 ? ` OR ${c.id} IN (${mine.join(\",\")})` : \"\";\n const spatial = `(${bbox})${pins}`;\n /**\n * The page's predicate outside the pin, not inside it.\n *\n * A pin says *where to look*; the filters say *what exists*. Written the other way round —\n * `bbox AND filter OR pinned` — a pinned node would survive a filter that excludes it, and the\n * canvas would draw a vertex the rest of the page has agreed is not there.\n */\n const where = (filter: FilterExpr) => both(spatial, predicateSql(filter));\n const size = c.size ? \", size\" : \"\";\n const subject = c.subject ? \", subject\" : \"\";\n // The relation the marks were read from *is* what the reader is holding, so the anchor CTE is\n // unconditional: there is no source left that fetches a window without also fetching the tiles\n // around it.\n const anchors = (filter: FilterExpr) =>\n anchorCte(nodes, edges, c, spatial, predicateSql(filter), perPixel, minLinkPixels);\n\n return {\n /**\n * The marks, then the anchors, in one answer — because they are one buffer.\n *\n * `ORDER BY local` is load-bearing rather than tidy: `local` runs `0..marks-1` over the sample\n * and continues past it over the anchors, so ordering by it puts every mark before every anchor\n * and makes `marks` a prefix length. `matched` is then still read off row zero.\n */\n points: (filter) =>\n `${visibleCte(nodes, c, where(filter), limit)}${anchors(filter)}\n SELECT local, id, x, y, category, matched, TRUE AS mark${size}${subject} FROM vis\n UNION ALL\n SELECT local, id, x, y, 0::INTEGER, NULL::BIGINT, FALSE AS mark${\n c.size ? \", CAST(NULL AS DOUBLE)\" : \"\"\n }${c.subject ? \", CAST(NULL AS VARCHAR)\" : \"\"} FROM anchor\n ORDER BY local`,\n /**\n * One end drawn and both ends positioned.\n *\n * There was a second form here — both ends drawn, edges leaving the window dropped — for a source\n * that fetched no bytes beyond the rectangle and therefore had no position to put a far end at.\n * It went with that source, and `longEnough` with it — the length predicate lives in `anchorCte`\n * now, on the same span, which is where it has to sit once the join runs over the held rows\n * rather than over the visible ones.\n */\n links: (filter) =>\n `${visibleCte(nodes, c, where(filter), limit, false)}${anchors(filter)}\n SELECT coalesce(sp.src, sa.local) AS src, coalesce(sp.dst, da.local) AS dst\n FROM span sp\n LEFT JOIN anchor sa ON sa.id = sp.src_id\n LEFT JOIN anchor da ON da.id = sp.dst_id`,\n assemble: (points, links) => ({\n /**\n * How many matched, separately from how many came back.\n *\n * Without it the view cannot tell a reader \"there is more here than I am showing you\", and a\n * truncated slice looks exactly like a complete one — which is the failure this whole branch\n * has been about. Read off the first row rather than asked for: `matched` is constant down the\n * column, and an empty answer has no row and no matches, which agree.\n */\n n: Number(numbers(points, \"matched\")[0] ?? 0),\n ...arrays(points, links, c.size ? \"size\" : undefined, c.subject !== undefined),\n }),\n };\n}\n\n/**\n * The half of a source that runs a [`Plan`], and the half that answers when nobody asked.\n *\n * Both sources need exactly this and neither should own a second copy of it — which is why it is a\n * function over the reads rather than two blocks of the same bookkeeping. What it holds is the\n * *standing* plan: the last question the camera put, kept so an answer arriving because the page\n * filtered something can be put back together the same way.\n */\nfunction watcher(reads: Reads) {\n let standing: Plan | null = null;\n let listener: ((slice: Slice) => void) | null = null;\n /** The half-answers of a push, waiting for their sibling. */\n const landed = new Map<\"points\" | \"links\", unknown>();\n\n const arrived = (half: \"points\" | \"links\") => (data: unknown) => {\n if (!standing || !listener) return;\n landed.set(half, data);\n if (landed.size < 2) return;\n const points = landed.get(\"points\");\n const links = landed.get(\"links\");\n landed.clear();\n listener(standing.assemble(points, links));\n };\n\n return {\n api: {\n /**\n * Say when the answer changes for a reason the camera cannot see.\n *\n * The reason it hands over a whole `Slice` rather than a nudge to ask again: the coordinator\n * has *already* re-run both reads with the new predicate by the time we hear about it. Asking\n * again would run the same two queries a second time to learn what is in hand.\n */\n watch(answered: (slice: Slice) => void): () => void {\n listener = answered;\n reads.points.onAnswer = arrived(\"points\");\n reads.links.onAnswer = arrived(\"links\");\n return () => {\n listener = null;\n landed.clear();\n standing = null;\n reads.points.release();\n reads.links.release();\n reads.meta.release();\n };\n },\n },\n async run(plan: Plan): Promise<Slice> {\n standing = plan;\n // Cleared because these two are halves of the *previous* question: keeping one would pair a\n // stale rectangle's points with the new rectangle's links the next time the page filters.\n landed.clear();\n const [points, links] = await Promise.all([\n reads.points.ask(plan.points),\n reads.links.ask(plan.links),\n ]);\n return plan.assemble(points, links);\n },\n };\n}\n\n/**\n * The reader's own gesture, as a clause — and the graph exempted from it.\n *\n * `clausePoints` defaults `clients` to the clause's source when that source is itself a client, which\n * is the exemption a crossfilter is built on. Here the source is a plain object and there are two\n * clients to exempt, so the set is written out: a lasso must filter the page's charts and leave the\n * canvas showing what the reader lassoed *in context*, rather than deleting everything else.\n */\nfunction publishSelection(\n reads: Reads,\n filterBy: Selection | undefined,\n idField: string,\n vertices: readonly VertexId[] | null,\n): void {\n if (!filterBy) return;\n filterBy.update(\n clausePoints([idField], vertices?.map((vertex) => [denseOf(vertex)]), {\n source: reads.points,\n clients: new Set([reads.points, reads.links]),\n }),\n );\n}\n\n/**\n * Arrow columns to the parallel typed arrays the renderer takes — sized once, filled in place.\n *\n * `fillColumn` rather than `numbers`: Arrow already hands back a typed buffer, and the obvious route\n * through `Array.from(...).map(Number)` allocates two full-length boxed arrays on the way to a third\n * that was the actual destination. Three copies to move nothing. Here the point buffers are sized\n * against the query's own `LIMIT` and each column is written straight into its stride, so `x` and\n * `y` interleave with no seam between them and no intermediate at all.\n */\nfunction arrays(\n points: unknown,\n links: unknown,\n sizeField?: string,\n withSubjects = false,\n): Omit<Slice, \"n\"> {\n // Sized from the answer rather than from the request's `limit`, which it used to be: an answer\n // carries the window's marks *and* the anchors the edges leaving it end at, so `limit` is no longer\n // an upper bound on the rows. Under-sizing here would drop the anchors silently and leave every\n // link that pointed at one indexing past the buffer.\n const rows = countOf(points, \"x\");\n const positions = new Float32Array(rows * 2);\n const n = fillColumn(points, \"x\", positions, 0, 2);\n fillColumn(points, \"y\", positions, 1, 2);\n\n // Where the marks stop. `mark` is `TRUE` down the sample and `FALSE` down the anchors, and the\n // query orders by `local`, so this is a prefix length rather than a count — which is what lets\n // `residentOf` and `buffers` treat \"is this a mark\" as an index comparison.\n const marked = new Int32Array(rows);\n fillColumn(points, \"mark\", marked);\n let marks = 0;\n while (marks < n && marked[marks] !== 0) marks++;\n\n // The dense ids land in a scratch and are widened into identities as they are copied across.\n //\n // In place, into the destination, would be better and is not available: `fillColumn` writes\n // `Number(…)`, and a `BigUint64Array` element takes a `bigint` only — assigning a `number` to one\n // throws rather than coercing. That refusal is the same guarantee this whole change is for, so the\n // extra `n`-long buffer is the price of the boundary being enforced by the runtime and not by us.\n const dense = new Float64Array(n);\n fillColumn(points, \"id\", dense);\n const vertices = new BigUint64Array(n);\n for (let i = 0; i < n; i++) vertices[i] = vertexId(VERTEX_TYPE, dense[i] as number);\n const categories = new Uint16Array(n);\n fillColumn(points, \"category\", categories);\n\n // `column` rather than `fillColumn`: an IRI is a string, so there is no typed buffer to write\n // into and no interleaving to express. It is the one thing a slice carries that never reaches the\n // GPU, which is why asking for it is a decision rather than a default.\n const subjects = withSubjects ? (column(points, \"subject\") as string[]) : undefined;\n\n let sizes: Float32Array | undefined;\n if (sizeField) {\n sizes = new Float32Array(n);\n fillColumn(points, sizeField, sizes);\n }\n\n // The edge count is not bounded by the point limit, so it is asked for rather than assumed.\n const edgeCount = countOf(links, \"src\");\n const edges = new Float32Array(edgeCount * 2);\n const wrote = fillColumn(links, \"src\", edges, 0, 2);\n fillColumn(links, \"dst\", edges, 1, 2);\n\n return {\n marks,\n vertices,\n subjects,\n positions: positions.subarray(0, n * 2),\n links: edges.subarray(0, wrote * 2),\n categories,\n sizes,\n };\n}\n\n/** A row count for a result that does not advertise one, without materialising the rows. */\nfunction countOf(rows: unknown, field: string): number {\n const advertised = (rows as { numRows?: number } | null)?.numRows;\n if (typeof advertised === \"number\") return advertised;\n const child = (rows as { getChild?: (f: string) => { length: number } | null })?.getChild?.(field);\n if (child) return child.length;\n return Array.from(rows as Iterable<unknown>).length;\n}\n\n/**\n * A corpus that fossil wrote, read by address — **and the addressing is fossil's.**\n *\n * **The five things a call site used to know, and now does not.** Drawing a corpus meant deriving\n * the chunk URLs from a `chunk_size` copied by hand, knowing how a tile is named, knowing what the\n * edge directory is called, knowing GraphAr's column names, and knowing that a glob cannot work over\n * a plain HTTP origin because there is no listing. Five conventions and about forty lines, none of\n * it the business of something that wants to draw a graph. `/docs/design/graph` carries the\n * argument; the copied `chunk_size` carries the evidence, because it went stale and read\n * a fraction of a corpus in silence for as long as it did.\n *\n * The consumer knows one thing: **where the corpus is.**\n *\n * ```ts\n * const { source } = await openCorpus({ coordinator, dest: \"/bench/1000000\", wasmUrl });\n * ```\n *\n * **And this side no longer knows the conventions either, which is the change.** It used to remove\n * that defect for its callers by committing it one level down — a YAML line-scanner, a\n * `chunk{k}.parquet` spelling, a `by_source/tile{k}.parquet` spelling and a `HEAD`-probing search\n * for a tile count — and by the time those were deleted all four were **wrong**: fossil writes\n * `container: rowgroups`, one `tiles.parquet` whose row groups are the tiles, and `vertex_count`\n * had been in the manifest the whole time the search was probing for it. Fossil's `open` is\n * `fossil_graph::plan` compiled to wasm32: the arithmetic the native reader runs, not a second\n * implementation of it that agrees until it does not.\n *\n * **Addressed, not queried.** The manifests and the per-tile boxes are read once and kept; after\n * that a camera move is arithmetic over boxes and a list of URLs. There is deliberately no request\n * on the path between the camera moving and a URL being computable — the moment there is one, this\n * has become the `viewport` verb fossil deleted.\n *\n * **What it is not.** It takes no column names and no type index. Those come from the manifest or\n * they do not come. There used to be a second source here for the case this is not — an arbitrary\n * relation with `x`/`y` that nobody wrote as a corpus — and taking `idField` here would have been\n * that one with extra steps. It is gone, so the rule is simpler than the guard against it: a column\n * name reaching this door is a name the manifest should have carried.\n */\n\nexport interface OpenCorpusOptions {\n coordinator: Coordinator;\n /**\n * The crossfilter this graph draws inside — the same `Selection` the page's charts filter by.\n *\n * Given, the predicate rides in the slice query and the canvas draws what survives.\n */\n filterBy?: Selection;\n /** Where the corpus lives, without a trailing slash — the directory holding `graph.graph.yml`. */\n dest: string;\n /**\n * Where `fossil_graph_wasm_bg.wasm` is.\n *\n * **The one thing about fossil's reader a caller still has to say, and not ours to default.** The\n * addressing runs in WASM, so the module has to be up before a URL can be composed, and only the\n * caller knows how its bundler resolves an asset — `?url` under Vite, an asset import under Next,\n * a `Response` over the bytes in Node. Passed straight through, spelled as fossil spells it.\n * Omitted, the boot is left to whoever already did it: it is memoised for the session, so a host\n * on its second corpus need not say it again.\n */\n wasmUrl?: OpenOptions[\"wasmUrl\"];\n /**\n * Which vertex type to draw, when a corpus carries more than one.\n *\n * Defaults to the first the manifest names. A corpus of one type never passes it; a corpus of\n * several has to, because *which graph do you mean* is not a question a reader can answer.\n */\n vertexType?: string;\n /**\n * Read the identity column as well. **Off by default, and the same trade as everywhere else:**\n * `subject` costs about twice the drawing tile, so the drawing path carries addresses and a host\n * asks for names when something has to be *named* rather than painted.\n */\n subjects?: boolean;\n /**\n * How a manifest is read, when a plain `fetch` of its URL is not how this host reads one.\n *\n * **The default is `manifest` below, and it is the whole of what this file knows about reading a\n * corpus** — right for a corpus served off an origin the page can already read, and wrong for a\n * host whose blobs sit behind a signature. There the URL fossil composes is correct and\n * unreadable, and nothing else on these options carries a credential.\n *\n * Passed straight through to fossil's `open`, which is where the capability belongs: `open`\n * composes every address and lends the reader to each one, so a host that signs a URL signs the\n * index and the per-type manifests the index names **without knowing which files those are**.\n * That is the point of lending a reader rather than handing over bytes — and it is what keeps a\n * signing host on this door, because the alternative it otherwise reaches for is composing the\n * addresses itself, which is the convention-copying `openCorpus` exists to end.\n *\n * It reads manifests and nothing else. The payload is read by the coordinator's own connector,\n * which is the host's already.\n */\n readText?: OpenOptions[\"readText\"];\n}\n\n/** One tile's bounding box, from the footer. A tile with no `x`/`y` statistics is not in the list. */\ninterface TileBox {\n tile: number;\n x0: number;\n x1: number;\n y0: number;\n y1: number;\n}\n\n/**\n * One manifest, as text — **the whole of what this file still knows about reading a corpus.**\n *\n * It takes an absolute URL because that is what fossil's door hands it. `open(dest, { readText })`\n * composes every address itself: it reads the index, then the per-type manifests the index names,\n * and nothing else. Which files those are is no longer a question asked on this side — the\n * twelve-line scan of the index's `vertices:`/`edges:` lists that used to stand here was the third\n * copy of a sequence the door now publishes, and the index's own file name left this file with it.\n *\n * A file that is not there raises here and reaches the caller as a `CorpusManifestError` naming the\n * URL, which is a better error than any invented on this side.\n *\n * The default rather than the only one: `OpenCorpusOptions.readText` replaces it, and a host behind\n * signed URLs is the case that needs to.\n */\nasync function manifest(url: string): Promise<string> {\n const response = await fetch(url);\n if (!response.ok) throw new Error(`corpus: ${url} is not readable (${response.status})`);\n return response.text();\n}\n\n/**\n * An opened corpus: the half that **draws** and the half that **answers**.\n *\n * A host needs both over the same bytes and they are not the same access. The canvas reads tiles by\n * address — a handful of files per camera move, chosen from the footer's boxes, with no query. A\n * chart, a crossfilter clause or a verb reads the *relation*: every row, by column name, in SQL.\n * Hiding the URLs behind `source` is right for the first and leaves the second with nothing to\n * query, so opening a corpus registers views for it.\n *\n * This is the other side's own shape. `fossil-mcp` describes itself as opening a dataset,\n * *registering views over the Parquet the corpus already holds*, and dispatching a verb — the same\n * two halves, named the same way, one call apart.\n */\nexport interface OpenedCorpus {\n /**\n * For the canvas: `<GraphCanvas source={…}>`. Reads tiles, never the whole relation.\n *\n * A `DuckSource`, and `CorpusSource` is gone with the reason it existed: `extent()` was declared\n * there because a source over an unlaid-out relation has no answer to it, and *optional on the\n * base contract* says that better than a second interface — a relation with `x`/`y` has an extent\n * too, and it was the one host that could not frame its opening view.\n */\n source: DuckSource;\n /**\n * The vertex relation, registered and ready to query by name.\n *\n * Every column the manifest declares, including the corpus' own properties — so a clause a chart\n * publishes over `kind` or `region` lands here with no translation, which is what makes one\n * crossfilter serve the canvas and the charts.\n */\n nodes: string;\n /** The source-ordered edge relation, or `undefined` when the corpus declares no edge for this type. */\n edges: string | undefined;\n}\n\nexport async function openCorpus(options: OpenCorpusOptions): Promise<OpenedCorpus> {\n const { coordinator, dest, filterBy, readText = manifest, subjects = false, vertexType, wasmUrl } = options;\n\n const reads = openReads(coordinator, filterBy);\n const meta = metaAsker(reads.meta);\n const watching = watcher(reads);\n\n /**\n * The addressing — **one `await`, one lent capability, and no arithmetic of ours.**\n *\n * Fossil's `open` takes where the corpus is and a way to read text, and hands back every URL the\n * corpus can produce, whichever container it declares. It needs no engine on this rung: lent a\n * reader, it fetches the index and the per-type manifests the index names, and nothing more.\n * What used to stand here — a line-scanning YAML reader, a `chunk{k}` spelling, an\n * edge-directory spelling and a `HEAD`-probing search for a tile count — was four conventions\n * fossil owns, written down on this side, and stale in all four by the time they were deleted.\n * A fifth went the same way once the door published the sequence rather than the file name: the\n * scan that worked out which manifests to ask for.\n *\n * **`readText` and not `manifestFiles`**, which is the other engine-free rung: holding the bytes\n * is what that one is for, and this never held them for its own sake — it fetched them only to\n * hand them over. It is the host's reader where one was given, and `fetch` where none was; either\n * way the addresses it is lent to are fossil's, which is the half that does not move.\n */\n const addressing = await openFossilCorpus(dest, { readText, wasmUrl });\n\n const type = addressing.vertexType(vertexType);\n /**\n * The relation whose SOURCE is this type — the CSR orientation, which is the drawing read.\n *\n * `by_target` is not asked for and its absence is reported rather than hidden: a window's\n * drawable edges all have their source on screen, so the source-aligned tiles are complete for\n * drawing and incomplete for incidence. `tilesFor` says which below, in `gaps`.\n */\n const edge = addressing.incident(type.type).find((e) => e.srcType === type.type);\n const adjacency = edge?.adjacency(\"src\") ?? null;\n\n /**\n * The boxes, and the one query this source makes that is not a slice.\n *\n * Read on first use rather than in the factory: a host that constructs a source and never draws\n * should not pay for it, and the cost is a footer read over the payload. Kept forever after —\n * tiles are precomputed and their boxes cannot move without the corpus being rewritten.\n *\n * **Held as the promise rather than as the answer**, which the move to one shared metadata client\n * forced and which was a latent defect before it: `total()` and `extent()` are called by different\n * effects with nothing ordering them, so two loads used to run concurrently and probe the whole\n * tile range twice.\n */\n let loading: Promise<TileBox[]> | null = null;\n\n /**\n * The tiles a window needs, held as bytes so that **panning back is free**.\n *\n * This is the one item on `BENCHMARKS.md`'s fix list that no amount of query tuning substitutes\n * for, and the measurement that puts it there is blunt: two of six drag steps at ten million\n * transfer zero new bytes and still cost 247 requests each, because every visit re-reads the same\n * footers and column chunks over HTTP. A tile fetched once and registered as a file is read from\n * memory forever after — no request, no range negotiation, no metadata round trip.\n *\n * **Keyed on the URL and not on the tile**, which is what makes it container-independent for\n * free: under `files` a tile is a file and the two keys agree, and under `rowgroups` every tile\n * names one `tiles.parquet`, which a tile-keyed cache would fetch once per tile.\n *\n * **The trade is honest and not free.** A registered file is the *whole* file, where DuckDB over\n * HTTP reads only the column chunks a query projects — the first visit costs more bytes and every\n * later one costs none. Which way that nets out depends on how the corpus is cut, which is the\n * corpus's decision and not ours.\n *\n * Discovered rather than required: a coordinator whose connector is not DuckDB-WASM has no\n * filesystem to register into, and reads by URL exactly as before. `@duckdb/duckdb-wasm` is\n * deliberately not a dependency of this package, so the capability is named structurally.\n */\n interface Registrar {\n registerFileBuffer(name: string, buffer: Uint8Array): Promise<void>;\n dropFile(name: string): Promise<void>;\n }\n const registrar = async (): Promise<Registrar | null> => {\n const connector = coordinator.databaseConnector?.() as\n | { getDuckDB?: () => Promise<Registrar> }\n | null\n | undefined;\n if (!connector?.getDuckDB) return null;\n try {\n return await connector.getDuckDB();\n } catch {\n return null;\n }\n };\n\n /** Registered name → how many bytes it is holding. Insertion order is the eviction order. */\n const resident = new Map<string, number>();\n /** URL → the name DuckDB should read it from, once that has been decided one way or the other. */\n const decided = new Map<string, string>();\n let held = 0;\n\n /**\n * Sixty-four megabytes of tiles, evicted oldest-first.\n *\n * A budget rather than a count, because a tile's size is the corpus's decision and a count would\n * mean something different for every one. Oldest-first rather than least-recently-used: a reader\n * pans, and a pan revisits what it just left, so recency and insertion order agree where it\n * matters — and an LRU's bookkeeping is a second structure to keep correct for a difference nobody\n * has measured.\n */\n const BUDGET = 64 * 1024 * 1024;\n\n /**\n * A file is worth holding when it is **cheap to fetch whole** — 256 KB, and the number is\n * measured.\n *\n * Registering a file means downloading all of it. On the million-node corpus a vertex chunk was\n * 74 KB and the edge chunks far larger, and caching both took a cold window from about 200 ms to\n * **17,979 ms** while a repeat fell to **11 ms** — a thousandfold win on revisit paid for with an\n * eighteen-second first paint, which is not a trade anybody would take.\n *\n * So the rule is a property of the file rather than a flag, and the corpus decides: one\n * `tiles.parquet` per set decides against, which is right for it — the row groups a window wants\n * are a byte range, and a range read is what DuckDB already does.\n */\n const WORTH_HOLDING = 256 * 1024;\n\n /**\n * The names DuckDB should read these URLs from — registered buffers where possible, URLs where\n * not, quoted either way.\n *\n * Fetched concurrently, because a window is a handful of files and they are independent; a\n * sequential loop here would make the first paint the sum of them rather than the slowest.\n */\n async function readable(urls: readonly string[]): Promise<string[]> {\n if (urls.length === 0) return [];\n const db = await registrar();\n if (!db) return urls.map((url) => `'${url}'`);\n const names = await Promise.all(\n urls.map(async (url) => {\n const already = decided.get(url);\n if (already !== undefined && (!already.startsWith(\"'\") ? resident.has(already) : true)) {\n return already;\n }\n // A name DuckDB can hold a buffer under, derived from the URL so that two tiles of two\n // types never collide and the same tile twice never registers twice.\n const name = `corpus_${url.replace(/[^A-Za-z0-9]+/g, \"_\")}`;\n // The size first, which is one metadata request against a body that may be megabytes — and\n // the same request a footer read already makes, so the shape is not new here.\n const probe = await fetch(url, { method: \"HEAD\" });\n const size = Number(probe.headers.get(\"content-length\"));\n if (!probe.ok || !Number.isFinite(size) || size > WORTH_HOLDING) {\n const plain = `'${url}'`;\n decided.set(url, plain);\n return plain;\n }\n const response = await fetch(url);\n // A file that will not load is not a reason to fail the whole window: fall back to the URL\n // and let DuckDB report whatever it finds there, which is the error a reader can act on.\n if (!response.ok) return `'${url}'`;\n const bytes = new Uint8Array(await response.arrayBuffer());\n await db.registerFileBuffer(name, bytes);\n resident.set(name, bytes.byteLength);\n decided.set(url, name);\n held += bytes.byteLength;\n return name;\n }),\n );\n while (held > BUDGET && resident.size > 0) {\n const [oldest, size] = resident.entries().next().value as [string, number];\n // Never evict a file this very window is about to read, or the query reads a dropped file.\n if (names.includes(oldest)) break;\n resident.delete(oldest);\n held -= size;\n await db.dropFile(oldest);\n }\n return names.map((n) => (n.startsWith(\"'\") ? n : `'${n}'`));\n }\n\n /**\n * Where the payload is, as the list of files that hold it.\n *\n * One file per tile under `container: files`, one file in total under `rowgroups` — and this side\n * does not know or care which, because the address is asked for rather than composed. Throws when\n * the manifest declares no `vertex_count`, which is the one absence that makes a corpus\n * un-enumerable.\n */\n const payloadFiles = type.files();\n /** A URL list as a SQL list literal. Every read below composes one and none of them composes a URL. */\n const quoted = (urls: readonly string[]) =>\n urls.map((url) => `'${url.replace(/'/g, \"''\")}'`).join(\", \");\n\n /**\n * The boxes, from the footer — **which tile a row group is, asked of the addressing.**\n *\n * `parquet_metadata` reports a `file_name` and a `row_group_id`, and which of the two names the\n * tile is the container's business: under `rowgroups` row group `k` IS tile `k`; under `files`\n * the file is, and its row groups are the writer's business, so their boxes are merged. This used\n * to read the tile out of the URL with `regexp_extract(file_name, 'chunk(\\d+)')` — one\n * container's spelling hard-coded into a query, and `NULL` for every row of a corpus fossil\n * writes today.\n *\n * `min_value`/`max_value`, never `min`/`max`. Parquet's original statistics fields are defined by\n * *signed* byte comparison, which is meaningless for an unsigned column — a writer that gets this\n * right leaves them empty. A reader that only knows the deprecated pair concludes the footer\n * carries no box for the column the whole address is built on. `coalesce` keeps the float columns\n * working either way.\n */\n function load(): Promise<TileBox[]> {\n loading ??= (async () => {\n const files = quoted(payloadFiles);\n const [xCol, yCol] = PAYLOAD_COORDINATES;\n // Which of `file_name` and `row_group_id` names the tile, as an expression rather than as a\n // branch in JavaScript: the grouping has to happen where the rows are either way.\n const tile =\n addressing.container === \"rowgroups\"\n ? \"row_group_id\"\n : `list_position([${files}], file_name) - 1`;\n const stats = await meta(\n `SELECT ${tile} AS tile,\n min(CASE WHEN path_in_schema = '${xCol}' THEN coalesce(stats_min_value, stats_min)::DOUBLE END) AS x0,\n max(CASE WHEN path_in_schema = '${xCol}' THEN coalesce(stats_max_value, stats_max)::DOUBLE END) AS x1,\n min(CASE WHEN path_in_schema = '${yCol}' THEN coalesce(stats_min_value, stats_min)::DOUBLE END) AS y0,\n max(CASE WHEN path_in_schema = '${yCol}' THEN coalesce(stats_max_value, stats_max)::DOUBLE END) AS y1\n FROM parquet_metadata([${files}])\n WHERE path_in_schema IN ('${xCol}', '${yCol}') GROUP BY 1 ORDER BY 1`,\n );\n const tiles = numbers(stats, \"tile\");\n const [x0, x1, y0, y1] = [\"x0\", \"x1\", \"y0\", \"y1\"].map((f) => numbers(stats, f));\n return tiles.map((t, i) => ({\n tile: t as number,\n x0: x0?.[i] as number,\n x1: x1?.[i] as number,\n y0: y0?.[i] as number,\n y1: y1?.[i] as number,\n }));\n })();\n return loading;\n }\n\n /** The tiles a rectangle touches. Pure — this is the whole of the selection, and it makes no call. */\n function intersecting(all: TileBox[], view: Viewport): number[] {\n return all\n .filter((b) => b.x1 >= view.xMin && b.x0 <= view.xMax && b.y1 >= view.yMin && b.y0 <= view.yMax)\n .map((b) => b.tile);\n }\n\n /**\n * Everything the corpus fixes, and nothing it does not — **by ROLE, not by name.**\n *\n * The address, the identity and the coordinates are facts of the format, and the names they are\n * written under come off `@fossil-lang/corpus`'s generated column table rather than four string\n * literals here. The endpoint columns come off the adjacency's own address. What colours and what\n * sizes are channels, so they arrive with the request and are filled in per slice.\n */\n const fixed = {\n id: PAYLOAD_ADDRESS[0] as string,\n subject: subjects ? (PAYLOAD_IDENTITY[0] as string) : undefined,\n x: PAYLOAD_COORDINATES[0] as string,\n y: PAYLOAD_COORDINATES[1] as string,\n source: adjacency?.column ?? \"src_dense\",\n target: edge?.adjacency(\"dst\")?.column ?? \"dst_dense\",\n } as const;\n\n /**\n * The relation half, registered once at open.\n *\n * A view rather than a table: `CREATE TABLE AS` would pull the corpus into memory, which is the\n * working set the whole bounded path exists to refuse. A view leaves the bytes where they are and\n * lets each query fetch the ranges it needs.\n *\n * Over **every** file of the payload, deliberately — this is the surface that answers *what does\n * it mean*, and a count, a histogram or a crossfilter clause is a question about the corpus rather\n * than about the window. The addressed reading is `slice`, beside it, and the two are different\n * access to the same bytes rather than two versions of one.\n */\n const nodesView = `corpus_${type.type}`;\n const edgesView = adjacency ? `corpus_${type.type}_edges` : undefined;\n await coordinator.exec(\n `CREATE OR REPLACE VIEW ${nodesView} AS SELECT * FROM read_parquet([${quoted(payloadFiles)}])`,\n );\n if (edgesView && edge) {\n await coordinator.exec(\n `CREATE OR REPLACE VIEW ${edgesView} AS\n SELECT * FROM read_parquet([${quoted(edge.projectionFiles(1, \"src\"))}])`,\n );\n }\n\n const source: DuckSource = {\n ...watching.api,\n\n publish(vertices) {\n publishSelection(reads, filterBy, fixed.id, vertices);\n },\n\n /**\n * How many vertices there are — **read, not probed.**\n *\n * The manifest declares `vertex_count`. What stood here was a note saying no manifest carries\n * it, and a doubling-then-bisecting `HEAD` search for the last chunk plus a\n * `parquet_file_metadata` read of its row count — about a dozen requests and a query, per\n * corpus, for a number already in hand.\n */\n async total() {\n if (type.count !== null) return Number(type.count);\n const rows = await meta(`SELECT count(*) AS n FROM ${nodesView}`);\n return Number(numbers(rows, \"n\")[0] ?? 0);\n },\n\n async extent() {\n const all = await load();\n return {\n xMin: Math.min(...all.map((b) => b.x0)),\n yMin: Math.min(...all.map((b) => b.y0)),\n xMax: Math.max(...all.map((b) => b.x1)),\n yMax: Math.max(...all.map((b) => b.y1)),\n };\n },\n\n // Regions only, and it says so by being the only method there is. A neighbourhood needs\n // adjacency this source does not index; `ExploringSource` declared the second question here and\n // was deleted with it — `index.test.ts` carries why, and fossil's `expand` is what answers it.\n async slice(request: SliceRequest): Promise<Slice> {\n const { fill, limit, minLinkPixels, perPixel, pinned, r, signal, view } = request;\n const columns: Columns = { ...fixed, category: fill, size: r };\n const all = await load();\n /**\n * The tiles the rectangle touches — **and the far view is not a special case of this.**\n *\n * It used to be: past a zoom threshold the selection was replaced by every tile, because the\n * aggregate branch was going to read the whole relation anyway. A window that covers the\n * extent already intersects every box, so the branch was arithmetic restating itself, and it\n * is the reason `Viewport` carried a `zoom` at all.\n */\n const selected = intersecting(all, view);\n // Nothing selected is a legitimate answer — the camera is over empty space — and asking\n // `read_parquet([])` is a syntax error rather than an empty result. Checked before the files\n // are fetched, so an empty window costs no bytes at all.\n if (selected.length === 0) {\n return {\n n: 0,\n marks: 0,\n vertices: new BigUint64Array(0),\n positions: new Float32Array(0),\n links: new Float32Array(0),\n categories: new Uint16Array(0),\n };\n }\n\n /**\n * The URLs those tiles are in — **asked, not composed**, and distinct.\n *\n * `directions: ['src']` is the drawing read: every edge a window can draw has its source on\n * screen, therefore in one of these files. The answer says so — `complete: false` with a\n * `not-requested` gap for `dst`.\n */\n const addressed = addressing.tilesFor({\n type: type.type,\n tiles: selected,\n directions: [\"src\"],\n });\n // Both halves at once: the vertex files and the edge files a window touches are independent\n // reads, and the window is not drawable until both have landed.\n const [vertexFiles, edgeFiles] = await Promise.all([\n readable(addressed.vertexUrls),\n readable(addressed.edgeUrls),\n ]);\n /**\n * The camera moved while the tiles were arriving, so this question is already the wrong one.\n *\n * Checked here rather than left to the caller because of what comes next: the reads hold one\n * standing question each, so a request that resumes after the loop moved on would *supersede*\n * the newer one and reject it — the stale question winning the race against the live one.\n *\n * `signal.reason` and not a sentinel of ours: an aborted signal already carries what it was\n * aborted with, which for `AbortController.abort()` is a `DOMException` named `AbortError`.\n * Rethrowing it is the whole of the cancellation contract a source owes — see `SliceRequest`.\n */\n if (signal?.aborted) throw signal.reason;\n const nodes = `read_parquet([${vertexFiles.join(\", \")}])`;\n // A corpus that declares no adjacency for this type still has to answer: the links query is\n // built either way, so what it reads is an empty relation of the right shape rather than a\n // `read_parquet([])`, which is a syntax error, or the vertex view, which has neither column.\n const relation =\n edgeFiles.length > 0\n ? `read_parquet([${edgeFiles.join(\", \")}])`\n : `(SELECT NULL::BIGINT AS ${columns.source}, NULL::BIGINT AS ${columns.target} WHERE FALSE)`;\n\n // `nodes` is both the window's rows and the bytes the reader holds: the tiles that answer\n // \"what is in the rectangle\" are the tiles that answer \"where is the far end of an edge that\n // leaves it\".\n return watching.run(region(nodes, relation, columns, view, limit, pinned, perPixel, minLinkPixels));\n },\n };\n\n return { source, nodes: nodesView, edges: edgesView };\n}\n"],"names":[],"mappings":";;;;;AAsEA;AAwCA;AACE;AAAO;AACsC;AACD;AACX;AAEnC;AAUA;AACE;AACA;AACE;AACA;AAAmB;AACZ;AAEX;AAUA;AACE;AAEA;AACA;AACF;AAGA;AACE;AAGF;AAuBA;AAOE;AAN2C;AACpB;AACA;AACA;AACA;AAKvB;AACF;AAeA;AAyBA;AAOE;AAYA;AAAO;AAGL;AAAA;AAEY;AACC;AAAA;AAIb;AACiB;AAAA;AAAA;AAGY;AAChB;AAEjB;AAYA;AAME;AACA;AACA;AACF;AA+BA;AASE;AAEA;AAAO;AACsE;AAAA;AAAA;AAAA;AAAA;AAK/D;AACgB;AACA;AAAA;AAAA;AAG8C;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAQ9E;AAsBA;AAUE;AA2BA;AAAO;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAS4D;AACS;AAAA;AAI1B;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAYwB;AAAA;AAAA;AAAA;AAAA;AAK1C;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AASgB;AACiC;AAAA;AAGnF;AAUA;AACE;AAGA;AAKE;AACA;AAEA;AACyC;AAG3C;AAAO;AACA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AASD;AAIE;AAKW;AACb;AACF;AAAA;AAGA;AAIA;AAA0C;AACZ;AACF;AAE5B;AAAkC;AACpC;AAEJ;AAUA;AAME;AACS;AAC+D;AACtD;AAC8B;AAC7C;AAEL;AAWA;AAUE;AAGA;AAKA;AACA;AACA;AACA;AAQA;AACA;AACA;AACA;AACA;AACA;AAKA;AAEA;AACA;AAMA;AAGA;AAEO;AACL;AACA;AACA;AACsC;AACJ;AAClC;AACA;AAEJ;AAGA;;AACE;AACA;AACA;AACA;AAEF;AAuHA;AACE;AACA;AACA;AACF;AAqCA;;AACE;AAgDA;AA4BA;;AACE;AAIA;AACA;AACE;AAAuB;AAEvB;AAAO;AACT;AAOF;AAWA;AAwBA;AACE;AACA;AACA;AACA;AAA4B;AAExB;AACA;AACE;AAIF;AAKA;AACE;AACA;AACO;AAET;AAGA;AACA;AACA;AAIO;AACR;AAEH;AACE;AAEA;AACA;AAEwB;AAE1B;AAA0D;AAW5D;AAqBA;AACE;AACE;AAQoB;AACJ;AAC2B;AACA;AACA;AACA;AACV;AACa;AAI9C;AAA4B;AACpB;AACG;AACA;AACA;AACA;AACT;AAEG;AAIT;AACE;AAEoB;AAWtB;AAAc;AACS;AACiC;AAC9B;AACA;AACK;AACa;AAiB5C;AAAkB;AAC0E;AAGxE;AACmB;AACoC;AAIhD;AACb;AAGV;AAAoD;AACtD;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAWE;AACA;AACA;AAAwC;AAC1C;AAGE;AACA;AAAO;AACiC;AACA;AACA;AACA;AAAA;AAE1C;AAAA;AAAA;AAAA;AAME;AAeA;AACE;AAAO;AACF;AACI;AACuB;AACD;AACJ;AACI;AAWjC;AAAsC;AACzB;AACJ;AACW;AAI+B;AACpB;AACF;AAa7B;AACA;AAYA;AAAkG;AACpG;AAIJ;;;;"}
|
package/dist/graph-canvas.d.ts
CHANGED
|
@@ -38,7 +38,7 @@ export interface GraphCanvasProps extends UseGraphProps {
|
|
|
38
38
|
*
|
|
39
39
|
* **Overlays and selection are not in here on purpose.** `useGraphOverlays` and `useGraphSelection`
|
|
40
40
|
* need callbacks only the product can write — what a click means, what a lasso commits to. They also
|
|
41
|
-
* need
|
|
41
|
+
* need the api from *above* this element, where a context cannot be read, which
|
|
42
42
|
* is the whole reason `useGraph` and `GraphRootProvider` exist beside this shortcut. A host with
|
|
43
43
|
* overlays calls those two; a host with only chrome calls this one and reads `useGraphContext` from
|
|
44
44
|
* a child.
|
package/dist/graph-canvas.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"graph-canvas.js","sources":["../src/graph-canvas.tsx"],"sourcesContent":["\"use client\";\n\nimport { cn } from \"@kanzo-tech/ui\";\nimport { createContext, useContext, type ReactNode } from \"react\";\nimport { useGraph, type GraphApi, type UseGraphProps } from \"./use-graph\";\n\n/**\n * The three pieces that turn `useGraph` into something you can put on a page — Ark's shape, minus\n * the part of it that only makes sense when a component has parts.\n *\n * `GraphRootProvider` takes an api built above it and renders the surface. `GraphCanvas` is the\n * shortcut that builds one for you, which is what a host wants until it needs to call a hook beside\n * the canvas. `useGraphContext` is how the chrome reads either of them.\n *\n * **`useGraph` creates and `useGraphContext` reads**, which is the convention Ark states and this\n * package used to invert: the reader was called `useGraphCanvas` and there was no creator at all, so\n * anyone arriving from Ark would have read it as the factory and got the opposite.\n */\n\nconst GraphContext = createContext<GraphApi | null>(null);\n\n/**\n * Reach the graph from its chrome. Strict: outside a provider there is nothing to answer with, and a\n * `null` here would surface as a legend that silently counts zero.\n */\nexport function useGraphContext(): GraphApi {\n const value = useContext(GraphContext);\n if (!value) throw new Error(\"useGraphContext must be used inside a <GraphCanvas> or <GraphRootProvider>\");\n return value;\n}\n\nexport interface GraphRootProviderProps {\n /** The api from `useGraph`, built wherever the host needs to reach it. */\n value: GraphApi;\n className?: string;\n /** The chrome — a legend, a toolbar, a zoom control. Positioned over the surface by the host. */\n children?: ReactNode;\n slot?: string;\n}\n\n/**\n * The element, and the context over it.\n *\n * Two divs rather than one, and the split is load-bearing: cosmos.gl takes the inner one and fills\n * it with a canvas of its own, so chrome parented there would be a sibling of that canvas inside an\n * element the renderer resizes. The outer one is the positioning context — `isolate`, so a host's\n * `z-index` on a legend cannot escape into the page — and `children` go there.\n */\nexport function GraphRootProvider(props: GraphRootProviderProps) {\n const { children, className, slot, value } = props;\n return (\n <div\n className={cn(\"relative isolate size-full overflow-hidden\", className)}\n data-slot={slot ?? \"graph-canvas\"}\n >\n <div className=\"size-full\" data-slot=\"graph-canvas-surface\" ref={value.hostRef} />\n <GraphContext.Provider value={value}>{children}</GraphContext.Provider>\n </div>\n );\n}\n\nexport interface GraphCanvasProps extends UseGraphProps {\n className?: string;\n children?: ReactNode;\n slot?: string;\n}\n\n/**\n * A bounded WebGL graph, wired.\n *\n * **This is `ChartRoot`'s counterpart, and it is deliberately not the whole screen.** It owns the\n * renderer's lifetime, the query loop that follows the camera, and the buffers a look implies — the\n * three that are the same in every product and were being re-wired by hand at each call site.\n * Everything else stays where it differs: a legend, an inspector, a hover card and a rule builder are\n * arrangements, and `children` is where they go.\n *\n * **Overlays and selection are not in here on purpose.** `useGraphOverlays` and `useGraphSelection`\n * need callbacks only the product can write — what a click means, what a lasso commits to. They also\n * need
|
|
1
|
+
{"version":3,"file":"graph-canvas.js","sources":["../src/graph-canvas.tsx"],"sourcesContent":["\"use client\";\n\nimport { cn } from \"@kanzo-tech/ui\";\nimport { createContext, useContext, type ReactNode } from \"react\";\nimport { useGraph, type GraphApi, type UseGraphProps } from \"./use-graph\";\n\n/**\n * The three pieces that turn `useGraph` into something you can put on a page — Ark's shape, minus\n * the part of it that only makes sense when a component has parts.\n *\n * `GraphRootProvider` takes an api built above it and renders the surface. `GraphCanvas` is the\n * shortcut that builds one for you, which is what a host wants until it needs to call a hook beside\n * the canvas. `useGraphContext` is how the chrome reads either of them.\n *\n * **`useGraph` creates and `useGraphContext` reads**, which is the convention Ark states and this\n * package used to invert: the reader was called `useGraphCanvas` and there was no creator at all, so\n * anyone arriving from Ark would have read it as the factory and got the opposite.\n */\n\nconst GraphContext = createContext<GraphApi | null>(null);\n\n/**\n * Reach the graph from its chrome. Strict: outside a provider there is nothing to answer with, and a\n * `null` here would surface as a legend that silently counts zero.\n */\nexport function useGraphContext(): GraphApi {\n const value = useContext(GraphContext);\n if (!value) throw new Error(\"useGraphContext must be used inside a <GraphCanvas> or <GraphRootProvider>\");\n return value;\n}\n\nexport interface GraphRootProviderProps {\n /** The api from `useGraph`, built wherever the host needs to reach it. */\n value: GraphApi;\n className?: string;\n /** The chrome — a legend, a toolbar, a zoom control. Positioned over the surface by the host. */\n children?: ReactNode;\n slot?: string;\n}\n\n/**\n * The element, and the context over it.\n *\n * Two divs rather than one, and the split is load-bearing: cosmos.gl takes the inner one and fills\n * it with a canvas of its own, so chrome parented there would be a sibling of that canvas inside an\n * element the renderer resizes. The outer one is the positioning context — `isolate`, so a host's\n * `z-index` on a legend cannot escape into the page — and `children` go there.\n */\nexport function GraphRootProvider(props: GraphRootProviderProps) {\n const { children, className, slot, value } = props;\n return (\n <div\n className={cn(\"relative isolate size-full overflow-hidden\", className)}\n data-slot={slot ?? \"graph-canvas\"}\n >\n <div className=\"size-full\" data-slot=\"graph-canvas-surface\" ref={value.hostRef} />\n <GraphContext.Provider value={value}>{children}</GraphContext.Provider>\n </div>\n );\n}\n\nexport interface GraphCanvasProps extends UseGraphProps {\n className?: string;\n children?: ReactNode;\n slot?: string;\n}\n\n/**\n * A bounded WebGL graph, wired.\n *\n * **This is `ChartRoot`'s counterpart, and it is deliberately not the whole screen.** It owns the\n * renderer's lifetime, the query loop that follows the camera, and the buffers a look implies — the\n * three that are the same in every product and were being re-wired by hand at each call site.\n * Everything else stays where it differs: a legend, an inspector, a hover card and a rule builder are\n * arrangements, and `children` is where they go.\n *\n * **Overlays and selection are not in here on purpose.** `useGraphOverlays` and `useGraphSelection`\n * need callbacks only the product can write — what a click means, what a lasso commits to. They also\n * need the api from *above* this element, where a context cannot be read, which\n * is the whole reason `useGraph` and `GraphRootProvider` exist beside this shortcut. A host with\n * overlays calls those two; a host with only chrome calls this one and reads `useGraphContext` from\n * a child.\n *\n * This component exists against an earlier decision that there should be no canvas component, and\n * `/docs/design/graph` carries what changed and what would reverse it.\n */\nexport function GraphCanvas(props: GraphCanvasProps) {\n const { children, className, slot, ...rest } = props;\n const api = useGraph(rest);\n return (\n <GraphRootProvider className={className} slot={slot} value={api}>\n {children}\n </GraphRootProvider>\n );\n}\n"],"names":[],"mappings":";;;;;AAmBA;AAMO;AACL;AACA;AACA;AACF;AAmBO;AACL;AACA;AACE;AAAC;AAAA;AACsE;AAClD;AAEnB;AAAgF;AACjC;AAAA;AAAA;AAGrD;AA2BO;AACL;AAEA;AAKF;;;;;;"}
|
package/dist/graph-looks.d.ts
CHANGED
|
@@ -1,19 +1,29 @@
|
|
|
1
1
|
/**
|
|
2
|
-
*
|
|
2
|
+
* The glyphs this canvas draws — **by name**, because a name is what the concept is.
|
|
3
3
|
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
4
|
+
* This was five exports of one idea: a `SHAPE` object mapping names to numbers, a `ShapeId` type
|
|
5
|
+
* that was the union of those numbers, `SHAPE_ORDER`, `SHAPE_OTHER` and `SHAPE_PATH` keyed by them.
|
|
6
|
+
* The numbers were never ours. They are cosmos.gl's `setPointShapes` enum indices, and the jump from
|
|
7
|
+
* `3` (diamond) to `7` (cross) is what gives that away — there is no `4`, `5` or `6` here because
|
|
8
|
+
* Pentagon, Hexagon and Star are members this canvas does not draw. Publishing them made the
|
|
9
|
+
* renderer's internal numbering part of a contract, so a host held `3` where it meant *diamond*, and
|
|
10
|
+
* an upstream enum that renumbered would have moved every legend on every page silently.
|
|
11
|
+
*
|
|
12
|
+
* A string union says the same thing, reads at the call site — `<ShapeGlyph shape="cross" />` — and
|
|
13
|
+
* makes the mapping this file's private business, which is what it always was.
|
|
14
|
+
*/
|
|
15
|
+
export type Shape = "circle" | "square" | "triangle" | "diamond" | "cross";
|
|
16
|
+
/**
|
|
17
|
+
* The name → cosmos.gl enum index, and the one place that translation happens.
|
|
18
|
+
*
|
|
19
|
+
* `None` (`8`) has no name here on purpose: the point fragment shader `discard`s a `NONE` point that
|
|
20
|
+
* carries no image, so "past capacity" spelled as `None` would delete the node from the picture, and
|
|
21
|
+
* a category nobody can name is still a node with edges. `obligations.ts` grades that.
|
|
22
|
+
*
|
|
23
|
+
* Module-scoped and off the barrel. `buffers` reads it on the way to the GPU; nothing else needs it,
|
|
24
|
+
* and anything that did would be reaching for the enum this type exists to hide.
|
|
8
25
|
*/
|
|
9
|
-
export declare const
|
|
10
|
-
readonly circle: 0;
|
|
11
|
-
readonly square: 1;
|
|
12
|
-
readonly triangle: 2;
|
|
13
|
-
readonly diamond: 3;
|
|
14
|
-
readonly cross: 7;
|
|
15
|
-
};
|
|
16
|
-
export type ShapeId = (typeof SHAPE)[keyof typeof SHAPE];
|
|
26
|
+
export declare const SHAPE_INDEX: Record<Shape, number>;
|
|
17
27
|
/**
|
|
18
28
|
* The shape scale, in slot order — the sibling of the colour scale.
|
|
19
29
|
*
|
|
@@ -35,7 +45,7 @@ export type ShapeId = (typeof SHAPE)[keyof typeof SHAPE];
|
|
|
35
45
|
* is the most dangerous shape a comment can have — see `OBLIGATIONS` in `./obligations.ts` for what
|
|
36
46
|
* the floor actually protects.
|
|
37
47
|
*/
|
|
38
|
-
export declare const SHAPE_ORDER:
|
|
48
|
+
export declare const SHAPE_ORDER: Shape[];
|
|
39
49
|
/**
|
|
40
50
|
* What a category past the scale wears — the shape channel's `--muted-foreground`.
|
|
41
51
|
*
|
|
@@ -43,7 +53,7 @@ export declare const SHAPE_ORDER: ShapeId[];
|
|
|
43
53
|
* the first one. This is the half that used to be missing, and the comment above was false without
|
|
44
54
|
* it — `SHAPE_ORDER[4]` is `undefined`, and the fallback was `circle`.
|
|
45
55
|
*/
|
|
46
|
-
export declare const SHAPE_OTHER:
|
|
56
|
+
export declare const SHAPE_OTHER: Shape;
|
|
47
57
|
/**
|
|
48
58
|
* The geometry a canvas draws — and nothing else.
|
|
49
59
|
*
|
|
@@ -139,8 +149,43 @@ export interface Look {
|
|
|
139
149
|
* default. Those defaults are what a graph drew before any of this existed.
|
|
140
150
|
*/
|
|
141
151
|
export declare function lookFrom(values?: Readonly<Record<string, string | undefined>>): Look;
|
|
142
|
-
/**
|
|
152
|
+
/**
|
|
153
|
+
* What a canvas draws when nobody has chosen anything.
|
|
154
|
+
*
|
|
155
|
+
* **Not on the barrel, and it used to be.** It is literally `lookFrom()` — a second public name for
|
|
156
|
+
* a value the package already hands out on request — and the reason it was exported is the one
|
|
157
|
+
* `resolveLook` below removes: a host that wanted one field different had to start from the whole
|
|
158
|
+
* object, because `look` took a whole `Look`. Spreading a default you were given is a copy of it,
|
|
159
|
+
* and a copy is what stops tracking the original the next time a number here moves.
|
|
160
|
+
*/
|
|
143
161
|
export declare const DEFAULT_LOOK: Look;
|
|
144
|
-
/**
|
|
145
|
-
|
|
162
|
+
/**
|
|
163
|
+
* A look in the pieces a caller wants different — everything else is this package's answer.
|
|
164
|
+
*
|
|
165
|
+
* Two levels, because a `Look` has exactly two: the fields, and `link`. Deep-merging arbitrarily
|
|
166
|
+
* would be a guess about a shape that is right here in this file, and `size` and `fade` are tuples
|
|
167
|
+
* that must be replaced whole rather than merged element-wise.
|
|
168
|
+
*/
|
|
169
|
+
export interface LookPatch extends Partial<Omit<Look, "link">> {
|
|
170
|
+
link?: Partial<Look["link"]>;
|
|
171
|
+
}
|
|
172
|
+
/**
|
|
173
|
+
* A patch over the package's own default — the merge `DEFAULT_LOOK` existed so a host could do by
|
|
174
|
+
* hand.
|
|
175
|
+
*
|
|
176
|
+
* **Nothing, and it is the shared constant rather than a copy of it.** That identity matters: the
|
|
177
|
+
* buffers are rebuilt whenever the look's reference changes, so a fresh object per render would
|
|
178
|
+
* mean a full colour/size/shape upload on every render of every canvas that never asked for one.
|
|
179
|
+
*/
|
|
180
|
+
export declare function resolveLook(patch?: LookPatch): Look;
|
|
181
|
+
/**
|
|
182
|
+
* The SVG path for a shape glyph inside a 12×12 box — the legend draws what the canvas draws.
|
|
183
|
+
*
|
|
184
|
+
* **Off the barrel, and `ShapeGlyph` is what replaced it.** One host imported this, to fill a
|
|
185
|
+
* `<path>` in a legend key and a hover card. What that host actually wanted was *the glyph*, and
|
|
186
|
+
* handing it the path data made it responsible for the viewBox, the fill and the fact that the box
|
|
187
|
+
* is twelve units — three facts it had to keep in step with this file by reading the comment above.
|
|
188
|
+
* A component carries all three and cannot fall out of step with itself.
|
|
189
|
+
*/
|
|
190
|
+
export declare const SHAPE_PATH: Record<Shape, string>;
|
|
146
191
|
//# sourceMappingURL=graph-looks.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"graph-looks.d.ts","sourceRoot":"","sources":["../src/graph-looks.ts"],"names":[],"mappings":"AAmBA
|
|
1
|
+
{"version":3,"file":"graph-looks.d.ts","sourceRoot":"","sources":["../src/graph-looks.ts"],"names":[],"mappings":"AAmBA;;;;;;;;;;;;;GAaG;AACH,MAAM,MAAM,KAAK,GAAG,QAAQ,GAAG,QAAQ,GAAG,UAAU,GAAG,SAAS,GAAG,OAAO,CAAC;AAE3E;;;;;;;;;GASG;AACH,eAAO,MAAM,WAAW,EAAE,MAAM,CAAC,KAAK,EAAE,MAAM,CAM7C,CAAC;AAEF;;;;;;;;;;;;;;;;;;;;GAoBG;AACH,eAAO,MAAM,WAAW,EAAE,KAAK,EAAgD,CAAC;AAEhF;;;;;;GAMG;AACH,eAAO,MAAM,WAAW,EAAE,KAAe,CAAC;AAE1C;;;;;;GAMG;AACH,MAAM,WAAW,IAAI;IACnB,qEAAqE;IACrE,IAAI,EAAE,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IACvB,IAAI,EAAE;QACJ;;;;;WAKG;QACH,MAAM,EAAE,OAAO,CAAC;QAChB,OAAO,EAAE,MAAM,CAAC;QAChB,KAAK,EAAE,MAAM,CAAC;QACd;;;;;;WAMG;QACH,KAAK,EAAE,MAAM,CAAC;QACd;;;;;;;;;;WAUG;QACH,KAAK,EAAE,OAAO,CAAC;QACf;;;;;;WAMG;QACH,IAAI,EAAE,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;KACxB,CAAC;IACF,mEAAmE;IACnE,MAAM,EAAE,MAAM,CAAC;IACf,+FAA+F;IAC/F,QAAQ,EAAE,OAAO,CAAC;IAClB;;;;;;OAMG;IACH,IAAI,EAAE,OAAO,CAAC;CAuBf;AAED;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GA+BG;AACH,wBAAgB,QAAQ,CAAC,MAAM,GAAE,QAAQ,CAAC,MAAM,CAAC,MAAM,EAAE,MAAM,GAAG,SAAS,CAAC,CAAM,GAAG,IAAI,CA0BxF;AAED;;;;;;;;GAQG;AACH,eAAO,MAAM,YAAY,EAAE,IAAiB,CAAC;AAE7C;;;;;;GAMG;AACH,MAAM,WAAW,SAAU,SAAQ,OAAO,CAAC,IAAI,CAAC,IAAI,EAAE,MAAM,CAAC,CAAC;IAC5D,IAAI,CAAC,EAAE,OAAO,CAAC,IAAI,CAAC,MAAM,CAAC,CAAC,CAAC;CAC9B;AAED;;;;;;;GAOG;AACH,wBAAgB,WAAW,CAAC,KAAK,CAAC,EAAE,SAAS,GAAG,IAAI,CAGnD;AAED;;;;;;;;GAQG;AACH,eAAO,MAAM,UAAU,EAAE,MAAM,CAAC,KAAK,EAAE,MAAM,CAQ5C,CAAC"}
|
package/dist/graph-looks.js
CHANGED
|
@@ -1,44 +1,55 @@
|
|
|
1
|
-
const
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
1
|
+
const a = {
|
|
2
|
+
circle: 0,
|
|
3
|
+
square: 1,
|
|
4
|
+
triangle: 2,
|
|
5
|
+
diamond: 3,
|
|
6
|
+
cross: 7
|
|
7
|
+
}, d = ["circle", "square", "triangle", "diamond"], u = "cross";
|
|
8
|
+
function c(e = {}) {
|
|
9
|
+
const n = (o, s) => {
|
|
10
|
+
const l = e[o];
|
|
11
|
+
return l === void 0 ? s : l === "true";
|
|
12
|
+
}, r = e.marks === "legible", t = Number.parseFloat(e.labels ?? "");
|
|
7
13
|
return {
|
|
8
|
-
size:
|
|
14
|
+
size: r ? [4, 13] : [2, 8],
|
|
9
15
|
link: {
|
|
10
|
-
render:
|
|
11
|
-
opacity:
|
|
12
|
-
width:
|
|
16
|
+
render: n("links", !0),
|
|
17
|
+
opacity: r ? 0.28 : 0.42,
|
|
18
|
+
width: r ? 0.5 : 0.6,
|
|
13
19
|
// A hint, and `obligations.ts` says why: every link bows the same way, so cosmos.gl's default
|
|
14
20
|
// of 0.5 reads as one pinwheel. A toggle rather than a range because a reader wants two
|
|
15
21
|
// pictures — straight, and told apart — not a number to tune.
|
|
16
|
-
curve:
|
|
17
|
-
blend:
|
|
22
|
+
curve: n("bowed-links", !0) ? 0.12 : 0,
|
|
23
|
+
blend: n("additive-links", !1),
|
|
18
24
|
// Shared by every form. It was three ranges within ±10% of each other, which is the
|
|
19
25
|
// definition of a field nobody chose.
|
|
20
26
|
fade: [200, 1400]
|
|
21
27
|
},
|
|
22
28
|
labels: Number.isFinite(t) ? t : 26,
|
|
23
|
-
vignette:
|
|
24
|
-
grid:
|
|
29
|
+
vignette: n("vignette", !1),
|
|
30
|
+
grid: n("grid", !0)
|
|
25
31
|
};
|
|
26
32
|
}
|
|
27
|
-
const
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
33
|
+
const i = c();
|
|
34
|
+
function k(e) {
|
|
35
|
+
return e ? { ...i, ...e, link: { ...i.link, ...e.link } } : i;
|
|
36
|
+
}
|
|
37
|
+
const v = {
|
|
38
|
+
circle: "M6 1.6a4.4 4.4 0 1 0 0 8.8 4.4 4.4 0 0 0 0-8.8Z",
|
|
39
|
+
square: "M2 2h8v8H2Z",
|
|
40
|
+
triangle: "M6 1.6 10.6 10H1.4Z",
|
|
41
|
+
diamond: "M6 1 11 6l-5 5-5-5Z",
|
|
32
42
|
// The proportions are cosmos.gl's own `crossDistance`: a plus with arms at 0.8 of the radius and
|
|
33
43
|
// a bar 0.3 thick, so the legend's glyph is the shape the shader draws.
|
|
34
|
-
|
|
44
|
+
cross: "M4.2 1.2h3.6v3h3v3.6h-3v3H4.2v-3h-3V4.2h3Z"
|
|
35
45
|
};
|
|
36
46
|
export {
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
c as lookFrom
|
|
47
|
+
i as DEFAULT_LOOK,
|
|
48
|
+
a as SHAPE_INDEX,
|
|
49
|
+
d as SHAPE_ORDER,
|
|
50
|
+
u as SHAPE_OTHER,
|
|
51
|
+
v as SHAPE_PATH,
|
|
52
|
+
c as lookFrom,
|
|
53
|
+
k as resolveLook
|
|
43
54
|
};
|
|
44
55
|
//# sourceMappingURL=graph-looks.js.map
|