mcp-context-cost 0.3.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +202 -43
- package/dist/audit/audit.d.ts +101 -0
- package/dist/audit/audit.js +492 -16
- package/dist/audit/config.d.ts +38 -0
- package/dist/audit/config.js +64 -0
- package/dist/audit/deferral.d.ts +346 -0
- package/dist/audit/deferral.js +376 -0
- package/dist/audit/diff.d.ts +124 -0
- package/dist/audit/diff.js +318 -0
- package/dist/audit/run.d.ts +34 -0
- package/dist/audit/run.js +45 -2
- package/dist/cli.d.ts +21 -0
- package/dist/cli.js +141 -7
- package/dist/core/adoption.d.ts +226 -0
- package/dist/core/adoption.js +432 -0
- package/dist/core/canonical.d.ts +6 -0
- package/dist/core/canonical.js +3 -0
- package/dist/core/index.d.ts +1 -0
- package/dist/core/index.js +1 -0
- package/dist/core/session-start.d.ts +102 -0
- package/dist/core/session-start.js +186 -0
- package/dist/core/types.d.ts +8 -0
- package/dist/sweep/client.d.ts +6 -1
- package/dist/sweep/client.js +1 -0
- package/dist/sweep/dashboard.d.ts +18 -0
- package/dist/sweep/dashboard.js +74 -10
- package/dist/sweep/docker.d.ts +31 -0
- package/dist/sweep/docker.js +20 -11
- package/dist/sweep/harness-guard.d.ts +57 -0
- package/dist/sweep/harness-guard.js +144 -0
- package/dist/sweep/history.d.ts +33 -1
- package/dist/sweep/history.js +60 -5
- package/dist/sweep/regen.js +6 -1
- package/dist/sweep/report.d.ts +18 -0
- package/dist/sweep/report.js +79 -5
- package/dist/sweep/run.d.ts +43 -0
- package/dist/sweep/run.js +135 -37
- package/dist/sweep/server-pages.js +31 -6
- package/dist/sweep/session-start.d.ts +3 -0
- package/dist/sweep/session-start.js +103 -0
- package/dist/sweep/shard.d.ts +41 -0
- package/dist/sweep/shard.js +58 -0
- package/dist/sweep/sweep-all.js +57 -2
- package/package.json +3 -1
|
@@ -0,0 +1,432 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Badge adoption — how many projects outside this one actually display the
|
|
3
|
+
* badge, and on what day someone last looked.
|
|
4
|
+
*
|
|
5
|
+
* Every other number this project publishes is about MCP servers. This one is
|
|
6
|
+
* about the project itself, and it exists because the alternative is a launch
|
|
7
|
+
* that produces a number nobody can attribute. "Nobody is using the badge" and
|
|
8
|
+
* "nobody has checked whether anybody is using the badge" are the same sentence
|
|
9
|
+
* to a reader, and only one of them is a measurement. So the reading published
|
|
10
|
+
* here is a dated observation with its own working shown: the exact queries
|
|
11
|
+
* that were run, every file they turned up, and what each of those files was
|
|
12
|
+
* judged to be. A zero from this instrument means *these queries ran on this
|
|
13
|
+
* date and found none*, which is a fact. A missing reading says so in those
|
|
14
|
+
* words and publishes no number at all.
|
|
15
|
+
*
|
|
16
|
+
* ## What counts as displaying the badge
|
|
17
|
+
*
|
|
18
|
+
* A file, in a repository owned by someone else, carrying a shields.io endpoint
|
|
19
|
+
* badge that is auditable — and the project publishes two ways to make one, so
|
|
20
|
+
* the rule counts both. Either the badge's JSON is served from this
|
|
21
|
+
* repository's `badges/` directory, or the author hosts their own
|
|
22
|
+
* `badges/<name>.json` — which is what `npm run sweep` produces, and what every
|
|
23
|
+
* badge snippet this project ships tells a reader to point shields at — and
|
|
24
|
+
* links the badge back at the measurement here, which the same snippet
|
|
25
|
+
* requires.
|
|
26
|
+
*
|
|
27
|
+
* Counting only the first form would have published a zero about a spelling
|
|
28
|
+
* nobody was ever told to write: no snippet in README, in the dashboard or in
|
|
29
|
+
* the staged upstream patch produces a URL inside this repository's `badges/`,
|
|
30
|
+
* because each of them is addressed to an author measuring their own server.
|
|
31
|
+
*
|
|
32
|
+
* What is still outside the rule is a self-hosted badge that links back to
|
|
33
|
+
* nothing. Nothing in such a file names this project, so no query can nominate
|
|
34
|
+
* it — and this project's own README says a badge nobody can audit is
|
|
35
|
+
* decoration. That limit is published on the page rather than papered over; see
|
|
36
|
+
* `renderAdoptionPage`.
|
|
37
|
+
*
|
|
38
|
+
* The link is paired with the image it wraps rather than looked for anywhere in
|
|
39
|
+
* the file, because a README carrying an unrelated shields badge and, elsewhere,
|
|
40
|
+
* a sentence naming this project is not a project displaying this badge.
|
|
41
|
+
*
|
|
42
|
+
* ## Why the search is a net and not the judgement
|
|
43
|
+
*
|
|
44
|
+
* Code search matches text, and the same badge is written two ways: shields
|
|
45
|
+
* percent-encodes the `url` parameter, so the published snippet carries
|
|
46
|
+
* `raw.githubusercontent.com%2Fathakur3%2F…`, while a hand-written badge may
|
|
47
|
+
* carry the plain path. Measured against GitHub code search on 2026-08-20, the
|
|
48
|
+
* two forms do not find each other: a repository whose README carries only the
|
|
49
|
+
* encoded form of a raw URL returns 0 for the plain form of that same path,
|
|
50
|
+
* while the encoded literal returns matches. Neither query alone is the
|
|
51
|
+
* question being asked.
|
|
52
|
+
*
|
|
53
|
+
* So the queries only nominate candidates. What a candidate *is* gets decided
|
|
54
|
+
* by reading the file — `classifyFile` below, applied to content that has had
|
|
55
|
+
* its percent-encoding undone, so one rule covers both spellings. Files that
|
|
56
|
+
* name the project without displaying the badge are kept in the reading as
|
|
57
|
+
* rejections, because a zero is worth much more next to the list of things that
|
|
58
|
+
* were examined and turned down.
|
|
59
|
+
*/
|
|
60
|
+
/** Method identifier, versioned independently of the o200k methodology. */
|
|
61
|
+
export const ADOPTION_METHOD = 'badge-sightings/v1';
|
|
62
|
+
export const BADGE_SOURCE = {
|
|
63
|
+
owner: 'athakur3',
|
|
64
|
+
repo: 'mcp-context-cost',
|
|
65
|
+
branch: 'main',
|
|
66
|
+
};
|
|
67
|
+
/**
|
|
68
|
+
* The published query set. Both spellings of the badge URL are asked for
|
|
69
|
+
* separately (see the header), plus the click-through the badge is supposed to
|
|
70
|
+
* carry, plus the project's own name as the widest net — anything that names
|
|
71
|
+
* the project becomes a candidate and is then judged by its contents.
|
|
72
|
+
*/
|
|
73
|
+
export function adoptionQueries(src = BADGE_SOURCE) {
|
|
74
|
+
const raw = `raw.githubusercontent.com/${src.owner}/${src.repo}/${src.branch}/badges`;
|
|
75
|
+
return [
|
|
76
|
+
{
|
|
77
|
+
name: 'badge-endpoint-encoded',
|
|
78
|
+
q: `"${raw.replace(/\//g, '%2F')}"`,
|
|
79
|
+
why: 'the form shields produces when the published snippet is used verbatim',
|
|
80
|
+
},
|
|
81
|
+
{
|
|
82
|
+
name: 'badge-endpoint-plain',
|
|
83
|
+
q: `"${raw}"`,
|
|
84
|
+
why: 'a badge written without percent-encoding, which the encoded query cannot find',
|
|
85
|
+
},
|
|
86
|
+
{
|
|
87
|
+
name: 'link-target',
|
|
88
|
+
q: `"${src.owner}.github.io/${src.repo}"`,
|
|
89
|
+
why: 'the measurement page a badge is required to link to, however the image is written',
|
|
90
|
+
},
|
|
91
|
+
{
|
|
92
|
+
name: 'project-name',
|
|
93
|
+
q: `"${src.repo}"`,
|
|
94
|
+
why: 'the widest net: anything naming the project, judged by its contents rather than by the query',
|
|
95
|
+
},
|
|
96
|
+
];
|
|
97
|
+
}
|
|
98
|
+
/**
|
|
99
|
+
* Undo percent-encoding without throwing on the malformed sequences that turn
|
|
100
|
+
* up in real files. Decoded per-escape rather than over the whole string, so
|
|
101
|
+
* one bad `%zz` costs that escape and nothing around it.
|
|
102
|
+
*/
|
|
103
|
+
export function decodeLoose(text) {
|
|
104
|
+
return text.replace(/(?:%[0-9a-fA-F]{2})+/g, (seq) => {
|
|
105
|
+
try {
|
|
106
|
+
return decodeURIComponent(seq);
|
|
107
|
+
}
|
|
108
|
+
catch {
|
|
109
|
+
return seq;
|
|
110
|
+
}
|
|
111
|
+
});
|
|
112
|
+
}
|
|
113
|
+
/** How far either side of an image to look for the link wrapping it. */
|
|
114
|
+
const LINK_WINDOW = 400;
|
|
115
|
+
/**
|
|
116
|
+
* Every shields endpoint badge in a file, decoded, each paired with its own
|
|
117
|
+
* link target. Both spellings a README uses are read: markdown
|
|
118
|
+
* `[](target)`, which is the snippet this project publishes, and an
|
|
119
|
+
* HTML anchor wrapping an `<img>`.
|
|
120
|
+
*/
|
|
121
|
+
export function endpointBadges(text) {
|
|
122
|
+
const decoded = decodeLoose(text);
|
|
123
|
+
const out = [];
|
|
124
|
+
const re = /img\.shields\.io\/endpoint\?url=([^\s)"'<>\]]+)/gi;
|
|
125
|
+
for (const m of decoded.matchAll(re)) {
|
|
126
|
+
const start = m.index ?? 0;
|
|
127
|
+
const after = decoded.slice(start + m[0].length, start + m[0].length + LINK_WINDOW);
|
|
128
|
+
// The image's own closing paren, an optional markdown title, then the link
|
|
129
|
+
// it is nested in. Anchored, so a stray `)](` later in the file is not read
|
|
130
|
+
// as this badge's link.
|
|
131
|
+
const md = after.match(/^\s*(?:"[^"]*"|'[^']*')?\s*\)\s*\]\s*\(\s*([^\s)]+)/);
|
|
132
|
+
let linkTarget = md ? md[1] : null;
|
|
133
|
+
if (linkTarget === null) {
|
|
134
|
+
const before = decoded.slice(Math.max(0, start - LINK_WINDOW), start);
|
|
135
|
+
const anchors = [...before.matchAll(/<a\b[^>]*?href\s*=\s*["']([^"']+)["']/gi)];
|
|
136
|
+
const last = anchors[anchors.length - 1];
|
|
137
|
+
if (last && !before.slice(last.index ?? 0).includes('</a>'))
|
|
138
|
+
linkTarget = last[1];
|
|
139
|
+
}
|
|
140
|
+
out.push({ url: m[1], linkTarget });
|
|
141
|
+
}
|
|
142
|
+
return out;
|
|
143
|
+
}
|
|
144
|
+
/**
|
|
145
|
+
* Whether a badge's JSON is served from this repository — auditable by
|
|
146
|
+
* construction, whatever the badge links to. Matched without regard to case:
|
|
147
|
+
* GitHub owner and repository names are case-insensitive, a badge written
|
|
148
|
+
* `MCP-Context-Cost` renders exactly the same one, and code search found the
|
|
149
|
+
* file that way too.
|
|
150
|
+
*/
|
|
151
|
+
export function hostedHere(url, src = BADGE_SOURCE) {
|
|
152
|
+
const lower = decodeLoose(url).toLowerCase();
|
|
153
|
+
return lower.includes(`${src.owner}/${src.repo}/`.toLowerCase()) && lower.includes('/badges/');
|
|
154
|
+
}
|
|
155
|
+
/**
|
|
156
|
+
* Whether a badge's link target leads back to this project — the repository,
|
|
157
|
+
* the published pages, or a raw file in either. This is exactly what the
|
|
158
|
+
* published snippet asks an author to do with a badge whose JSON they host
|
|
159
|
+
* themselves, and it is what makes that badge auditable rather than decoration.
|
|
160
|
+
*/
|
|
161
|
+
export function linksBackToProject(target, src = BADGE_SOURCE) {
|
|
162
|
+
const lower = decodeLoose(target).toLowerCase();
|
|
163
|
+
return (lower.includes(`${src.owner}.github.io/${src.repo}`.toLowerCase()) ||
|
|
164
|
+
lower.includes(`${src.owner}/${src.repo}`.toLowerCase()));
|
|
165
|
+
}
|
|
166
|
+
/**
|
|
167
|
+
* Whether a file displays this project's badge, in either published form: the
|
|
168
|
+
* JSON served from here, or self-hosted JSON with the badge linked back at the
|
|
169
|
+
* measurement here. See the header for why both count and why the link is
|
|
170
|
+
* paired with the image rather than looked for anywhere in the file.
|
|
171
|
+
*/
|
|
172
|
+
export function displaysBadge(text, src = BADGE_SOURCE) {
|
|
173
|
+
return endpointBadges(text).some((b) => hostedHere(b.url, src) || (b.linkTarget !== null && linksBackToProject(b.linkTarget, src)));
|
|
174
|
+
}
|
|
175
|
+
/**
|
|
176
|
+
* Every shields endpoint `url` in a file whose JSON is served from this
|
|
177
|
+
* repository's `badges/` directory, decoded. The first of the two forms above,
|
|
178
|
+
* on its own.
|
|
179
|
+
*/
|
|
180
|
+
export function endpointUrls(text, src = BADGE_SOURCE) {
|
|
181
|
+
return endpointBadges(text)
|
|
182
|
+
.map((b) => b.url)
|
|
183
|
+
.filter((u) => hostedHere(u, src));
|
|
184
|
+
}
|
|
185
|
+
/**
|
|
186
|
+
* What a candidate file is. `null` when it turns out to be neither — a search
|
|
187
|
+
* index can be older than the file it points at.
|
|
188
|
+
*
|
|
189
|
+
* Case is ignored here for the same reason it is ignored above, and the first
|
|
190
|
+
* real run is why it is stated rather than assumed: a file discussing
|
|
191
|
+
* "MCP-context-cost" was found by the search and would have been thrown out by
|
|
192
|
+
* an exact-case test, which is a rejection that looks identical to a file that
|
|
193
|
+
* genuinely stopped mentioning the project.
|
|
194
|
+
*/
|
|
195
|
+
export function classifyFile(text, src = BADGE_SOURCE) {
|
|
196
|
+
if (displaysBadge(text, src))
|
|
197
|
+
return 'badge';
|
|
198
|
+
const decoded = decodeLoose(text).toLowerCase();
|
|
199
|
+
return decoded.includes(src.repo.toLowerCase()) ? 'mention' : null;
|
|
200
|
+
}
|
|
201
|
+
/** `owner/repo` → is that owner someone other than this project's? */
|
|
202
|
+
export function isThirdParty(repoFullName, src = BADGE_SOURCE) {
|
|
203
|
+
const owner = repoFullName.split('/')[0] ?? '';
|
|
204
|
+
return owner.toLowerCase() !== src.owner.toLowerCase();
|
|
205
|
+
}
|
|
206
|
+
function sightingKey(s) {
|
|
207
|
+
return JSON.stringify([s.repo, s.path]);
|
|
208
|
+
}
|
|
209
|
+
/**
|
|
210
|
+
* Date this run's sightings against the last one. A file seen before keeps its
|
|
211
|
+
* `firstSeenAt`; a file no longer found is kept with the date it was last seen
|
|
212
|
+
* rather than deleted, so a badge that disappears is visible as a badge that
|
|
213
|
+
* disappeared instead of as one that never existed.
|
|
214
|
+
*/
|
|
215
|
+
export function mergeSightings(previous, fresh, checkedAt) {
|
|
216
|
+
const byKey = new Map();
|
|
217
|
+
for (const s of previous)
|
|
218
|
+
byKey.set(sightingKey(s), s);
|
|
219
|
+
for (const f of fresh) {
|
|
220
|
+
const key = sightingKey(f);
|
|
221
|
+
const before = byKey.get(key);
|
|
222
|
+
byKey.set(key, {
|
|
223
|
+
...f,
|
|
224
|
+
firstSeenAt: before?.firstSeenAt ?? checkedAt,
|
|
225
|
+
lastSeenAt: checkedAt,
|
|
226
|
+
});
|
|
227
|
+
}
|
|
228
|
+
return [...byKey.values()].sort((a, b) => sightingKey(a).localeCompare(sightingKey(b)));
|
|
229
|
+
}
|
|
230
|
+
/** Repositories displaying the badge as of `checkedAt` — sorted, deduplicated. */
|
|
231
|
+
export function badgeRepos(sightings, checkedAt) {
|
|
232
|
+
const repos = new Set();
|
|
233
|
+
for (const s of sightings) {
|
|
234
|
+
if (s.kind === 'badge' && s.lastSeenAt === checkedAt)
|
|
235
|
+
repos.add(s.repo);
|
|
236
|
+
}
|
|
237
|
+
return [...repos].sort();
|
|
238
|
+
}
|
|
239
|
+
/**
|
|
240
|
+
* Whether this run may publish a number, and if not, why not. A query that did
|
|
241
|
+
* not answer, or one whose results were cut short, means the set of files that
|
|
242
|
+
* carry the badge was never established — and a count taken from an incomplete
|
|
243
|
+
* search is a zero that means "we did not finish", which is the exact confusion
|
|
244
|
+
* this instrument exists to remove.
|
|
245
|
+
*/
|
|
246
|
+
export function resolveCount(queries, sightings, checkedAt, unreadableCandidates = 0) {
|
|
247
|
+
if (queries.length === 0)
|
|
248
|
+
return { thirdPartyRepos: null, unresolved: 'no-query-was-run' };
|
|
249
|
+
const failed = queries.filter((q) => q.state !== 'ok');
|
|
250
|
+
if (failed.length > 0) {
|
|
251
|
+
return { thirdPartyRepos: null, unresolved: `query-did-not-answer: ${failed.map((q) => q.name).join(', ')}` };
|
|
252
|
+
}
|
|
253
|
+
const truncated = queries.filter((q) => q.truncated);
|
|
254
|
+
if (truncated.length > 0) {
|
|
255
|
+
return { thirdPartyRepos: null, unresolved: `more-results-than-collected: ${truncated.map((q) => q.name).join(', ')}` };
|
|
256
|
+
}
|
|
257
|
+
if (unreadableCandidates > 0) {
|
|
258
|
+
return { thirdPartyRepos: null, unresolved: `candidate-could-not-be-read: ${unreadableCandidates}` };
|
|
259
|
+
}
|
|
260
|
+
return { thirdPartyRepos: badgeRepos(sightings, checkedAt).length, unresolved: null };
|
|
261
|
+
}
|
|
262
|
+
/**
|
|
263
|
+
* The last completed reading to carry into this run's record: this one if it
|
|
264
|
+
* completed, otherwise whatever the previous run was carrying.
|
|
265
|
+
*/
|
|
266
|
+
export function carryResolved(previous, current) {
|
|
267
|
+
if (typeof current.thirdPartyRepos === 'number') {
|
|
268
|
+
return { checkedAt: current.checkedAt, thirdPartyRepos: current.thirdPartyRepos };
|
|
269
|
+
}
|
|
270
|
+
if (!previous)
|
|
271
|
+
return null;
|
|
272
|
+
if (typeof previous.thirdPartyRepos === 'number') {
|
|
273
|
+
return { checkedAt: previous.checkedAt, thirdPartyRepos: previous.thirdPartyRepos };
|
|
274
|
+
}
|
|
275
|
+
return previous.lastResolved ?? null;
|
|
276
|
+
}
|
|
277
|
+
/** Parse results/badge-adoption.json; anything malformed yields null, never throws. */
|
|
278
|
+
export function parseAdoption(text) {
|
|
279
|
+
let run;
|
|
280
|
+
try {
|
|
281
|
+
run = JSON.parse(text);
|
|
282
|
+
}
|
|
283
|
+
catch {
|
|
284
|
+
return null;
|
|
285
|
+
}
|
|
286
|
+
const r = run;
|
|
287
|
+
if (!r || typeof r.checkedAt !== 'string' || !Array.isArray(r.sightings))
|
|
288
|
+
return null;
|
|
289
|
+
if (!Array.isArray(r.queries) || !r.source)
|
|
290
|
+
return null;
|
|
291
|
+
return {
|
|
292
|
+
method: typeof r.method === 'string' ? r.method : ADOPTION_METHOD,
|
|
293
|
+
checkedAt: r.checkedAt,
|
|
294
|
+
source: r.source,
|
|
295
|
+
queries: r.queries,
|
|
296
|
+
candidates: typeof r.candidates === 'number' ? r.candidates : 0,
|
|
297
|
+
sightings: r.sightings,
|
|
298
|
+
thirdPartyRepos: typeof r.thirdPartyRepos === 'number' ? r.thirdPartyRepos : null,
|
|
299
|
+
unresolved: typeof r.unresolved === 'string' ? r.unresolved : null,
|
|
300
|
+
lastResolved: r.lastResolved ?? null,
|
|
301
|
+
};
|
|
302
|
+
}
|
|
303
|
+
function mdCell(s) {
|
|
304
|
+
return String(s ?? '')
|
|
305
|
+
.replace(/[|`[\]<>]/g, (c) => `\\${c}`)
|
|
306
|
+
.replace(/\r?\n/g, ' ')
|
|
307
|
+
.slice(0, 160);
|
|
308
|
+
}
|
|
309
|
+
/**
|
|
310
|
+
* A link destination, escaped but never shortened. `mdCell`'s 160-character
|
|
311
|
+
* cap is right for text a table has to hold and wrong for a URL: a truncated
|
|
312
|
+
* one is a broken link, and the whole point of listing a file is that a reader
|
|
313
|
+
* can go and look at it.
|
|
314
|
+
*/
|
|
315
|
+
function mdUrl(s) {
|
|
316
|
+
return String(s ?? '')
|
|
317
|
+
.replace(/\s/g, '%20')
|
|
318
|
+
.replace(/\(/g, '%28')
|
|
319
|
+
.replace(/\)/g, '%29')
|
|
320
|
+
.replace(/[|<>]/g, encodeURIComponent);
|
|
321
|
+
}
|
|
322
|
+
/**
|
|
323
|
+
* The page a reader opens. `null` is the state that matters most: no run on
|
|
324
|
+
* record renders as "nobody has looked", in those words, with no number — which
|
|
325
|
+
* is the whole distinction this instrument exists to make readable.
|
|
326
|
+
*/
|
|
327
|
+
export function renderAdoptionPage(run, src = BADGE_SOURCE) {
|
|
328
|
+
const out = [];
|
|
329
|
+
out.push('# Who displays the badge');
|
|
330
|
+
out.push('');
|
|
331
|
+
out.push('*Generated by `tools/measure-adoption.ts` (`npm run adoption`). Do not edit: this page is ' +
|
|
332
|
+
'rebuilt from `results/badge-adoption.json` every time someone looks.*');
|
|
333
|
+
out.push('');
|
|
334
|
+
if (!run) {
|
|
335
|
+
out.push('**Nobody has looked yet.** No reading has been taken, so there is no number here —');
|
|
336
|
+
out.push('not a zero, which would say something different and would not be true.');
|
|
337
|
+
out.push('');
|
|
338
|
+
out.push('Run `npm run adoption` to take one.');
|
|
339
|
+
out.push('');
|
|
340
|
+
return out.join('\n') + '\n';
|
|
341
|
+
}
|
|
342
|
+
const current = badgeRepos(run.sightings, run.checkedAt);
|
|
343
|
+
if (run.unresolved) {
|
|
344
|
+
out.push(`**The count could not be established on ${run.checkedAt}.** Reason: \`${run.unresolved}\`.`);
|
|
345
|
+
out.push('');
|
|
346
|
+
out.push('No number is published rather than a zero that might only mean the search stopped early.');
|
|
347
|
+
out.push('');
|
|
348
|
+
out.push(run.lastResolved
|
|
349
|
+
? `The last reading that did complete found **${run.lastResolved.thirdPartyRepos}** on ` +
|
|
350
|
+
`${run.lastResolved.checkedAt}. That one still stands; this one adds nothing to it.`
|
|
351
|
+
: 'No reading has ever completed, so nothing is known yet either way.');
|
|
352
|
+
}
|
|
353
|
+
else if (run.thirdPartyRepos === 0) {
|
|
354
|
+
out.push(`**Zero projects outside this repository display the badge**, as of ${run.checkedAt}.`);
|
|
355
|
+
out.push('');
|
|
356
|
+
out.push(`That zero was looked for: ${run.queries.length} queries ran and turned up ` +
|
|
357
|
+
`${run.candidates} third-party file(s), listed below, none of which carries the badge.`);
|
|
358
|
+
}
|
|
359
|
+
else {
|
|
360
|
+
out.push(`**${run.thirdPartyRepos} project(s) outside this repository display the badge**, as of ${run.checkedAt}:`);
|
|
361
|
+
out.push('');
|
|
362
|
+
for (const r of current)
|
|
363
|
+
out.push(`- [${mdCell(r)}](https://github.com/${r})`);
|
|
364
|
+
}
|
|
365
|
+
out.push('');
|
|
366
|
+
out.push('A reading is a dated observation, not a live counter. It is worth exactly as much as');
|
|
367
|
+
out.push('its date, and re-running it is one command.');
|
|
368
|
+
out.push('');
|
|
369
|
+
out.push('## What counts as displaying it');
|
|
370
|
+
out.push('');
|
|
371
|
+
out.push('A file in somebody else\'s repository carrying a shields.io endpoint badge that can be');
|
|
372
|
+
out.push('audited, in either of the two forms this project publishes instructions for:');
|
|
373
|
+
out.push('');
|
|
374
|
+
out.push('- the badge\'s JSON is served from this repository\'s `badges/` directory; or');
|
|
375
|
+
out.push('- the author hosts their own `badges/<name>.json` — what `npm run sweep` writes, and');
|
|
376
|
+
out.push(' what every badge snippet here tells you to point shields at — **and links the badge**');
|
|
377
|
+
out.push(' **back at the measurement**, which the same snippet requires.');
|
|
378
|
+
out.push('');
|
|
379
|
+
out.push('The link is read from the badge itself, not from anywhere in the file: a README with an');
|
|
380
|
+
out.push('unrelated shields badge that elsewhere names this project is counted as naming it, not');
|
|
381
|
+
out.push('as displaying the badge.');
|
|
382
|
+
out.push('');
|
|
383
|
+
out.push('## What was asked');
|
|
384
|
+
out.push('');
|
|
385
|
+
out.push('| query | what it is for | files found |');
|
|
386
|
+
out.push('|---|---|---|');
|
|
387
|
+
for (const q of run.queries) {
|
|
388
|
+
const hits = q.state === 'ok' ? String(q.hits ?? 0) + (q.truncated ? ' (truncated)' : '') : `not answered — ${mdCell(q.error)}`;
|
|
389
|
+
out.push(`| \`${mdCell(q.q)}\` | ${mdCell(q.why)} | ${hits} |`);
|
|
390
|
+
}
|
|
391
|
+
out.push('');
|
|
392
|
+
out.push('Run against GitHub code search, which indexes default branches of public');
|
|
393
|
+
out.push('repositories. Files in this project\'s own repositories are excluded before anything');
|
|
394
|
+
out.push('is counted.');
|
|
395
|
+
out.push('');
|
|
396
|
+
out.push('## What was found');
|
|
397
|
+
out.push('');
|
|
398
|
+
if (run.sightings.length === 0) {
|
|
399
|
+
out.push('No file outside this project named it at all.');
|
|
400
|
+
}
|
|
401
|
+
else {
|
|
402
|
+
out.push('| repository | file | what it is | first seen | last seen |');
|
|
403
|
+
out.push('|---|---|---|---|---|');
|
|
404
|
+
for (const s of run.sightings) {
|
|
405
|
+
const what = s.kind === 'badge' ? '**displays the badge**' : 'names the project, no badge';
|
|
406
|
+
out.push(`| [${mdCell(s.repo)}](https://github.com/${s.repo}) | [${mdCell(s.path)}](${mdUrl(s.url)}) | ` +
|
|
407
|
+
`${what} | ${s.firstSeenAt} | ${s.lastSeenAt} |`);
|
|
408
|
+
}
|
|
409
|
+
out.push('');
|
|
410
|
+
out.push('A row whose *last seen* is older than the date above was found by an earlier reading');
|
|
411
|
+
out.push('and not by this one.');
|
|
412
|
+
}
|
|
413
|
+
out.push('');
|
|
414
|
+
out.push('## What this cannot see');
|
|
415
|
+
out.push('');
|
|
416
|
+
out.push('- A self-hosted badge that links back to nothing — the one badge shape above that is');
|
|
417
|
+
out.push(' not counted. Nothing in such a file names this project, so no query can nominate it');
|
|
418
|
+
out.push(' either, and a badge carrying no route to the measurement behind it is the kind this');
|
|
419
|
+
out.push(' project calls decoration.');
|
|
420
|
+
out.push('- Anything outside public GitHub: private repositories, other forges, documentation');
|
|
421
|
+
out.push(' sites whose source is not on GitHub, and repositories the code search index has not');
|
|
422
|
+
out.push(' reached.');
|
|
423
|
+
out.push('- Whether anybody looked at a badge. This counts files that display one, which is a');
|
|
424
|
+
out.push(' different question from reach.');
|
|
425
|
+
out.push('');
|
|
426
|
+
out.push(`Method \`${run.method}\`, against \`${src.owner}/${src.repo}\` on branch \`${src.branch}\`. ` +
|
|
427
|
+
'The raw reading, including every query and every file examined, is in ' +
|
|
428
|
+
'[`results/badge-adoption.json`](https://github.com/' +
|
|
429
|
+
`${src.owner}/${src.repo}/blob/${src.branch}/results/badge-adoption.json).`);
|
|
430
|
+
out.push('');
|
|
431
|
+
return out.join('\n');
|
|
432
|
+
}
|
package/dist/core/canonical.d.ts
CHANGED
|
@@ -20,6 +20,12 @@ export declare function measureTools(tools: unknown[], meta: {
|
|
|
20
20
|
launchCommand?: string;
|
|
21
21
|
envVarNames?: string[];
|
|
22
22
|
measuredAt?: string;
|
|
23
|
+
/**
|
|
24
|
+
* The initialize `instructions` string, or null when the server returned
|
|
25
|
+
* none. Omit it only when nothing was captured: an omitted field records
|
|
26
|
+
* "never asked", which session-start.ts refuses to read as zero.
|
|
27
|
+
*/
|
|
28
|
+
instructions?: string | null;
|
|
23
29
|
}): Measurement;
|
|
24
30
|
export declare function failedMeasurement(status: Exclude<MeasurementStatus, 'measured' | 'dynamic'>, meta: {
|
|
25
31
|
serverName: string;
|
package/dist/core/canonical.js
CHANGED
|
@@ -52,6 +52,9 @@ export function measureTools(tools, meta) {
|
|
|
52
52
|
serverVersion: meta.serverVersion,
|
|
53
53
|
launchCommand: meta.launchCommand,
|
|
54
54
|
envVarNames: meta.envVarNames,
|
|
55
|
+
// Left undefined (and so absent from the JSON) when the caller had nothing
|
|
56
|
+
// to record, which is exactly how a pre-field measurement reads.
|
|
57
|
+
serverInstructions: meta.instructions,
|
|
55
58
|
};
|
|
56
59
|
}
|
|
57
60
|
export function failedMeasurement(status, meta) {
|
package/dist/core/index.d.ts
CHANGED
package/dist/core/index.js
CHANGED
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
import type { Measurement } from './types.js';
|
|
2
|
+
/** Method identifier, versioned independently of METHODOLOGY_VERSION. */
|
|
3
|
+
export declare const SESSION_START_METHOD = "deferred-load/v1";
|
|
4
|
+
/**
|
|
5
|
+
* Tool names in server-returned order. A tool without a usable name is dropped
|
|
6
|
+
* rather than given a placeholder — the same rule `toAnthropicTools` follows,
|
|
7
|
+
* for the same reason: an invented name would change the count being published.
|
|
8
|
+
*/
|
|
9
|
+
export declare function toolNames(raw: unknown[]): string[];
|
|
10
|
+
/**
|
|
11
|
+
* o200k count of the canonical tool-name list: `JSON.stringify` over the name
|
|
12
|
+
* array, the same serialization discipline as the headline number, so both
|
|
13
|
+
* re-derive from the same capture with the same five lines.
|
|
14
|
+
*/
|
|
15
|
+
export declare function toolNameTokens(raw: unknown[]): number;
|
|
16
|
+
/**
|
|
17
|
+
* The two halves are counted separately and added, rather than counted over one
|
|
18
|
+
* concatenated string. Concatenation would let boundary tokens merge across the
|
|
19
|
+
* seam, so the published total would not equal the sum of the parts printed
|
|
20
|
+
* beside it — a discrepancy of a token or two that no reader could account for.
|
|
21
|
+
*/
|
|
22
|
+
export declare function sessionStartTokens(raw: unknown[], instructions: string | null): number;
|
|
23
|
+
/** One server's instructions, captured beside — not inside — a measurement. */
|
|
24
|
+
export interface SessionStartRow {
|
|
25
|
+
/** Exactly what `initialize` returned; '' when the server returned none. */
|
|
26
|
+
instructions: string;
|
|
27
|
+
instructionsTokens: number;
|
|
28
|
+
/** SHA-256 of the instructions bytes — the dispute artifact, as ever. */
|
|
29
|
+
instructionsSha256: string;
|
|
30
|
+
/**
|
|
31
|
+
* `canonicalSha256` of the measurement this capture stood beside. A re-sweep
|
|
32
|
+
* moves that hash, which marks the row stale: instructions are a property of
|
|
33
|
+
* the same server build that produced the tools, so a changed tool set is
|
|
34
|
+
* reason enough to stop trusting the instructions captured with the old one.
|
|
35
|
+
*/
|
|
36
|
+
capturedSha256: string | null;
|
|
37
|
+
serverVersion?: string;
|
|
38
|
+
/** Set when the server could not be reached; no numbers are published. */
|
|
39
|
+
error?: string;
|
|
40
|
+
}
|
|
41
|
+
export interface SessionStartRun {
|
|
42
|
+
method: string;
|
|
43
|
+
/** UTC day the instructions were captured (YYYY-MM-DD). */
|
|
44
|
+
measuredAt: string;
|
|
45
|
+
/** How the servers were isolated during the capture. */
|
|
46
|
+
isolation?: string;
|
|
47
|
+
servers: Record<string, SessionStartRow>;
|
|
48
|
+
}
|
|
49
|
+
/** Where the instructions half of a figure came from — or that it is missing. */
|
|
50
|
+
export type InstructionsSource = 'measurement' | 'capture' | 'not-captured';
|
|
51
|
+
export interface SessionStartLoad {
|
|
52
|
+
toolCount: number;
|
|
53
|
+
toolNameTokens: number;
|
|
54
|
+
/** null when instructions have never been captured for this server. */
|
|
55
|
+
instructionsTokens: number | null;
|
|
56
|
+
/** Names plus instructions; names alone when instructions are unknown. */
|
|
57
|
+
totalTokens: number;
|
|
58
|
+
/** True when `totalTokens` is a lower bound, not a measurement. */
|
|
59
|
+
isFloor: boolean;
|
|
60
|
+
instructionsSource: InstructionsSource;
|
|
61
|
+
}
|
|
62
|
+
/**
|
|
63
|
+
* The instructions recorded *inside* a measurement, distinguishing three states
|
|
64
|
+
* that JSON round-trips faithfully:
|
|
65
|
+
* - a string → captured, the server sent this
|
|
66
|
+
* - null → captured, the server sent none
|
|
67
|
+
* - undefined → never captured (every measurement predating the field)
|
|
68
|
+
*
|
|
69
|
+
* Absent-means-unknown rather than absent-means-zero, on the precedent set by
|
|
70
|
+
* history.csv's `isolation` column: an old row reads as unknown and is never
|
|
71
|
+
* back-filled from a value it did not record.
|
|
72
|
+
*/
|
|
73
|
+
export declare function measuredInstructions(m: Measurement): string | undefined;
|
|
74
|
+
/**
|
|
75
|
+
* Resolve one server's session-start load from its measurement and, if it has
|
|
76
|
+
* one, the instructions captured beside it.
|
|
77
|
+
*
|
|
78
|
+
* Precedence is measurement over capture and never the other way round: the
|
|
79
|
+
* measurement's instructions came off the same server process as its tools, in
|
|
80
|
+
* the same run, so it cannot be stale relative to itself. The side capture is
|
|
81
|
+
* the backfill for measurements taken before the field existed, and it is used
|
|
82
|
+
* only while it still points at the measurement on disk.
|
|
83
|
+
*
|
|
84
|
+
* Returns null for a measurement with no capture to read names from — a
|
|
85
|
+
* `startup-failure` has no session-start load because it has no session.
|
|
86
|
+
*/
|
|
87
|
+
export declare function sessionStartLoad(m: Measurement, row?: SessionStartRow): SessionStartLoad | null;
|
|
88
|
+
/**
|
|
89
|
+
* A side capture is usable only if it carries a number and still points at the
|
|
90
|
+
* measurement on disk. Unlike a stale divergence row — which is hidden, because
|
|
91
|
+
* there is nothing else to print — a stale row here degrades the figure to its
|
|
92
|
+
* names-only floor. The column never blanks; it only ever stops claiming to
|
|
93
|
+
* know the half it no longer knows.
|
|
94
|
+
*/
|
|
95
|
+
export declare function isCurrentInstructions(row: SessionStartRow | undefined, canonicalSha256: string | null): row is SessionStartRow;
|
|
96
|
+
/** Build a row from a freshly captured instructions string. */
|
|
97
|
+
export declare function toSessionStartRow(instructions: string | null, meta: {
|
|
98
|
+
capturedSha256: string | null;
|
|
99
|
+
serverVersion?: string;
|
|
100
|
+
}): SessionStartRow;
|
|
101
|
+
/** Parse results/session-start.json; anything malformed yields null, never throws. */
|
|
102
|
+
export declare function parseSessionStart(text: string): SessionStartRun | null;
|