litura-app 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,38 @@
1
+ import { completeText } from './pi.js';
2
+ import { mergeReviewFindings, parseReviewResponse, validateReviewFindings } from './review.js';
3
+
4
+ async function requestReviewPass({ systemPrompt, userPrompt, allowedCodes, source, selection, attempts }) {
5
+ let lastError;
6
+ for (let attempt = 0; attempt < attempts; attempt++) {
7
+ try {
8
+ const raw = await completeText({
9
+ systemPrompt,
10
+ userPrompt: attempt === 0
11
+ ? userPrompt
12
+ : `${userPrompt}\n\n---\n\nRESPONSE RETRY: ${lastError.message}. Return the valid JSON array required by the system prompt, even when it is empty.`,
13
+ selection,
14
+ maxTokens: 4000,
15
+ signal: AbortSignal.timeout(90_000),
16
+ });
17
+ const findings = parseReviewResponse(raw);
18
+ validateReviewFindings(findings, allowedCodes, source);
19
+ return findings;
20
+ } catch (error) {
21
+ lastError = error;
22
+ if (attempt + 1 < attempts) {
23
+ await new Promise(resolve => setTimeout(resolve, 300 * (attempt + 1)));
24
+ }
25
+ }
26
+ }
27
+ throw lastError;
28
+ }
29
+
30
+ export async function requestReview({ systemPrompt, userPrompt, prompts, selection, attempts = 3 }) {
31
+ const requests = prompts ?? [{ systemPrompt, userPrompt }];
32
+ const groups = await Promise.all(requests.map(prompt => requestReviewPass({
33
+ ...prompt,
34
+ selection,
35
+ attempts,
36
+ })));
37
+ return mergeReviewFindings(groups);
38
+ }
@@ -0,0 +1,93 @@
1
+ export const REVIEW_CODES = [
2
+ 'level-1-whole-essay',
3
+ 'level-2-introduction',
4
+ 'level-3-index-discussion',
5
+ 'level-4-fractal-structure',
6
+ 'level-5-point-placement',
7
+ 'level-6-key-terms',
8
+ 'level-7-paragraph-flow',
9
+ 'generic-prose',
10
+ ];
11
+
12
+ export function reviewCodesForPass(targeted, phase) {
13
+ return phase === 'global' ? REVIEW_CODES.slice(0, 4)
14
+ : phase === 'local' ? REVIEW_CODES.slice(targeted ? 5 : 4) : REVIEW_CODES;
15
+ }
16
+
17
+ export function buildReviewTask({ targeted = false, phase = 'all' } = {}) {
18
+ const allowedCodes = reviewCodesForPass(targeted, phase);
19
+ const scope = targeted
20
+ ? 'Audit ONLY the passages under PASSAGES TO AUDIT. The draft is context you must read but must not flag. ' +
21
+ 'Every quote must be copied from those passages. For structure, use only level-6-key-terms, ' +
22
+ 'level-7-paragraph-flow, or generic-prose; reserve larger structural findings for a full review. '
23
+ : '';
24
+ const phaseScope = phase === 'global'
25
+ ? 'This is the global structure pass. Use only levels 1 through 4; do not report paragraph-level or generic-prose findings. '
26
+ : phase === 'local'
27
+ ? `This is the local prose and paragraph pass. Use only ${allowedCodes.join(', ')}; do not diagnose whole-essay structure. `
28
+ : '';
29
+ const globalRules = phase === 'local' ? '' :
30
+ 'A recommendation introduced with should or must but no destabilizing problem or cost is a clear level-2-introduction failure. ' +
31
+ 'At level 2, cost means the consequence of the opening problem for the reader or subject; do not require an obstacle, objection, or implementation cost for the recommendation. Assess the opening unit as a whole, not every later recommendation paragraph as a new introduction. ' +
32
+ 'At level 3, an enumeration is an index only when its wording promises the structure or contents that follow. A list whose items are all developed in the immediately following sentence or clauses has kept its local promise; do not require the rest of the essay to use that list as its outline. ' +
33
+ 'Do not use level 3 or 4 for one isolated unrelated sentence inside an otherwise relevant paragraph; the local pass handles that as level 7. Reserve level 4 for a whole paragraph or section that fails to support its containing unit. ' +
34
+ 'Final whole-essay check: when a title recommends an action but the ending says that action remains undecided or unavailable, report level-1-whole-essay because the opening problem remains unresolved. ';
35
+ const localRules = phase === 'global' ? '' :
36
+ 'Run these three local diagnostic passes unless the genre is excluded; do not return [] merely because grammar is clean. ' +
37
+ 'At level 5, mechanically locate the main claim of every explanatory paragraph with at least five sentences. If it is neither the first nor the last sentence, report level-5-point-placement even when the paragraph is otherwise coherent. ' +
38
+ 'In an argumentative paragraph, treat its primary should or must recommendation as the main claim; do not mistake an earlier descriptive topic sentence for that claim. ' +
39
+ 'At level 6, list the central terms established by an opening and verify that the discussion carries them through repetition, a clear synonym, a pronoun, or an unambiguous conceptual continuation. Report an unannounced replacement term only when it makes the referent or organising promise unclear. Terms introduced and explained inside the same opening unit are fulfilled locally; do not require them to recur later unless the draft explicitly announces them as continuing threads. ' +
40
+ 'If a missing term is itself an item in an explicit opening index, omit the level-6 finding because the global pass handles the broken promise at level 3. ' +
41
+ 'At level 7, compare every pair of consecutive sentence themes. If a sentence introduces an unrelated subject with no shared term, pronoun, synonym, or conceptual bridge, report level-7-paragraph-flow even in a three-sentence paragraph. A clear contrast, cause and effect, example, general-to-specific move, instruction, or catalogue supplies a conceptual bridge without repeated wording. ' +
42
+ 'An unexplained switch to a different entity or domain is a clear level-7 failure; do not assume intentional disorientation unless the draft signals a narrative or artistic purpose. ' +
43
+ 'Before reporting level 7, test the ideas rather than vocabulary: a move from a misuse or problem to its proper use or remedy, or from a claim to an instruction about the same activity, is a conceptual bridge. ' +
44
+ 'In aphoristic, historical, or deliberately list-like prose, do not flag a grammatical subject change by itself when the ideas remain plainly related. If an unrelated passage forms a whole paragraph or section rather than one sentence, omit level 7 because the global pass handles it at level 4. ' +
45
+ 'Final local check before answering: in every five-sentence-or-longer argumentative paragraph, a should/must recommendation with at least two sentences before and after it requires level-5-point-placement. ';
46
+
47
+ return (
48
+ 'You are a sharp human editor auditing a draft for writing and reader-structure problems. ' +
49
+ scope +
50
+ phaseScope +
51
+ 'Detect only; do not rewrite the draft, score it, or guess who wrote it. ' +
52
+ 'Inspect only the levels assigned to this pass, in order. Definitions of other levels below are for disambiguation, never permission to report them. When one break could fit several levels, ' +
53
+ 'use the earliest affected level: if a final paragraph changes subject instead of resolving the opening problem, always use level-1-whole-essay, never level-4-fractal-structure; ' +
54
+ 'if an explicit opening index promises items the discussion omits, use level-3-index-discussion rather than level-4-fractal-structure or level-6-key-terms. ' +
55
+ 'Use exactly one code for each finding: ' +
56
+ 'level-1-whole-essay for a misleading title, an essay that does not make readers care, or an ending that fails to resolve the opening problem; ' +
57
+ 'level-2-introduction for missing common ground, a weak or unfair status quo, no destabilizing problem or cost, or no clear point/solution; ' +
58
+ 'level-3-index-discussion when a paragraph or section opening fails to set expectations or its discussion breaks that promise; ' +
59
+ 'level-4-fractal-structure when a section or paragraph does not support the point of its containing unit; ' +
60
+ 'level-5-point-placement when a unit has no point, buries it in the middle, or states it at both beginning and end; ' +
61
+ 'level-6-key-terms when central terms announced by an opening disappear, arrive unannounced, or fail to form a coherent thematic string; ' +
62
+ 'level-7-paragraph-flow for an abrupt old-to-new information break. Constant-topic, linking, super-theme, and preview-and-develop are all valid flows; ' +
63
+ 'generic-prose for throat-clearing, vague attribution, empty puffery, faux insight, generic filler, binary contrast, robotic rhythm, repetitive recap, dramatic fragmentation, stacked hedging, clustered scare quotes or all-caps emphasis, or an abstraction given mind-like agency. ' +
64
+ 'Apply essay and introduction codes only to argumentative or explanatory drafts with enough text to support the diagnosis. ' +
65
+ globalRules +
66
+ localRules +
67
+ 'Do not force these patterns onto a short answer, list, reference material, dialogue, narrative turn, or intentional disorientation. ' +
68
+ 'Do not flag polished grammar, formal vocabulary, proper names, quotations, or one isolated stylistic choice. ' +
69
+ 'Passive voice is valid when the actor is unknown or irrelevant. Ordinary product verbs such as report shows, form submits, and filter narrows describe real behavior and are not mind-like agency. ' +
70
+ 'Preserve unusual details, humor, uncertainty, bluntness, cadence, and useful roughness. ' +
71
+ 'Prefer the most specific code and do not report the same underlying problem at several levels. ' +
72
+ 'Before returning, deduplicate by cause: if a higher-level finding already explains a lower-level symptom, omit the lower-level finding. ' +
73
+ 'A titleless single-paragraph excerpt is not a whole essay or a section, so do not assign levels 1, 2, or 4 to it. ' +
74
+ 'A structural fix may tell the writer to move, connect, add, remove, or rename material; it need not be achievable by replacing the quote alone. ' +
75
+ 'Return ONLY a JSON array of at most 8 objects with string fields code, quote, pattern, reason, fix. ' +
76
+ `code must be one of: ${allowedCodes.join(', ')}. ` +
77
+ 'The response must be valid JSON: use double-quoted keys and strings and escape any quotation marks inside a string. ' +
78
+ 'quote must be the shortest exact contiguous quote that identifies the problem. fix is a brief direction, not a rewrite. ' +
79
+ // The card is read at a glance while the draft stays on screen; a
80
+ // paragraph of explanation there is never read at all.
81
+ 'Keep them short: pattern at most 4 words, reason at most 12 words, fix at most 12 words. ' +
82
+ 'No preamble, no restating the quote, no hedging — name the problem and the move. ' +
83
+ 'Use the language of the draft for pattern, reason, and fix. Return [] when there are no strong findings.'
84
+ );
85
+ }
86
+
87
+ export function buildReviewUser({ document, target = '', context = '' }) {
88
+ return [
89
+ context ? `VOICE OR REFERENCE CONTEXT (do not audit):\n${context}` : '',
90
+ `${target ? 'DRAFT (context only)' : 'DRAFT TO AUDIT'}:\n${document}`,
91
+ target ? `PASSAGES TO AUDIT:\n${target}` : '',
92
+ ].filter(Boolean).join('\n\n---\n\n');
93
+ }
package/review.js ADDED
@@ -0,0 +1,265 @@
1
+ import { REVIEW_CODES } from './review-prompt.js';
2
+
3
+ export function parseReviewResponse(raw) {
4
+ const match = raw.match(/\[[\s\S]*\]/);
5
+ if (!match && /^\s*no (?:strong )?findings?\.?\s*$/i.test(raw)) return [];
6
+ if (!match) throw new Error('No findings array in response');
7
+ const value = JSON.parse(match[0]);
8
+ if (!Array.isArray(value)) throw new Error('Expected findings array');
9
+ return value.map(item => {
10
+ if (!item || typeof item !== 'object') throw new Error('Each finding must be an object');
11
+ const fields = ['quote', 'pattern', 'reason', 'fix'];
12
+ if (!fields.every(key => typeof item[key] === 'string' && item[key].trim())) {
13
+ throw new Error('Each finding needs non-empty quote, pattern, reason, and fix strings');
14
+ }
15
+ const finding = Object.fromEntries(fields.map(key => [key, item[key].trim()]));
16
+ const code = String(item.code ?? '').trim();
17
+ return { code: REVIEW_CODES.includes(code) ? code : 'unclassified', ...finding };
18
+ }).slice(0, 8);
19
+ }
20
+
21
+ export function validateReviewFindings(findings, allowedCodes = REVIEW_CODES, source) {
22
+ for (const finding of findings) {
23
+ if (!allowedCodes.includes(finding.code)) throw new Error(`Code ${finding.code} is outside this pass; allowed: ${allowedCodes.join(', ')}`);
24
+ if (source !== undefined && !source.includes(finding.quote)) throw new Error('Quote must be copied exactly from the passages being audited');
25
+ }
26
+ }
27
+
28
+ export function mergeReviewFindings(groups) {
29
+ const findings = groups.flat();
30
+ const promises = findings.filter(finding => finding.code === 'level-3-index-discussion');
31
+ const seen = [];
32
+ return findings.filter(finding => {
33
+ // Both passes can diagnose the same opening list. Its broken promise owns
34
+ // the finding; retain key-term findings elsewhere in the document.
35
+ if (finding.code === 'level-6-key-terms' && promises.some(promise =>
36
+ promise.quote.includes(finding.quote) || finding.quote.includes(promise.quote))) return false;
37
+ if (seen.some(other => other.code === finding.code &&
38
+ (other.quote.includes(finding.quote) || finding.quote.includes(other.quote)))) return false;
39
+ seen.push(finding);
40
+ return true;
41
+ }).slice(0, 8);
42
+ }
43
+
44
+ // ─── Local style scoring ─────────────────────────────────────────────────────
45
+ //
46
+ // A cheap, deterministic read on how much a passage smells of AI, computed
47
+ // without a model call. Two uses: a live readout for the writer, and a filter
48
+ // that keeps obviously-clean sentences from costing a review request.
49
+ //
50
+ // Word lists and anchors below are heuristics, not trained thresholds.
51
+
52
+ const TELL_WORDS_STRONG = [
53
+ 'delve', 'delves', 'delving', 'tapestry', 'testament', 'underscore', 'underscores',
54
+ 'underscoring', 'leverage', 'leverages', 'leveraging', 'multifaceted', 'realm',
55
+ 'interplay', 'seamless', 'seamlessly', 'groundbreaking', 'nestled',
56
+ ];
57
+
58
+ const TELL_WORDS_WEAK = [
59
+ 'crucial', 'pivotal', 'vibrant', 'robust', 'foster', 'fosters', 'fostering',
60
+ 'enhance', 'enhances', 'enhancing', 'showcase', 'showcases', 'showcasing',
61
+ 'garner', 'bolster', 'utilize', 'utilizes', 'moreover', 'furthermore', 'notably',
62
+ 'transformative', 'innovative', 'boasts', 'renowned', 'breathtaking', 'stunning',
63
+ ];
64
+
65
+ const TELL_PHRASES = [
66
+ /\bin today'?s [a-z-]+ world\b/i,
67
+ /\bat the end of the day\b/i,
68
+ /\bexperts? (?:agree|say|believe)\b/i,
69
+ /\bstudies show\b/i,
70
+ /\bit is important to note\b/i,
71
+ /\bnot (?:just|only)\b[^.!?]{0,60}\bbut\b/i,
72
+ /\b(?:serves|stands) as a\b/i,
73
+ /\bplays? a (?:vital|crucial|pivotal|key|significant) role\b/i,
74
+ /\bat its core\b/i,
75
+ /\bthe real question is\b/i,
76
+ /\blet'?s (?:dive|explore|break this down)\b/i,
77
+ /\bhere'?s what you need to know\b/i,
78
+ /\ba testament to\b/i,
79
+ /\bevolving landscape\b/i,
80
+ /\bin order to\b/i,
81
+ /\bdue to the fact that\b/i,
82
+ /\b(?:dashboard|data|design|roadmap|platform|system|tool|app|algorithm) (?:understands?|knows?|decides?|wants?|believes?|cares?)\b/i,
83
+ /\b(?:could|may|might|arguably|potentially|possibly)(?:\s+\w+){0,2}\s+(?:potentially|possibly|arguably|perhaps|may|might)\b/i,
84
+ /(?:["“][^"”\n]{1,24}["”][\s,]*){3,}|\b[A-Z]{3,}(?:\s+[A-Z]{3,}){2,}\b/,
85
+ ];
86
+
87
+ const WORD_RE = /\p{L}[\p{L}\p{N}'’-]*/gu;
88
+
89
+ // The word lists are English. On a Cyrillic or other non-Latin draft they would
90
+ // report zero tells, so callers must not gate on a score that cannot see them.
91
+ export function isLatinScript(text) {
92
+ const letters = text.match(/\p{L}/gu) ?? [];
93
+ if (!letters.length) return true;
94
+ return letters.filter(ch => /[\p{Script=Latin}]/u.test(ch)).length / letters.length > 0.5;
95
+ }
96
+
97
+ function splitAllSentences(text) {
98
+ return text.split(/(?<=[.!?…])\s+|\n+/).map(part => part.trim()).filter(Boolean);
99
+ }
100
+
101
+ export function styleMetrics(text) {
102
+ const words = (text.match(WORD_RE) ?? []).map(word => word.toLowerCase());
103
+ const sentences = splitAllSentences(text);
104
+ const lengths = sentences.map(sentence => (sentence.match(WORD_RE) ?? []).length).filter(Boolean);
105
+
106
+ const mean = lengths.length ? lengths.reduce((a, b) => a + b, 0) / lengths.length : 0;
107
+ const variance = lengths.length > 1
108
+ ? lengths.reduce((a, b) => a + (b - mean) ** 2, 0) / lengths.length
109
+ : 0;
110
+ // Coefficient of variation of sentence length. Human prose swings; AI prose
111
+ // settles into an even mid-length cadence.
112
+ const burstiness = mean ? Math.sqrt(variance) / mean : 0;
113
+
114
+ // Moving-average type-token ratio: lexical variety that does not sag purely
115
+ // because the text got longer, unlike raw TTR.
116
+ const window = 50;
117
+ let diversity;
118
+ if (words.length <= window) {
119
+ diversity = words.length ? new Set(words).size / words.length : 0;
120
+ } else {
121
+ let sum = 0;
122
+ for (let i = 0; i + window <= words.length; i++) {
123
+ sum += new Set(words.slice(i, i + window)).size / window;
124
+ }
125
+ diversity = sum / (words.length - window + 1);
126
+ }
127
+
128
+ let repetition = 0;
129
+ if (words.length >= 3) {
130
+ const grams = [];
131
+ for (let i = 0; i + 3 <= words.length; i++) grams.push(words.slice(i, i + 3).join(' '));
132
+ repetition = 1 - new Set(grams).size / grams.length;
133
+ }
134
+
135
+ const strongHits = words.filter(word => TELL_WORDS_STRONG.includes(word));
136
+ const weakHits = words.filter(word => TELL_WORDS_WEAK.includes(word));
137
+ // The matched text itself, not the pattern — a score the writer cannot trace
138
+ // back to words in their own draft is just a number to argue with.
139
+ const phraseHits = TELL_PHRASES.map(pattern => text.match(pattern)?.[0]).filter(Boolean);
140
+ const strong = strongHits.length;
141
+ const weak = weakHits.length;
142
+ const phrases = phraseHits.length;
143
+ const tells = strong + weak + phrases;
144
+ const hits = [...new Set([...phraseHits, ...strongHits, ...weakHits].map(hit => hit.trim()))];
145
+ // Phrases weigh most: "in today's fast-paced world" is a whole tell, `robust`
146
+ // on its own is barely one.
147
+ const tellDensity = words.length ? (strong * 2 + weak + phrases * 4) / words.length : 0;
148
+
149
+ return { words: words.length, sentences: lengths.length, burstiness, diversity, repetition, tells, tellDensity, hits };
150
+ }
151
+
152
+ const ANCHORS = { burstinessHuman: 0.6, diversityLow: 0.3, diversityHigh: 0.72, repetitionMax: 0.18, tellMax: 0.05 };
153
+ const WEIGHTS = { tells: 0.45, burstiness: 0.25, diversity: 0.17, repetition: 0.13 };
154
+
155
+ // 0 = reads clean, 100 = every axis maxed. Structural axes need a paragraph to
156
+ // mean anything, so on short passages the score leans on the lexical axis alone.
157
+ export function styleScore(text) {
158
+ const metrics = styleMetrics(text);
159
+ if (!metrics.words) return { ...metrics, score: 0, structural: false };
160
+
161
+ const clamp = value => Math.max(0, Math.min(1, value));
162
+ const lexical = clamp(metrics.tellDensity / ANCHORS.tellMax);
163
+ const structural = metrics.words >= 40 && metrics.sentences >= 3;
164
+
165
+ if (!structural) {
166
+ return { ...metrics, score: Math.round(100 * lexical), structural };
167
+ }
168
+
169
+ const raw =
170
+ WEIGHTS.tells * lexical +
171
+ WEIGHTS.burstiness * clamp((ANCHORS.burstinessHuman - metrics.burstiness) / ANCHORS.burstinessHuman) +
172
+ WEIGHTS.diversity * clamp((ANCHORS.diversityHigh - metrics.diversity) / (ANCHORS.diversityHigh - ANCHORS.diversityLow)) +
173
+ WEIGHTS.repetition * clamp(metrics.repetition / ANCHORS.repetitionMax);
174
+ return { ...metrics, score: Math.round(100 * raw), structural };
175
+ }
176
+
177
+ // ── Rewrite scope ──
178
+ //
179
+ // A short selection sent alongside a whole draft reads to the model as "clean
180
+ // this up", and it returns a rewritten draft that cannot be substituted back.
181
+ // Marking the span and bounding the answer keeps a replacement a replacement.
182
+ //
183
+ export const SELECT_OPEN = '⟦';
184
+ export const SELECT_CLOSE = '⟧';
185
+
186
+ export const SELECT_SLOT = '___';
187
+
188
+ // The client sends the caret-accurate offset; a bare quote may occur more than
189
+ // once, so fall back to the first match only when the offset does not fit.
190
+ function selectionAt(document, selected, from) {
191
+ if (!selected) return -1;
192
+ return Number.isInteger(from) && document.slice(from, from + selected.length) === selected
193
+ ? from
194
+ : document.indexOf(selected);
195
+ }
196
+
197
+ export function markSelection(document, selected, from) {
198
+ const at = selectionAt(document, selected, from);
199
+ if (at < 0) return document;
200
+ return document.slice(0, at) + SELECT_OPEN + selected + SELECT_CLOSE + document.slice(at + selected.length);
201
+ }
202
+
203
+ // The sentence with the selection cut out. Scope alone is not enough: told only
204
+ // to keep it short, the model returns wording that repeats what already follows
205
+ // ("experts agree" → "collaboration is key" in front of "collaboration is key").
206
+ // Shown the gap it has to fill, it stops writing the neighbours.
207
+ export function selectionSlot(document, selected, from) {
208
+ const at = selectionAt(document, selected, from);
209
+ if (at < 0) return null;
210
+ const end = at + selected.length;
211
+ const sentence = completedSentences(document).find(item => item.from <= at && item.to >= end)
212
+ ?? { from: Math.max(0, at - 90), to: Math.min(document.length, end + 90) };
213
+ return document.slice(sentence.from, at).trimStart() + SELECT_SLOT + document.slice(end, sentence.to).trimEnd();
214
+ }
215
+
216
+ // Room to rephrase, not to absorb the sentences around it.
217
+ export const variantLimit = selected => Math.max(selected.length * 3, selected.length + 60);
218
+
219
+ export function parseVariants(raw) {
220
+ const match = raw.match(/\[[\s\S]*\]/);
221
+ if (!match) throw new Error('No JSON array in response');
222
+ const variants = JSON.parse(match[0]);
223
+ if (!Array.isArray(variants) || variants.length < 3) throw new Error('Expected 3 variants');
224
+ return variants.slice(0, 3).map(String);
225
+ }
226
+
227
+ // Models restate the last words before continuing ("…the queue" → "queue began
228
+ // draining"). Asking them not to is unreliable; cutting the overlap is not.
229
+ export function trimOverlap(prefix, suggestion) {
230
+ const max = Math.min(prefix.length, suggestion.length, 60);
231
+ for (let n = max; n >= 3; n--) {
232
+ if (prefix.slice(-n).toLowerCase() === suggestion.slice(0, n).toLowerCase()) {
233
+ return suggestion.slice(n);
234
+ }
235
+ }
236
+ return suggestion;
237
+ }
238
+
239
+ // Sentences that are finished, with offsets. A trailing fragment with no
240
+ // terminal punctuation is still being typed and is deliberately left out.
241
+ // ponytail: naive split, "т.д." and "Dr." break it — swap in Intl.Segmenter
242
+ // if false splits ever cost a real review.
243
+ export function completedSentences(document) {
244
+ return [...document.matchAll(/[^.!?…\n]*[.!?…]+["'»”’)\]]*/g)]
245
+ .map(match => ({ text: match[0].trim(), from: match.index, to: match.index + match[0].length }))
246
+ .filter(sentence => sentence.text.length > 0);
247
+ }
248
+
249
+ // Anchor each quote to a free spot in the document. `occupied` holds ranges
250
+ // already carrying a finding — incremental reviews must not double-mark them.
251
+ export function locateFindings(document, findings, occupied = []) {
252
+ const taken = occupied.map(({ from, to }) => ({ from, to }));
253
+ const located = [];
254
+ for (const finding of findings) {
255
+ let from = document.indexOf(finding.quote);
256
+ while (from >= 0 && taken.some(range => from < range.to && from + finding.quote.length > range.from)) {
257
+ from = document.indexOf(finding.quote, from + 1);
258
+ }
259
+ if (from < 0) continue;
260
+ const range = { ...finding, from, to: from + finding.quote.length };
261
+ taken.push(range);
262
+ located.push(range);
263
+ }
264
+ return located.sort((a, b) => a.from - b.from);
265
+ }
package/style.md ADDED
@@ -0,0 +1,150 @@
1
+ # Writing style guide
2
+
3
+ Write so the text reads like a person wrote it, not a chatbot. Never change what
4
+ the text claims. Never add a fact, name, number, date, quote, or citation that
5
+ is not already in the draft or the user's material.
6
+
7
+ Pattern list adapted from Wikipedia's "Signs of AI writing" (WikiProject AI
8
+ Cleanup), which catalogues tells found in real AI edits.
9
+ Additional copy-pattern categories are inspired by the MIT-licensed
10
+ [anti-slop](https://github.com/miqdadbadjuber/anti-slop) project.
11
+
12
+ ## Patterns to avoid
13
+
14
+ **Inflated importance.** stands as, serves as, is a testament to, plays a vital
15
+ / crucial / pivotal role, underscores its significance, marks a turning point,
16
+ evolving landscape, leaves an indelible mark. Ordinary facts get described as
17
+ historic shifts.
18
+
19
+ **Shallow -ing tails.** A clause bolted onto a plain fact to make it sound
20
+ deep: highlighting, underscoring, ensuring, reflecting, symbolizing,
21
+ showcasing, fostering, contributing to.
22
+
23
+ **Sales language.** boasts, vibrant, rich (figurative), nestled, in the heart
24
+ of, breathtaking, stunning, renowned, must-visit, commitment to, groundbreaking.
25
+
26
+ **Vague attribution.** experts agree, studies show, industry reports, observers
27
+ have noted, some critics argue. Name a real source from the material or drop
28
+ the claim. Never invent one.
29
+
30
+ **Avoiding plain verbs.** Prefer is, are, has. Replace serves as, stands as,
31
+ represents, features, boasts with the simple verb.
32
+
33
+ **Not X but Y.** "not just innovative, but transformative", "not only… but
34
+ also…", and clipped negative endings like "no guessing" instead of a full
35
+ clause.
36
+
37
+ **Forced triples.** Three parallel items where two or four would be honest.
38
+ Same for three parallel clauses inside one sentence.
39
+
40
+ **False ranges.** "from X to Y" where X and Y are not the ends of any range.
41
+
42
+ **Filler openers and announcements.** Let's dive in, here's what you need to
43
+ know, it is important to note that, in order to, due to the fact that, at this
44
+ point in time. State the point instead of announcing it.
45
+
46
+ **Fake candour.** Honestly? Look. Here's the thing. Let's be honest. Real talk.
47
+ Used as a staged pause before an ordinary claim.
48
+
49
+ **Pretend depth.** at its core, the real question is, what really matters,
50
+ fundamentally, the deeper issue. Also formulaic sayings: "X is the language of
51
+ Y", "X is not a tool but a mirror", "the currency of".
52
+
53
+ **Straw objections.** "This isn't really about…", "Don't get me wrong",
54
+ "A tempting approach would be…" — answering an objection nobody raised, or
55
+ raising an option nobody would pick, then dropping it.
56
+
57
+ **Dramatic fragmentation.** A row of short punchy fragments used for drama. One
58
+ short sentence for emphasis is fine.
59
+
60
+ **Stacked qualifiers.** could potentially possibly, might arguably, in some
61
+ cases it may. Keep a qualifier only when the meaning needs it.
62
+
63
+ **Mind-reading objects.** A dashboard may show or filter; it does not
64
+ understand, want, believe, care, or decide. Name the person making the choice,
65
+ unless the verb describes an ordinary product action.
66
+
67
+ **Manufactured emphasis.** Several ALL-CAPS words, repeated scare quotes, or
68
+ decorative emoji do not make a weak sentence stronger. Preserve real dialogue,
69
+ titles, acronyms, and one deliberate accent.
70
+
71
+ **Generic positive endings.** "The future looks bright", "exciting times
72
+ ahead", "a step in the right direction". End on the last concrete fact.
73
+
74
+ **Chatbot residue.** I hope this helps, Certainly!, Great question, Would you
75
+ like me to…, as of my last update, while specific details are limited.
76
+
77
+ **Overused words.** delve, tapestry, testament, underscore, leverage, seamless,
78
+ multifaceted, realm, interplay, crucial, pivotal, vibrant, robust, foster,
79
+ enhance, showcase, garner, bolster, utilize, moreover, furthermore. These are
80
+ tells when they cluster, not one at a time.
81
+
82
+ **Repetition handled by rule.** Cycling synonyms for the same subject
83
+ (protagonist / main character / central figure), or several sentences opening
84
+ with the same subject. Fix the pattern, not the word.
85
+
86
+ ## What not to flag
87
+
88
+ These are not evidence on their own. Flag them only when several real tells
89
+ appear together.
90
+
91
+ - Clean grammar and consistent style. That is editing, not AI.
92
+ - Formal or academic vocabulary outside the overused list above.
93
+ - A single em dash. Many writers use them constantly.
94
+ - Curly quotes. macOS, Word, and most editors insert them automatically.
95
+ - One "however" or "additionally". The tell is the pile-up.
96
+ - One short sentence used for emphasis.
97
+ - Repetition that builds deliberate rhythm.
98
+ - Passive voice when the actor is unknown or beside the point.
99
+ - Ordinary product verbs such as "the report shows" or "the form submits".
100
+ - "Honestly" or "look" inside a sentence, as opposed to a staged opener.
101
+ - Missing citations. Most writing is unsourced.
102
+ - Scope notes, safety warnings, real corrections, and answers to named
103
+ objections. These carry information.
104
+ - Watched phrases inside quotations, titles, or proper names, where the phrase
105
+ is being discussed rather than used.
106
+
107
+ ## What to preserve
108
+
109
+ - Concrete, odd, specific details. A street name, an exact figure, a strange
110
+ aside.
111
+ - Mixed feelings and unresolved tension. "I think it's good and it still
112
+ bothers me."
113
+ - Uncertainty the writer actually holds.
114
+ - Humour, bluntness, and mild rudeness.
115
+ - Dated references, slang, and in-jokes tied to a moment.
116
+ - Self-interruptions and parentheticals.
117
+ - Uneven rhythm. Real writing alternates short and long sentences; AI drifts
118
+ toward an even mid-length cadence.
119
+
120
+ ## Keep the reader oriented
121
+
122
+ Open each paragraph or section by setting expectations for what follows, then
123
+ keep that promise. Carry its central terms through the sentences that develop
124
+ them; do not announce a subject and quietly switch to another.
125
+
126
+ Put familiar information before new information when that makes the connection
127
+ between sentences easier to follow. A paragraph may hold one subject steady,
128
+ pick up the previous sentence's new information as the next subject, or preview
129
+ several subjects and develop them in order. Use whichever movement fits the
130
+ material rather than forcing one pattern everywhere.
131
+
132
+ Make a unit's main point once, near its beginning or its end. Do not bury it in
133
+ the middle or repeat it as a recap. For argumentative and explanatory writing,
134
+ show the reader the problem and its cost before presenting the point or
135
+ solution, and let the ending resolve the problem opened at the start.
136
+
137
+ These are diagnostics, not templates. Do not force essay structure onto a short
138
+ answer, a list, reference material, dialogue, or a passage whose deliberate
139
+ disorientation is part of its voice.
140
+
141
+ ## Removing tells is only half the job
142
+
143
+ Text stripped of AI patterns can still read as flat. Where the genre allows —
144
+ essays, posts, opinion, personal writing — let the writer's voice through.
145
+ Keep reference, technical, and legal text neutral. Never invent a fact or an
146
+ experience to make prose feel personal.
147
+
148
+ If the user supplies a sample of their own writing, it outranks every rule
149
+ here. Match its sentence length, vocabulary, punctuation habits, and quirks
150
+ rather than correcting them.