aidetect 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
aidetect/__init__.py ADDED
@@ -0,0 +1 @@
1
+ """Local, offline AI-writing checks and IB word counts for my own drafts."""
aidetect/binoculars.py ADDED
@@ -0,0 +1,243 @@
1
+ """
2
+ Binoculars AI-text detector — a second-opinion scorer for my drafts.
3
+
4
+ Binoculars (Hans et al. 2024, https://arxiv.org/abs/2401.12070) needs no
5
+ training. It runs the text through TWO causal language models that share a
6
+ tokenizer — an "observer" (base) and a "performer" (instruct) — and compares
7
+ how surprised each is:
8
+
9
+ score = perplexity(performer) / cross_perplexity(observer, performer)
10
+
11
+ Human text tends to score HIGHER, machine text LOWER. It's the ratio that
12
+ matters: LLM text is unsurprising to a model (low perplexity) but the two
13
+ models also agree closely on it (low cross-perplexity), and dividing cancels
14
+ out the "this topic is just easy/hard" effect that fools plain perplexity.
15
+
16
+ The paper used a Falcon-7B pair (~28GB, won't fit an 18GB Mac). Small Qwen2.5
17
+ pairs fit but barely separate human from AI text (~63%, near chance) — which is
18
+ why this was shelved as a negative result. A Gemma 4 pair fixes that: it hits
19
+ ~96% on the calibration set. Gemma 4 ships as a multimodal checkpoint, so the
20
+ --mlx path loads it 4-bit via mlx-vlm and runs it text-only.
21
+
22
+ aidetect bino path/to/draft.docx
23
+ aidetect bino notes.txt --text "some sentence to score"
24
+ aidetect bino notes.txt --pair big # bigger Qwen pair
25
+ aidetect bino notes.txt --mlx --pair gemma # Gemma 4 E2B on a Mac
26
+
27
+ Threshold: the AI/human boundary is PER MODEL PAIR. If `aidetect calibrate` has saved a
28
+ threshold for the active pair it's loaded automatically; otherwise it falls back
29
+ to Falcon's 0.90 placeholder. Lower score = more AI-like.
30
+ """
31
+
32
+ import argparse
33
+ import os
34
+
35
+ import torch
36
+ import torch.nn.functional as F
37
+ from transformers import AutoTokenizer, AutoModelForCausalLM
38
+
39
+ from .detect import pick_device
40
+ from .paths import threshold_path, user_threshold_path
41
+ from .text import MIN_WORDS, bar, read_paragraphs
42
+
43
+ # Same-tokenizer pairs. base = observer, instruct = performer.
44
+ # Gemma 4 is Apache-2 (no HF gate). E2B is the smallest but ~5B raw params each,
45
+ # so the fp16 pair is ~20GB — over an 18GB Mac. Use these on a CUDA box; on a Mac
46
+ # use --mlx (4-bit quantized, fits in ~6GB).
47
+ PAIRS = {
48
+ "small": ("Qwen/Qwen2.5-0.5B", "Qwen/Qwen2.5-0.5B-Instruct"),
49
+ "big": ("Qwen/Qwen2.5-1.5B", "Qwen/Qwen2.5-1.5B-Instruct"),
50
+ "gemma": ("google/gemma-4-E2B", "google/gemma-4-E2B-it"),
51
+ "gemma+": ("google/gemma-4-E4B", "google/gemma-4-E4B-it"),
52
+ }
53
+
54
+ # --mlx pairs. observer = google base repo (no pre-quant exists, so we quantize it
55
+ # locally on first run); performer = mlx-community's pre-quantized instruct repo.
56
+ MLX_PAIRS = {
57
+ "gemma": ("google/gemma-4-E2B", "mlx-community/gemma-4-e2b-it-qat-OptiQ-4bit"),
58
+ "gemma+": ("google/gemma-4-E4B", "mlx-community/gemma-4-e4b-it-qat-OptiQ-4bit"),
59
+ }
60
+ MLX_CACHE = os.path.expanduser("~/.cache/ai-detect-mlx")
61
+ MAX_LEN = 1024 # token window; longer paragraphs get truncated
62
+ # ponytail: Falcon's tuned boundary as a placeholder. Wrong for our pair —
63
+ # recalibrate on known-human text. Lower score = more AI-like.
64
+ DEFAULT_THRESHOLD = 0.90
65
+
66
+
67
+ def pair_tag(pair_key, backend):
68
+ return f"{pair_key}-mlx" if backend == "mlx" else pair_key
69
+
70
+
71
+ def load_saved_threshold(pair_key, backend):
72
+ """Return the saved threshold for this pair, or None if never calibrated.
73
+
74
+ A threshold you fitted yourself (~/.config/aidetect) wins over the one
75
+ shipped in the wheel, so recalibrating survives an upgrade.
76
+ """
77
+ import json
78
+ tag = pair_tag(pair_key, backend)
79
+ for path in (user_threshold_path(tag), threshold_path(tag)):
80
+ if path and os.path.exists(path):
81
+ return json.load(open(path)).get("threshold")
82
+ return None
83
+
84
+
85
+ def load_pair(pair_key, device):
86
+ obs_id, perf_id = PAIRS[pair_key]
87
+ tokenizer = AutoTokenizer.from_pretrained(obs_id)
88
+ dtype = torch.float16 if device.type == "mps" else torch.float32
89
+ observer = AutoModelForCausalLM.from_pretrained(obs_id, torch_dtype=dtype).to(device).eval()
90
+ performer = AutoModelForCausalLM.from_pretrained(perf_id, torch_dtype=dtype).to(device).eval()
91
+ return tokenizer, observer, performer
92
+
93
+
94
+ def _mlx_quantized(repo):
95
+ """Return a local 4-bit MLX copy of an fp16 HF repo, converting once and reusing.
96
+ Gemma 4 ships as a VLM checkpoint, so this goes through mlx-vlm, not mlx-lm."""
97
+ from mlx_vlm import convert
98
+ out = os.path.join(MLX_CACHE, repo.replace("/", "_") + "-4bit")
99
+ if not os.path.isdir(out):
100
+ print(f"converting {repo} to 4-bit MLX (one-time, downloads fp16 weights)...")
101
+ convert(repo, mlx_path=out, quantize=True)
102
+ return out
103
+
104
+
105
+ def load_pair_mlx(pair_key):
106
+ """MLX backend for Macs. observer is quantized locally; performer is pre-quantized.
107
+ Returns (tokenizer, observer, performer) — models are mlx-vlm VLMs run text-only."""
108
+ from mlx_vlm import load
109
+ obs_src, perf_repo = MLX_PAIRS[pair_key]
110
+ observer, processor = load(_mlx_quantized(obs_src)) # base tokenizer drives encoding
111
+ performer, _ = load(perf_repo)
112
+ tokenizer = getattr(processor, "tokenizer", processor)
113
+ return tokenizer, observer, performer
114
+
115
+
116
+ def add_backend_args(ap):
117
+ """The --pair/--mlx flags, shared by binoculars.py and calibrate.py."""
118
+ ap.add_argument("--pair", choices=list(PAIRS), default="small",
119
+ help="model pair to use (default %(default)s)")
120
+ ap.add_argument("--mlx", action="store_true",
121
+ help="use the 4-bit MLX gemma pair (Apple Silicon; needs mlx-vlm)")
122
+
123
+
124
+ def load_backend(args):
125
+ """Turn parsed CLI args into a loaded model pair.
126
+ Returns (pair_key, backend, device, tokenizer, observer, performer)."""
127
+ pair_key = args.pair
128
+ if args.mlx:
129
+ if pair_key not in MLX_PAIRS:
130
+ raise SystemExit(f"--mlx only supports {list(MLX_PAIRS)}; pass e.g. --pair gemma")
131
+ print(f"loading MLX pair {' + '.join(MLX_PAIRS[pair_key])}...")
132
+ tokenizer, observer, performer = load_pair_mlx(pair_key)
133
+ return pair_key, "mlx", None, tokenizer, observer, performer
134
+ device = pick_device()
135
+ print(f"loading {' + '.join(PAIRS[pair_key])} on {device}... (first run downloads the models)")
136
+ tokenizer, observer, performer = load_pair(pair_key, device)
137
+ return pair_key, "torch", device, tokenizer, observer, performer
138
+
139
+
140
+ def _mlx_logits(model, ids):
141
+ """Run an mlx-vlm model text-only (no pixel_values) and return logits as torch."""
142
+ import mlx.core as mx
143
+ import numpy as np
144
+ out = model(mx.array([ids])) # text-only path; returns LanguageModelOutput
145
+ logits = out.logits if hasattr(out, "logits") else out # (1, seq, vocab)
146
+ logits = logits.astype(mx.float32) # numpy can't read mlx bfloat16 buffers
147
+ return torch.from_numpy(np.array(logits)).float()
148
+
149
+
150
+ def perplexity(input_ids, logits):
151
+ """Mean next-token cross-entropy: how surprised the performer is by the text."""
152
+ # predict token t+1 from position t, so line the logits up one step ahead
153
+ shift_logits = logits[..., :-1, :]
154
+ shift_labels = input_ids[..., 1:]
155
+ ce = F.cross_entropy(
156
+ shift_logits.reshape(-1, shift_logits.size(-1)),
157
+ shift_labels.reshape(-1),
158
+ reduction="mean",
159
+ )
160
+ return ce.item()
161
+
162
+
163
+ def cross_perplexity(observer_logits, performer_logits):
164
+ """Mean cross-entropy of the performer's predictions against the observer's
165
+ full distribution — how much the two models DISAGREE per token."""
166
+ # observer's probabilities are the soft targets; score the performer against them
167
+ p = F.softmax(observer_logits, dim=-1)
168
+ log_q = F.log_softmax(performer_logits, dim=-1)
169
+ ce = -(p * log_q).sum(dim=-1) # per-token cross-entropy
170
+ return ce.mean().item()
171
+
172
+
173
+ def score_text(text, tokenizer, observer, performer, device, backend="torch"):
174
+ """Return the Binoculars score for one chunk. Lower = more AI-like.
175
+ Both backends end up feeding torch logits into the same perplexity math."""
176
+ if backend == "mlx":
177
+ ids = tokenizer.encode(text)[:MAX_LEN] # base+instruct share this vocab
178
+ obs_logits = _mlx_logits(observer, ids)
179
+ perf_logits = _mlx_logits(performer, ids)
180
+ input_ids = torch.tensor([ids])
181
+ else:
182
+ enc = tokenizer(text, truncation=True, max_length=MAX_LEN, return_tensors="pt")
183
+ input_ids = enc["input_ids"].to(device)
184
+ attention_mask = enc["attention_mask"].to(device)
185
+ with torch.no_grad():
186
+ obs_logits = observer(input_ids=input_ids, attention_mask=attention_mask).logits.float()
187
+ perf_logits = performer(input_ids=input_ids, attention_mask=attention_mask).logits.float()
188
+ ppl = perplexity(input_ids, perf_logits)
189
+ x_ppl = cross_perplexity(obs_logits, perf_logits)
190
+ return ppl / x_ppl
191
+
192
+
193
+ def report(paragraphs, threshold, tokenizer, observer, performer, device, backend="torch"):
194
+ if not paragraphs:
195
+ print("No paragraphs with >= %d words found." % MIN_WORDS)
196
+ return
197
+ scores = []
198
+ for i, para in enumerate(paragraphs, 1):
199
+ s = score_text(para, tokenizer, observer, performer, device, backend)
200
+ scores.append(s)
201
+ flag = " <-- AI-ish" if s < threshold else ""
202
+ # bar: lower score = more AI, so invert for a "how AI-ish" bar
203
+ aiish = max(0.0, min(1.0, (threshold * 1.3 - s) / (threshold * 1.3)))
204
+ print(f"P{i:>3} {s:5.2f} [{bar(aiish)}]{flag}")
205
+ print(f" {para[:70].strip()}...")
206
+ avg = sum(scores) / len(scores)
207
+ low = sum(1 for s in scores if s < threshold)
208
+ print("-" * 60)
209
+ print(f"average Binoculars score: {avg:.2f} | {low}/{len(scores)} paragraphs flagged (< {threshold})")
210
+ print("reminder: directional only. Lower = more AI-like. Recalibrate with `aidetect calibrate`.")
211
+
212
+
213
+ def main(argv=None):
214
+ ap = argparse.ArgumentParser(prog="aidetect bino",
215
+ description="Binoculars AI-text detector (second opinion).")
216
+ ap.add_argument("path", nargs="?", help=".docx or .txt file to score")
217
+ ap.add_argument("--text", help="score a single string instead of a file")
218
+ add_backend_args(ap)
219
+ ap.add_argument("--threshold", type=float, default=None,
220
+ help="flag paragraphs below this (default: calibrated pair threshold, else %s)"
221
+ % DEFAULT_THRESHOLD)
222
+ args = ap.parse_args(argv)
223
+
224
+ if not args.path and not args.text:
225
+ ap.error("give a file path or --text")
226
+
227
+ pair_key, backend, device, tokenizer, observer, performer = load_backend(args)
228
+
229
+ # explicit --threshold wins; else use the calibrated one; else Falcon's placeholder
230
+ threshold = args.threshold
231
+ if threshold is None:
232
+ threshold = load_saved_threshold(pair_key, backend)
233
+ if threshold is not None:
234
+ print(f"using calibrated threshold {threshold}")
235
+ if threshold is None:
236
+ threshold = DEFAULT_THRESHOLD
237
+
238
+ if args.text:
239
+ s = score_text(args.text, tokenizer, observer, performer, device, backend)
240
+ verdict = "AI-ish" if s < threshold else "human-ish"
241
+ print(f"Binoculars score: {s:.2f} ({verdict}, threshold {threshold})")
242
+ else:
243
+ report(read_paragraphs(args.path), threshold, tokenizer, observer, performer, device, backend)
aidetect/calibrate.py ADDED
@@ -0,0 +1,123 @@
1
+ """
2
+ Calibrate the Binoculars threshold for a model pair, then optionally check a draft.
3
+
4
+ Binoculars outputs a raw score, not a probability, and the AI/human boundary is
5
+ specific to the model pair. This scores a labelled set (a folder of known-human
6
+ prose against a folder of known-LLM prose on the same topics), finds the cutoff
7
+ that best separates them, and saves it to ~/.config/aidetect.
8
+
9
+ My own calibration set is not shipped with the package: it is 12 real pre-2020 IB
10
+ Extended Essays and 12 LLM-written imitations, and it lives in corpora/ in the
11
+ repo. Point --human-dir and --ai-dir at your own.
12
+
13
+ aidetect calibrate --human-dir corpora/human --ai-dir corpora/ai
14
+ aidetect calibrate --human-dir corpora/human --ai-dir corpora/ai --mlx --pair gemma
15
+ aidetect calibrate --human-dir h --ai-dir a --check draft.docx
16
+
17
+ The human class has to be provably human or the threshold is meaningless. Do not
18
+ point --human-dir at anything written after ChatGPT that you cannot certify.
19
+ """
20
+
21
+ import argparse
22
+ import glob
23
+ import json
24
+ import os
25
+
26
+ from .binoculars import (MLX_PAIRS, PAIRS, add_backend_args, load_backend,
27
+ pair_tag, report, score_text)
28
+ from .paths import ensure_user_dir, user_threshold_path
29
+ from .text import read_paragraphs
30
+
31
+
32
+ def score_folder(folder, tokenizer, observer, performer, device, backend="torch"):
33
+ """Score every .txt in a folder. Returns list of (id, score)."""
34
+ out = []
35
+ for path in sorted(glob.glob(os.path.join(folder, "*.txt"))):
36
+ text = open(path, encoding="utf-8").read().strip()
37
+ if not text:
38
+ continue
39
+ sid = os.path.splitext(os.path.basename(path))[0]
40
+ out.append((sid, score_text(text, tokenizer, observer, performer, device, backend)))
41
+ return out
42
+
43
+
44
+ def pick_threshold(human_scores, ai_scores):
45
+ """Cutoff that best separates the clusters (predict AI if score < cutoff).
46
+ Human text scores higher, AI lower. Ties broken toward the most centered
47
+ cutoff (largest gap to the nearest sample) so it generalizes better."""
48
+ everything = sorted(set(human_scores + ai_scores))
49
+ candidates = [everything[0] - 0.01]
50
+ for a, b in zip(everything, everything[1:]):
51
+ candidates.append((a + b) / 2)
52
+ candidates.append(everything[-1] + 0.01)
53
+
54
+ best = None
55
+ for c in candidates:
56
+ correct = sum(s >= c for s in human_scores) + sum(s < c for s in ai_scores)
57
+ acc = correct / (len(human_scores) + len(ai_scores))
58
+ margin = min(abs(s - c) for s in human_scores + ai_scores)
59
+ key = (acc, margin)
60
+ if best is None or key > best[0]:
61
+ best = (key, c)
62
+ (acc, _margin), cutoff = best
63
+ return cutoff, acc
64
+
65
+
66
+ def main(argv=None):
67
+ ap = argparse.ArgumentParser(
68
+ prog="aidetect calibrate",
69
+ description="Fit the Binoculars threshold for a model pair on your own labelled set.")
70
+ ap.add_argument("--human-dir", required=True,
71
+ help="folder of .txt files you can certify are human-written")
72
+ ap.add_argument("--ai-dir", required=True,
73
+ help="folder of .txt files you know are LLM-written")
74
+ ap.add_argument("--check", help="optionally score this draft with the new threshold")
75
+ add_backend_args(ap)
76
+ args = ap.parse_args(argv)
77
+
78
+ for label, folder in (("--human-dir", args.human_dir), ("--ai-dir", args.ai_dir)):
79
+ if not glob.glob(os.path.join(folder, "*.txt")):
80
+ ap.error(f"{label}: no .txt files in {folder}")
81
+
82
+ pair_key, backend, device, tokenizer, observer, performer = load_backend(args)
83
+
84
+ # --- score the labelled set ---
85
+ human = score_folder(args.human_dir, tokenizer, observer, performer, device, backend)
86
+ ai = score_folder(args.ai_dir, tokenizer, observer, performer, device, backend)
87
+ human_scores = [s for _, s in human]
88
+ ai_scores = [s for _, s in ai]
89
+
90
+ print("\n=== calibration scores (higher = more human) ===")
91
+ print(f"{'HUMAN':<22}{'AI (LLM-written)':<22}")
92
+ for i in range(max(len(human), len(ai))):
93
+ h = f"{human[i][0]} {human[i][1]:.3f}" if i < len(human) else ""
94
+ a = f"{ai[i][0]} {ai[i][1]:.3f}" if i < len(ai) else ""
95
+ print(f"{h:<22}{a:<22}")
96
+ print(f"\nhuman: min {min(human_scores):.3f} mean {sum(human_scores)/len(human_scores):.3f} max {max(human_scores):.3f}")
97
+ print(f"ai : min {min(ai_scores):.3f} mean {sum(ai_scores)/len(ai_scores):.3f} max {max(ai_scores):.3f}")
98
+
99
+ cutoff, acc = pick_threshold(human_scores, ai_scores)
100
+ print(f"\n>>> calibrated threshold: {cutoff:.3f} (separates {acc*100:.0f}% of the {len(human)+len(ai)} samples)")
101
+ print(" flag as AI-ish when score < threshold")
102
+
103
+ # save for reuse / future recalibration
104
+ active_pairs = MLX_PAIRS if backend == "mlx" else PAIRS
105
+ out = {
106
+ "pair": active_pairs[pair_key],
107
+ "backend": backend,
108
+ "threshold": round(cutoff, 4),
109
+ "accuracy": round(acc, 4),
110
+ "n_human": len(human),
111
+ "n_ai": len(ai),
112
+ "human_mean": round(sum(human_scores)/len(human_scores), 4),
113
+ "ai_mean": round(sum(ai_scores)/len(ai_scores), 4),
114
+ }
115
+ ensure_user_dir()
116
+ thr_path = user_threshold_path(pair_tag(pair_key, backend))
117
+ json.dump(out, open(thr_path, "w"), indent=2)
118
+ print(f" saved -> {thr_path}")
119
+
120
+ # --- run the calibrated check on a draft ---
121
+ if args.check:
122
+ print(f"\n=== checking {args.check} with threshold {cutoff:.3f} ===")
123
+ report(read_paragraphs(args.check), cutoff, tokenizer, observer, performer, device, backend)
aidetect/cli.py ADDED
@@ -0,0 +1,46 @@
1
+ """
2
+ `aidetect` entry point.
3
+
4
+ Subcommand modules are imported lazily and on demand: `aidetect count` must not
5
+ pay a ~2s torch import to count words, and on a machine without mlx-vlm the
6
+ torch-free subcommands still have to work.
7
+ """
8
+
9
+ import sys
10
+
11
+ COMMANDS = {
12
+ "count": ("aidetect.count", "IB-rules word count for a draft"),
13
+ "score": ("aidetect.detect", "score paragraphs with the desklib detector"),
14
+ "bino": ("aidetect.binoculars", "second opinion via Binoculars (needs a model pair)"),
15
+ "extract": ("aidetect.extract", "pull finished prose out of a .docx into a .txt"),
16
+ "calibrate": ("aidetect.calibrate", "fit a Binoculars threshold on your own labelled set"),
17
+ }
18
+
19
+ USAGE = "usage: aidetect <command> [options]\n\ncommands:\n" + "".join(
20
+ f" {name:<10} {help}\n" for name, (_mod, help) in COMMANDS.items()
21
+ ) + "\nrun `aidetect <command> --help` for a command's options.\n"
22
+
23
+
24
+ def main():
25
+ if len(sys.argv) < 2 or sys.argv[1] in ("-h", "--help", "help"):
26
+ print(USAGE, end="")
27
+ return 0
28
+ if sys.argv[1] in ("-V", "--version"):
29
+ from importlib.metadata import version
30
+ print(version("aidetect"))
31
+ return 0
32
+
33
+ command = sys.argv[1]
34
+ if command not in COMMANDS:
35
+ print(f"aidetect: unknown command {command!r}\n", file=sys.stderr)
36
+ print(USAGE, end="", file=sys.stderr)
37
+ return 2
38
+
39
+ module_name = COMMANDS[command][0]
40
+ from importlib import import_module
41
+ module = import_module(module_name)
42
+ return module.main(sys.argv[2:])
43
+
44
+
45
+ if __name__ == "__main__":
46
+ sys.exit(main())
aidetect/count.py ADDED
@@ -0,0 +1,118 @@
1
+ """
2
+ IB word count for a draft.
3
+
4
+ Word's own count is wrong for IB: it counts headings, block quotes, tables,
5
+ footnotes and citations, none of which are assessed. This counts what the IB
6
+ counts, and splits the total by section so an over-long draft says WHERE.
7
+
8
+ Excluded, per the IB rules that are common to the EE and the subject IAs:
9
+ - headings (skipped by style)
10
+ - the bibliography onward (stop at the Bibliography/Works Cited heading)
11
+ - block quotes and bullets (same first-character rule extract.py uses)
12
+ - tables and footnotes (python-docx never puts them in doc.paragraphs)
13
+ - in-text citations (parentheticals containing a year / ibid / et al)
14
+
15
+ aidetect count draft.docx
16
+ aidetect count draft.docx --limit 4000
17
+ aidetect count draft.docx --limit 4000 --json
18
+ """
19
+
20
+ import argparse
21
+ import json
22
+ import re
23
+
24
+ from .text import is_end_heading, is_heading, is_prose
25
+
26
+ # A parenthetical is a citation if it carries a 4-digit year, "ibid" or "et al".
27
+ # ponytail: narrow on purpose. "(the second of these)" is prose and stays counted;
28
+ # widening this to every parenthetical would silently eat hundreds of real words.
29
+ CITATION = re.compile(
30
+ r"\((?:[^()]*(?:\b(?:1[6-9]|20)\d{2}\b|\bibid\b|\bet al\b)[^()]*)\)",
31
+ re.IGNORECASE,
32
+ )
33
+
34
+
35
+ def count_words(text):
36
+ """Words in one paragraph, with in-text citations removed.
37
+
38
+ Tokens with no letter or digit are dropped: removing "(Smith, 2024)" from
39
+ "...sharply (Smith, 2024)." strands the full stop as its own token, and a
40
+ lone "." is not a word.
41
+ """
42
+ stripped = CITATION.sub(" ", text)
43
+ return sum(1 for token in stripped.split() if any(c.isalnum() for c in token))
44
+
45
+
46
+ def count_docx(path):
47
+ """Return (sections, total). sections is a list of (heading, words)."""
48
+ import docx
49
+
50
+ sections = [("(untitled)", 0)]
51
+ for p in docx.Document(path).paragraphs:
52
+ if is_end_heading(p):
53
+ break
54
+ if is_heading(p):
55
+ title = p.text.strip()
56
+ if title:
57
+ sections.append((title, 0))
58
+ continue
59
+ # min_words=1: the 25-word floor is a detector heuristic. A one-line
60
+ # paragraph is still words the examiner counts.
61
+ if is_prose(p, min_words=1):
62
+ title, words = sections[-1]
63
+ sections[-1] = (title, words + count_words(p.text))
64
+ sections = [s for s in sections if s[1] > 0]
65
+ return sections, sum(w for _, w in sections)
66
+
67
+
68
+ def report(sections, total, limit):
69
+ width = max((len(t) for t, _ in sections), default=10)
70
+ for title, words in sections:
71
+ share = words / total if total else 0
72
+ print(f" {title:<{width}} {words:>5} {'#' * round(share * 30)}")
73
+ print("-" * (width + 40))
74
+ if limit:
75
+ over = total - limit
76
+ verdict = f"{over:+d} over" if over > 0 else f"{-over} to spare"
77
+ print(f" {'TOTAL':<{width}} {total:>5} / {limit} limit, {verdict}")
78
+ else:
79
+ print(f" {'TOTAL':<{width}} {total:>5}")
80
+ print("excludes headings, quotes, bullets, tables, footnotes, "
81
+ "citations and everything from the bibliography on.")
82
+
83
+
84
+ def as_json(sections, total, limit):
85
+ """The machine interface. Every key is always present; `null` means the
86
+ question does not apply, not that it could not be answered. A draft with no
87
+ prose is an empty list and exit 0, because parsing succeeded and the honest
88
+ answer is zero."""
89
+ return {
90
+ "sections": [{"title": t, "words": w} for t, w in sections],
91
+ "total": total,
92
+ "limit": limit,
93
+ "over": None if limit is None else total - limit,
94
+ }
95
+
96
+
97
+ def main(argv=None):
98
+ ap = argparse.ArgumentParser(prog="aidetect count",
99
+ description="IB-rules word count for a draft.")
100
+ ap.add_argument("path", help=".docx file to count")
101
+ ap.add_argument("--limit", type=int, default=None,
102
+ help="word limit to measure against (e.g. 4000 for the EE)")
103
+ ap.add_argument("--json", action="store_true",
104
+ help="emit one JSON object on stdout instead of the table")
105
+ args = ap.parse_args(argv)
106
+
107
+ if not args.path.lower().endswith(".docx"):
108
+ ap.error("count needs a .docx (a .txt has no headings or styles to go on)")
109
+
110
+ sections, total = count_docx(args.path)
111
+
112
+ if args.json:
113
+ print(json.dumps(as_json(sections, total, args.limit)))
114
+ return
115
+ if not sections:
116
+ print("No prose found. Check the file is a draft and not an outline.")
117
+ return
118
+ report(sections, total, args.limit)
aidetect/detect.py ADDED
@@ -0,0 +1,121 @@
1
+ """
2
+ Local AI-writing detector for my drafts.
3
+
4
+ Scores each paragraph of a .docx (or a .txt / --text string) with the
5
+ desklib DeBERTa detector. Everything runs on my Mac, offline after the
6
+ first model download. The score is a rough, directional read, NOT Turnitin:
7
+ high just means "this paragraph reads AI-ish, maybe reword it".
8
+
9
+ Usage:
10
+ aidetect score path/to/draft.docx
11
+ aidetect score notes.txt
12
+ aidetect score --text "some sentence to score"
13
+ """
14
+
15
+ import argparse
16
+
17
+ import torch
18
+ import torch.nn as nn
19
+ from transformers import AutoTokenizer, AutoConfig, AutoModel, PreTrainedModel
20
+
21
+ from .text import MIN_WORDS, bar, read_paragraphs
22
+
23
+ MODEL_ID = "desklib/ai-text-detector-v1.01"
24
+ MAX_LEN = 768 # model's token window; longer paragraphs get truncated
25
+
26
+
27
+ # --- the model class, copied verbatim from the desklib model card ---
28
+ # It's a normal transformer with mean-pooling + a 1-unit classifier head.
29
+ class DesklibAIDetectionModel(PreTrainedModel):
30
+ config_class = AutoConfig
31
+ # transformers 5.x reads this during from_pretrained; the model has no tied
32
+ # weights (encoder + linear head), so an empty mapping is correct. Without it,
33
+ # loading raises AttributeError on 5.x. ponytail: needed once mlx pulled 5.x in.
34
+ all_tied_weights_keys = {}
35
+
36
+ def __init__(self, config):
37
+ super().__init__(config)
38
+ self.model = AutoModel.from_config(config)
39
+ self.classifier = nn.Linear(config.hidden_size, 1)
40
+ self.init_weights()
41
+
42
+ def forward(self, input_ids, attention_mask=None, labels=None):
43
+ outputs = self.model(input_ids, attention_mask=attention_mask)
44
+ last_hidden_state = outputs[0]
45
+ # mean-pool the token vectors, ignoring padding
46
+ mask = attention_mask.unsqueeze(-1).expand(last_hidden_state.size()).float()
47
+ summed = torch.sum(last_hidden_state * mask, dim=1)
48
+ counts = torch.clamp(mask.sum(dim=1), min=1e-9)
49
+ pooled = summed / counts
50
+ logits = self.classifier(pooled)
51
+ return {"logits": logits}
52
+
53
+
54
+ def pick_device():
55
+ if torch.backends.mps.is_available():
56
+ return torch.device("mps") # Apple Silicon GPU
57
+ return torch.device("cpu")
58
+
59
+
60
+ def load_model(device):
61
+ tokenizer = AutoTokenizer.from_pretrained(MODEL_ID)
62
+ model = DesklibAIDetectionModel.from_pretrained(MODEL_ID)
63
+ model.to(device)
64
+ model.eval()
65
+ return tokenizer, model
66
+
67
+
68
+ def score_text(text, tokenizer, model, device):
69
+ """Return P(AI-generated) in [0, 1] for one chunk of text."""
70
+ encoded = tokenizer(
71
+ text,
72
+ padding="max_length",
73
+ truncation=True,
74
+ max_length=MAX_LEN,
75
+ return_tensors="pt",
76
+ )
77
+ input_ids = encoded["input_ids"].to(device)
78
+ attention_mask = encoded["attention_mask"].to(device)
79
+ with torch.no_grad():
80
+ logits = model(input_ids=input_ids, attention_mask=attention_mask)["logits"]
81
+ return torch.sigmoid(logits).item()
82
+
83
+
84
+ def report(paragraphs, tokenizer, model, device):
85
+ if not paragraphs:
86
+ print("No paragraphs with >= %d words found." % MIN_WORDS)
87
+ return
88
+ scores = []
89
+ for i, para in enumerate(paragraphs, 1):
90
+ prob = score_text(para, tokenizer, model, device)
91
+ scores.append(prob)
92
+ flag = " <-- AI-ish" if prob >= 0.5 else ""
93
+ preview = para[:70].replace("\n", " ")
94
+ print(f"P{i:>3} {prob:5.2f} [{bar(prob)}]{flag}")
95
+ print(f" {preview}...")
96
+ avg = sum(scores) / len(scores)
97
+ high = sum(1 for s in scores if s >= 0.5)
98
+ print("-" * 60)
99
+ print(f"average AI score: {avg:.2f} | {high}/{len(scores)} paragraphs flagged")
100
+ print("reminder: directional only, not a Turnitin score.")
101
+
102
+
103
+ def main(argv=None):
104
+ ap = argparse.ArgumentParser(prog="aidetect score",
105
+ description="Local AI-writing detector (desklib).")
106
+ ap.add_argument("path", nargs="?", help=".docx or .txt file to score")
107
+ ap.add_argument("--text", help="score a single string instead of a file")
108
+ args = ap.parse_args(argv)
109
+
110
+ if not args.path and not args.text:
111
+ ap.error("give a file path or --text")
112
+
113
+ device = pick_device()
114
+ print(f"loading {MODEL_ID} on {device}... (first run downloads ~1.5GB)")
115
+ tokenizer, model = load_model(device)
116
+
117
+ if args.text:
118
+ prob = score_text(args.text, tokenizer, model, device)
119
+ print(f"AI score: {prob:.2f} [{bar(prob)}]")
120
+ else:
121
+ report(read_paragraphs(args.path), tokenizer, model, device)
aidetect/extract.py ADDED
@@ -0,0 +1,40 @@
1
+ """
2
+ Pull just the finished PROSE out of a .docx into a clean .txt, so the
3
+ detector scores real writing instead of headings, bullets and note-scaffolding.
4
+
5
+ Drops: headings, cover/TOC lines, footnotes (python-docx never reads those),
6
+ tables (also not in doc.paragraphs), bullet lines, and my note-labels
7
+ (NOTES / Verdict / Analysis: / Criticism / Mini-conclusion ...).
8
+
9
+ aidetect extract "in.docx" out.txt
10
+ """
11
+
12
+ import argparse
13
+
14
+ import docx
15
+
16
+ from .text import MIN_WORDS, is_end_heading, is_prose
17
+
18
+
19
+ def main(argv=None):
20
+ ap = argparse.ArgumentParser(prog="aidetect extract",
21
+ description="Extract finished prose from a .docx.")
22
+ ap.add_argument("in_path", help="source .docx")
23
+ ap.add_argument("out_path", help="destination .txt")
24
+ args = ap.parse_args(argv)
25
+
26
+ doc = docx.Document(args.in_path)
27
+ prose = []
28
+ for p in doc.paragraphs:
29
+ # Bibliography is the last section; once its heading shows up, the rest
30
+ # is citations, not prose. Stop here.
31
+ if is_end_heading(p):
32
+ break
33
+ if is_prose(p, min_words=MIN_WORDS):
34
+ prose.append(p.text.strip())
35
+ with open(args.out_path, "w", encoding="utf-8") as f:
36
+ f.write("\n\n".join(prose))
37
+ words = sum(len(p.split()) for p in prose)
38
+ print(f"kept {len(prose)} paragraphs, {words} words -> {args.out_path}")
39
+ print("note: this is the detector's prose filter, not an IB count. "
40
+ "Use `aidetect count` for that.")
aidetect/paths.py ADDED
@@ -0,0 +1,32 @@
1
+ """
2
+ Where thresholds live.
3
+
4
+ Two locations, checked in this order:
5
+ 1. ~/.config/aidetect/threshold-<tag>.json — what `aidetect calibrate` writes.
6
+ 2. the thresholds/ folder inside the installed package — the ones shipped.
7
+
8
+ The package folder is read-only in spirit (it is site-packages on an install and
9
+ gets replaced on upgrade), so calibration output never goes there.
10
+ """
11
+
12
+ import os
13
+ from importlib import resources
14
+
15
+ USER_DIR = os.path.expanduser("~/.config/aidetect")
16
+
17
+
18
+ def user_threshold_path(tag):
19
+ return os.path.join(USER_DIR, f"threshold-{tag}.json")
20
+
21
+
22
+ def threshold_path(tag):
23
+ """Path to the shipped threshold for this pair, or None if there isn't one."""
24
+ ref = resources.files("aidetect") / "thresholds" / f"threshold-{tag}.json"
25
+ # as_file would be needed for a zipped install; these are always unpacked
26
+ # because the wheel has no zip-safe flag, so a plain path is fine.
27
+ return str(ref) if ref.is_file() else None
28
+
29
+
30
+ def ensure_user_dir():
31
+ os.makedirs(USER_DIR, exist_ok=True)
32
+ return USER_DIR
aidetect/text.py ADDED
@@ -0,0 +1,64 @@
1
+ """
2
+ Reading prose out of .docx and .txt files. No torch in here on purpose:
3
+ `aidetect count` uses this module and should not pay a model import to count words.
4
+ """
5
+
6
+ MIN_WORDS = 25 # ponytail: skip fragments; short text scores as noise
7
+
8
+ # Lines that begin with any of these are note-scaffolding, not prose.
9
+ NOTE_STARTS = (
10
+ "NOTES", "Verdict", "Analysis:", "Criticism", "Mini-conclusion",
11
+ "Tool definition", "Why chosen", "Overall synthesis", "Force =",
12
+ "POINTS", "Safest", "Moderate", "Research Question", "Word count",
13
+ "Table of Contents",
14
+ )
15
+ # ponytail: prefix heuristic, not a parser. If a real sentence ever starts
16
+ # with one of these words it gets dropped too — check the output if a section vanishes.
17
+
18
+ # Headings that mean "the prose is over, the rest is citations".
19
+ END_HEADINGS = ("bibliography", "works cited", "references")
20
+
21
+
22
+ def is_heading(p):
23
+ return p.style.name.startswith("Heading")
24
+
25
+
26
+ def is_end_heading(p):
27
+ return is_heading(p) and p.text.strip().lower() in END_HEADINGS
28
+
29
+
30
+ def is_prose(p, min_words=MIN_WORDS):
31
+ """True if this python-docx paragraph is finished prose, not scaffolding.
32
+
33
+ Tables and footnotes need no handling here: doc.paragraphs excludes table
34
+ cells, and python-docx never exposes footnotes at all. IB excludes both.
35
+ """
36
+ text = p.text.strip()
37
+ if len(text.split()) < min_words:
38
+ return False
39
+ if is_heading(p):
40
+ return False
41
+ if text[0] in "•-*“\"[»": # bullet, quoted snippet, or [SCAFFOLD]/» note marker
42
+ return False
43
+ if text.startswith(NOTE_STARTS):
44
+ return False
45
+ return True
46
+
47
+
48
+ def read_paragraphs(path, min_words=MIN_WORDS):
49
+ """Pull prose paragraphs out of a .docx or .txt file."""
50
+ if path.lower().endswith(".docx"):
51
+ import docx # python-docx, only needed for Word files
52
+ doc = docx.Document(path)
53
+ chunks = [p.text for p in doc.paragraphs]
54
+ else:
55
+ with open(path, encoding="utf-8") as f:
56
+ # blank line separates paragraphs in a plain-text file
57
+ chunks = f.read().split("\n\n")
58
+ # keep only real paragraphs, not headings/blanks/fragments
59
+ return [c.strip() for c in chunks if len(c.split()) >= min_words]
60
+
61
+
62
+ def bar(prob, width=20):
63
+ filled = round(prob * width)
64
+ return "#" * filled + "-" * (width - filled)
@@ -0,0 +1,12 @@
1
+ {
2
+ "pair": [
3
+ "Qwen/Qwen2.5-1.5B",
4
+ "Qwen/Qwen2.5-1.5B-Instruct"
5
+ ],
6
+ "threshold": 0.9777,
7
+ "accuracy": 0.6667,
8
+ "n_human": 12,
9
+ "n_ai": 12,
10
+ "human_mean": 1.0185,
11
+ "ai_mean": 0.9891
12
+ }
@@ -0,0 +1,13 @@
1
+ {
2
+ "pair": [
3
+ "google/gemma-4-E2B",
4
+ "mlx-community/gemma-4-e2b-it-qat-OptiQ-4bit"
5
+ ],
6
+ "backend": "mlx",
7
+ "threshold": 0.8264,
8
+ "accuracy": 0.9583,
9
+ "n_human": 12,
10
+ "n_ai": 12,
11
+ "human_mean": 0.9005,
12
+ "ai_mean": 0.7126
13
+ }
@@ -0,0 +1,12 @@
1
+ {
2
+ "pair": [
3
+ "Qwen/Qwen2.5-0.5B",
4
+ "Qwen/Qwen2.5-0.5B-Instruct"
5
+ ],
6
+ "threshold": 0.9662,
7
+ "accuracy": 0.625,
8
+ "n_human": 12,
9
+ "n_ai": 12,
10
+ "human_mean": 1.0024,
11
+ "ai_mean": 0.9962
12
+ }
@@ -0,0 +1,236 @@
1
+ Metadata-Version: 2.5
2
+ Name: aidetect
3
+ Version: 0.1.0
4
+ Summary: Local, offline AI-writing detector and IB word counter for your own drafts
5
+ Project-URL: Homepage, https://github.com/nitrimandylis/aidetect
6
+ Project-URL: Source, https://github.com/nitrimandylis/aidetect
7
+ Author: Nikolas Trimandylis
8
+ License-Expression: MIT
9
+ License-File: LICENSE
10
+ Keywords: ai-detection,binoculars,docx,ib,word-count
11
+ Classifier: Environment :: Console
12
+ Classifier: Programming Language :: Python :: 3
13
+ Classifier: Topic :: Text Processing :: Linguistic
14
+ Requires-Python: >=3.10
15
+ Requires-Dist: mlx-vlm; sys_platform == 'darwin' and platform_machine == 'arm64'
16
+ Requires-Dist: python-docx
17
+ Requires-Dist: torch
18
+ Requires-Dist: transformers
19
+ Description-Content-Type: text/markdown
20
+
21
+ ```
22
+ █████╗ ██╗ ██████╗ ███████╗████████╗███████╗ ██████╗████████╗
23
+ ██╔══██╗██║ ██╔══██╗██╔════╝╚══██╔══╝██╔════╝██╔════╝╚══██╔══╝
24
+ ███████║██║█████╗██║ ██║█████╗ ██║ █████╗ ██║ ██║
25
+ ██╔══██║██║╚════╝██║ ██║██╔══╝ ██║ ██╔══╝ ██║ ██║
26
+ ██║ ██║██║ ██████╔╝███████╗ ██║ ███████╗╚██████╗ ██║
27
+ ╚═╝ ╚═╝╚═╝ ╚═════╝ ╚══════╝ ╚═╝ ╚══════╝ ╚═════╝ ╚═╝
28
+ ```
29
+ <div align="center">
30
+
31
+ ### `SCORE YOUR OWN PROSE // BEFORE A TEACHER SCORES IT FOR YOU`
32
+
33
+ *a local, offline AI-writing detector and IB word counter for drafts you actually wrote*
34
+
35
+ ![pypi](https://img.shields.io/badge/pypi-aidetect-3775A9?style=flat-square&labelColor=111111)
36
+ ![language](https://img.shields.io/badge/language-python-3776AB?style=flat-square&labelColor=111111)
37
+ ![runs](https://img.shields.io/badge/runs-offline-2ea043?style=flat-square&labelColor=111111)
38
+ ![model](https://img.shields.io/badge/model-desklib_DeBERTa-8957e5?style=flat-square&labelColor=111111)
39
+ ![license](https://img.shields.io/badge/license-MIT-2ea043?style=flat-square&labelColor=111111)
40
+ ![telemetry](https://img.shields.io/badge/telemetry-0_(it's_your_essay)-111111?style=flat-square&labelColor=111111)
41
+
42
+ </div>
43
+
44
+ ---
45
+
46
+ ## 🔍 What is this
47
+
48
+ A command-line tool that reads a `.docx` or `.txt` and scores each paragraph
49
+ 0–1 on how AI-generated it reads, using the `desklib/ai-text-detector-v1.01`
50
+ DeBERTa model — the one sitting at #1 on the RAID benchmark. Everything runs
51
+ on your own machine; after the first model download it never touches the
52
+ network. You point it at your Extended Essay, it tells you which paragraphs
53
+ sound like a language model wrote them.
54
+
55
+ The point isn't to cheat a detector. It's the opposite: I write my own drafts,
56
+ and sometimes my own honest prose still trips these classifiers because that's
57
+ what earnest formal writing looks like to them. This flags those paragraphs so
58
+ I can reword before a teacher runs Turnitin and has an awkward conversation
59
+ with me about it.
60
+
61
+ It is directional, not oracular. A high score means "reword this," not "you're
62
+ caught." It is not, and cannot be, the number Turnitin shows a teacher.
63
+
64
+ ```console
65
+ nick@aidetect:~$ aidetect score EE-clean.txt
66
+ P 1 0.08 [##------------------]
67
+ P 2 0.71 [##############------] <-- AI-ish
68
+ average AI score: 0.34 | 1/6 paragraphs flagged
69
+ reminder: directional only, not a Turnitin score.
70
+ ```
71
+
72
+ ## 🧠 The detection engine
73
+
74
+ | | feature | what it actually does |
75
+ |---|---|---|
76
+ | 01 | **per-paragraph scoring** | what it actually catches — splits your draft and scores each paragraph, so you fix the two bad ones instead of rewriting everything |
77
+ | 02 | **docx + txt input** | reads Word files straight (paragraphs, no headings) or plain text split on blank lines |
78
+ | 03 | **prose extractor** | `aidetect extract` strips headings, bullets, footnotes and your own note-scaffolding first, so the score is about writing, not structure |
79
+ | 04 | **offline after setup** | first run pulls ~1.5GB of model, every run after is airgapped — your essay never leaves the laptop |
80
+ | 05 | **second opinion** | cross-check against the lighter [Ejhfast/fast-ai-detector] when one model's paranoia isn't enough |
81
+ | 06 | **Binoculars (Gemma 4)** | a training-free perplexity-ratio detector — near chance with small Qwen pairs, but 96% on the labelled set once swapped to a Gemma 4 pair; see below |
82
+ | 07 | **IB word count** | `aidetect count` (`--json` for scripts and agents) counts what the IB counts — no headings, quotes, tables, footnotes, citations or bibliography — and splits the total by section, so an over-long draft tells you *where* |
83
+
84
+ ## 🚀 Run it
85
+
86
+ ```bash
87
+ uv tool install aidetect # or: pipx install aidetect
88
+ ```
89
+
90
+ ```bash
91
+ aidetect # list the five subcommands
92
+ aidetect count "draft.docx" --limit 4000 # IB word count, by section
93
+ aidetect count "draft.docx" --json # same, as one JSON object
94
+ aidetect score "draft.docx" # score a whole draft
95
+ aidetect score --text "one sentence" # score a single string
96
+ aidetect bino "draft.docx" --mlx --pair gemma
97
+ ```
98
+
99
+ `count` and `extract` are instant and need no model. `score` and `bino` need a
100
+ machine that can hold a transformer: built and tested on an 18GB Apple Silicon
101
+ Mac, MPS-accelerated. Their first run downloads the model and will sit there for
102
+ a minute — that's normal, not a hang. Every run after is fast and offline.
103
+
104
+ On Apple Silicon the Gemma 4 MLX pair installs automatically. Elsewhere it is
105
+ skipped and the Qwen pairs still work.
106
+
107
+ ### `--json`
108
+
109
+ `count` takes `--json` and prints exactly one object on stdout, nothing else:
110
+
111
+ ```json
112
+ {"sections": [{"title": "Introduction", "words": 812}], "total": 3940, "limit": 4000, "over": -60}
113
+ ```
114
+
115
+ Every key is always present. `limit` and `over` are `null` when no `--limit` was
116
+ given — that means "does not apply", not "could not be read". A draft with no
117
+ prose is an empty `sections` list and exit 0. Errors go to stderr with a non-zero
118
+ exit, so consumers branch on the exit code rather than parsing error text.
119
+
120
+ ## 🔩 Under the hood
121
+
122
+ ```mermaid
123
+ flowchart LR
124
+ A[.docx / .txt] --> B[aidetect extract<br/>strip non-prose]
125
+ B --> C[read_paragraphs<br/>>= 25 words]
126
+ C --> D[desklib DeBERTa<br/>mean-pool + sigmoid]
127
+ D --> E[per-paragraph<br/>0-1 score + flags]
128
+ ```
129
+
130
+ | file | job |
131
+ |---|---|
132
+ | `src/aidetect/cli.py` | the `aidetect` entry point — dispatches subcommands, importing each lazily so `count` never loads torch |
133
+ | `src/aidetect/text.py` | shared, torch-free: what counts as prose, what ends a document, how a `.docx` is read |
134
+ | `src/aidetect/count.py` | the IB word count — sections, citation stripping, budget |
135
+ | `src/aidetect/detect.py` | loads the desklib model, scores each paragraph, prints the bars and flags |
136
+ | `src/aidetect/extract.py` | pulls clean prose out of a `.docx` into a `.txt` — drops headings, bullets, note-labels |
137
+ | `src/aidetect/binoculars.py` | training-free perplexity-ratio scorer over a base+instruct LM pair (Qwen, or Gemma 4 via `--mlx`; see below) |
138
+ | `src/aidetect/calibrate.py` | fits a threshold on a labelled set you supply, saves it to `~/.config/aidetect` |
139
+ | `src/aidetect/paths.py` | where thresholds are looked up — `~/.config/aidetect` first, then the ones in the package |
140
+ | `src/aidetect/thresholds/` | the thresholds shipped with the package; a threshold you fit yourself wins over these |
141
+ | `corpora/` | my labelled calibration sets. Repo-only, deliberately not shipped in the package |
142
+ | `tests/` | count rules and Binoculars math, both self-checking, no model download |
143
+ | `pyproject.toml` | package metadata and dependencies — torch · transformers · python-docx, plus mlx-vlm on Apple Silicon |
144
+
145
+ ## 🔭 Binoculars: shelved, then revived by Gemma 4
146
+
147
+ [Binoculars](https://arxiv.org/abs/2401.12070) is a training-free detector: run
148
+ text through two LMs that share a tokenizer (a base "observer" and an instruct
149
+ "performer") and divide perplexity by cross-perplexity. Its designed model pair
150
+ is Falcon-7B ×2 (~28GB) — too big for an 18GB Mac, so the fallback was a small
151
+ same-family pair (Qwen2.5-0.5B or 1.5B) that fits.
152
+
153
+ `aidetect calibrate` scores a labelled set — mine is 12 real pre-2020 IB Extended
154
+ Essay paragraphs vs 12 LLM-written ones on the same topics — and finds the best
155
+ separating threshold. Measured across pairs:
156
+
157
+ | pair | best separation | chance |
158
+ |---|---|---|
159
+ | Qwen2.5-0.5B | 62% | 50% |
160
+ | Qwen2.5-1.5B | 67% | 50% |
161
+ | **Gemma 4 E2B** | **96%** | 50% |
162
+
163
+ The Qwen pairs sit near a coin flip: their human and AI score clusters almost
164
+ completely overlap, because the perplexity gap Binoculars exploits is sharp in
165
+ larger models and mush in sub-2B ones. That was the original negative result.
166
+ Swapping in a **Gemma 4** pair opens a clean gap (human mean 0.90 vs AI 0.71)
167
+ and separates the set at 96%. Gemma 4 ships as a multimodal checkpoint, so
168
+ `--mlx` quantizes it to 4-bit and runs it text-only through
169
+ [mlx-vlm](https://github.com/Blaizzy/mlx-vlm), fitting the 18GB Mac in ~6GB:
170
+
171
+ ```bash
172
+ aidetect bino IA-clean.txt --mlx --pair gemma # uses the shipped threshold
173
+
174
+ # refit the threshold on your own labelled set
175
+ aidetect calibrate --human-dir corpora/human --ai-dir corpora/ai --mlx --pair gemma
176
+ ```
177
+
178
+ desklib stays the primary detector; Binoculars is now a usable second opinion
179
+ rather than a dead end. The calibration sets are not shipped with the package —
180
+ clone the repo to reproduce the numbers, or point `--human-dir`/`--ai-dir` at
181
+ your own. Your fitted threshold lands in `~/.config/aidetect` and takes
182
+ precedence over the shipped one, so it survives an upgrade.
183
+
184
+ ## 🧪 The peer set: genre context, not a human class
185
+
186
+ The 12 human samples are Extended Essays: English, History, Biology, Physics,
187
+ Philosophy. A CS IA is a different animal, all database schemas, GUI components and
188
+ method-by-method justification, and technical prose is inherently more
189
+ predictable token-by-token, which drags perplexity-ratio scores down no matter
190
+ who typed it. So a CS IA scoring below the EE human mean means less than it
191
+ looks.
192
+
193
+ `corpora/peer/` holds 12 paragraphs of real IB Computer Science IA prose
194
+ (5 projects, 5 authors: sudokuMaster, IBOrganizer, MyCalendar, and two
195
+ IBO-published new-syllabus specimens). Measured against the same anchors:
196
+
197
+ | set | Binoculars mean | desklib mean |
198
+ |---|---|---|
199
+ | human (2008 EEs, verified pre-2020) | **0.90** | n/a |
200
+ | **peer (CS IAs, 2021–2025)** | **0.85** | **0.46** |
201
+ | ai (LLM-written, matched topics) | **0.71** | n/a |
202
+
203
+ The genre gap is real and it is about 0.05 on Binoculars. Score your IA against
204
+ `peer`, not against `human`.
205
+
206
+ **It is not a human class and it never fits a threshold.** Every source
207
+ postdates ChatGPT; three of the five were written in 2025. None carries an
208
+ authorship attestation, and in 2025 a fair share of student IAs were not
209
+ written unaided. Fold that into `human/` and any AI-assisted sample drags the
210
+ mean down, lowers the threshold, and the tool starts clearing drafts for the
211
+ wrong reason, a detector that reassures instead of measures. `aidetect calibrate`
212
+ reads only the two folders you name, so `peer/` stays out of threshold fitting
213
+ by construction, not by discipline.
214
+
215
+ What it can tell you: *"my prose scores like other IAs in this genre."* What it
216
+ can never tell you: *"my prose is human."* Matching a set you cannot vouch for
217
+ proves you are not an outlier, nothing more.
218
+
219
+ **Stack:** python · pytorch · transformers · mlx-vlm · desklib DeBERTa
220
+
221
+ The lighter cross-check tool lives at [Ejhfast/fast-ai-detector] — it's a
222
+ separate repo, not vendored here.
223
+
224
+ ---
225
+
226
+ <div align="center">
227
+
228
+ **[Nick Trimandylis](https://github.com/nitrimandylis)**
229
+
230
+ `I WRITE MY OWN ESSAYS — THIS JUST CHECKS THEY STILL READ LIKE IT`
231
+
232
+ MIT licensed — see [LICENSE](LICENSE).
233
+
234
+ </div>
235
+
236
+ [Ejhfast/fast-ai-detector]: https://github.com/Ejhfast/fast-ai-detector
@@ -0,0 +1,17 @@
1
+ aidetect/__init__.py,sha256=JmB2DMVDsV7YJ9mWFRPMCM9gryATeppjQdyT6cPzibk,77
2
+ aidetect/binoculars.py,sha256=eParkZIjAfHy0EgcO00-Pz28c1pQiXKW0Lwx4eEWNno,11140
3
+ aidetect/calibrate.py,sha256=gmwM9P9sCdS2y3obMmANqnNbP5Lzdz-gs--X5t6U5cY,5524
4
+ aidetect/cli.py,sha256=PZo5Y-guqI5GZdv1iU3ff484A6Oc0vlXS8A1zj_mVVo,1597
5
+ aidetect/count.py,sha256=_2DfSk-8laIk6UBUWEmyRZCQmA8gJa6IVdwv9Wv6nNo,4556
6
+ aidetect/detect.py,sha256=sRTQbD4wrdR_Ok9xk3Vhm_2mwBwyev9I9zeg-Y-Cl6Y,4463
7
+ aidetect/extract.py,sha256=hLUFXfErkk5sEG2BhkX9hmfRs6WHsAkDAGaMbDtV3Vw,1480
8
+ aidetect/paths.py,sha256=4mfijCfA-fmJ_vsRUyovAxrVvew6dTxjVkELafwGCWA,1039
9
+ aidetect/text.py,sha256=WMhP9Jyo_7tSw9bNb3OXUQZHqa-9IP4iCgSjSuOOcNE,2329
10
+ aidetect/thresholds/threshold-big.json,sha256=zUbg8AkNJWTzoadrPb7gEf5f1S3emg67IAlTR8CtKdU,198
11
+ aidetect/thresholds/threshold-gemma-mlx.json,sha256=uAr1cH4uYTVuAamNqJ9H9E6qBqgwFJA1UKkdr42RQXo,236
12
+ aidetect/thresholds/threshold-small.json,sha256=WE53N5jJ-hTYlMXlXIxy9TUwUgVdp-FxHEiHAdUp3Fs,197
13
+ aidetect-0.1.0.dist-info/METADATA,sha256=EZK2F3iue1rUmE_uEv1NvmnHpgzSgE66ecLqbUiKxw4,12235
14
+ aidetect-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
15
+ aidetect-0.1.0.dist-info/entry_points.txt,sha256=1ZHejHnZZ8c5gbi8KGxzgtSx8TUqCqy9ARhamI9KJVg,47
16
+ aidetect-0.1.0.dist-info/licenses/LICENSE,sha256=wMe8eI4xhFhIOGRWTFaIq0mTgeCpY099f91fHxofxCI,1073
17
+ aidetect-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ aidetect = aidetect.cli:main
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Nick Trimandylis
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.