@hraness/message-like-me 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/CHANGELOG.md +151 -0
  2. package/LICENSE +21 -0
  3. package/README.md +698 -0
  4. package/SECURITY.md +318 -0
  5. package/dist/agentic-messaging-v1.d.ts +179 -0
  6. package/dist/agentic-messaging-v1.js +52 -0
  7. package/dist/canonical-json.d.ts +3 -0
  8. package/dist/cli-bs3db5jr.js +643 -0
  9. package/dist/cli-d7qv38ab.js +485 -0
  10. package/dist/cli-kw20gkk3.js +5 -0
  11. package/dist/cli-qqafdvz9.js +5 -0
  12. package/dist/cli-ry4128kz.js +584 -0
  13. package/dist/cli-ththzwja.js +20 -0
  14. package/dist/cli-x1qncxm7.js +1078 -0
  15. package/dist/cli.js +9436 -0
  16. package/dist/ensoul-source-v1.d.ts +121 -0
  17. package/dist/ensoul-source-v1.js +24 -0
  18. package/dist/index.d.ts +3 -0
  19. package/dist/index.js +47 -0
  20. package/dist/message-bundle-v1-identity.d.ts +2 -0
  21. package/dist/message-bundle-v1.d.ts +205 -0
  22. package/dist/message-bundle-v1.js +37 -0
  23. package/dist/message-bundle-v2-identity.d.ts +2 -0
  24. package/dist/message-bundle-v2.d.ts +215 -0
  25. package/dist/message-bundle-v2.js +43 -0
  26. package/dist/metrics.d.ts +41 -0
  27. package/dist/types.d.ts +568 -0
  28. package/docs/local-message-bundle-v1.md +245 -0
  29. package/docs/local-message-bundle-v2.md +167 -0
  30. package/docs/methodology.md +322 -0
  31. package/docs/research.md +164 -0
  32. package/package.json +82 -0
  33. package/schema/ensoul-messages-source-v1.schema.json +248 -0
  34. package/schema/local-message-bundle-v1.schema.json +449 -0
  35. package/schema/local-message-bundle-v2.schema.json +462 -0
  36. package/schema/style-profile-v1.schema.json +223 -0
  37. package/schema/style-profile-v2.schema.json +202 -0
  38. package/skills/ensoul/LICENSE +23 -0
  39. package/skills/ensoul/NOTICE.md +7 -0
  40. package/skills/ensoul/SKILL.md +226 -0
  41. package/skills/ensoul/VENDORED_FROM.md +7 -0
  42. package/skills/ensoul/agents/openai.yaml +4 -0
  43. package/skills/ensoul/references/ensoul-source-packet-v1.schema.json +187 -0
  44. package/skills/ensoul/references/evidence-method.md +148 -0
  45. package/skills/ensoul/references/output-blueprint.md +143 -0
  46. package/skills/ensoul/references/source-packets.md +139 -0
  47. package/skills/ensoul/scripts/prepare_x_archive.py +467 -0
  48. package/skills/ensoul/scripts/validate_source_packet.py +477 -0
  49. package/skills/message-like-me/SKILL.md +229 -0
  50. package/skills/message-like-me/agents/openai.yaml +4 -0
  51. package/skills/message-like-me/references/analysis.md +159 -0
  52. package/skills/message-like-me/references/drafting.md +86 -0
  53. package/skills/message-like-me/references/ensoul.md +94 -0
  54. package/skills/message-like-me/references/evaluation.md +81 -0
  55. package/skills/message-like-me/references/privacy.md +106 -0
  56. package/skills/message-like-me/references/profile-schema.md +148 -0
@@ -0,0 +1,322 @@
1
+ # Methodology
2
+
3
+ Message Like Me separates deterministic measurement from semantic judgment.
4
+ The CLI reads local data, constructs versioned artifacts, and reports counts
5
+ and distributions. An Agent Skill interprets a bounded sample and drafts
6
+ unsent text. Neither component establishes that a draft is what the user would
7
+ have written.
8
+
9
+ ## Data boundary
10
+
11
+ The Messages database and optional AddressBook databases remain authoritative.
12
+ The CLI makes stable private snapshots and opens only those snapshots through
13
+ SQLite. It does not modify Messages, Contacts, their databases, or their
14
+ transactional sidecars.
15
+
16
+ A caller-owned local message bundle is a separate versioned source
17
+ observation. The CLI verifies its complete fixed inventory and digests before
18
+ ingest, never obtains its provider credential, and does not call its producer.
19
+ Frozen v1 carries bounded Beeper observations. V2 carries one native WhatsApp
20
+ account exported through Wrench's official Wacli adapter. It requires exact
21
+ WhatsApp JIDs and projects a phone handle only from an E.164-backed user JID.
22
+
23
+ A caller-owned X data archive is another offline source observation. The CLI
24
+ parses bounded supported entries directly from the owner-only ZIP without
25
+ extracting files, evaluating archive JavaScript, accessing a network, or
26
+ downloading media. It preserves exact archive and account provenance. The
27
+ archive source covers direct messages, not X Chat. Bounded reply and mention
28
+ identity observations from selected tweet members may associate a provider user
29
+ ID with an X handle or display name; tweet prose is not added to the messaging
30
+ corpus or used as style evidence.
31
+
32
+ The normalized corpus, private installation key, aggregate metrics, profiles,
33
+ and drafting context stay in the local data root. Study, Ensoul source,
34
+ evaluation, and agent handoff files are written only to explicit paths.
35
+ Ordinary views use keyed pseudonymous IDs and omit bodies, handles, contact
36
+ names, and group titles.
37
+
38
+ This is a process boundary, not encryption. Another process running as the
39
+ same user, a compromised host, a device backup, or an agent provider that is
40
+ given a packet may still receive private data. Incoming messages also belong
41
+ to other participants. They provide response context but never become samples
42
+ of the owner's prose in style analysis. A separately requested contact-subject
43
+ Ensoul packet may attribute incoming text to that exact direct person after
44
+ rebasing direction; its owner-authored records then become counterpart context.
45
+
46
+ ## Normalized observations
47
+
48
+ The corpus preserves source, account, network, message direction, provider
49
+ ordering, timestamp, body availability and source, message kind, attachment
50
+ metadata, edit or retraction metadata, reply target and reply observability,
51
+ service, and conversation membership where the source supports them.
52
+ Unsupported, deleted, or truncated text remains unavailable rather than being
53
+ reconstructed. A truncated text record still represents a message bubble for
54
+ tempo and any observable reply evidence, but never contributes prose.
55
+
56
+ Native iMessage history and each connected bundle account have distinct source
57
+ namespaces. A bounded, truncated, or unknown bundle is not an authoritative
58
+ statement that omitted history no longer exists. Reimport merges present
59
+ records with retained state. Only explicit deletion, removal, replacement, or
60
+ tombstone state suppresses evidence, and a later record reappearance clears
61
+ that suppression. Bundle creation times are monotonic per source, so an older
62
+ snapshot cannot resurrect or overwrite newer state.
63
+
64
+ An X archive normally has its own source namespace and retains its exact ZIP
65
+ and account provenance. The caller may name an existing Beeper X source as an
66
+ overlap only when both sources describe the same exact account. Reconciliation
67
+ is limited to one-to-one direct conversations and requires an exact peer handle
68
+ plus exact shared-message evidence. It retains both provenances and lets one
69
+ proven exact message contribute once. Group DMs remain separate because the
70
+ legacy archive cannot establish cross-provider sender identity strongly enough.
71
+ Missing or contradictory evidence fails closed. Proven equivalence survives
72
+ later reingests, so the same message does not return as a duplicate. Archive
73
+ absence does not suppress retained history.
74
+
75
+ A native Wacli WhatsApp bundle normally has its own source namespace. If the
76
+ caller names an existing Beeper WhatsApp source, the two reconcile only after
77
+ exact self-account E.164, exact direct-peer E.164, and unambiguous shared text
78
+ message proof. Groups and bodyless records cannot prove equivalence. Both
79
+ provenances and unique history remain. Exact duplicates contribute once, the
80
+ native conversation becomes the preferred private `whatsappJid` route, and its
81
+ proven Beeper route remains evidence-only. This route preference does not grant
82
+ Message Like Me provider access or sending authority.
83
+
84
+ The analysis uses several operational units:
85
+
86
+ - A **message** is one source record. Only outgoing text bodies contribute to
87
+ surface-style measurements.
88
+ - A **burst** is a run of messages in one direction whose neighboring records
89
+ remain within the configured burst gap. The default gap is five minutes.
90
+ - A **session** is a run of conversation activity without a gap longer than
91
+ the configured session gap. The default gap is eight hours.
92
+ - A **response episode** pairs an incoming burst with the next outgoing burst
93
+ in the same session.
94
+ - An **explicit reply** is source metadata linking a message to an earlier
95
+ message. It is distinct from a reaction or an ordinary adjacent response.
96
+ Reply observability is recorded separately: X data archives do not expose
97
+ reply links, so their messages are unavailable rather than observed
98
+ non-replies.
99
+ - A **reaction** is counted as interaction behavior, not authored prose. A
100
+ reaction without a provider timestamp contributes to counts and direction
101
+ but not to temporal order, sessions, bursts, or response episodes. Raw
102
+ provider reaction values remain private and are not categorical dimensions
103
+ in aggregate metrics or drafting context.
104
+
105
+ Five minutes and eight hours are reproducible segmentation parameters, not
106
+ claims about natural conversational boundaries. Every metrics artifact records
107
+ the parameters used. Comparisons are meaningful only when their definitions
108
+ match.
109
+
110
+ ## Deterministic metrics
111
+
112
+ For each conversation, the CLI reports the evidence window and counts of
113
+ incoming, outgoing, text, session, burst, and response records. Tempo metrics
114
+ include response-latency quantiles, outgoing messages per response, the ratio
115
+ of single-message to multi-message responses, multi-message inbound contexts,
116
+ visible multi-question contexts, and explicit reply frequency. Reply metrics
117
+ report explicit, eligible, and unavailable messages separately, and calculate
118
+ the ratio only from eligible messages. Session, burst, and response
119
+ construction runs independently for each source conversation before
120
+ person-scope results are combined. Adjacent timestamps in two apps or threads
121
+ never create one artificial episode. Mixed person scopes expose their sorted
122
+ service breakdown.
123
+
124
+ Surface measurements cover characters and words, lowercase starts, terminal
125
+ punctuation, question and exclamation marks, emoji-bearing messages, and
126
+ multiline messages. These are observable features, not explanations. For
127
+ example, visible question marks are only a proxy for questions, and a long
128
+ latency cannot reveal whether the user was busy, asleep, deciding what to say,
129
+ or simply missing local history.
130
+
131
+ Session starts and ends are likewise structural facts under the configured
132
+ threshold. They do not identify who cares more, who is avoiding whom, or the
133
+ nature of a relationship.
134
+
135
+ ## Bounded semantic study
136
+
137
+ `study prepare` selects response episodes that contain both incoming and
138
+ outgoing text. The version-one selector favors coverage of different response
139
+ shapes, tags, lengths, reply use, latency bands, and positions across the
140
+ available time window. It is deterministic for the same corpus, bounds, and
141
+ parameters.
142
+
143
+ The CLI default packet limit is 24 examples. Each emitted body is capped at 4 KiB,
144
+ each direction keeps at most 12 text messages per example, and total emitted
145
+ body text is capped at 256 KiB. Coverage metadata states what was truncated or
146
+ omitted. A packet is a sample of response contexts, not a transcript.
147
+
148
+ Version 0.2 added temporal bounds to study selection. A profile intended for
149
+ held-out evaluation should use only examples before the chosen cutoff. The
150
+ cutoff, corpus revision, selection parameters, packet receipt, and evidence
151
+ window form part of the analysis provenance. Profile validity uses a digest of
152
+ the selected person or conversation scope within those exact time bounds, so a
153
+ later message outside a closed study window does not rewrite its evidence.
154
+
155
+ The agent studies prose, delivery shape, multi-point response strategy, reply
156
+ use, and exceptions. It must keep measured facts separate from interpretations
157
+ and cite study-example IDs instead of copying private phrases into a profile.
158
+ The resulting profile records its contact scope, corpus revision, exact packet
159
+ digest, analysis time, contextual rules, limitations, and confidence.
160
+
161
+ Profile provenance proves which artifact was analyzed. It does not prove that
162
+ the agent interpreted that artifact correctly.
163
+
164
+ ## Subject-relative Ensoul source packets
165
+
166
+ `ensoul prepare` applies the same bounded diverse response selector to one
167
+ pinned contact corpus, but its unit of attribution is the explicitly selected
168
+ subject. For `--subject owner`, stored outgoing messages remain subject-authored
169
+ and incoming messages remain counterpart context. For `--subject contact`, the
170
+ adapter first reverses direction, then recomputes sessions, bursts, responses,
171
+ and selection. This makes clearly attributable incoming prose the contact's
172
+ subject evidence and the owner's outgoing prose its response context.
173
+
174
+ Contact attribution is valid only for an exact `person_` scope created by a
175
+ complete one-to-one AddressBook match. Conversation aliases, unmatched direct
176
+ threads, shared handles, and groups are rejected because the normalized corpus
177
+ cannot establish a safe contact author for them. Owner packets may use a named
178
+ conversation or person scope, but groups and any multi-participant scope are
179
+ rejected. The packet retains redacted scope kind, conversation count, and sorted
180
+ service labels so channel dependence stays visible without names or coordinates.
181
+
182
+ The output specializes `ensoul.source-packet.v1` with payload schema
183
+ `ensoul.messages-source.v1`. Records contain only selected active text bodies,
184
+ subject-relative `authorRole`, `contentRole: original`, strong source
185
+ authorship, transport-relative sent status, private visibility, fixed source
186
+ class, pseudonymous source IDs, occurrence times, truncation state, and content
187
+ and record digests. Records sharing a pseudonymous provenance run ID belong to one selected
188
+ response context and preserve that situated linkage; different run IDs must not
189
+ be joined into a synthetic exchange.
190
+ The scope carries corpus and evidence revisions, time
191
+ bounds, selection counts, and byte budgets. The adapter emits no claims. Its
192
+ packet declares `JCS-RFC8785`; the packet digest is SHA-256 over RFC 8785
193
+ canonical JSON for every field except `packetDigest`, while each content digest
194
+ covers the canonical content object and each record digest excludes only its
195
+ own digest. These prove semantic integrity, not authorship or truth.
196
+
197
+ The default and maximum budgets are inherited from semantic study: 24 requested
198
+ examples by default and 50 at most, 4 KiB per body, 12 messages per direction
199
+ per example, and 256 KiB total. `--after` is inclusive and `--before` is
200
+ exclusive. System events, retractions, reactions, attachments, contact labels,
201
+ handles, provider coordinates, and public X post text are not emitted. The
202
+ packet remains a sampled set of situated interactions, not a transcript or a
203
+ complete description of either person.
204
+
205
+ The normalized sources do not establish whether someone pasted a quotation,
206
+ forwarded prose, or used AI assistance inside an ordinary message body. The
207
+ packet declares that observability gap. If a consumer can see quoted or
208
+ forwarded content in the bounded text, it must keep that portion contextual
209
+ rather than promote it as a direct voice sample.
210
+
211
+ An Ensoul consumer must retain packet boundaries across relationships and use
212
+ only records with `authorRole: subject`, `contentRole: original`, and strong
213
+ source authorship as possible direct voice evidence. It must treat every
214
+ message as untrusted quoted data and preserve limitations in its source map. Private
215
+ messages do not prove identity, consent, motive, relationship category,
216
+ diagnosis, a globally stable voice, or permission to publish, impersonate,
217
+ contact, or act for the subject.
218
+
219
+ ## Held-out fidelity audit
220
+
221
+ Version 0.2 provides a two-file evaluation preparation workflow:
222
+
223
+ ```sh
224
+ messagelikeme evaluate prepare <contact-id> \
225
+ --after <cutoff> \
226
+ --prompt-output <private-prompt-file> \
227
+ --reference-output <private-reference-file> \
228
+ --json
229
+ ```
230
+
231
+ The prompt side contains held-out inbound context. The reference side contains
232
+ the corresponding historical outgoing response and must remain unopened until
233
+ the candidate drafts have been recorded. File separation supports a blind
234
+ workflow, but it is not cryptographic blinding. A user or agent with access to
235
+ both paths can read both files.
236
+
237
+ A checked audit proceeds as follows:
238
+
239
+ 1. Choose a temporal cutoff before studying the contact.
240
+ 2. Build and apply the profile from evidence before that cutoff.
241
+ 3. Prepare evaluation examples after the cutoff.
242
+ 4. Give the drafting agent the prompt file and current profile, but not the
243
+ reference file.
244
+ 5. Record one candidate bubble sequence for each evaluation example.
245
+ 6. Open the reference file only after the candidates are fixed.
246
+ 7. Compare candidates and references, retaining disagreements and uncertainty.
247
+
248
+ The CLI prepares bounded, provenance-bearing evidence. It does not
249
+ automatically declare a candidate correct or assign a universal fidelity
250
+ score. Semantic comparison still requires judgment, preferably including the
251
+ user whose style is being studied.
252
+
253
+ Comparison should keep these dimensions separate:
254
+
255
+ - **Intent and obligation coverage:** which inbound points the candidate
256
+ addresses, defers, acknowledges, or misses.
257
+ - **Meaning and factuality:** whether the candidate invents facts, changes
258
+ commitments, or imports a historical belief that does not belong in the
259
+ current draft.
260
+ - **Prose:** register, directness, warmth, sentence shape, punctuation, and
261
+ other supported tendencies.
262
+ - **Delivery shape:** bubble count, ordering, per-bubble length, follow-ups,
263
+ and explicit reply choice.
264
+ - **Privacy:** reuse of names, distinctive phrases, anecdotes, or details that
265
+ came from another historical context.
266
+ - **Calibration:** whether sparse or contradictory evidence should have caused
267
+ the agent to use a neutral default or ask the user.
268
+
269
+ The historical response is a reference observation, not a unique correct
270
+ answer. The user might reasonably respond differently now. Results should be
271
+ reported by dimension and example, alongside an unprofiled drafting baseline
272
+ when possible. A single similarity score hides the failures that matter most.
273
+
274
+ ## Drafting method
275
+
276
+ Ordinary drafting uses the current deterministic context and the applicable
277
+ validated profile. The user's present intent, supplied facts, uncertainty,
278
+ and requested format outrank historical style. Contact-specific rules apply
279
+ only to their supported scope; context rules can override broad tendencies.
280
+
281
+ The output may be one message or several separately presented bubbles. A
282
+ historical latency distribution never instructs the agent to delay its answer.
283
+ Every result remains an unsent candidate. Message Like Me has no command for
284
+ sending, reacting, scheduling, or operating a messaging application.
285
+
286
+ ## Sources of error
287
+
288
+ Reported behavior can be distorted by:
289
+
290
+ - Messages that are not synchronized to the Mac or are no longer present;
291
+ - partial local provider exports whose completeness bounds exclude older or
292
+ remote history;
293
+ - X archives that predate recent messages, omit X Chat, or cannot report
294
+ explicit reply links;
295
+ - unsupported body encodings, attachments, edits, retractions, or source
296
+ schema changes;
297
+ - ambiguous or stale Contacts labels;
298
+ - group conversations whose audience changes over time;
299
+ - timezone changes, work schedules, sleep, travel, notification settings, and
300
+ device availability;
301
+ - a bounded packet that underrepresents rare but important contexts;
302
+ - simple surface proxies that miss pragmatic meaning;
303
+ - model and prompt differences in the agent interpreting a packet; and
304
+ - genuine drift in how the user communicates.
305
+
306
+ For reproducibility, retain schema versions, corpus revision, packet digest,
307
+ time bounds, segmentation parameters, budget and coverage fields, and the
308
+ agent environment used for semantic analysis. Re-ingest and re-evaluate when
309
+ the corpus changes materially. Describe uncertainty instead of broadening a
310
+ contact-specific observation into an identity claim.
311
+
312
+ ## What the method can support
313
+
314
+ The checked artifacts can support statements such as "within this evidence
315
+ window, multi-message responses were more common for this conversation" or
316
+ "the held-out candidate reproduced the historical bubble count but missed one
317
+ inbound question."
318
+
319
+ They cannot support statements that the system has cloned the user, recovered
320
+ their personality, diagnosed a relationship, proved authorship, predicted a
321
+ future decision, obtained a contact's consent, or produced a message approved
322
+ by the user.
@@ -0,0 +1,164 @@
1
+ # Research and prior art
2
+
3
+ Message Like Me is a local measurement and profiling layer for message
4
+ drafting. It is not a model, an autonomous messaging agent, or a claim that a
5
+ software system represents a person. This review explains the evidence behind
6
+ that boundary and the neighboring open-source work that informed it.
7
+
8
+ The cited papers are primary research publications or preprints. Project
9
+ descriptions link to their official repositories. A paper result is evidence
10
+ about the task and population it evaluated, not proof that the same result
11
+ holds for private conversations across the messaging sources a user imports.
12
+
13
+ ## Personalization is contextual
14
+
15
+ [PersonaChat](https://aclanthology.org/P18-1205/) found that conditioning a
16
+ dialogue system on both its assigned profile and information about its
17
+ interlocutor improved next-utterance prediction. More recent work on
18
+ [linguistic accommodation](https://aclanthology.org/2025.sigdial-1.16/) found
19
+ that human answers aligned more with a partner's style than LLM answers did,
20
+ while the LLM answers aligned more closely in semantic content.
21
+
22
+ These results do not establish a universal method for imitating a person.
23
+ They support a narrower design inference: a useful messaging profile should
24
+ separate broadly repeated tendencies from contact-specific and
25
+ context-specific adjustments. Message Like Me therefore treats incoming
26
+ messages as response context and only the user's outgoing messages as evidence
27
+ of the user's prose.
28
+
29
+ [Catch Me If You Can? Not Yet](https://aclanthology.org/2025.findings-emnlp.532/)
30
+ evaluates nuanced individual style in informal communication, a task close to
31
+ private messaging. Its scope reinforces the same boundary: measured tendencies
32
+ can guide a draft without establishing a faithful digital copy of its author.
33
+
34
+ [LaMP](https://aclanthology.org/2024.acl-long.399/) evaluated personalized
35
+ classification and generation from user histories and found retrieval-based
36
+ personalization useful across most of its tasks. Its experiments included
37
+ term, semantic, and time-aware retrieval. [PEARL](https://aclanthology.org/2024.customnlp4u-1.16/)
38
+ studied personalized writing assistance and trained a retriever to select
39
+ historical documents according to their downstream generation value. PEARL
40
+ also used retrieval quality to identify outputs likely to need revision.
41
+
42
+ The project inference is selective rather than exhaustive use of history. A
43
+ small response-context sample with explicit coverage limits exposes less text
44
+ than placing an entire transcript in an agent prompt. Recency,
45
+ relationship, and current conversational purpose can all change which past
46
+ examples apply.
47
+
48
+ ## Personalization needs held-out evaluation
49
+
50
+ [ExPerT](https://aclanthology.org/2025.findings-acl.900/) evaluates
51
+ personalized long-form generation by comparing evidence-bearing aspects of
52
+ content and writing style separately. Its reported agreement with human
53
+ judgment improved over the comparison methods in that study. It does not
54
+ measure message timing, bubble boundaries, or reply-link behavior.
55
+
56
+ [Can You Make It Sound Like You?](https://aclanthology.org/2026.acl-long.2030/)
57
+ studies personalized writing through human review and post-editing. That
58
+ workflow supports Message Like Me's product boundary: the output is an unsent
59
+ candidate for the user to inspect and revise, not an autonomous act on the
60
+ user's behalf.
61
+
62
+ [Münker, Schwager, and Rettinger](https://arxiv.org/abs/2506.21974) tested
63
+ LLM-based imitation of social-network communication and argue that a
64
+ simulation must be validated for empirical realism in the setting where it
65
+ was fitted. This supports holding later conversations out of profile creation,
66
+ drafting from their inbound context without seeing the historical response,
67
+ and only then comparing the candidate with the reference. A match on surface
68
+ features alone does not establish semantic equivalence, authorship, or
69
+ identity.
70
+
71
+ ## Digital-agent results are task-bounded
72
+
73
+ [Generative Agents](https://arxiv.org/abs/2304.03442) showed that stored
74
+ experiences, retrieval, reflection, and planning can produce believable agent
75
+ behavior in a simulated town. Believability in that environment is not the
76
+ same as fidelity to a real individual.
77
+
78
+ [Generative Agent Simulations of 1,000 People](https://arxiv.org/abs/2411.10109)
79
+ built agents from two-hour interviews with 1,052 participants and evaluated
80
+ them on surveys, personality measures, and experimental replications. The
81
+ reported survey result is relative to how consistently participants repeated
82
+ their own answers two weeks later. It is not evidence that the agents could
83
+ write private messages like those participants or act on their behalf.
84
+
85
+ The accompanying [official repository](https://github.com/StanfordHCI/genagents)
86
+ does not publish the interview-derived individual agent bank. It describes
87
+ aggregated access for fixed tasks and reviewed access for individual outputs
88
+ because of participant privacy. That is useful precedent: an open-source
89
+ method can remain public while real person-level evidence stays private.
90
+
91
+ Message Like Me consequently uses the terms *style profile*, *historical
92
+ tendency*, and *draft candidate*. It does not use a messaging profile to infer
93
+ beliefs, personality, relationship quality, future behavior, or authority to
94
+ represent the user.
95
+
96
+ ## Private text can remain revealing
97
+
98
+ [Quantifying Memorization Across Neural Language Models](https://arxiv.org/abs/2202.07646)
99
+ found that extractable memorization increased with model capacity, repeated
100
+ examples, and longer prompting context in the evaluated model families.
101
+ [Beyond Memorization](https://arxiv.org/abs/2310.07298) showed that LLMs could
102
+ infer personal attributes from text even when the task was not extraction of a
103
+ memorized training example. Removing obvious names is therefore not a complete
104
+ privacy defense.
105
+
106
+ [When Personalization Misleads](https://aclanthology.org/2026.findings-acl.395/)
107
+ found that personalization could steer factual answers toward a user's prior
108
+ history rather than objective truth in its evaluated settings. The project
109
+ inference is that meaning, current intent, and factual correctness must outrank
110
+ style fidelity.
111
+
112
+ [NIST's digital identity risk guidance](https://pages.nist.gov/800-63-4/sp800-63/dirm/)
113
+ lists impersonation, privacy loss, and reputational damage among relevant
114
+ harms. Message Like Me reduces those risks through local storage, bounded
115
+ exports, pseudonymous ordinary views, explicit provenance, and the absence of
116
+ message-sending commands. Those controls do not create consent from a
117
+ conversation partner or make a hosted agent local.
118
+
119
+ ## Tempo is relational and descriptive
120
+
121
+ A 2026 preprint on [response times in donated WhatsApp and Instagram chats](https://arxiv.org/abs/2605.03687)
122
+ reported persistent response-speed similarity between chat partners in its
123
+ sample. This is preliminary evidence from different platforms and cannot set a
124
+ norm for users of any supported messaging source. It does support comparing
125
+ tempo within a dyad instead of treating one global latency distribution as a
126
+ personal rule.
127
+
128
+ Historical latency is affected by sleep, work, travel, notifications, device
129
+ availability, urgency, and missing data. Message Like Me reports it as
130
+ descriptive metadata. It does not tell an agent to wait before returning a
131
+ draft or portray a historical delay as a preference or promise.
132
+
133
+ ## Open-source landscape
134
+
135
+ | Project | Public scope | Relevant distinction |
136
+ | --- | --- | --- |
137
+ | [OpenSelf](https://github.com/Open-Self/Open-Self) | Self-hosted profile and memory system that can automatically reply through WhatsApp, Telegram, and Discord | Message Like Me deliberately ends at an inspectable, unsent draft and does not simulate typing, delay delivery, or operate a messaging account. |
138
+ | [Second-Me](https://github.com/mindverse/Second-Me) | Locally trained and hosted "AI self" with memory, model alignment, and a network | Message Like Me does not train a model, construct an identity, or join an agent network. |
139
+ | [Doppelganger](https://github.com/NotYuSheng/Doppelganger) | LoRA fine-tuning from chat exports | Its documentation warns that trained models can reproduce private data and other participants' text. Message Like Me keeps analysis in inspectable profiles instead of weights. |
140
+ | [Write Like Me](https://github.com/Hiro-Inagawa/write-like-me) | Stylometric profiles for several writing registers with held-out verification | It is useful prior art for measured profiles and verification. Message Like Me focuses on dyadic response context, bubble shape, tempo, and reply behavior. |
141
+ | [imessage-exporter](https://github.com/ReagentX/imessage-exporter) | Broad read-only parsing and export of modern Messages features | It is a valuable compatibility reference. Its GPL-3.0 implementation is not copied into this MIT project. |
142
+ | [iMessageAnalyzer](https://github.com/dsouzarc/imessageanalyzer) | Conversation statistics, starters, and successive-message analysis | Its fixed time thresholds are prior art, not universal conversational facts. Message Like Me records its own versioned thresholds with results. |
143
+ | [iMCP](https://github.com/mattt/iMCP) | Sandboxed native access to Messages and the Contacts framework | It demonstrates a possible future native Contacts boundary. The current CLI uses bounded read-only database snapshots. |
144
+ | [imessage-rag](https://github.com/sapochat/imessage-rag) | Local retrieval and question answering over message history | Retrieval of facts from conversations is a different task from measuring the user's outgoing style. |
145
+
146
+ ## Explicit limitations
147
+
148
+ The research above does not establish that:
149
+
150
+ - one stable profile captures how a person writes to every contact;
151
+ - linguistic similarity implies the same intent, judgment, or factual answer;
152
+ - a historical response is the response the user would choose now;
153
+ - contact frequency, latency, or warmth reveals relationship quality;
154
+ - a model-generated draft was authored, approved, or sent by the user;
155
+ - pseudonymous identifiers anonymize a corpus against someone with access to
156
+ the store or installation key;
157
+ - local CLI processing controls the data practices of the agent environment
158
+ that opens a study packet; or
159
+ - possession of a conversation database grants permission to publish,
160
+ fine-tune on, or impersonate its participants.
161
+
162
+ The defensible claim is smaller: Message Like Me measures selected historical
163
+ messaging behavior, keeps semantic interpretations tied to bounded evidence,
164
+ and helps an already-running agent produce an unsent candidate for user review.
package/package.json ADDED
@@ -0,0 +1,82 @@
1
+ {
2
+ "name": "@hraness/message-like-me",
3
+ "version": "0.8.0",
4
+ "description": "A local-first CLI and Agent Skill for studying private messaging history and drafting messages that sound like you.",
5
+ "license": "MIT",
6
+ "type": "module",
7
+ "packageManager": "bun@1.3.14",
8
+ "engines": {
9
+ "bun": ">=1.3.14"
10
+ },
11
+ "repository": {
12
+ "type": "git",
13
+ "url": "git+https://github.com/hraness/message-like-me.git"
14
+ },
15
+ "homepage": "https://messagelikeme.com",
16
+ "bugs": {
17
+ "url": "https://github.com/hraness/message-like-me/issues"
18
+ },
19
+ "keywords": [
20
+ "agent-skill",
21
+ "beeper",
22
+ "bun",
23
+ "cli",
24
+ "imessage",
25
+ "local-first",
26
+ "messaging",
27
+ "messaging-style",
28
+ "privacy",
29
+ "wacli",
30
+ "whatsapp"
31
+ ],
32
+ "bin": {
33
+ "messagelikeme": "./dist/cli.js"
34
+ },
35
+ "exports": {
36
+ ".": {
37
+ "types": "./dist/index.d.ts",
38
+ "import": "./dist/index.js"
39
+ },
40
+ "./message-bundle-v1": {
41
+ "types": "./dist/message-bundle-v1.d.ts",
42
+ "import": "./dist/message-bundle-v1.js"
43
+ },
44
+ "./message-bundle-v2": {
45
+ "types": "./dist/message-bundle-v2.d.ts",
46
+ "import": "./dist/message-bundle-v2.js"
47
+ },
48
+ "./agentic-messaging-v1": {
49
+ "types": "./dist/agentic-messaging-v1.d.ts",
50
+ "import": "./dist/agentic-messaging-v1.js"
51
+ },
52
+ "./ensoul-source-v1": {
53
+ "types": "./dist/ensoul-source-v1.d.ts",
54
+ "import": "./dist/ensoul-source-v1.js"
55
+ }
56
+ },
57
+ "files": [
58
+ "dist",
59
+ "docs",
60
+ "schema",
61
+ "skills",
62
+ "CHANGELOG.md",
63
+ "README.md",
64
+ "SECURITY.md",
65
+ "LICENSE"
66
+ ],
67
+ "scripts": {
68
+ "build": "bun scripts/build-dist.ts",
69
+ "test": "bun test ./src ./scripts",
70
+ "typecheck": "tsc --noEmit -p tsconfig.json",
71
+ "check:skill": "bun scripts/validate-skill.ts",
72
+ "check:standalone": "bun scripts/check-standalone.ts",
73
+ "check:dist": "bun scripts/check-dist.ts",
74
+ "check:package": "bun scripts/package-smoke.ts",
75
+ "check": "bun run typecheck && bun run test && bun run check:skill && bun run check:standalone && bun run build && bun run check:dist && bun run check:package",
76
+ "prepack": "bun run check"
77
+ },
78
+ "devDependencies": {
79
+ "@types/bun": "1.3.14",
80
+ "typescript": "6.0.3"
81
+ }
82
+ }