twittertools-mcp 0.0.0-stage → 1.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +67 -2
- package/package.json +38 -4
- package/server.js +206 -0
- package/xrules.js +241 -0
package/README.md
CHANGED
|
@@ -1,3 +1,68 @@
|
|
|
1
|
-
#
|
|
1
|
+
# twittertools-mcp
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
[Model Context Protocol](https://modelcontextprotocol.io) server for
|
|
4
|
+
[TwitterTools](https://twittertools.com) — brings the toolkit to AI agents
|
|
5
|
+
(Claude, ChatGPT, Cursor, ZCode, …). No login, no tracking, no API keys.
|
|
6
|
+
|
|
7
|
+
## Tools
|
|
8
|
+
|
|
9
|
+
| Tool | What it does |
|
|
10
|
+
| --- | --- |
|
|
11
|
+
| `get_tweet` | Fetch a public post by URL or ID: text, author, media with direct CDN URLs |
|
|
12
|
+
| `get_thread` | Unroll a thread into its posts, in order (partial results are flagged) |
|
|
13
|
+
| `count_chars` | Measure text against X's real 280-character limit (CJK ×2, links = 23) |
|
|
14
|
+
| `split_thread` | Split long text into numbered, limit-fitting posts |
|
|
15
|
+
| `parse_tweet_url` | URL/ID → ID, author handle, posting time, canonical permalink |
|
|
16
|
+
|
|
17
|
+
`get_tweet` / `get_thread` call the stateless [twittertools.com](https://twittertools.com)
|
|
18
|
+
API, which resolves public posts through X's free syndication endpoint. The other
|
|
19
|
+
three run entirely offline. Media is **not** proxied: tweet results carry the
|
|
20
|
+
direct `pbs.twimg.com` / `video.twimg.com` URLs, which any client can fetch on its own.
|
|
21
|
+
|
|
22
|
+
## Install
|
|
23
|
+
|
|
24
|
+
```sh
|
|
25
|
+
# Claude Code
|
|
26
|
+
claude mcp add twittertools -- npx -y twittertools-mcp
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
Or add it to any client that reads an `mcpServers` config
|
|
30
|
+
(Claude Desktop's `claude_desktop_config.json`, Cursor's `.cursor/mcp.json`, …):
|
|
31
|
+
|
|
32
|
+
```json
|
|
33
|
+
{
|
|
34
|
+
"mcpServers": {
|
|
35
|
+
"twittertools": {
|
|
36
|
+
"command": "npx",
|
|
37
|
+
"args": ["-y", "twittertools-mcp"]
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
Requires Node 18+.
|
|
44
|
+
|
|
45
|
+
## Self-hosting
|
|
46
|
+
|
|
47
|
+
Point `TWITTERTOOLS_API_BASE` at your own instance of
|
|
48
|
+
[`server/downloader_server.py`](../server/) to keep lookups entirely on your
|
|
49
|
+
infrastructure:
|
|
50
|
+
|
|
51
|
+
```json
|
|
52
|
+
{ "command": "npx", "args": ["-y", "twittertools-mcp"], "env": { "TWITTERTOOLS_API_BASE": "https://your-instance.example" } }
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
## Privacy
|
|
56
|
+
|
|
57
|
+
The server is stateless: nothing is stored, no telemetry, no accounts. It talks
|
|
58
|
+
only to the configured API base. The API itself is rate-limited per IP and keeps
|
|
59
|
+
only aggregate, anonymous counters — see [twittertools.com/privacy](https://twittertools.com/privacy).
|
|
60
|
+
|
|
61
|
+
## Development
|
|
62
|
+
|
|
63
|
+
```sh
|
|
64
|
+
npm install
|
|
65
|
+
cd .. && npx vitest run mcp # unit + stdio protocol tests (offline, mock API)
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
MIT — same as the [main project](https://github.com/steley/twittertools).
|
package/package.json
CHANGED
|
@@ -1,6 +1,40 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "twittertools-mcp",
|
|
3
|
-
"version": "
|
|
4
|
-
"
|
|
5
|
-
"description": "
|
|
6
|
-
|
|
3
|
+
"version": "1.0.1",
|
|
4
|
+
"mcpName": "io.github.steley/twittertools-mcp",
|
|
5
|
+
"description": "MCP server for twittertools.com — fetch X (Twitter) posts and threads, count and split posts with X's real rules. No login, no tracking.",
|
|
6
|
+
"license": "MIT",
|
|
7
|
+
"type": "module",
|
|
8
|
+
"bin": {
|
|
9
|
+
"twittertools-mcp": "server.js"
|
|
10
|
+
},
|
|
11
|
+
"files": [
|
|
12
|
+
"server.js",
|
|
13
|
+
"xrules.js"
|
|
14
|
+
],
|
|
15
|
+
"engines": {
|
|
16
|
+
"node": ">=18"
|
|
17
|
+
},
|
|
18
|
+
"keywords": [
|
|
19
|
+
"mcp",
|
|
20
|
+
"model-context-protocol",
|
|
21
|
+
"twitter",
|
|
22
|
+
"x",
|
|
23
|
+
"tweet",
|
|
24
|
+
"thread",
|
|
25
|
+
"twittertools"
|
|
26
|
+
],
|
|
27
|
+
"repository": {
|
|
28
|
+
"type": "git",
|
|
29
|
+
"url": "git+https://github.com/steley/twittertools.git",
|
|
30
|
+
"directory": "mcp"
|
|
31
|
+
},
|
|
32
|
+
"homepage": "https://twittertools.com",
|
|
33
|
+
"bugs": "https://github.com/steley/twittertools/issues",
|
|
34
|
+
"scripts": {
|
|
35
|
+
"start": "node server.js"
|
|
36
|
+
},
|
|
37
|
+
"dependencies": {
|
|
38
|
+
"@modelcontextprotocol/sdk": "^1.0.0"
|
|
39
|
+
}
|
|
40
|
+
}
|
package/server.js
ADDED
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* TwitterTools MCP server (stdio) — brings twittertools.com to AI agents.
|
|
4
|
+
*
|
|
5
|
+
* Tools backed by the public API (stateless, no keys, rate-limited per IP):
|
|
6
|
+
* get_tweet -> GET {base}/api/tweet?id=...
|
|
7
|
+
* get_thread -> GET {base}/api/thread?url=...
|
|
8
|
+
* Local tools (offline, same engine as the website):
|
|
9
|
+
* count_chars, split_thread, parse_tweet_url
|
|
10
|
+
*
|
|
11
|
+
* Media is NOT proxied through the API here: get_tweet results carry the
|
|
12
|
+
* direct pbs.twimg.com / video.twimg.com URLs, which any client can fetch.
|
|
13
|
+
*
|
|
14
|
+
* Set TWITTERTOOLS_API_BASE to point at a self-hosted instance.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
import { Server } from "@modelcontextprotocol/sdk/server/index.js";
|
|
18
|
+
import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js";
|
|
19
|
+
import { ListToolsRequestSchema, CallToolRequestSchema } from "@modelcontextprotocol/sdk/types.js";
|
|
20
|
+
import {
|
|
21
|
+
countTweet,
|
|
22
|
+
splitThread,
|
|
23
|
+
parseTweetInput,
|
|
24
|
+
snowflakeToDate,
|
|
25
|
+
permalinkFor,
|
|
26
|
+
} from "./xrules.js";
|
|
27
|
+
|
|
28
|
+
const NAME = "twittertools";
|
|
29
|
+
const VERSION = "1.0.1"; // keep in sync with package.json
|
|
30
|
+
const API_BASE = (process.env.TWITTERTOOLS_API_BASE || "https://twittertools.com").replace(/\/+$/, "");
|
|
31
|
+
const ATTRIBUTION = "\n\nvia twittertools.com";
|
|
32
|
+
const TEXT_INPUT_MAX = 100_000; // generous, but caps local work per call
|
|
33
|
+
|
|
34
|
+
const urlOrIdSchema = {
|
|
35
|
+
type: "object",
|
|
36
|
+
properties: {
|
|
37
|
+
url_or_id: {
|
|
38
|
+
type: "string",
|
|
39
|
+
description: "Post URL (x.com or twitter.com), bare numeric ID, or text containing one",
|
|
40
|
+
},
|
|
41
|
+
},
|
|
42
|
+
required: ["url_or_id"],
|
|
43
|
+
};
|
|
44
|
+
|
|
45
|
+
const TOOLS = [
|
|
46
|
+
{
|
|
47
|
+
name: "get_tweet",
|
|
48
|
+
description:
|
|
49
|
+
"Fetch one public X (Twitter) post by URL or ID: text, author, engagement counts, " +
|
|
50
|
+
"and media with direct CDN URLs (photos, or MP4 video variants sorted by quality). " +
|
|
51
|
+
"Deleted, protected, or age-restricted posts are not available without an X login.",
|
|
52
|
+
inputSchema: urlOrIdSchema,
|
|
53
|
+
},
|
|
54
|
+
{
|
|
55
|
+
name: "get_thread",
|
|
56
|
+
description:
|
|
57
|
+
"Unroll a public X thread into its posts in order. Expensive upstream — call sparingly, " +
|
|
58
|
+
"and prefer pasting the thread's LAST post, which returns the whole chain. Results may " +
|
|
59
|
+
"be partial; the response says so when they are.",
|
|
60
|
+
inputSchema: urlOrIdSchema,
|
|
61
|
+
},
|
|
62
|
+
{
|
|
63
|
+
name: "count_chars",
|
|
64
|
+
description:
|
|
65
|
+
"Count text against X's real 280-character limit: CJK and emoji weigh 2, any link " +
|
|
66
|
+
"weighs exactly 23. Same engine as twittertools.com/tweet-character-counter/.",
|
|
67
|
+
inputSchema: {
|
|
68
|
+
type: "object",
|
|
69
|
+
properties: {
|
|
70
|
+
text: { type: "string", description: "The text to measure" },
|
|
71
|
+
},
|
|
72
|
+
required: ["text"],
|
|
73
|
+
},
|
|
74
|
+
},
|
|
75
|
+
{
|
|
76
|
+
name: "split_thread",
|
|
77
|
+
description:
|
|
78
|
+
"Split long text into numbered posts (1/, 2/, …) that each fit X's 280 weighted-character " +
|
|
79
|
+
"limit, preferring sentence boundaries and never breaking a URL. Same engine as " +
|
|
80
|
+
"twittertools.com/tweet-splitter/.",
|
|
81
|
+
inputSchema: {
|
|
82
|
+
type: "object",
|
|
83
|
+
properties: {
|
|
84
|
+
text: { type: "string", description: "The text to split" },
|
|
85
|
+
numbering: { type: "boolean", description: "Prefix each post with 'n/ ' (default true)" },
|
|
86
|
+
},
|
|
87
|
+
required: ["text"],
|
|
88
|
+
},
|
|
89
|
+
},
|
|
90
|
+
{
|
|
91
|
+
name: "parse_tweet_url",
|
|
92
|
+
description:
|
|
93
|
+
"Parse a post URL, bare ID, or free text containing an ID: returns the numeric ID, " +
|
|
94
|
+
"author handle, posting timestamp (decoded from the Snowflake ID), and canonical permalink. Offline.",
|
|
95
|
+
inputSchema: {
|
|
96
|
+
type: "object",
|
|
97
|
+
properties: {
|
|
98
|
+
input: { type: "string", description: "URL, ID, or text containing either" },
|
|
99
|
+
},
|
|
100
|
+
required: ["input"],
|
|
101
|
+
},
|
|
102
|
+
},
|
|
103
|
+
];
|
|
104
|
+
|
|
105
|
+
function textResult(text, { error = false } = {}) {
|
|
106
|
+
return { content: [{ type: "text", text }], isError: error };
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
function strArg(args, field, max = 2000) {
|
|
110
|
+
const v = args?.[field];
|
|
111
|
+
if (typeof v !== "string" || !v.trim()) {
|
|
112
|
+
throw new Error(`Missing required argument: ${field} (string)`);
|
|
113
|
+
}
|
|
114
|
+
if (v.length > max) {
|
|
115
|
+
throw new Error(`${field} is too long (max ${max} characters)`);
|
|
116
|
+
}
|
|
117
|
+
return v;
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
async function callApi(endpoint, params) {
|
|
121
|
+
let res;
|
|
122
|
+
try {
|
|
123
|
+
res = await fetch(`${API_BASE}/api/${endpoint}?${new URLSearchParams(params)}`, {
|
|
124
|
+
headers: { accept: "application/json" },
|
|
125
|
+
signal: AbortSignal.timeout(20_000),
|
|
126
|
+
});
|
|
127
|
+
} catch (e) {
|
|
128
|
+
const timedOut = e?.name === "TimeoutError" || e?.name === "AbortError";
|
|
129
|
+
throw new Error(
|
|
130
|
+
timedOut
|
|
131
|
+
? `twittertools API timed out after 20s (${API_BASE})`
|
|
132
|
+
: `twittertools API unreachable (${API_BASE}) — check the network or TWITTERTOOLS_API_BASE`
|
|
133
|
+
);
|
|
134
|
+
}
|
|
135
|
+
let body = null;
|
|
136
|
+
try {
|
|
137
|
+
body = await res.json();
|
|
138
|
+
} catch {
|
|
139
|
+
// non-JSON body (proxy error page, HTML) — fall through to the status message
|
|
140
|
+
}
|
|
141
|
+
if (!res.ok) {
|
|
142
|
+
throw new Error(body?.error || `twittertools API returned HTTP ${res.status}`);
|
|
143
|
+
}
|
|
144
|
+
return body;
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
async function callTool(name, args) {
|
|
148
|
+
switch (name) {
|
|
149
|
+
case "get_tweet": {
|
|
150
|
+
const data = await callApi("tweet", { id: strArg(args, "url_or_id") });
|
|
151
|
+
if (!data?.tweet) throw new Error("twittertools API returned no tweet");
|
|
152
|
+
return textResult(JSON.stringify(data.tweet, null, 2) + ATTRIBUTION);
|
|
153
|
+
}
|
|
154
|
+
case "get_thread": {
|
|
155
|
+
const data = await callApi("thread", { url: strArg(args, "url_or_id") });
|
|
156
|
+
if (!data?.tweets) throw new Error("twittertools API returned no thread");
|
|
157
|
+
const head = data?.partial
|
|
158
|
+
? `Note: this thread result is PARTIAL (${data.reason || "incomplete"}) — posts may be missing.\n\n`
|
|
159
|
+
: "";
|
|
160
|
+
return textResult(head + JSON.stringify({ count: data.tweets.length, tweets: data.tweets }, null, 2) + ATTRIBUTION);
|
|
161
|
+
}
|
|
162
|
+
case "count_chars":
|
|
163
|
+
return textResult(JSON.stringify(countTweet(strArg(args, "text", TEXT_INPUT_MAX)), null, 2));
|
|
164
|
+
case "split_thread": {
|
|
165
|
+
const numbering = args?.numbering !== false;
|
|
166
|
+
const posts = splitThread(strArg(args, "text", TEXT_INPUT_MAX), numbering);
|
|
167
|
+
return textResult(JSON.stringify({ count: posts.length, posts }, null, 2));
|
|
168
|
+
}
|
|
169
|
+
case "parse_tweet_url": {
|
|
170
|
+
const parsed = parseTweetInput(strArg(args, "input"));
|
|
171
|
+
if (!parsed) {
|
|
172
|
+
return textResult("No post URL or ID found in the input.", { error: true });
|
|
173
|
+
}
|
|
174
|
+
const created = snowflakeToDate(parsed.id);
|
|
175
|
+
return textResult(
|
|
176
|
+
JSON.stringify(
|
|
177
|
+
{
|
|
178
|
+
id: parsed.id,
|
|
179
|
+
screenName: parsed.screenName,
|
|
180
|
+
createdAt: Number.isNaN(created.getTime()) ? null : created.toISOString(),
|
|
181
|
+
permalink: permalinkFor(parsed.id, parsed.screenName),
|
|
182
|
+
},
|
|
183
|
+
null,
|
|
184
|
+
2
|
|
185
|
+
)
|
|
186
|
+
);
|
|
187
|
+
}
|
|
188
|
+
default:
|
|
189
|
+
return textResult(`Unknown tool: ${name}`, { error: true });
|
|
190
|
+
}
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
const server = new Server({ name: NAME, version: VERSION }, { capabilities: { tools: {} } });
|
|
194
|
+
|
|
195
|
+
server.setRequestHandler(ListToolsRequestSchema, async () => ({ tools: TOOLS }));
|
|
196
|
+
|
|
197
|
+
server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
198
|
+
const { name, arguments: args } = request.params;
|
|
199
|
+
try {
|
|
200
|
+
return await callTool(name, args);
|
|
201
|
+
} catch (e) {
|
|
202
|
+
return textResult(`Error: ${e?.message || String(e)}`, { error: true });
|
|
203
|
+
}
|
|
204
|
+
});
|
|
205
|
+
|
|
206
|
+
await server.connect(new StdioServerTransport());
|
package/xrules.js
ADDED
|
@@ -0,0 +1,241 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* X/Twitter text rules, ported 1:1 from the repo's TypeScript originals
|
|
3
|
+
* (src/lib/tweet-count.ts, tweet-split.ts, snowflake.ts) so this package can
|
|
4
|
+
* ship standalone. The repo vitest suite pins the originals; the mcp tests
|
|
5
|
+
* re-run the same cases against this port — keep the two in sync.
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
// --------------------------------------------------------------------------- //
|
|
9
|
+
// Weighted counting (twitter-text confit.json v3): most Latin/ASCII code
|
|
10
|
+
// points weigh 1, everything else (CJK, emoji, most non-Latin scripts)
|
|
11
|
+
// weighs 2, and any URL counts as exactly 23 (t.co wrapping). Limit 280.
|
|
12
|
+
// --------------------------------------------------------------------------- //
|
|
13
|
+
|
|
14
|
+
const WEIGHT_100_RANGES = [
|
|
15
|
+
[0x0000, 0x10ff],
|
|
16
|
+
[0x2000, 0x200d],
|
|
17
|
+
[0x2010, 0x201f],
|
|
18
|
+
[0x2032, 0x2037],
|
|
19
|
+
];
|
|
20
|
+
|
|
21
|
+
/** http(s) and www. links — unambiguous. */
|
|
22
|
+
const URL_REGEX = /(https?:\/\/[^\s<>"']+|www\.[^\s<>"']+)/gi;
|
|
23
|
+
/**
|
|
24
|
+
* Scheme-less domains (X counts "x.com/foo" as a link too). The real
|
|
25
|
+
* twitter-text regex validates every IANA TLD; this approximation covers the
|
|
26
|
+
* common ones, which is accurate for virtually all real posts.
|
|
27
|
+
*/
|
|
28
|
+
const BARE_DOMAIN_REGEX =
|
|
29
|
+
/\b(?:[a-z0-9-]+\.)+(?:com|org|net|io|co|dev|app|xyz|me|gov|edu|info|cn|jp|uk|de|fr|ru|br|in|nl|it|es|se|no|fi|ca|au|us)(?:\/[^\s<>"']*)?/gi;
|
|
30
|
+
/** X never includes trailing punctuation in a link. */
|
|
31
|
+
const TRAILING_PUNCT = /[.,;:!?)\]}'"。,!?)、》…]+$/;
|
|
32
|
+
|
|
33
|
+
const MAX_WEIGHTED = 28_000; // 280 chars x 100
|
|
34
|
+
|
|
35
|
+
function isWeight100(cp) {
|
|
36
|
+
return WEIGHT_100_RANGES.some(([lo, hi]) => cp >= lo && cp <= hi);
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
export function extractUrls(text) {
|
|
40
|
+
const primary = text.match(URL_REGEX) ?? [];
|
|
41
|
+
// mask the unambiguous matches, then count scheme-less domains in what's
|
|
42
|
+
// left so no URL is counted twice
|
|
43
|
+
const masked = text.replace(URL_REGEX, (m) => '\u0000'.repeat(m.length));
|
|
44
|
+
const bare = masked.match(BARE_DOMAIN_REGEX) ?? [];
|
|
45
|
+
return [...primary, ...bare].map((u) => u.replace(TRAILING_PUNCT, '')).filter(Boolean);
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
export function countTweet(text) {
|
|
49
|
+
const urls = extractUrls(text);
|
|
50
|
+
let withoutUrls = text;
|
|
51
|
+
for (const u of urls) withoutUrls = withoutUrls.replace(u, '');
|
|
52
|
+
|
|
53
|
+
let units = urls.length * 2_300; // each link counts as 23 chars
|
|
54
|
+
let codePoints = 0;
|
|
55
|
+
let heavyCodePoints = 0;
|
|
56
|
+
|
|
57
|
+
for (const ch of withoutUrls) {
|
|
58
|
+
codePoints++;
|
|
59
|
+
const cp = ch.codePointAt(0) ?? 0;
|
|
60
|
+
if (isWeight100(cp)) {
|
|
61
|
+
units += 100;
|
|
62
|
+
} else {
|
|
63
|
+
units += 200;
|
|
64
|
+
heavyCodePoints++;
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
const weightedLength = Math.round(units / 100);
|
|
69
|
+
return {
|
|
70
|
+
weightedLength,
|
|
71
|
+
codePoints,
|
|
72
|
+
heavyCodePoints,
|
|
73
|
+
urlCount: urls.length,
|
|
74
|
+
limit: 280,
|
|
75
|
+
remaining: 280 - weightedLength,
|
|
76
|
+
valid: units <= MAX_WEIGHTED,
|
|
77
|
+
};
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
// --------------------------------------------------------------------------- //
|
|
81
|
+
// Thread splitting at the weighted limit, preferring sentence boundaries
|
|
82
|
+
// over line breaks over word breaks. Links are atomic.
|
|
83
|
+
// --------------------------------------------------------------------------- //
|
|
84
|
+
|
|
85
|
+
const LIMIT = 280;
|
|
86
|
+
/** Reserved for the "12/ " prefix when numbering is on (covers up to 999 parts). */
|
|
87
|
+
const NUMBERING_RESERVE = 5;
|
|
88
|
+
|
|
89
|
+
const SENTENCE_END = new Set(['.', '!', '?', '…', '。', '!', '?']);
|
|
90
|
+
const CLOSERS = new Set(['.', '!', '?', '…', '。', '!', '?', ',', ';', ':', '"', "'", ')', ']', '」', '』', '”', '’']);
|
|
91
|
+
|
|
92
|
+
/** Break a line into sentences; decimals ("3.5") and abbreviations without a
|
|
93
|
+
* following space don't count as boundaries. */
|
|
94
|
+
function sentencesOf(line) {
|
|
95
|
+
const out = [];
|
|
96
|
+
const chars = Array.from(line);
|
|
97
|
+
let start = 0;
|
|
98
|
+
for (let i = 0; i < chars.length; i++) {
|
|
99
|
+
if (!SENTENCE_END.has(chars[i])) continue;
|
|
100
|
+
let j = i + 1;
|
|
101
|
+
while (j < chars.length && CLOSERS.has(chars[j])) j++;
|
|
102
|
+
if (j >= chars.length || chars[j] === ' ') {
|
|
103
|
+
const s = chars.slice(start, j).join('').trim();
|
|
104
|
+
if (s) out.push(s);
|
|
105
|
+
while (j < chars.length && chars[j] === ' ') j++;
|
|
106
|
+
start = j;
|
|
107
|
+
i = j - 1;
|
|
108
|
+
} else {
|
|
109
|
+
i = j - 1; // "3.5", "x.com" — resume scanning after the punctuation run
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
const rest = chars.slice(start).join('').trim();
|
|
113
|
+
if (rest) out.push(rest);
|
|
114
|
+
return out;
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
/** Split an oversized piece on whitespace; a single unbreakable word longer
|
|
118
|
+
* than the limit gets cut by code points (URLs keep working — they carry no
|
|
119
|
+
* spaces, and their 23-char weight is what counts, not their raw length). */
|
|
120
|
+
function hardSplit(s, eff) {
|
|
121
|
+
const out = [];
|
|
122
|
+
let cur = '';
|
|
123
|
+
for (const word of s.split(/\s+/)) {
|
|
124
|
+
const pieces =
|
|
125
|
+
countTweet(word).weightedLength > eff
|
|
126
|
+
? chunkCodePoints(word, eff)
|
|
127
|
+
: [word];
|
|
128
|
+
for (const piece of pieces) {
|
|
129
|
+
const cand = cur ? cur + ' ' + piece : piece;
|
|
130
|
+
if (countTweet(cand).weightedLength <= eff) cur = cand;
|
|
131
|
+
else {
|
|
132
|
+
if (cur) out.push(cur);
|
|
133
|
+
cur = piece;
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
if (cur) out.push(cur);
|
|
138
|
+
return out;
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
function chunkCodePoints(s, eff) {
|
|
142
|
+
const out = [];
|
|
143
|
+
let cur = '';
|
|
144
|
+
for (const ch of Array.from(s)) {
|
|
145
|
+
const cand = cur + ch;
|
|
146
|
+
if (countTweet(cand).weightedLength <= eff) cur = cand;
|
|
147
|
+
else {
|
|
148
|
+
out.push(cur);
|
|
149
|
+
cur = ch;
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
if (cur) out.push(cur);
|
|
153
|
+
return out;
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
function segment(text, eff) {
|
|
157
|
+
const segs = [];
|
|
158
|
+
let pendingGlue = '';
|
|
159
|
+
for (const rawLine of text.replace(/\r\n?/g, '\n').split('\n')) {
|
|
160
|
+
const line = rawLine.trim();
|
|
161
|
+
if (!line) {
|
|
162
|
+
pendingGlue = '\n\n'; // paragraph break
|
|
163
|
+
continue;
|
|
164
|
+
}
|
|
165
|
+
for (const sentence of sentencesOf(line)) {
|
|
166
|
+
const pieces =
|
|
167
|
+
countTweet(sentence).weightedLength > eff ? hardSplit(sentence, eff) : [sentence];
|
|
168
|
+
pieces.forEach((piece, i) => {
|
|
169
|
+
segs.push({ text: piece, glue: segs.length === 0 ? '' : i === 0 ? pendingGlue || ' ' : ' ' });
|
|
170
|
+
});
|
|
171
|
+
pendingGlue = '';
|
|
172
|
+
}
|
|
173
|
+
pendingGlue = pendingGlue || '\n'; // next line joins with a line break
|
|
174
|
+
}
|
|
175
|
+
return segs;
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
/** Split `text` into posts that each fit X's 280-character weighted limit.
|
|
179
|
+
* With `numbering`, every post gets an "n/ " prefix and the usable budget is
|
|
180
|
+
* reserved for it. Returns the ready-to-post texts. */
|
|
181
|
+
export function splitThread(text, numbering = true) {
|
|
182
|
+
const input = text.trim();
|
|
183
|
+
if (!input) return [];
|
|
184
|
+
const eff = LIMIT - (numbering ? NUMBERING_RESERVE : 0);
|
|
185
|
+
const parts = [];
|
|
186
|
+
let cur = '';
|
|
187
|
+
for (const seg of segment(input, eff)) {
|
|
188
|
+
const cand = cur ? cur + seg.glue + seg.text : seg.text;
|
|
189
|
+
if (countTweet(cand).weightedLength <= eff) cur = cand;
|
|
190
|
+
else {
|
|
191
|
+
if (cur.trim()) parts.push(cur.trim());
|
|
192
|
+
cur = seg.text;
|
|
193
|
+
}
|
|
194
|
+
}
|
|
195
|
+
if (cur.trim()) parts.push(cur.trim());
|
|
196
|
+
if (!numbering) return parts;
|
|
197
|
+
return parts.map((p, i) => `${i + 1}/ ${p}`);
|
|
198
|
+
}
|
|
199
|
+
|
|
200
|
+
// --------------------------------------------------------------------------- //
|
|
201
|
+
// Snowflake IDs: the top 41 bits are milliseconds since X's custom epoch.
|
|
202
|
+
// --------------------------------------------------------------------------- //
|
|
203
|
+
|
|
204
|
+
export const X_EPOCH = 1288834974657;
|
|
205
|
+
|
|
206
|
+
export function snowflakeToDate(id) {
|
|
207
|
+
const n = typeof id === 'bigint' ? id : BigInt(id);
|
|
208
|
+
const ms = Number(n >> 22n) + X_EPOCH;
|
|
209
|
+
// IDs past X's real range decode beyond the max JS Date (±8.64e15 ms) —
|
|
210
|
+
// surface them as Invalid Date instead of a silently-wrong value
|
|
211
|
+
return new Date(ms > 8.64e15 || ms < -8.64e15 ? NaN : ms);
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
// The (?:^|[^A-Za-z0-9-]) boundary keeps lookalike hosts (notx.com, foo-x.com)
|
|
215
|
+
// from matching as x.com — the host must start at a word boundary.
|
|
216
|
+
const STATUS_RE =
|
|
217
|
+
/(?:^|[^A-Za-z0-9-])(?:https?:\/\/)?(?:www\.|mobile\.)?(?:x|twitter)\.com\/([A-Za-z0-9_]{1,15})\/status(?:es)?\/(\d{5,25})/i;
|
|
218
|
+
const BARE_ID_RE = /^\d{5,25}$/;
|
|
219
|
+
|
|
220
|
+
/** Extract a status id (and screen name when present) from a URL, bare ID, or free text. */
|
|
221
|
+
export function parseTweetInput(input) {
|
|
222
|
+
const text = (input ?? '').trim();
|
|
223
|
+
if (!text) return null;
|
|
224
|
+
|
|
225
|
+
const m = text.match(STATUS_RE);
|
|
226
|
+
if (m) return { id: m[2], screenName: m[1].toLowerCase() === 'i' ? null : m[1] };
|
|
227
|
+
|
|
228
|
+
if (BARE_ID_RE.test(text)) return { id: text, screenName: null };
|
|
229
|
+
|
|
230
|
+
// A bare ID hidden inside longer text
|
|
231
|
+
const idMatch = text.match(/(\d{15,25})/);
|
|
232
|
+
if (idMatch) return { id: idMatch[1], screenName: null };
|
|
233
|
+
|
|
234
|
+
return null;
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
export function permalinkFor(id, screenName) {
|
|
238
|
+
return screenName
|
|
239
|
+
? `https://x.com/${screenName}/status/${id}`
|
|
240
|
+
: `https://x.com/i/web/status/${id}`;
|
|
241
|
+
}
|