trendcite 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- trendcite/__init__.py +4 -0
- trendcite/__main__.py +3 -0
- trendcite/briefs.py +234 -0
- trendcite/cli.py +128 -0
- trendcite/cluster.py +244 -0
- trendcite/config.py +68 -0
- trendcite/fixtures/demo_github.json +54 -0
- trendcite/fixtures/demo_hackernews.json +13 -0
- trendcite/fixtures/demo_meta.json +12 -0
- trendcite/fixtures/demo_reddit_devs.xml +22 -0
- trendcite/fixtures/demo_reddit_saas.xml +38 -0
- trendcite/fixtures/demo_rss.xml +68 -0
- trendcite/history.py +192 -0
- trendcite/http.py +111 -0
- trendcite/identity.py +140 -0
- trendcite/lexicon.py +144 -0
- trendcite/llm.py +226 -0
- trendcite/models.py +228 -0
- trendcite/normalize.py +126 -0
- trendcite/observation.py +169 -0
- trendcite/pipeline.py +245 -0
- trendcite/py.typed +0 -0
- trendcite/render.py +105 -0
- trendcite/scoring.py +314 -0
- trendcite/security.py +231 -0
- trendcite/signal.py +408 -0
- trendcite/signal_scoring.py +651 -0
- trendcite/sources/__init__.py +18 -0
- trendcite/sources/base.py +31 -0
- trendcite/sources/github.py +86 -0
- trendcite/sources/hackernews.py +78 -0
- trendcite/sources/reddit.py +55 -0
- trendcite/sources/rss.py +118 -0
- trendcite/sources/x.py +27 -0
- trendcite/text.py +274 -0
- trendcite/versions.py +45 -0
- trendcite-0.1.0.dist-info/METADATA +377 -0
- trendcite-0.1.0.dist-info/RECORD +42 -0
- trendcite-0.1.0.dist-info/WHEEL +5 -0
- trendcite-0.1.0.dist-info/entry_points.txt +2 -0
- trendcite-0.1.0.dist-info/licenses/LICENSE +21 -0
- trendcite-0.1.0.dist-info/top_level.txt +1 -0
trendcite/__init__.py
ADDED
trendcite/__main__.py
ADDED
trendcite/briefs.py
ADDED
|
@@ -0,0 +1,234 @@
|
|
|
1
|
+
"""Turn scored clusters into Content Opportunity Briefs (deterministic templates).
|
|
2
|
+
|
|
3
|
+
Everything in a brief except the draft outline is derived from captured evidence
|
|
4
|
+
and computed scores. The outline is a writing aid and is labelled as such.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from datetime import datetime
|
|
10
|
+
|
|
11
|
+
from .models import Brief, EvidenceItem, TopicCluster
|
|
12
|
+
from .scoring import HALF_LIFE_HOURS, WEIGHTS, age_hours, select_evidence
|
|
13
|
+
from .signal import SignalBrief
|
|
14
|
+
from .text import CRITICAL_RE
|
|
15
|
+
|
|
16
|
+
SOURCE_NAMES = {
|
|
17
|
+
"hackernews": "Hacker News",
|
|
18
|
+
"github": "GitHub",
|
|
19
|
+
"rss": "RSS/Atom",
|
|
20
|
+
"reddit": "Reddit",
|
|
21
|
+
"x": "X",
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def metric_summary(item: EvidenceItem) -> str:
|
|
26
|
+
m = item.metrics
|
|
27
|
+
if item.source == "hackernews":
|
|
28
|
+
return f"{int(m.get('points', 0))} points, {int(m.get('comments', 0))} comments"
|
|
29
|
+
if item.source == "github":
|
|
30
|
+
return f"{int(m.get('stars', 0))} stars, {int(m.get('forks', 0))} forks"
|
|
31
|
+
return "no engagement metrics available from this source"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _source_list(sources: list[str]) -> str:
|
|
35
|
+
return ", ".join(SOURCE_NAMES.get(s, s) for s in sources)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _angle(cluster: TopicCluster) -> str:
|
|
39
|
+
topic = cluster.label
|
|
40
|
+
assert cluster.score is not None
|
|
41
|
+
s = cluster.score
|
|
42
|
+
sources = set(s.sources)
|
|
43
|
+
if "github" in sources and sources & {"hackernews", "reddit"}:
|
|
44
|
+
return (
|
|
45
|
+
f"Builders are already shipping around {topic} while the discussion is still "
|
|
46
|
+
f"unsettled: share what actually works (and what doesn't) from first-hand use."
|
|
47
|
+
)
|
|
48
|
+
if "reddit" in sources and s.engagement < 0.6:
|
|
49
|
+
return (
|
|
50
|
+
f"Practitioners are asking open questions about {topic}; answer them with your "
|
|
51
|
+
f"own operating numbers and decisions rather than a summary of the debate."
|
|
52
|
+
)
|
|
53
|
+
if s.engagement >= 0.6 and s.corroboration >= 0.5:
|
|
54
|
+
return (
|
|
55
|
+
f"{topic} is drawing strong attention across communities; the useful angle is "
|
|
56
|
+
f"what it changes for a small team this quarter."
|
|
57
|
+
)
|
|
58
|
+
return (
|
|
59
|
+
f"{topic} is being covered by publishers; add a first-hand or contrarian take "
|
|
60
|
+
f"that the existing coverage lacks."
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _why_now(
|
|
65
|
+
cluster: TopicCluster,
|
|
66
|
+
evidence: list[EvidenceItem],
|
|
67
|
+
now: datetime,
|
|
68
|
+
percentiles: dict[str, float],
|
|
69
|
+
) -> list[str]:
|
|
70
|
+
assert cluster.score is not None
|
|
71
|
+
s = cluster.score
|
|
72
|
+
items = evidence
|
|
73
|
+
ages = [age_hours(i, now) for i in items]
|
|
74
|
+
lines = [
|
|
75
|
+
f"{len(items)} evidence item(s) shown and scored: {s.unique_urls} unique URL(s) from "
|
|
76
|
+
f"{len(s.sources)} source(s) ({_source_list(list(s.sources))}); newest "
|
|
77
|
+
f"{min(ages):.0f} h old, oldest {max(ages):.0f} h old."
|
|
78
|
+
]
|
|
79
|
+
if len(cluster.items) > len(items):
|
|
80
|
+
lines.append(
|
|
81
|
+
f"{len(cluster.items)} items matched this topic in total; the rest are not "
|
|
82
|
+
f"shown and do not affect the score."
|
|
83
|
+
)
|
|
84
|
+
measured = [i for i in items if i.item_id in percentiles]
|
|
85
|
+
if measured:
|
|
86
|
+
top = max(measured, key=lambda i: (percentiles[i.item_id], i.item_id))
|
|
87
|
+
lines.append(
|
|
88
|
+
f'Strongest engagement signal: "{top.title}" '
|
|
89
|
+
f"({SOURCE_NAMES.get(top.source, top.source)}: "
|
|
90
|
+
f"{metric_summary(top)}; engagement percentile {percentiles[top.item_id] * 100:.0f} "
|
|
91
|
+
f"within that source this run)."
|
|
92
|
+
)
|
|
93
|
+
if cluster.niche_matches:
|
|
94
|
+
lines.append("Matches your niche terms: " + ", ".join(cluster.niche_matches) + ".")
|
|
95
|
+
if cluster.related_terms:
|
|
96
|
+
lines.append("Co-occurring terms: " + ", ".join(cluster.related_terms) + ".")
|
|
97
|
+
return lines
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _score_explanation(cluster: TopicCluster) -> list[str]:
|
|
101
|
+
assert cluster.score is not None
|
|
102
|
+
s = cluster.score
|
|
103
|
+
w = WEIGHTS
|
|
104
|
+
rows = [
|
|
105
|
+
(
|
|
106
|
+
"recency",
|
|
107
|
+
w.recency,
|
|
108
|
+
s.recency,
|
|
109
|
+
f"{HALF_LIFE_HOURS:.0f} h half-life, averaged over items",
|
|
110
|
+
),
|
|
111
|
+
("engagement", w.engagement, s.engagement, "top-3 within-source percentiles"),
|
|
112
|
+
(
|
|
113
|
+
"corroboration",
|
|
114
|
+
w.corroboration,
|
|
115
|
+
s.corroboration,
|
|
116
|
+
f"{s.independent_sources} independent source(s): sources that each link a "
|
|
117
|
+
f"different URL",
|
|
118
|
+
),
|
|
119
|
+
(
|
|
120
|
+
"relevance",
|
|
121
|
+
w.relevance,
|
|
122
|
+
s.relevance,
|
|
123
|
+
f"{len(cluster.niche_matches)} niche term(s) matched in the evidence",
|
|
124
|
+
),
|
|
125
|
+
("diversity", w.diversity, s.diversity, f"{s.publishers} distinct publisher(s)"),
|
|
126
|
+
]
|
|
127
|
+
lines = [
|
|
128
|
+
f"{name}: {value:.2f} x {weight:.2f} = {100 * value * weight:.1f} pts ({why})"
|
|
129
|
+
for name, weight, value, why in rows
|
|
130
|
+
]
|
|
131
|
+
lines.append(f"total: {s.total:.1f}/100, confidence {s.confidence}")
|
|
132
|
+
lines.append(
|
|
133
|
+
"all components are computed from the evidence items listed in this brief; high "
|
|
134
|
+
"confidence requires >= 3 independent sources, >= 4 unique URLs, >= 3 publishers, "
|
|
135
|
+
"a niche match (when a niche is set) and evidence that agrees on more than one word"
|
|
136
|
+
)
|
|
137
|
+
return lines
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _counterpoints(cluster: TopicCluster, evidence: list[EvidenceItem], now: datetime) -> list[str]:
|
|
141
|
+
assert cluster.score is not None
|
|
142
|
+
s = cluster.score
|
|
143
|
+
notes: list[str] = []
|
|
144
|
+
if s.independent_sources <= 1:
|
|
145
|
+
where = _source_list(list(s.sources))
|
|
146
|
+
if len(s.sources) > 1:
|
|
147
|
+
notes.append(
|
|
148
|
+
f"Seen via {where}, but they link the same underlying URL; that is one "
|
|
149
|
+
f"story mirrored across channels, not independent corroboration."
|
|
150
|
+
)
|
|
151
|
+
else:
|
|
152
|
+
notes.append(
|
|
153
|
+
f"Only seen on {where}; this may be one community's echo chamber rather "
|
|
154
|
+
f"than a broad trend."
|
|
155
|
+
)
|
|
156
|
+
mirrored = len(evidence) - s.unique_urls
|
|
157
|
+
if mirrored > 0 and s.independent_sources > 1:
|
|
158
|
+
notes.append(
|
|
159
|
+
f"{mirrored} evidence item(s) repeat a URL already shown via another channel; "
|
|
160
|
+
f"repeats are counted once for corroboration."
|
|
161
|
+
)
|
|
162
|
+
if s.unique_urls <= 2:
|
|
163
|
+
notes.append(
|
|
164
|
+
f"Small sample ({s.unique_urls} unique URL(s)). Treat as an early signal, not a trend."
|
|
165
|
+
)
|
|
166
|
+
if not any(i.engagement() is not None for i in evidence):
|
|
167
|
+
notes.append("No engagement metrics are available for these sources; reach is unknown.")
|
|
168
|
+
stale = [i for i in evidence if age_hours(i, now) > 7 * 24]
|
|
169
|
+
if stale:
|
|
170
|
+
notes.append(f"{len(stale)} item(s) are over a week old; part of this signal is not new.")
|
|
171
|
+
critical = [i for i in evidence if CRITICAL_RE.search(i.title)]
|
|
172
|
+
if critical:
|
|
173
|
+
notes.append(
|
|
174
|
+
f'Part of the evidence is critical or cautionary (e.g. "{critical[0].title}"); '
|
|
175
|
+
f"address the downside explicitly instead of only the upside."
|
|
176
|
+
)
|
|
177
|
+
notes.append(
|
|
178
|
+
"Clustering is keyword-based: confirm the linked items really discuss the same thing "
|
|
179
|
+
"before relying on the corroboration count."
|
|
180
|
+
)
|
|
181
|
+
return notes
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def _questions(topic: str) -> list[str]:
|
|
185
|
+
return [
|
|
186
|
+
f"What have you personally shipped, broken or decided about {topic} in the last 90 days?",
|
|
187
|
+
"Where does this evidence disagree with what you hear from your own customers?",
|
|
188
|
+
f"For a 3-person team, is {topic} a 'do now', 'watch' or 'ignore' this quarter, and why?",
|
|
189
|
+
"Which number from your own business could you share to make the point concrete?",
|
|
190
|
+
]
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def _outline(cluster: TopicCluster, evidence: list[EvidenceItem]) -> list[str]:
|
|
194
|
+
first = evidence[0]
|
|
195
|
+
return [
|
|
196
|
+
f'Hook: open with the concrete signal from evidence [1] ("{first.title}").',
|
|
197
|
+
"Context: summarise what the linked evidence shows, citing items by number.",
|
|
198
|
+
"Your POV: answer the first founder question with a specific story or decision.",
|
|
199
|
+
"Counterpoint: name the strongest objection listed above and respond to it honestly.",
|
|
200
|
+
f"Takeaway: one practical recommendation about {cluster.label} for small teams.",
|
|
201
|
+
]
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def build_brief(
|
|
205
|
+
cluster: TopicCluster,
|
|
206
|
+
rank: int,
|
|
207
|
+
now: datetime,
|
|
208
|
+
percentiles: dict[str, float],
|
|
209
|
+
signal: SignalBrief | None = None,
|
|
210
|
+
) -> Brief:
|
|
211
|
+
"""Project one scored cluster (and, when available, its signal) into a public brief."""
|
|
212
|
+
assert cluster.score is not None
|
|
213
|
+
evidence = cluster.evidence or select_evidence(cluster.items, percentiles, now)
|
|
214
|
+
flags: list[str] = []
|
|
215
|
+
flagged = [i for i in evidence if i.flags]
|
|
216
|
+
if flagged:
|
|
217
|
+
flags.append(
|
|
218
|
+
f"{len(flagged)} evidence item(s) contain text that looks like instructions to an AI "
|
|
219
|
+
f"system (possible prompt injection). Shown as inert data only; nothing was executed."
|
|
220
|
+
)
|
|
221
|
+
return Brief(
|
|
222
|
+
rank=rank,
|
|
223
|
+
topic=cluster.label,
|
|
224
|
+
angle=_angle(cluster),
|
|
225
|
+
why_now=_why_now(cluster, evidence, now, percentiles),
|
|
226
|
+
evidence=evidence,
|
|
227
|
+
score=cluster.score,
|
|
228
|
+
score_explanation=_score_explanation(cluster),
|
|
229
|
+
counterpoints=_counterpoints(cluster, evidence, now),
|
|
230
|
+
founder_questions=_questions(cluster.label),
|
|
231
|
+
outline=_outline(cluster, evidence),
|
|
232
|
+
flags=flags,
|
|
233
|
+
signal=signal,
|
|
234
|
+
)
|
trendcite/cli.py
ADDED
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
"""Command-line interface: ``trendcite demo | live | sources``."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import sys
|
|
7
|
+
from collections.abc import Sequence
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
from . import __version__
|
|
11
|
+
from .config import Config, ConfigError, load_config
|
|
12
|
+
from .llm import LLMError, enhance_briefs, provider_from_env
|
|
13
|
+
from .models import Report
|
|
14
|
+
from .pipeline import make_adapters, run_demo, run_live
|
|
15
|
+
from .render import to_json, to_markdown
|
|
16
|
+
from .security import configure_logging, redact
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _split(value: str | None) -> list[str] | None:
|
|
20
|
+
if value is None:
|
|
21
|
+
return None
|
|
22
|
+
return [v.strip() for v in value.split(",") if v.strip()]
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _add_common(p: argparse.ArgumentParser) -> None:
|
|
26
|
+
p.add_argument("--format", choices=["markdown", "json"], default="markdown")
|
|
27
|
+
p.add_argument("--out", type=Path, help="write the report to this file instead of stdout")
|
|
28
|
+
p.add_argument("--top", type=int, default=None, help="number of briefs, clamped to 3-5")
|
|
29
|
+
p.add_argument("--niche", help="comma-separated niche phrases, e.g. 'ai agents,pricing'")
|
|
30
|
+
p.add_argument(
|
|
31
|
+
"--llm",
|
|
32
|
+
action="store_true",
|
|
33
|
+
help="refine angles/outlines with the provider set in TRENDCITE_LLM_PROVIDER",
|
|
34
|
+
)
|
|
35
|
+
p.add_argument("-v", "--verbose", action="store_true")
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
39
|
+
parser = argparse.ArgumentParser(
|
|
40
|
+
prog="trendcite",
|
|
41
|
+
description="Evidence-first trend intelligence: traceable content opportunity briefs.",
|
|
42
|
+
)
|
|
43
|
+
parser.add_argument("--version", action="version", version=f"trendcite {__version__}")
|
|
44
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
45
|
+
|
|
46
|
+
demo = sub.add_parser("demo", help="offline demo on bundled synthetic fixtures (no network)")
|
|
47
|
+
_add_common(demo)
|
|
48
|
+
|
|
49
|
+
live = sub.add_parser("live", help="collect from public read-only sources and build briefs")
|
|
50
|
+
_add_common(live)
|
|
51
|
+
live.add_argument("--config", type=Path, help="path to trendcite.toml")
|
|
52
|
+
live.add_argument("--sources", help="comma-separated: hackernews,github,rss,reddit,x")
|
|
53
|
+
live.add_argument("--feeds", help="comma-separated RSS/Atom URLs (replaces configured feeds)")
|
|
54
|
+
live.add_argument("--subreddits", help="comma-separated subreddit names")
|
|
55
|
+
live.add_argument("--github-queries", help="comma-separated GitHub search queries")
|
|
56
|
+
|
|
57
|
+
sub.add_parser("sources", help="list source adapters and their status requirements")
|
|
58
|
+
return parser
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _emit(text: str, out: Path | None) -> None:
|
|
62
|
+
if out is None:
|
|
63
|
+
sys.stdout.write(text)
|
|
64
|
+
return
|
|
65
|
+
out.write_text(text, encoding="utf-8")
|
|
66
|
+
sys.stderr.write(f"trendcite: wrote {out}\n")
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _maybe_llm(report: Report, enabled: bool) -> None:
|
|
70
|
+
if not enabled:
|
|
71
|
+
return
|
|
72
|
+
try:
|
|
73
|
+
provider = provider_from_env()
|
|
74
|
+
except LLMError as exc:
|
|
75
|
+
report.notes.append(f"LLM synthesis disabled: {redact(str(exc))}")
|
|
76
|
+
return
|
|
77
|
+
if provider is None:
|
|
78
|
+
report.notes.append(
|
|
79
|
+
"--llm given but TRENDCITE_LLM_PROVIDER is not set; deterministic output only."
|
|
80
|
+
)
|
|
81
|
+
return
|
|
82
|
+
report.notes.extend(enhance_briefs(report.briefs, provider))
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _live_config(args: argparse.Namespace) -> Config:
|
|
86
|
+
cfg = load_config(args.config)
|
|
87
|
+
for attr, value in (
|
|
88
|
+
("sources", _split(args.sources)),
|
|
89
|
+
("feeds", _split(args.feeds)),
|
|
90
|
+
("subreddits", _split(args.subreddits)),
|
|
91
|
+
("github_queries", _split(args.github_queries)),
|
|
92
|
+
("niche", _split(args.niche)),
|
|
93
|
+
):
|
|
94
|
+
if value is not None:
|
|
95
|
+
setattr(cfg, attr, value)
|
|
96
|
+
if args.top is not None:
|
|
97
|
+
cfg.top = args.top
|
|
98
|
+
return cfg
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def main(argv: Sequence[str] | None = None) -> int:
|
|
102
|
+
for stream in (sys.stdout, sys.stderr):
|
|
103
|
+
reconfigure = getattr(stream, "reconfigure", None)
|
|
104
|
+
if reconfigure is not None:
|
|
105
|
+
reconfigure(encoding="utf-8", errors="replace")
|
|
106
|
+
args = build_parser().parse_args(argv)
|
|
107
|
+
configure_logging(getattr(args, "verbose", False))
|
|
108
|
+
try:
|
|
109
|
+
if args.command == "sources":
|
|
110
|
+
for adapter in make_adapters(
|
|
111
|
+
Config(sources=["hackernews", "github", "rss", "reddit", "x"])
|
|
112
|
+
):
|
|
113
|
+
print(f"{adapter.name:<11} {adapter.description}")
|
|
114
|
+
return 0
|
|
115
|
+
if args.command == "demo":
|
|
116
|
+
report = run_demo(top=args.top or 5, niche=_split(args.niche))
|
|
117
|
+
else:
|
|
118
|
+
report = run_live(_live_config(args))
|
|
119
|
+
_maybe_llm(report, args.llm)
|
|
120
|
+
rendered = to_json(report) if args.format == "json" else to_markdown(report)
|
|
121
|
+
_emit(rendered, args.out)
|
|
122
|
+
if args.command == "live" and not any(s.ok for s in report.source_status):
|
|
123
|
+
sys.stderr.write("trendcite: no source was reachable; see the source table.\n")
|
|
124
|
+
return 2
|
|
125
|
+
return 0
|
|
126
|
+
except (ConfigError, ValueError) as exc:
|
|
127
|
+
sys.stderr.write(f"trendcite: error: {redact(str(exc))}\n")
|
|
128
|
+
return 2
|
trendcite/cluster.py
ADDED
|
@@ -0,0 +1,244 @@
|
|
|
1
|
+
"""Deterministic topic clustering by shared key terms, with a coherence gate.
|
|
2
|
+
|
|
3
|
+
Algorithm (greedy, fully deterministic):
|
|
4
|
+
|
|
5
|
+
1. Extract unigram and bigram terms from each item's *feature view* (title + excerpt
|
|
6
|
+
head with URLs, domains and feed boilerplate removed; see ``text.item_text``).
|
|
7
|
+
2. A term is a *seed candidate* if it is not generic or boilerplate and occurs in items
|
|
8
|
+
with >= ``min_items`` distinct underlying URLs (``EvidenceItem.story_key``).
|
|
9
|
+
3. Coherence gate. The items a seed may claim depend on how distinctive the seed is:
|
|
10
|
+
|
|
11
|
+
- phrase (two words, at least one not generic): every item containing it. A shared
|
|
12
|
+
multi-word phrase is itself strong lexical agreement.
|
|
13
|
+
- specific word (not in the common-English reference lexicon ``lexicon.py``, e.g.
|
|
14
|
+
"mcp", "kubernetes"): every item containing it if they are cohesive
|
|
15
|
+
(``scoring.evidence_is_cohesive``: every URL shares a further specific term or
|
|
16
|
+
phrase with more than half of the other URLs).
|
|
17
|
+
- otherwise (a common word such as "decision", "agent", "memory", or a specific
|
|
18
|
+
word that failed the cohesion check): only the items that also share the seed's
|
|
19
|
+
best co-occurring *agreement terms* (``text.agreement_terms``). Uniform
|
|
20
|
+
principle: members must share terms worth >= 2 in total, where a phrase counts
|
|
21
|
+
2, a specific word 1 and a common word 0 (``text.specificity``). So "gpt" +
|
|
22
|
+
"astra" is enough, while "agent" needs a shared phrase or two more specific
|
|
23
|
+
words. The label names the extra terms. Inflected forms are checked against the
|
|
24
|
+
lexicon too ("testing" counts as "test").
|
|
25
|
+
- last resort, specific word from a single independent source: all its items form
|
|
26
|
+
a *candidate* cluster that claims no corroboration. The pipeline only publishes
|
|
27
|
+
clusters whose evidence is cohesive, so such a candidate is reported as
|
|
28
|
+
suppressed rather than turned into a brief.
|
|
29
|
+
|
|
30
|
+
The claimed set must still cover >= ``min_items`` distinct underlying URLs, so the
|
|
31
|
+
same link mirrored through two channels can never form a cluster by itself.
|
|
32
|
+
4. Repeatedly pick the candidate whose claimed set has the most independent sources
|
|
33
|
+
(sources that each contribute a different underlying URL), then the most unique
|
|
34
|
+
URLs, then the most items, preferring phrases, then seeds that needed no extra
|
|
35
|
+
term, then alphabetical order.
|
|
36
|
+
5. Stop when no candidate qualifies.
|
|
37
|
+
|
|
38
|
+
Tradeoffs: the gate trades recall for precision. Items that share only a common word
|
|
39
|
+
plus one other word are never clustered, even when they happen to be related; a
|
|
40
|
+
specific word with two senses can still merge unrelated items when they also share a
|
|
41
|
+
second specific term. Every brief keeps a counterpoint telling the reader to check
|
|
42
|
+
the evidence.
|
|
43
|
+
|
|
44
|
+
Items that never join a cluster are not turned into briefs: a single uncorroborated
|
|
45
|
+
item is not treated as a trend.
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
from __future__ import annotations
|
|
49
|
+
|
|
50
|
+
from collections import Counter
|
|
51
|
+
from itertools import combinations
|
|
52
|
+
|
|
53
|
+
from .models import EvidenceItem, TopicCluster
|
|
54
|
+
from .scoring import evidence_is_cohesive, independent_sources
|
|
55
|
+
from .text import (
|
|
56
|
+
STOPWORDS,
|
|
57
|
+
agreement_terms,
|
|
58
|
+
extract_terms,
|
|
59
|
+
is_seed_candidate,
|
|
60
|
+
is_specific,
|
|
61
|
+
item_text,
|
|
62
|
+
specificity,
|
|
63
|
+
tokenize,
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
DISPLAY_WORDS = {
|
|
67
|
+
"ai": "AI",
|
|
68
|
+
"mcp": "MCP",
|
|
69
|
+
"ci": "CI",
|
|
70
|
+
"crdt": "CRDT",
|
|
71
|
+
"sqlite": "SQLite",
|
|
72
|
+
"api": "API",
|
|
73
|
+
"llm": "LLM",
|
|
74
|
+
"saas": "SaaS",
|
|
75
|
+
"hn": "HN",
|
|
76
|
+
"github": "GitHub",
|
|
77
|
+
"gpt": "GPT",
|
|
78
|
+
"rag": "RAG",
|
|
79
|
+
"rl": "RL",
|
|
80
|
+
"ui": "UI",
|
|
81
|
+
"ux": "UX",
|
|
82
|
+
"css": "CSS",
|
|
83
|
+
"sql": "SQL",
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def display_label(term: str) -> str:
|
|
88
|
+
label = " ".join(DISPLAY_WORDS.get(w, w) for w in term.split())
|
|
89
|
+
return label[:1].upper() + label[1:]
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _best_label(seed: str, items: list[EvidenceItem]) -> str:
|
|
93
|
+
"""Longest phrase (<= 3 words) containing the seed shared by >= 60% of items."""
|
|
94
|
+
seed_tokens = seed.split()
|
|
95
|
+
threshold = max(2, -(-len(items) * 3 // 5))
|
|
96
|
+
counts: Counter[str] = Counter()
|
|
97
|
+
for item in items:
|
|
98
|
+
toks = tokenize(item_text(item))
|
|
99
|
+
seen: set[str] = set()
|
|
100
|
+
for n in (3, 2):
|
|
101
|
+
for i in range(len(toks) - n + 1):
|
|
102
|
+
gram = toks[i : i + n]
|
|
103
|
+
if gram[0] in STOPWORDS or gram[-1] in STOPWORDS or len(set(gram)) < n:
|
|
104
|
+
continue # also skips tag runs like "mcp mcp-server"
|
|
105
|
+
for j in range(n - len(seed_tokens) + 1):
|
|
106
|
+
if gram[j : j + len(seed_tokens)] == seed_tokens:
|
|
107
|
+
seen.add(" ".join(gram))
|
|
108
|
+
break
|
|
109
|
+
counts.update(seen)
|
|
110
|
+
best = seed
|
|
111
|
+
for phrase, count in sorted(
|
|
112
|
+
counts.items(), key=lambda kv: (-len(kv[0].split()), -kv[1], kv[0])
|
|
113
|
+
):
|
|
114
|
+
if count >= threshold and is_seed_candidate(phrase):
|
|
115
|
+
best = phrase
|
|
116
|
+
break
|
|
117
|
+
return best
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def coherent_members(
|
|
121
|
+
seed: str, candidates: set[int], items: list[EvidenceItem], term_sets: list[set[str]]
|
|
122
|
+
) -> tuple[list[int], str | None]:
|
|
123
|
+
"""Indices (into ``items``) that ``seed`` may claim among ``candidates``.
|
|
124
|
+
|
|
125
|
+
Returns ``(members, qualifier)``; ``qualifier`` is the second term the members were
|
|
126
|
+
required to share, or ``None`` when the seed alone was sufficient.
|
|
127
|
+
"""
|
|
128
|
+
ids = sorted(candidates)
|
|
129
|
+
if " " in seed:
|
|
130
|
+
return ids, None
|
|
131
|
+
specific = is_specific(seed)
|
|
132
|
+
if specific and evidence_is_cohesive(
|
|
133
|
+
seed, [items[i] for i in ids], [term_sets[i] for i in ids]
|
|
134
|
+
):
|
|
135
|
+
return ids, None
|
|
136
|
+
# The seed alone is not enough: claim the items that also share the best
|
|
137
|
+
# co-occurring agreement term(s), so members share terms worth >= 2 in total
|
|
138
|
+
# (see ``text.specificity``). A specific seed needs one more term; a common seed
|
|
139
|
+
# needs a shared phrase or two shared specific words.
|
|
140
|
+
needed = 2 - specificity(seed)
|
|
141
|
+
by_term: dict[str, list[int]] = {}
|
|
142
|
+
for i in ids:
|
|
143
|
+
terms = sorted(agreement_terms(term_sets[i], seed))
|
|
144
|
+
for term in terms:
|
|
145
|
+
if specificity(term) >= needed:
|
|
146
|
+
by_term.setdefault(term, []).append(i)
|
|
147
|
+
if needed == 2:
|
|
148
|
+
words = [t for t in terms if specificity(t) == 1]
|
|
149
|
+
for a, b in combinations(words, 2):
|
|
150
|
+
by_term.setdefault(f"{a} + {b}", []).append(i)
|
|
151
|
+
best: tuple[tuple[int, int, int, int, str], list[int]] | None = None
|
|
152
|
+
for term, group in by_term.items():
|
|
153
|
+
members = [items[i] for i in group]
|
|
154
|
+
n_urls = len({m.story_key for m in members})
|
|
155
|
+
if n_urls < 2:
|
|
156
|
+
continue
|
|
157
|
+
rank = (
|
|
158
|
+
-independent_sources(members),
|
|
159
|
+
-n_urls,
|
|
160
|
+
-len(group),
|
|
161
|
+
0 if " " in term else 1,
|
|
162
|
+
term,
|
|
163
|
+
)
|
|
164
|
+
if best is None or rank < best[0]:
|
|
165
|
+
best = (rank, group)
|
|
166
|
+
if best is not None:
|
|
167
|
+
return best[1], best[0][4]
|
|
168
|
+
if specific and independent_sources([items[i] for i in ids]) <= 1:
|
|
169
|
+
# One community using one specific word: kept as a *candidate* (it claims no
|
|
170
|
+
# corroboration), but not published unless its evidence is cohesive.
|
|
171
|
+
return ids, None
|
|
172
|
+
return [], None
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def cluster_items(
|
|
176
|
+
items: list[EvidenceItem], min_items: int = 2, max_clusters: int = 25
|
|
177
|
+
) -> list[TopicCluster]:
|
|
178
|
+
ordered = sorted(items, key=lambda i: (i.source, i.url, i.title))
|
|
179
|
+
term_sets = [extract_terms(item_text(i)) for i in ordered]
|
|
180
|
+
postings: dict[str, set[int]] = {}
|
|
181
|
+
for idx, terms in enumerate(term_sets):
|
|
182
|
+
for term in terms:
|
|
183
|
+
if is_seed_candidate(term):
|
|
184
|
+
postings.setdefault(term, set()).add(idx)
|
|
185
|
+
postings = {
|
|
186
|
+
t: ids
|
|
187
|
+
for t, ids in postings.items()
|
|
188
|
+
if len({ordered[i].story_key for i in ids}) >= min_items
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
assigned: set[int] = set()
|
|
192
|
+
clusters: list[TopicCluster] = []
|
|
193
|
+
while len(clusters) < max_clusters:
|
|
194
|
+
best: tuple[tuple[int, int, int, int, int, str], list[int], str | None] | None = None
|
|
195
|
+
for term, ids in postings.items():
|
|
196
|
+
free = ids - assigned
|
|
197
|
+
if len(free) < min_items:
|
|
198
|
+
continue
|
|
199
|
+
claimed, qualifier = coherent_members(term, free, ordered, term_sets)
|
|
200
|
+
claimed_items = [ordered[i] for i in claimed]
|
|
201
|
+
n_urls = len({i.story_key for i in claimed_items})
|
|
202
|
+
if n_urls < min_items:
|
|
203
|
+
continue
|
|
204
|
+
# More independent sources, then more unique URLs, then more items,
|
|
205
|
+
# then phrases, then seeds that needed no qualifier, then alphabetical.
|
|
206
|
+
rank = (
|
|
207
|
+
-independent_sources(claimed_items),
|
|
208
|
+
-n_urls,
|
|
209
|
+
-len(claimed),
|
|
210
|
+
0 if " " in term else 1,
|
|
211
|
+
0 if qualifier is None else 1,
|
|
212
|
+
term,
|
|
213
|
+
)
|
|
214
|
+
if best is None or rank < best[0]:
|
|
215
|
+
best = (rank, claimed, qualifier)
|
|
216
|
+
if best is None:
|
|
217
|
+
break
|
|
218
|
+
best_term = best[0][5]
|
|
219
|
+
members = best[1]
|
|
220
|
+
assigned.update(members)
|
|
221
|
+
cluster_items_ = [ordered[i] for i in members]
|
|
222
|
+
label = _best_label(best_term, cluster_items_)
|
|
223
|
+
qualifier = best[2]
|
|
224
|
+
if qualifier and not set(qualifier.split()) <= set(label.split()):
|
|
225
|
+
label = f"{label} + {qualifier}"
|
|
226
|
+
label_words = set(label.split())
|
|
227
|
+
related: Counter[str] = Counter()
|
|
228
|
+
for i in members:
|
|
229
|
+
related.update(
|
|
230
|
+
t
|
|
231
|
+
for t in term_sets[i]
|
|
232
|
+
if is_seed_candidate(t) and not set(t.split()) <= label_words
|
|
233
|
+
)
|
|
234
|
+
ranked = sorted(related.items(), key=lambda kv: (-kv[1], kv[0]))
|
|
235
|
+
related_terms = [t for t, c in ranked if c >= 2][:5]
|
|
236
|
+
clusters.append(
|
|
237
|
+
TopicCluster(
|
|
238
|
+
key=best_term,
|
|
239
|
+
label=display_label(label),
|
|
240
|
+
items=cluster_items_,
|
|
241
|
+
related_terms=related_terms,
|
|
242
|
+
)
|
|
243
|
+
)
|
|
244
|
+
return clusters
|
trendcite/config.py
ADDED
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
"""Configuration: defaults plus an optional TOML file (``trendcite.toml``).
|
|
2
|
+
|
|
3
|
+
Configuration only names *what* to read (feeds, subreddits, queries, niche terms).
|
|
4
|
+
It never contains credentials; optional LLM keys are read from the environment.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import tomllib
|
|
10
|
+
from dataclasses import dataclass, field, fields
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
DEFAULT_FEEDS = [
|
|
15
|
+
"https://github.blog/feed/",
|
|
16
|
+
"https://simonwillison.net/atom/everything/",
|
|
17
|
+
"https://blog.pragmaticengineer.com/rss/",
|
|
18
|
+
]
|
|
19
|
+
DEFAULT_SUBREDDITS = ["startups", "SaaS", "programming"]
|
|
20
|
+
DEFAULT_GITHUB_QUERIES = ["ai agent", "developer tools"]
|
|
21
|
+
DEFAULT_NICHE = ["ai agents", "developer tools", "startups", "saas"]
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class ConfigError(ValueError):
|
|
25
|
+
pass
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass
|
|
29
|
+
class Config:
|
|
30
|
+
niche: list[str] = field(default_factory=lambda: list(DEFAULT_NICHE))
|
|
31
|
+
feeds: list[str] = field(default_factory=lambda: list(DEFAULT_FEEDS))
|
|
32
|
+
subreddits: list[str] = field(default_factory=lambda: list(DEFAULT_SUBREDDITS))
|
|
33
|
+
github_queries: list[str] = field(default_factory=lambda: list(DEFAULT_GITHUB_QUERIES))
|
|
34
|
+
hn_limit: int = 30
|
|
35
|
+
sources: list[str] = field(default_factory=lambda: ["hackernews", "github", "rss", "reddit"])
|
|
36
|
+
top: int = 5
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _str_list(value: Any, key: str, limit: int = 50) -> list[str]:
|
|
40
|
+
if not isinstance(value, list) or not all(isinstance(v, str) for v in value):
|
|
41
|
+
raise ConfigError(f"'{key}' must be a list of strings")
|
|
42
|
+
return [v.strip() for v in value if v.strip()][:limit]
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def load_config(path: str | Path | None) -> Config:
|
|
46
|
+
cfg = Config()
|
|
47
|
+
if path is None:
|
|
48
|
+
return cfg
|
|
49
|
+
p = Path(path)
|
|
50
|
+
try:
|
|
51
|
+
data = tomllib.loads(p.read_text(encoding="utf-8"))
|
|
52
|
+
except FileNotFoundError as exc:
|
|
53
|
+
raise ConfigError(f"config file not found: {p}") from exc
|
|
54
|
+
except tomllib.TOMLDecodeError as exc:
|
|
55
|
+
raise ConfigError(f"invalid TOML in {p}: {exc}") from exc
|
|
56
|
+
known = {f.name for f in fields(Config)}
|
|
57
|
+
unknown = sorted(set(data) - known)
|
|
58
|
+
if unknown:
|
|
59
|
+
raise ConfigError(f"unknown config keys: {', '.join(unknown)}")
|
|
60
|
+
for key in ("niche", "feeds", "subreddits", "github_queries", "sources"):
|
|
61
|
+
if key in data:
|
|
62
|
+
setattr(cfg, key, _str_list(data[key], key))
|
|
63
|
+
for key in ("hn_limit", "top"):
|
|
64
|
+
if key in data:
|
|
65
|
+
if not isinstance(data[key], int) or isinstance(data[key], bool):
|
|
66
|
+
raise ConfigError(f"'{key}' must be an integer")
|
|
67
|
+
setattr(cfg, key, data[key])
|
|
68
|
+
return cfg
|