trendcite 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
trendcite/__init__.py ADDED
@@ -0,0 +1,4 @@
1
+ """TrendCite: evidence-first trend intelligence for founders."""
2
+
3
+ __version__ = "0.1.0"
4
+ __all__ = ["__version__"]
trendcite/__main__.py ADDED
@@ -0,0 +1,3 @@
1
+ from trendcite.cli import main
2
+
3
+ raise SystemExit(main())
trendcite/briefs.py ADDED
@@ -0,0 +1,234 @@
1
+ """Turn scored clusters into Content Opportunity Briefs (deterministic templates).
2
+
3
+ Everything in a brief except the draft outline is derived from captured evidence
4
+ and computed scores. The outline is a writing aid and is labelled as such.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from datetime import datetime
10
+
11
+ from .models import Brief, EvidenceItem, TopicCluster
12
+ from .scoring import HALF_LIFE_HOURS, WEIGHTS, age_hours, select_evidence
13
+ from .signal import SignalBrief
14
+ from .text import CRITICAL_RE
15
+
16
+ SOURCE_NAMES = {
17
+ "hackernews": "Hacker News",
18
+ "github": "GitHub",
19
+ "rss": "RSS/Atom",
20
+ "reddit": "Reddit",
21
+ "x": "X",
22
+ }
23
+
24
+
25
+ def metric_summary(item: EvidenceItem) -> str:
26
+ m = item.metrics
27
+ if item.source == "hackernews":
28
+ return f"{int(m.get('points', 0))} points, {int(m.get('comments', 0))} comments"
29
+ if item.source == "github":
30
+ return f"{int(m.get('stars', 0))} stars, {int(m.get('forks', 0))} forks"
31
+ return "no engagement metrics available from this source"
32
+
33
+
34
+ def _source_list(sources: list[str]) -> str:
35
+ return ", ".join(SOURCE_NAMES.get(s, s) for s in sources)
36
+
37
+
38
+ def _angle(cluster: TopicCluster) -> str:
39
+ topic = cluster.label
40
+ assert cluster.score is not None
41
+ s = cluster.score
42
+ sources = set(s.sources)
43
+ if "github" in sources and sources & {"hackernews", "reddit"}:
44
+ return (
45
+ f"Builders are already shipping around {topic} while the discussion is still "
46
+ f"unsettled: share what actually works (and what doesn't) from first-hand use."
47
+ )
48
+ if "reddit" in sources and s.engagement < 0.6:
49
+ return (
50
+ f"Practitioners are asking open questions about {topic}; answer them with your "
51
+ f"own operating numbers and decisions rather than a summary of the debate."
52
+ )
53
+ if s.engagement >= 0.6 and s.corroboration >= 0.5:
54
+ return (
55
+ f"{topic} is drawing strong attention across communities; the useful angle is "
56
+ f"what it changes for a small team this quarter."
57
+ )
58
+ return (
59
+ f"{topic} is being covered by publishers; add a first-hand or contrarian take "
60
+ f"that the existing coverage lacks."
61
+ )
62
+
63
+
64
+ def _why_now(
65
+ cluster: TopicCluster,
66
+ evidence: list[EvidenceItem],
67
+ now: datetime,
68
+ percentiles: dict[str, float],
69
+ ) -> list[str]:
70
+ assert cluster.score is not None
71
+ s = cluster.score
72
+ items = evidence
73
+ ages = [age_hours(i, now) for i in items]
74
+ lines = [
75
+ f"{len(items)} evidence item(s) shown and scored: {s.unique_urls} unique URL(s) from "
76
+ f"{len(s.sources)} source(s) ({_source_list(list(s.sources))}); newest "
77
+ f"{min(ages):.0f} h old, oldest {max(ages):.0f} h old."
78
+ ]
79
+ if len(cluster.items) > len(items):
80
+ lines.append(
81
+ f"{len(cluster.items)} items matched this topic in total; the rest are not "
82
+ f"shown and do not affect the score."
83
+ )
84
+ measured = [i for i in items if i.item_id in percentiles]
85
+ if measured:
86
+ top = max(measured, key=lambda i: (percentiles[i.item_id], i.item_id))
87
+ lines.append(
88
+ f'Strongest engagement signal: "{top.title}" '
89
+ f"({SOURCE_NAMES.get(top.source, top.source)}: "
90
+ f"{metric_summary(top)}; engagement percentile {percentiles[top.item_id] * 100:.0f} "
91
+ f"within that source this run)."
92
+ )
93
+ if cluster.niche_matches:
94
+ lines.append("Matches your niche terms: " + ", ".join(cluster.niche_matches) + ".")
95
+ if cluster.related_terms:
96
+ lines.append("Co-occurring terms: " + ", ".join(cluster.related_terms) + ".")
97
+ return lines
98
+
99
+
100
+ def _score_explanation(cluster: TopicCluster) -> list[str]:
101
+ assert cluster.score is not None
102
+ s = cluster.score
103
+ w = WEIGHTS
104
+ rows = [
105
+ (
106
+ "recency",
107
+ w.recency,
108
+ s.recency,
109
+ f"{HALF_LIFE_HOURS:.0f} h half-life, averaged over items",
110
+ ),
111
+ ("engagement", w.engagement, s.engagement, "top-3 within-source percentiles"),
112
+ (
113
+ "corroboration",
114
+ w.corroboration,
115
+ s.corroboration,
116
+ f"{s.independent_sources} independent source(s): sources that each link a "
117
+ f"different URL",
118
+ ),
119
+ (
120
+ "relevance",
121
+ w.relevance,
122
+ s.relevance,
123
+ f"{len(cluster.niche_matches)} niche term(s) matched in the evidence",
124
+ ),
125
+ ("diversity", w.diversity, s.diversity, f"{s.publishers} distinct publisher(s)"),
126
+ ]
127
+ lines = [
128
+ f"{name}: {value:.2f} x {weight:.2f} = {100 * value * weight:.1f} pts ({why})"
129
+ for name, weight, value, why in rows
130
+ ]
131
+ lines.append(f"total: {s.total:.1f}/100, confidence {s.confidence}")
132
+ lines.append(
133
+ "all components are computed from the evidence items listed in this brief; high "
134
+ "confidence requires >= 3 independent sources, >= 4 unique URLs, >= 3 publishers, "
135
+ "a niche match (when a niche is set) and evidence that agrees on more than one word"
136
+ )
137
+ return lines
138
+
139
+
140
+ def _counterpoints(cluster: TopicCluster, evidence: list[EvidenceItem], now: datetime) -> list[str]:
141
+ assert cluster.score is not None
142
+ s = cluster.score
143
+ notes: list[str] = []
144
+ if s.independent_sources <= 1:
145
+ where = _source_list(list(s.sources))
146
+ if len(s.sources) > 1:
147
+ notes.append(
148
+ f"Seen via {where}, but they link the same underlying URL; that is one "
149
+ f"story mirrored across channels, not independent corroboration."
150
+ )
151
+ else:
152
+ notes.append(
153
+ f"Only seen on {where}; this may be one community's echo chamber rather "
154
+ f"than a broad trend."
155
+ )
156
+ mirrored = len(evidence) - s.unique_urls
157
+ if mirrored > 0 and s.independent_sources > 1:
158
+ notes.append(
159
+ f"{mirrored} evidence item(s) repeat a URL already shown via another channel; "
160
+ f"repeats are counted once for corroboration."
161
+ )
162
+ if s.unique_urls <= 2:
163
+ notes.append(
164
+ f"Small sample ({s.unique_urls} unique URL(s)). Treat as an early signal, not a trend."
165
+ )
166
+ if not any(i.engagement() is not None for i in evidence):
167
+ notes.append("No engagement metrics are available for these sources; reach is unknown.")
168
+ stale = [i for i in evidence if age_hours(i, now) > 7 * 24]
169
+ if stale:
170
+ notes.append(f"{len(stale)} item(s) are over a week old; part of this signal is not new.")
171
+ critical = [i for i in evidence if CRITICAL_RE.search(i.title)]
172
+ if critical:
173
+ notes.append(
174
+ f'Part of the evidence is critical or cautionary (e.g. "{critical[0].title}"); '
175
+ f"address the downside explicitly instead of only the upside."
176
+ )
177
+ notes.append(
178
+ "Clustering is keyword-based: confirm the linked items really discuss the same thing "
179
+ "before relying on the corroboration count."
180
+ )
181
+ return notes
182
+
183
+
184
+ def _questions(topic: str) -> list[str]:
185
+ return [
186
+ f"What have you personally shipped, broken or decided about {topic} in the last 90 days?",
187
+ "Where does this evidence disagree with what you hear from your own customers?",
188
+ f"For a 3-person team, is {topic} a 'do now', 'watch' or 'ignore' this quarter, and why?",
189
+ "Which number from your own business could you share to make the point concrete?",
190
+ ]
191
+
192
+
193
+ def _outline(cluster: TopicCluster, evidence: list[EvidenceItem]) -> list[str]:
194
+ first = evidence[0]
195
+ return [
196
+ f'Hook: open with the concrete signal from evidence [1] ("{first.title}").',
197
+ "Context: summarise what the linked evidence shows, citing items by number.",
198
+ "Your POV: answer the first founder question with a specific story or decision.",
199
+ "Counterpoint: name the strongest objection listed above and respond to it honestly.",
200
+ f"Takeaway: one practical recommendation about {cluster.label} for small teams.",
201
+ ]
202
+
203
+
204
+ def build_brief(
205
+ cluster: TopicCluster,
206
+ rank: int,
207
+ now: datetime,
208
+ percentiles: dict[str, float],
209
+ signal: SignalBrief | None = None,
210
+ ) -> Brief:
211
+ """Project one scored cluster (and, when available, its signal) into a public brief."""
212
+ assert cluster.score is not None
213
+ evidence = cluster.evidence or select_evidence(cluster.items, percentiles, now)
214
+ flags: list[str] = []
215
+ flagged = [i for i in evidence if i.flags]
216
+ if flagged:
217
+ flags.append(
218
+ f"{len(flagged)} evidence item(s) contain text that looks like instructions to an AI "
219
+ f"system (possible prompt injection). Shown as inert data only; nothing was executed."
220
+ )
221
+ return Brief(
222
+ rank=rank,
223
+ topic=cluster.label,
224
+ angle=_angle(cluster),
225
+ why_now=_why_now(cluster, evidence, now, percentiles),
226
+ evidence=evidence,
227
+ score=cluster.score,
228
+ score_explanation=_score_explanation(cluster),
229
+ counterpoints=_counterpoints(cluster, evidence, now),
230
+ founder_questions=_questions(cluster.label),
231
+ outline=_outline(cluster, evidence),
232
+ flags=flags,
233
+ signal=signal,
234
+ )
trendcite/cli.py ADDED
@@ -0,0 +1,128 @@
1
+ """Command-line interface: ``trendcite demo | live | sources``."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import sys
7
+ from collections.abc import Sequence
8
+ from pathlib import Path
9
+
10
+ from . import __version__
11
+ from .config import Config, ConfigError, load_config
12
+ from .llm import LLMError, enhance_briefs, provider_from_env
13
+ from .models import Report
14
+ from .pipeline import make_adapters, run_demo, run_live
15
+ from .render import to_json, to_markdown
16
+ from .security import configure_logging, redact
17
+
18
+
19
+ def _split(value: str | None) -> list[str] | None:
20
+ if value is None:
21
+ return None
22
+ return [v.strip() for v in value.split(",") if v.strip()]
23
+
24
+
25
+ def _add_common(p: argparse.ArgumentParser) -> None:
26
+ p.add_argument("--format", choices=["markdown", "json"], default="markdown")
27
+ p.add_argument("--out", type=Path, help="write the report to this file instead of stdout")
28
+ p.add_argument("--top", type=int, default=None, help="number of briefs, clamped to 3-5")
29
+ p.add_argument("--niche", help="comma-separated niche phrases, e.g. 'ai agents,pricing'")
30
+ p.add_argument(
31
+ "--llm",
32
+ action="store_true",
33
+ help="refine angles/outlines with the provider set in TRENDCITE_LLM_PROVIDER",
34
+ )
35
+ p.add_argument("-v", "--verbose", action="store_true")
36
+
37
+
38
+ def build_parser() -> argparse.ArgumentParser:
39
+ parser = argparse.ArgumentParser(
40
+ prog="trendcite",
41
+ description="Evidence-first trend intelligence: traceable content opportunity briefs.",
42
+ )
43
+ parser.add_argument("--version", action="version", version=f"trendcite {__version__}")
44
+ sub = parser.add_subparsers(dest="command", required=True)
45
+
46
+ demo = sub.add_parser("demo", help="offline demo on bundled synthetic fixtures (no network)")
47
+ _add_common(demo)
48
+
49
+ live = sub.add_parser("live", help="collect from public read-only sources and build briefs")
50
+ _add_common(live)
51
+ live.add_argument("--config", type=Path, help="path to trendcite.toml")
52
+ live.add_argument("--sources", help="comma-separated: hackernews,github,rss,reddit,x")
53
+ live.add_argument("--feeds", help="comma-separated RSS/Atom URLs (replaces configured feeds)")
54
+ live.add_argument("--subreddits", help="comma-separated subreddit names")
55
+ live.add_argument("--github-queries", help="comma-separated GitHub search queries")
56
+
57
+ sub.add_parser("sources", help="list source adapters and their status requirements")
58
+ return parser
59
+
60
+
61
+ def _emit(text: str, out: Path | None) -> None:
62
+ if out is None:
63
+ sys.stdout.write(text)
64
+ return
65
+ out.write_text(text, encoding="utf-8")
66
+ sys.stderr.write(f"trendcite: wrote {out}\n")
67
+
68
+
69
+ def _maybe_llm(report: Report, enabled: bool) -> None:
70
+ if not enabled:
71
+ return
72
+ try:
73
+ provider = provider_from_env()
74
+ except LLMError as exc:
75
+ report.notes.append(f"LLM synthesis disabled: {redact(str(exc))}")
76
+ return
77
+ if provider is None:
78
+ report.notes.append(
79
+ "--llm given but TRENDCITE_LLM_PROVIDER is not set; deterministic output only."
80
+ )
81
+ return
82
+ report.notes.extend(enhance_briefs(report.briefs, provider))
83
+
84
+
85
+ def _live_config(args: argparse.Namespace) -> Config:
86
+ cfg = load_config(args.config)
87
+ for attr, value in (
88
+ ("sources", _split(args.sources)),
89
+ ("feeds", _split(args.feeds)),
90
+ ("subreddits", _split(args.subreddits)),
91
+ ("github_queries", _split(args.github_queries)),
92
+ ("niche", _split(args.niche)),
93
+ ):
94
+ if value is not None:
95
+ setattr(cfg, attr, value)
96
+ if args.top is not None:
97
+ cfg.top = args.top
98
+ return cfg
99
+
100
+
101
+ def main(argv: Sequence[str] | None = None) -> int:
102
+ for stream in (sys.stdout, sys.stderr):
103
+ reconfigure = getattr(stream, "reconfigure", None)
104
+ if reconfigure is not None:
105
+ reconfigure(encoding="utf-8", errors="replace")
106
+ args = build_parser().parse_args(argv)
107
+ configure_logging(getattr(args, "verbose", False))
108
+ try:
109
+ if args.command == "sources":
110
+ for adapter in make_adapters(
111
+ Config(sources=["hackernews", "github", "rss", "reddit", "x"])
112
+ ):
113
+ print(f"{adapter.name:<11} {adapter.description}")
114
+ return 0
115
+ if args.command == "demo":
116
+ report = run_demo(top=args.top or 5, niche=_split(args.niche))
117
+ else:
118
+ report = run_live(_live_config(args))
119
+ _maybe_llm(report, args.llm)
120
+ rendered = to_json(report) if args.format == "json" else to_markdown(report)
121
+ _emit(rendered, args.out)
122
+ if args.command == "live" and not any(s.ok for s in report.source_status):
123
+ sys.stderr.write("trendcite: no source was reachable; see the source table.\n")
124
+ return 2
125
+ return 0
126
+ except (ConfigError, ValueError) as exc:
127
+ sys.stderr.write(f"trendcite: error: {redact(str(exc))}\n")
128
+ return 2
trendcite/cluster.py ADDED
@@ -0,0 +1,244 @@
1
+ """Deterministic topic clustering by shared key terms, with a coherence gate.
2
+
3
+ Algorithm (greedy, fully deterministic):
4
+
5
+ 1. Extract unigram and bigram terms from each item's *feature view* (title + excerpt
6
+ head with URLs, domains and feed boilerplate removed; see ``text.item_text``).
7
+ 2. A term is a *seed candidate* if it is not generic or boilerplate and occurs in items
8
+ with >= ``min_items`` distinct underlying URLs (``EvidenceItem.story_key``).
9
+ 3. Coherence gate. The items a seed may claim depend on how distinctive the seed is:
10
+
11
+ - phrase (two words, at least one not generic): every item containing it. A shared
12
+ multi-word phrase is itself strong lexical agreement.
13
+ - specific word (not in the common-English reference lexicon ``lexicon.py``, e.g.
14
+ "mcp", "kubernetes"): every item containing it if they are cohesive
15
+ (``scoring.evidence_is_cohesive``: every URL shares a further specific term or
16
+ phrase with more than half of the other URLs).
17
+ - otherwise (a common word such as "decision", "agent", "memory", or a specific
18
+ word that failed the cohesion check): only the items that also share the seed's
19
+ best co-occurring *agreement terms* (``text.agreement_terms``). Uniform
20
+ principle: members must share terms worth >= 2 in total, where a phrase counts
21
+ 2, a specific word 1 and a common word 0 (``text.specificity``). So "gpt" +
22
+ "astra" is enough, while "agent" needs a shared phrase or two more specific
23
+ words. The label names the extra terms. Inflected forms are checked against the
24
+ lexicon too ("testing" counts as "test").
25
+ - last resort, specific word from a single independent source: all its items form
26
+ a *candidate* cluster that claims no corroboration. The pipeline only publishes
27
+ clusters whose evidence is cohesive, so such a candidate is reported as
28
+ suppressed rather than turned into a brief.
29
+
30
+ The claimed set must still cover >= ``min_items`` distinct underlying URLs, so the
31
+ same link mirrored through two channels can never form a cluster by itself.
32
+ 4. Repeatedly pick the candidate whose claimed set has the most independent sources
33
+ (sources that each contribute a different underlying URL), then the most unique
34
+ URLs, then the most items, preferring phrases, then seeds that needed no extra
35
+ term, then alphabetical order.
36
+ 5. Stop when no candidate qualifies.
37
+
38
+ Tradeoffs: the gate trades recall for precision. Items that share only a common word
39
+ plus one other word are never clustered, even when they happen to be related; a
40
+ specific word with two senses can still merge unrelated items when they also share a
41
+ second specific term. Every brief keeps a counterpoint telling the reader to check
42
+ the evidence.
43
+
44
+ Items that never join a cluster are not turned into briefs: a single uncorroborated
45
+ item is not treated as a trend.
46
+ """
47
+
48
+ from __future__ import annotations
49
+
50
+ from collections import Counter
51
+ from itertools import combinations
52
+
53
+ from .models import EvidenceItem, TopicCluster
54
+ from .scoring import evidence_is_cohesive, independent_sources
55
+ from .text import (
56
+ STOPWORDS,
57
+ agreement_terms,
58
+ extract_terms,
59
+ is_seed_candidate,
60
+ is_specific,
61
+ item_text,
62
+ specificity,
63
+ tokenize,
64
+ )
65
+
66
+ DISPLAY_WORDS = {
67
+ "ai": "AI",
68
+ "mcp": "MCP",
69
+ "ci": "CI",
70
+ "crdt": "CRDT",
71
+ "sqlite": "SQLite",
72
+ "api": "API",
73
+ "llm": "LLM",
74
+ "saas": "SaaS",
75
+ "hn": "HN",
76
+ "github": "GitHub",
77
+ "gpt": "GPT",
78
+ "rag": "RAG",
79
+ "rl": "RL",
80
+ "ui": "UI",
81
+ "ux": "UX",
82
+ "css": "CSS",
83
+ "sql": "SQL",
84
+ }
85
+
86
+
87
+ def display_label(term: str) -> str:
88
+ label = " ".join(DISPLAY_WORDS.get(w, w) for w in term.split())
89
+ return label[:1].upper() + label[1:]
90
+
91
+
92
+ def _best_label(seed: str, items: list[EvidenceItem]) -> str:
93
+ """Longest phrase (<= 3 words) containing the seed shared by >= 60% of items."""
94
+ seed_tokens = seed.split()
95
+ threshold = max(2, -(-len(items) * 3 // 5))
96
+ counts: Counter[str] = Counter()
97
+ for item in items:
98
+ toks = tokenize(item_text(item))
99
+ seen: set[str] = set()
100
+ for n in (3, 2):
101
+ for i in range(len(toks) - n + 1):
102
+ gram = toks[i : i + n]
103
+ if gram[0] in STOPWORDS or gram[-1] in STOPWORDS or len(set(gram)) < n:
104
+ continue # also skips tag runs like "mcp mcp-server"
105
+ for j in range(n - len(seed_tokens) + 1):
106
+ if gram[j : j + len(seed_tokens)] == seed_tokens:
107
+ seen.add(" ".join(gram))
108
+ break
109
+ counts.update(seen)
110
+ best = seed
111
+ for phrase, count in sorted(
112
+ counts.items(), key=lambda kv: (-len(kv[0].split()), -kv[1], kv[0])
113
+ ):
114
+ if count >= threshold and is_seed_candidate(phrase):
115
+ best = phrase
116
+ break
117
+ return best
118
+
119
+
120
+ def coherent_members(
121
+ seed: str, candidates: set[int], items: list[EvidenceItem], term_sets: list[set[str]]
122
+ ) -> tuple[list[int], str | None]:
123
+ """Indices (into ``items``) that ``seed`` may claim among ``candidates``.
124
+
125
+ Returns ``(members, qualifier)``; ``qualifier`` is the second term the members were
126
+ required to share, or ``None`` when the seed alone was sufficient.
127
+ """
128
+ ids = sorted(candidates)
129
+ if " " in seed:
130
+ return ids, None
131
+ specific = is_specific(seed)
132
+ if specific and evidence_is_cohesive(
133
+ seed, [items[i] for i in ids], [term_sets[i] for i in ids]
134
+ ):
135
+ return ids, None
136
+ # The seed alone is not enough: claim the items that also share the best
137
+ # co-occurring agreement term(s), so members share terms worth >= 2 in total
138
+ # (see ``text.specificity``). A specific seed needs one more term; a common seed
139
+ # needs a shared phrase or two shared specific words.
140
+ needed = 2 - specificity(seed)
141
+ by_term: dict[str, list[int]] = {}
142
+ for i in ids:
143
+ terms = sorted(agreement_terms(term_sets[i], seed))
144
+ for term in terms:
145
+ if specificity(term) >= needed:
146
+ by_term.setdefault(term, []).append(i)
147
+ if needed == 2:
148
+ words = [t for t in terms if specificity(t) == 1]
149
+ for a, b in combinations(words, 2):
150
+ by_term.setdefault(f"{a} + {b}", []).append(i)
151
+ best: tuple[tuple[int, int, int, int, str], list[int]] | None = None
152
+ for term, group in by_term.items():
153
+ members = [items[i] for i in group]
154
+ n_urls = len({m.story_key for m in members})
155
+ if n_urls < 2:
156
+ continue
157
+ rank = (
158
+ -independent_sources(members),
159
+ -n_urls,
160
+ -len(group),
161
+ 0 if " " in term else 1,
162
+ term,
163
+ )
164
+ if best is None or rank < best[0]:
165
+ best = (rank, group)
166
+ if best is not None:
167
+ return best[1], best[0][4]
168
+ if specific and independent_sources([items[i] for i in ids]) <= 1:
169
+ # One community using one specific word: kept as a *candidate* (it claims no
170
+ # corroboration), but not published unless its evidence is cohesive.
171
+ return ids, None
172
+ return [], None
173
+
174
+
175
+ def cluster_items(
176
+ items: list[EvidenceItem], min_items: int = 2, max_clusters: int = 25
177
+ ) -> list[TopicCluster]:
178
+ ordered = sorted(items, key=lambda i: (i.source, i.url, i.title))
179
+ term_sets = [extract_terms(item_text(i)) for i in ordered]
180
+ postings: dict[str, set[int]] = {}
181
+ for idx, terms in enumerate(term_sets):
182
+ for term in terms:
183
+ if is_seed_candidate(term):
184
+ postings.setdefault(term, set()).add(idx)
185
+ postings = {
186
+ t: ids
187
+ for t, ids in postings.items()
188
+ if len({ordered[i].story_key for i in ids}) >= min_items
189
+ }
190
+
191
+ assigned: set[int] = set()
192
+ clusters: list[TopicCluster] = []
193
+ while len(clusters) < max_clusters:
194
+ best: tuple[tuple[int, int, int, int, int, str], list[int], str | None] | None = None
195
+ for term, ids in postings.items():
196
+ free = ids - assigned
197
+ if len(free) < min_items:
198
+ continue
199
+ claimed, qualifier = coherent_members(term, free, ordered, term_sets)
200
+ claimed_items = [ordered[i] for i in claimed]
201
+ n_urls = len({i.story_key for i in claimed_items})
202
+ if n_urls < min_items:
203
+ continue
204
+ # More independent sources, then more unique URLs, then more items,
205
+ # then phrases, then seeds that needed no qualifier, then alphabetical.
206
+ rank = (
207
+ -independent_sources(claimed_items),
208
+ -n_urls,
209
+ -len(claimed),
210
+ 0 if " " in term else 1,
211
+ 0 if qualifier is None else 1,
212
+ term,
213
+ )
214
+ if best is None or rank < best[0]:
215
+ best = (rank, claimed, qualifier)
216
+ if best is None:
217
+ break
218
+ best_term = best[0][5]
219
+ members = best[1]
220
+ assigned.update(members)
221
+ cluster_items_ = [ordered[i] for i in members]
222
+ label = _best_label(best_term, cluster_items_)
223
+ qualifier = best[2]
224
+ if qualifier and not set(qualifier.split()) <= set(label.split()):
225
+ label = f"{label} + {qualifier}"
226
+ label_words = set(label.split())
227
+ related: Counter[str] = Counter()
228
+ for i in members:
229
+ related.update(
230
+ t
231
+ for t in term_sets[i]
232
+ if is_seed_candidate(t) and not set(t.split()) <= label_words
233
+ )
234
+ ranked = sorted(related.items(), key=lambda kv: (-kv[1], kv[0]))
235
+ related_terms = [t for t, c in ranked if c >= 2][:5]
236
+ clusters.append(
237
+ TopicCluster(
238
+ key=best_term,
239
+ label=display_label(label),
240
+ items=cluster_items_,
241
+ related_terms=related_terms,
242
+ )
243
+ )
244
+ return clusters
trendcite/config.py ADDED
@@ -0,0 +1,68 @@
1
+ """Configuration: defaults plus an optional TOML file (``trendcite.toml``).
2
+
3
+ Configuration only names *what* to read (feeds, subreddits, queries, niche terms).
4
+ It never contains credentials; optional LLM keys are read from the environment.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import tomllib
10
+ from dataclasses import dataclass, field, fields
11
+ from pathlib import Path
12
+ from typing import Any
13
+
14
+ DEFAULT_FEEDS = [
15
+ "https://github.blog/feed/",
16
+ "https://simonwillison.net/atom/everything/",
17
+ "https://blog.pragmaticengineer.com/rss/",
18
+ ]
19
+ DEFAULT_SUBREDDITS = ["startups", "SaaS", "programming"]
20
+ DEFAULT_GITHUB_QUERIES = ["ai agent", "developer tools"]
21
+ DEFAULT_NICHE = ["ai agents", "developer tools", "startups", "saas"]
22
+
23
+
24
+ class ConfigError(ValueError):
25
+ pass
26
+
27
+
28
+ @dataclass
29
+ class Config:
30
+ niche: list[str] = field(default_factory=lambda: list(DEFAULT_NICHE))
31
+ feeds: list[str] = field(default_factory=lambda: list(DEFAULT_FEEDS))
32
+ subreddits: list[str] = field(default_factory=lambda: list(DEFAULT_SUBREDDITS))
33
+ github_queries: list[str] = field(default_factory=lambda: list(DEFAULT_GITHUB_QUERIES))
34
+ hn_limit: int = 30
35
+ sources: list[str] = field(default_factory=lambda: ["hackernews", "github", "rss", "reddit"])
36
+ top: int = 5
37
+
38
+
39
+ def _str_list(value: Any, key: str, limit: int = 50) -> list[str]:
40
+ if not isinstance(value, list) or not all(isinstance(v, str) for v in value):
41
+ raise ConfigError(f"'{key}' must be a list of strings")
42
+ return [v.strip() for v in value if v.strip()][:limit]
43
+
44
+
45
+ def load_config(path: str | Path | None) -> Config:
46
+ cfg = Config()
47
+ if path is None:
48
+ return cfg
49
+ p = Path(path)
50
+ try:
51
+ data = tomllib.loads(p.read_text(encoding="utf-8"))
52
+ except FileNotFoundError as exc:
53
+ raise ConfigError(f"config file not found: {p}") from exc
54
+ except tomllib.TOMLDecodeError as exc:
55
+ raise ConfigError(f"invalid TOML in {p}: {exc}") from exc
56
+ known = {f.name for f in fields(Config)}
57
+ unknown = sorted(set(data) - known)
58
+ if unknown:
59
+ raise ConfigError(f"unknown config keys: {', '.join(unknown)}")
60
+ for key in ("niche", "feeds", "subreddits", "github_queries", "sources"):
61
+ if key in data:
62
+ setattr(cfg, key, _str_list(data[key], key))
63
+ for key in ("hn_limit", "top"):
64
+ if key in data:
65
+ if not isinstance(data[key], int) or isinstance(data[key], bool):
66
+ raise ConfigError(f"'{key}' must be an integer")
67
+ setattr(cfg, key, data[key])
68
+ return cfg