secondbrain-py 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (273) hide show
  1. brain/__init__.py +0 -0
  2. brain/__main__.py +18 -0
  3. brain/_capture_command.py +445 -0
  4. brain/_compose.py +52 -0
  5. brain/activity.py +206 -0
  6. brain/ask.py +631 -0
  7. brain/audio.py +591 -0
  8. brain/backfill/__init__.py +12 -0
  9. brain/backfill/search_extras.py +141 -0
  10. brain/backfill/source_rows.py +101 -0
  11. brain/bin/__init__.py +1 -0
  12. brain/bin/_launcher.py +107 -0
  13. brain/bin/down.py +8 -0
  14. brain/bin/launchd.py +268 -0
  15. brain/bin/monitor.py +570 -0
  16. brain/bin/rebuild.py +8 -0
  17. brain/bin/status.py +8 -0
  18. brain/bin/up.py +8 -0
  19. brain/brief.py +272 -0
  20. brain/capture.py +49 -0
  21. brain/chat.py +293 -0
  22. brain/cli.py +9760 -0
  23. brain/cli_claude.py +81 -0
  24. brain/cli_connect.py +285 -0
  25. brain/cli_demo.py +266 -0
  26. brain/config.py +1949 -0
  27. brain/connect.py +925 -0
  28. brain/db.py +540 -0
  29. brain/demo/__init__.py +452 -0
  30. brain/demo/corpus/manifest.json +403 -0
  31. brain/demo/embedder.py +74 -0
  32. brain/durations.py +84 -0
  33. brain/edit_session.py +156 -0
  34. brain/editor.py +67 -0
  35. brain/elicit/__init__.py +16 -0
  36. brain/elicit/detectors.py +250 -0
  37. brain/elicit/drafter.py +70 -0
  38. brain/elicit/queue.py +220 -0
  39. brain/elicit/schema.py +48 -0
  40. brain/elicit/session.py +445 -0
  41. brain/embedding_targets.py +54 -0
  42. brain/embeddings.py +424 -0
  43. brain/enrichment.py +808 -0
  44. brain/errors.py +357 -0
  45. brain/eval/__init__.py +129 -0
  46. brain/eval/answer_eval.py +281 -0
  47. brain/eval/baseline.py +265 -0
  48. brain/eval/concept_extraction.py +378 -0
  49. brain/eval/corpus.py +152 -0
  50. brain/eval/errors.py +19 -0
  51. brain/eval/graph_baseline.py +226 -0
  52. brain/eval/graph_retrieval.py +202 -0
  53. brain/eval/graph_runner.py +319 -0
  54. brain/eval/metrics.py +101 -0
  55. brain/eval/runner.py +223 -0
  56. brain/format.py +783 -0
  57. brain/gaps.py +390 -0
  58. brain/graph_rag/__init__.py +94 -0
  59. brain/graph_rag/_retrieval_common.py +113 -0
  60. brain/graph_rag/aggregates.py +303 -0
  61. brain/graph_rag/aliases/__init__.py +583 -0
  62. brain/graph_rag/backends/__init__.py +10 -0
  63. brain/graph_rag/backends/_age_helpers.py +473 -0
  64. brain/graph_rag/backends/age.py +782 -0
  65. brain/graph_rag/backends/base.py +272 -0
  66. brain/graph_rag/build.py +344 -0
  67. brain/graph_rag/communities.py +644 -0
  68. brain/graph_rag/communities_summary.py +437 -0
  69. brain/graph_rag/concepts.py +202 -0
  70. brain/graph_rag/cooccur.py +193 -0
  71. brain/graph_rag/cross_type.py +312 -0
  72. brain/graph_rag/extract.py +885 -0
  73. brain/graph_rag/fuse.py +371 -0
  74. brain/graph_rag/global_.py +412 -0
  75. brain/graph_rag/grouping.py +372 -0
  76. brain/graph_rag/person_resolver.py +167 -0
  77. brain/graph_rag/reconcile.py +792 -0
  78. brain/graph_rag/relational.py +353 -0
  79. brain/graph_rag/retrieve.py +526 -0
  80. brain/graph_rag/router.py +288 -0
  81. brain/graph_rag/schema.py +320 -0
  82. brain/graph_rag/sync.py +237 -0
  83. brain/graph_rag/tenancy.py +43 -0
  84. brain/graph_rag/themes.py +501 -0
  85. brain/graph_rag/weighting.py +202 -0
  86. brain/ingest/__init__.py +1926 -0
  87. brain/ingest/chunker.py +249 -0
  88. brain/ingest/docx.py +40 -0
  89. brain/ingest/gmail.py +621 -0
  90. brain/ingest/markdown.py +37 -0
  91. brain/ingest/pdf.py +61 -0
  92. brain/ingest/stdin.py +22 -0
  93. brain/ingest/sub_tokens.py +91 -0
  94. brain/ingest/text.py +16 -0
  95. brain/interactions.py +205 -0
  96. brain/maintenance.py +355 -0
  97. brain/mcp_server.py +3405 -0
  98. brain/migrations/001_init.sql +43 -0
  99. brain/migrations/002_qwen3_embedding.sql +17 -0
  100. brain/migrations/003_vault_model.sql +41 -0
  101. brain/migrations/004_relax_content_hash_uniqueness.sql +18 -0
  102. brain/migrations/005_derived_links.sql +67 -0
  103. brain/migrations/006_dedup_file_by_source_path.sql +25 -0
  104. brain/migrations/007_email_thread_and_draft.sql +15 -0
  105. brain/migrations/008_gmail_thread_unique.sql +11 -0
  106. brain/migrations/009_chunks_weighted_tsv.sql +28 -0
  107. brain/migrations/010_interactions.sql +30 -0
  108. brain/migrations/011_documents_summary.sql +23 -0
  109. brain/migrations/012_graphrag.sql +171 -0
  110. brain/migrations/013_graphrag_communities.sql +125 -0
  111. brain/migrations/014_graphrag_community_summary_hash.sql +33 -0
  112. brain/migrations/015_interactions_graph_targets.sql +89 -0
  113. brain/migrations/016_index_hygiene.sql +61 -0
  114. brain/migrations/017_elicit.sql +30 -0
  115. brain/migrations/018_review_gap_signal_kinds.sql +40 -0
  116. brain/migrations/019_search_queries.sql +35 -0
  117. brain/migrations/020_link_suggestions.sql +40 -0
  118. brain/migrations/021_timeline_doc_date.sql +34 -0
  119. brain/migrations/022_link_suggestions_undirected.sql +84 -0
  120. brain/migrations/023_search_queries_fts_count.sql +28 -0
  121. brain/quartz_overrides/__init__.py +8 -0
  122. brain/quartz_overrides/quartz/bootstrap-cli.mjs +65 -0
  123. brain/quartz_overrides/quartz/build.ts +568 -0
  124. brain/quartz_overrides/quartz/cli/args.js +152 -0
  125. brain/quartz_overrides/quartz/cli/build_partial_handler.js +544 -0
  126. brain/quartz_overrides/quartz/cli/handlers.js +636 -0
  127. brain/quartz_overrides/quartz/components/CommandPalette.tsx +172 -0
  128. brain/quartz_overrides/quartz/components/Explorer.tsx +198 -0
  129. brain/quartz_overrides/quartz/components/Footer.tsx +27 -0
  130. brain/quartz_overrides/quartz/components/Graph.tsx +468 -0
  131. brain/quartz_overrides/quartz/components/PageTitle.tsx +72 -0
  132. brain/quartz_overrides/quartz/components/RelatedDocs.tsx +38 -0
  133. brain/quartz_overrides/quartz/components/Search.tsx +161 -0
  134. brain/quartz_overrides/quartz/components/SummaryLede.tsx +72 -0
  135. brain/quartz_overrides/quartz/components/index.ts +92 -0
  136. brain/quartz_overrides/quartz/components/pages/TagContent.tsx +272 -0
  137. brain/quartz_overrides/quartz/components/scripts/commandPalette.inline.ts +665 -0
  138. brain/quartz_overrides/quartz/components/scripts/explorer.inline.ts +768 -0
  139. brain/quartz_overrides/quartz/components/scripts/graph.inline.ts +2302 -0
  140. brain/quartz_overrides/quartz/components/scripts/relatedDocs.inline.ts +163 -0
  141. brain/quartz_overrides/quartz/components/scripts/search.inline.ts +1011 -0
  142. brain/quartz_overrides/quartz/plugins/emitters/contentIndex.ts +546 -0
  143. brain/quartz_overrides/quartz/plugins/transformers/codeCopy.ts +94 -0
  144. brain/quartz_overrides/quartz/plugins/transformers/derivedFenceMark.ts +302 -0
  145. brain/quartz_overrides/quartz/plugins/transformers/emailThread.ts +148 -0
  146. brain/quartz_overrides/quartz/plugins/transformers/emptyDoorFilter.ts +213 -0
  147. brain/quartz_overrides/quartz/plugins/transformers/index.ts +114 -0
  148. brain/quartz_overrides/quartz/plugins/transformers/linkKindMark.ts +205 -0
  149. brain/quartz_overrides/quartz/plugins/transformers/linkSourceTag.ts +104 -0
  150. brain/quartz_overrides/quartz/plugins/transformers/relativeDate.ts +100 -0
  151. brain/quartz_overrides/quartz/plugins/transformers/reloadSignal.ts +131 -0
  152. brain/quartz_overrides/quartz/processors/parse.ts +371 -0
  153. brain/quartz_overrides/quartz/processors/parser_cache.ts +78 -0
  154. brain/quartz_overrides/quartz/static/brain-logo-dark.png +0 -0
  155. brain/quartz_overrides/quartz/static/brain-logo-light.png +0 -0
  156. brain/quartz_overrides/quartz/static/codeCopy.js +196 -0
  157. brain/quartz_overrides/quartz/static/emailThread.js +334 -0
  158. brain/quartz_overrides/quartz/static/favicon.ico +0 -0
  159. brain/quartz_overrides/quartz/static/icon.png +0 -0
  160. brain/quartz_overrides/quartz/static/linkSourceTag.js +104 -0
  161. brain/quartz_overrides/quartz/static/relativeDate.js +142 -0
  162. brain/quartz_overrides/quartz/static/reload.js +168 -0
  163. brain/quartz_overrides/quartz/styles/brain/_article.scss +252 -0
  164. brain/quartz_overrides/quartz/styles/brain/_atmosphere.scss +113 -0
  165. brain/quartz_overrides/quartz/styles/brain/_callouts.scss +180 -0
  166. brain/quartz_overrides/quartz/styles/brain/_cmdk.scss +7 -0
  167. brain/quartz_overrides/quartz/styles/brain/_code.scss +208 -0
  168. brain/quartz_overrides/quartz/styles/brain/_command_palette.scss +369 -0
  169. brain/quartz_overrides/quartz/styles/brain/_email_thread.scss +228 -0
  170. brain/quartz_overrides/quartz/styles/brain/_explorer.scss +142 -0
  171. brain/quartz_overrides/quartz/styles/brain/_home.scss +182 -0
  172. brain/quartz_overrides/quartz/styles/brain/_links.scss +322 -0
  173. brain/quartz_overrides/quartz/styles/brain/_marginalia.scss +117 -0
  174. brain/quartz_overrides/quartz/styles/brain/_motion.scss +175 -0
  175. brain/quartz_overrides/quartz/styles/brain/_people_hub.scss +100 -0
  176. brain/quartz_overrides/quartz/styles/brain/_related_docs.scss +137 -0
  177. brain/quartz_overrides/quartz/styles/brain/_search.scss +252 -0
  178. brain/quartz_overrides/quartz/styles/brain/_sidebar.scss +468 -0
  179. brain/quartz_overrides/quartz/styles/brain/_summary_lede.scss +56 -0
  180. brain/quartz_overrides/quartz/styles/brain/_surface.scss +43 -0
  181. brain/quartz_overrides/quartz/styles/brain/_tag_content.scss +118 -0
  182. brain/quartz_overrides/quartz/styles/brain/_tokens.scss +197 -0
  183. brain/quartz_overrides/quartz/styles/brain/_typography.scss +92 -0
  184. brain/quartz_overrides/quartz/styles/custom.scss +89 -0
  185. brain/quartz_overrides/quartz/styles/graph.scss +505 -0
  186. brain/quartz_overrides/quartz/util/ctx.ts +92 -0
  187. brain/quartz_overrides/quartz/util/fastpath_manifest.ts +608 -0
  188. brain/quartz_overrides/quartz/util/path.ts +358 -0
  189. brain/quartz_overrides/quartz/util/sourceIcons.ts +55 -0
  190. brain/quartz_overrides/quartz.config.ts +270 -0
  191. brain/quartz_overrides/quartz.layout.ts +314 -0
  192. brain/queries.py +1188 -0
  193. brain/rank_fusion.py +8 -0
  194. brain/resurface.py +210 -0
  195. brain/review/__init__.py +26 -0
  196. brain/review/emit.py +27 -0
  197. brain/review/queries.py +436 -0
  198. brain/review/render.py +196 -0
  199. brain/review/scans.py +355 -0
  200. brain/review/weekly.py +413 -0
  201. brain/search.py +704 -0
  202. brain/set_similarity.py +15 -0
  203. brain/setup.py +1205 -0
  204. brain/tags.py +56 -0
  205. brain/templates/Caddyfile.j2 +9 -0
  206. brain/templates/__init__.py +1 -0
  207. brain/templates/bin/__init__.py +1 -0
  208. brain/templates/bin/_brain-brief-fg.sh +25 -0
  209. brain/templates/bin/_brain-build-fg.sh +53 -0
  210. brain/templates/bin/_brain-watcher-fg.sh +65 -0
  211. brain/templates/bin/brain-down.sh +89 -0
  212. brain/templates/bin/brain-status.sh +83 -0
  213. brain/templates/bin/brain-up.sh +221 -0
  214. brain/templates/docker/age/Dockerfile +79 -0
  215. brain/templates/docker-compose.stock.yml.j2 +26 -0
  216. brain/templates/docker-compose.yml.j2 +34 -0
  217. brain/templates/env.example +190 -0
  218. brain/templates/launchd/__init__.py +1 -0
  219. brain/templates/launchd/com.brain.brief.plist.j2 +45 -0
  220. brain/templates/launchd/com.brain.build.plist.j2 +46 -0
  221. brain/templates/launchd/com.brain.watcher.plist.j2 +46 -0
  222. brain/templates/skill/SKILL.md +63 -0
  223. brain/templates/skill/__init__.py +1 -0
  224. brain/timeline.py +834 -0
  225. brain/todo.py +124 -0
  226. brain/uninstall.py +185 -0
  227. brain/vault/__init__.py +115 -0
  228. brain/vault/_atomic.py +25 -0
  229. brain/vault/daily_index.py +228 -0
  230. brain/vault/derived_links/__init__.py +50 -0
  231. brain/vault/derived_links/directory.py +683 -0
  232. brain/vault/derived_links/fence.py +408 -0
  233. brain/vault/derived_links/gws.py +64 -0
  234. brain/vault/derived_links/participants.py +143 -0
  235. brain/vault/derived_links/pass_runner.py +362 -0
  236. brain/vault/derived_links/rules.py +137 -0
  237. brain/vault/export.py +683 -0
  238. brain/vault/frontmatter.py +165 -0
  239. brain/vault/graph.py +620 -0
  240. brain/vault/graph_format.py +388 -0
  241. brain/vault/link_rewrite.py +235 -0
  242. brain/vault/links.py +260 -0
  243. brain/vault/note_builder.py +211 -0
  244. brain/vault/paths.py +55 -0
  245. brain/vault/quartz_overlay.py +236 -0
  246. brain/vault/rename.py +591 -0
  247. brain/vault/resolver.py +304 -0
  248. brain/vault/slug.py +127 -0
  249. brain/vault/sync.py +1513 -0
  250. brain/vault/sync_summaries.py +264 -0
  251. brain/vault/templates.py +145 -0
  252. brain/vault/watch.py +1052 -0
  253. brain/wiki/__init__.py +6 -0
  254. brain/wiki/_github_slugger.py +76 -0
  255. brain/wiki/_person_name.py +314 -0
  256. brain/wiki/build_homepage.py +541 -0
  257. brain/wiki/build_partial.py +273 -0
  258. brain/wiki/build_people.py +934 -0
  259. brain/wiki/build_related.py +758 -0
  260. brain/wiki/build_swap.py +585 -0
  261. brain/wiki/build_watcher.py +975 -0
  262. brain/wiki/edit_classifier.py +215 -0
  263. brain/wiki/errors.py +10 -0
  264. brain/wiki/fastpath_manifest.py +475 -0
  265. brain/wiki/fastpath_state.py +174 -0
  266. brain/wiki/install.py +296 -0
  267. brain/wiki/slug.py +111 -0
  268. secondbrain_py-0.2.1.dist-info/METADATA +195 -0
  269. secondbrain_py-0.2.1.dist-info/RECORD +273 -0
  270. secondbrain_py-0.2.1.dist-info/WHEEL +5 -0
  271. secondbrain_py-0.2.1.dist-info/entry_points.txt +11 -0
  272. secondbrain_py-0.2.1.dist-info/licenses/LICENSE +21 -0
  273. secondbrain_py-0.2.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,249 @@
1
+ """Paragraph-aware text chunker with token budget and overlap."""
2
+ import logging
3
+ import re
4
+ from collections.abc import Callable
5
+ from dataclasses import dataclass
6
+
7
+ logger = logging.getLogger(__name__)
8
+
9
+
10
+ @dataclass(frozen=True)
11
+ class Chunk:
12
+ """A single chunk of text produced by :func:`chunk_text`."""
13
+
14
+ index: int
15
+ content: str
16
+
17
+
18
+ _PARAGRAPH_SPLIT = re.compile(r"\n\s*\n")
19
+ _SENTENCE_SPLIT = re.compile(r"(?<=[.!?])\s+")
20
+ _LINE_SPLIT = re.compile(r"\n")
21
+ _WHITESPACE_SPLIT = re.compile(r"\s+")
22
+
23
+
24
+ def chunk_text(
25
+ text: str,
26
+ *,
27
+ target_tokens: int = 600,
28
+ overlap_tokens: int = 100,
29
+ count_tokens: Callable[[str], int],
30
+ ) -> list[Chunk]:
31
+ """Split text into paragraph-aware chunks under a token budget.
32
+
33
+ Strategy:
34
+ 1. Split on blank lines (paragraphs).
35
+ 2. Greedily pack paragraphs into a chunk until the next would exceed
36
+ ``target_tokens``.
37
+ 3. If a single paragraph exceeds ``target_tokens``, split it via a
38
+ fallback chain: sentence terminators → single newlines → whitespace
39
+ → characters.
40
+ 4. Add ``overlap_tokens`` worth of trailing content from chunk N onto
41
+ chunk N+1, capped so no chunk exceeds ``target_tokens + overlap_tokens``.
42
+
43
+ Every emitted chunk is guaranteed to satisfy
44
+ ``count_tokens(content) <= target_tokens + overlap_tokens``.
45
+ """
46
+ text = text.strip()
47
+ if not text:
48
+ return []
49
+
50
+ ceiling = target_tokens + overlap_tokens
51
+ paragraphs = [p.strip() for p in _PARAGRAPH_SPLIT.split(text) if p.strip()]
52
+ units: list[str] = []
53
+ for para in paragraphs:
54
+ if count_tokens(para) <= target_tokens:
55
+ units.append(para)
56
+ else:
57
+ units.extend(_split_long_paragraph(para, target_tokens, count_tokens))
58
+
59
+ chunks_text: list[str] = []
60
+ current: list[str] = []
61
+ current_tokens = 0
62
+ for unit in units:
63
+ unit_tokens = count_tokens(unit)
64
+ if current and current_tokens + unit_tokens > target_tokens:
65
+ chunks_text.append("\n\n".join(current))
66
+ current = []
67
+ current_tokens = 0
68
+ current.append(unit)
69
+ current_tokens += unit_tokens
70
+ if current:
71
+ chunks_text.append("\n\n".join(current))
72
+
73
+ if overlap_tokens > 0 and len(chunks_text) > 1:
74
+ chunks_text = _add_overlap(
75
+ chunks_text, overlap_tokens, count_tokens, ceiling=ceiling
76
+ )
77
+
78
+ # Defensive backstop: any chunk somehow over the ceiling gets hard-split.
79
+ # The fallback chain above should make this branch unreachable.
80
+ final: list[str] = []
81
+ for c in chunks_text:
82
+ if count_tokens(c) <= ceiling:
83
+ final.append(c)
84
+ else: # pragma: no cover - defensive
85
+ logger.warning(
86
+ "chunker backstop fired: chunk had %d tokens, ceiling=%d",
87
+ count_tokens(c),
88
+ ceiling,
89
+ )
90
+ final.extend(_split_long_paragraph(c, target_tokens, count_tokens))
91
+
92
+ return [Chunk(index=i, content=c) for i, c in enumerate(final)]
93
+
94
+
95
+ def _split_long_paragraph(
96
+ para: str, target_tokens: int, count_tokens: Callable[[str], int]
97
+ ) -> list[str]:
98
+ """Split an oversized paragraph into pieces each ``<= target_tokens``.
99
+
100
+ Cascades through progressively finer separators; each step only fires when
101
+ the previous step left a piece over budget, so well-formed prose flows
102
+ through the sentence-only fast path unchanged.
103
+ """
104
+ pieces = _pack_split(para, _SENTENCE_SPLIT, target_tokens, count_tokens, joiner=" ")
105
+ pieces = _refine(pieces, _LINE_SPLIT, target_tokens, count_tokens, joiner="\n")
106
+ pieces = _refine(pieces, _WHITESPACE_SPLIT, target_tokens, count_tokens, joiner=" ")
107
+ out: list[str] = []
108
+ for piece in pieces:
109
+ if count_tokens(piece) <= target_tokens:
110
+ out.append(piece)
111
+ else:
112
+ out.extend(_split_by_chars(piece, target_tokens, count_tokens))
113
+ return out
114
+
115
+
116
+ def _pack_split(
117
+ text: str,
118
+ pattern: re.Pattern[str],
119
+ target_tokens: int,
120
+ count_tokens: Callable[[str], int],
121
+ *,
122
+ joiner: str,
123
+ ) -> list[str]:
124
+ """Split ``text`` by ``pattern``, then greedily pack parts into pieces.
125
+
126
+ Each emitted piece tries to stay ``<= target_tokens``. A part that is
127
+ individually larger than ``target_tokens`` is emitted alone; the caller is
128
+ expected to refine it with a finer split.
129
+ """
130
+ parts = [p.strip() for p in pattern.split(text) if p.strip()]
131
+ pieces: list[str] = []
132
+ current: list[str] = []
133
+ current_tokens = 0
134
+ for part in parts:
135
+ part_tokens = count_tokens(part)
136
+ if current and current_tokens + part_tokens > target_tokens:
137
+ pieces.append(joiner.join(current))
138
+ current = []
139
+ current_tokens = 0
140
+ current.append(part)
141
+ current_tokens += part_tokens
142
+ if current:
143
+ pieces.append(joiner.join(current))
144
+ return pieces
145
+
146
+
147
+ def _refine(
148
+ pieces: list[str],
149
+ pattern: re.Pattern[str],
150
+ target_tokens: int,
151
+ count_tokens: Callable[[str], int],
152
+ *,
153
+ joiner: str,
154
+ ) -> list[str]:
155
+ """Re-split any piece that is still over budget using a finer pattern."""
156
+ out: list[str] = []
157
+ for piece in pieces:
158
+ if count_tokens(piece) <= target_tokens:
159
+ out.append(piece)
160
+ else:
161
+ out.extend(
162
+ _pack_split(piece, pattern, target_tokens, count_tokens, joiner=joiner)
163
+ )
164
+ return out
165
+
166
+
167
+ def _split_by_chars(
168
+ text: str, target_tokens: int, count_tokens: Callable[[str], int]
169
+ ) -> list[str]:
170
+ """Last-resort: split a single whitespace-free blob by character count.
171
+
172
+ Used when none of sentence/newline/whitespace splitting reduced a piece
173
+ below ``target_tokens`` — typical for base64 payloads or minified JSON.
174
+ Callers only invoke this when the input is already over budget.
175
+ """
176
+ total_tokens = count_tokens(text)
177
+ avg_chars = max(1, len(text) // total_tokens)
178
+ char_budget = max(1, int(target_tokens * avg_chars * 0.9))
179
+ pieces: list[str] = []
180
+ i = 0
181
+ n = len(text)
182
+ while i < n:
183
+ end = min(n, i + char_budget)
184
+ piece = text[i:end]
185
+ # Token estimate may overshoot; iteratively shrink until under budget.
186
+ while count_tokens(piece) > target_tokens and end - i > 1:
187
+ shrink = max(1, (end - i) // 8)
188
+ end -= shrink
189
+ piece = text[i:end]
190
+ pieces.append(piece)
191
+ i = end
192
+ return pieces
193
+
194
+
195
+ def _add_overlap(
196
+ chunks: list[str],
197
+ overlap_tokens: int,
198
+ count_tokens: Callable[[str], int],
199
+ *,
200
+ ceiling: int,
201
+ ) -> list[str]:
202
+ """Prepend a tail of the previous chunk onto each subsequent chunk.
203
+
204
+ The prepended tail is bounded so that
205
+ ``count_tokens(combined) <= ceiling``. If the chunk is already at the
206
+ ceiling, no overlap is added.
207
+ """
208
+ out = [chunks[0]]
209
+ for prev, cur in zip(chunks, chunks[1:], strict=False):
210
+ cur_tokens = count_tokens(cur)
211
+ room = ceiling - cur_tokens
212
+ if room <= 0: # pragma: no cover - defensive
213
+ out.append(cur)
214
+ continue
215
+ budget = min(overlap_tokens, room)
216
+ tail = _take_tail_tokens(prev, budget, count_tokens)
217
+ if not tail: # pragma: no cover - defensive
218
+ out.append(cur)
219
+ continue
220
+ combined = tail + "\n\n" + cur
221
+ if count_tokens(combined) > ceiling: # pragma: no cover - defensive
222
+ # Tail estimate overshot; drop overlap rather than violate ceiling.
223
+ out.append(cur)
224
+ else:
225
+ out.append(combined)
226
+ return out
227
+
228
+
229
+ def _take_tail_tokens(
230
+ text: str, n_tokens: int, count_tokens: Callable[[str], int]
231
+ ) -> str:
232
+ """Return at most ~``n_tokens`` worth of trailing content.
233
+
234
+ Paragraph-aligned where possible; falls back to hard-splitting the last
235
+ paragraph if it alone is larger than ``n_tokens``.
236
+ """
237
+ paragraphs = [p.strip() for p in _PARAGRAPH_SPLIT.split(text) if p.strip()]
238
+ selected: list[str] = []
239
+ total = 0
240
+ for para in reversed(paragraphs):
241
+ para_tokens = count_tokens(para)
242
+ if selected and total + para_tokens > n_tokens:
243
+ break
244
+ if not selected and para_tokens > n_tokens:
245
+ atoms = _split_long_paragraph(para, n_tokens, count_tokens)
246
+ return atoms[-1] if atoms else ""
247
+ selected.insert(0, para)
248
+ total += para_tokens
249
+ return "\n\n".join(selected)
brain/ingest/docx.py ADDED
@@ -0,0 +1,40 @@
1
+ """DOCX extractor — paragraphs + tables."""
2
+ from pathlib import Path
3
+
4
+ from docx import Document
5
+
6
+ from . import ExtractedDoc
7
+
8
+
9
+ def extract_docx(path: Path) -> ExtractedDoc:
10
+ """Extract an :class:`ExtractedDoc` from a ``.docx`` file on disk.
11
+
12
+ Paragraphs and table cell text are collected in document order.
13
+ The title is taken from the first Heading-styled paragraph; if none is
14
+ present, falls back to the file stem.
15
+ """
16
+ d = Document(str(path))
17
+ parts: list[str] = []
18
+ title: str | None = None
19
+
20
+ for para in d.paragraphs:
21
+ text = para.text.strip()
22
+ if not text:
23
+ continue
24
+ style_name = (para.style.name or "") if para.style is not None else ""
25
+ if title is None and style_name.startswith("Heading"):
26
+ title = text
27
+ parts.append(text)
28
+
29
+ for table in d.tables:
30
+ for row in table.rows:
31
+ cells = [cell.text.strip() for cell in row.cells]
32
+ parts.append("\t".join(cells))
33
+
34
+ return ExtractedDoc(
35
+ title=title or Path(path).stem,
36
+ content="\n\n".join(parts),
37
+ content_type="docx",
38
+ source_path=str(Path(path).resolve()),
39
+ metadata={},
40
+ )