secondbrain-py 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- brain/__init__.py +0 -0
- brain/__main__.py +18 -0
- brain/_capture_command.py +445 -0
- brain/_compose.py +52 -0
- brain/activity.py +206 -0
- brain/ask.py +631 -0
- brain/audio.py +591 -0
- brain/backfill/__init__.py +12 -0
- brain/backfill/search_extras.py +141 -0
- brain/backfill/source_rows.py +101 -0
- brain/bin/__init__.py +1 -0
- brain/bin/_launcher.py +107 -0
- brain/bin/down.py +8 -0
- brain/bin/launchd.py +268 -0
- brain/bin/monitor.py +570 -0
- brain/bin/rebuild.py +8 -0
- brain/bin/status.py +8 -0
- brain/bin/up.py +8 -0
- brain/brief.py +272 -0
- brain/capture.py +49 -0
- brain/chat.py +293 -0
- brain/cli.py +9760 -0
- brain/cli_claude.py +81 -0
- brain/cli_connect.py +285 -0
- brain/cli_demo.py +266 -0
- brain/config.py +1949 -0
- brain/connect.py +925 -0
- brain/db.py +540 -0
- brain/demo/__init__.py +452 -0
- brain/demo/corpus/manifest.json +403 -0
- brain/demo/embedder.py +74 -0
- brain/durations.py +84 -0
- brain/edit_session.py +156 -0
- brain/editor.py +67 -0
- brain/elicit/__init__.py +16 -0
- brain/elicit/detectors.py +250 -0
- brain/elicit/drafter.py +70 -0
- brain/elicit/queue.py +220 -0
- brain/elicit/schema.py +48 -0
- brain/elicit/session.py +445 -0
- brain/embedding_targets.py +54 -0
- brain/embeddings.py +424 -0
- brain/enrichment.py +808 -0
- brain/errors.py +357 -0
- brain/eval/__init__.py +129 -0
- brain/eval/answer_eval.py +281 -0
- brain/eval/baseline.py +265 -0
- brain/eval/concept_extraction.py +378 -0
- brain/eval/corpus.py +152 -0
- brain/eval/errors.py +19 -0
- brain/eval/graph_baseline.py +226 -0
- brain/eval/graph_retrieval.py +202 -0
- brain/eval/graph_runner.py +319 -0
- brain/eval/metrics.py +101 -0
- brain/eval/runner.py +223 -0
- brain/format.py +783 -0
- brain/gaps.py +390 -0
- brain/graph_rag/__init__.py +94 -0
- brain/graph_rag/_retrieval_common.py +113 -0
- brain/graph_rag/aggregates.py +303 -0
- brain/graph_rag/aliases/__init__.py +583 -0
- brain/graph_rag/backends/__init__.py +10 -0
- brain/graph_rag/backends/_age_helpers.py +473 -0
- brain/graph_rag/backends/age.py +782 -0
- brain/graph_rag/backends/base.py +272 -0
- brain/graph_rag/build.py +344 -0
- brain/graph_rag/communities.py +644 -0
- brain/graph_rag/communities_summary.py +437 -0
- brain/graph_rag/concepts.py +202 -0
- brain/graph_rag/cooccur.py +193 -0
- brain/graph_rag/cross_type.py +312 -0
- brain/graph_rag/extract.py +885 -0
- brain/graph_rag/fuse.py +371 -0
- brain/graph_rag/global_.py +412 -0
- brain/graph_rag/grouping.py +372 -0
- brain/graph_rag/person_resolver.py +167 -0
- brain/graph_rag/reconcile.py +792 -0
- brain/graph_rag/relational.py +353 -0
- brain/graph_rag/retrieve.py +526 -0
- brain/graph_rag/router.py +288 -0
- brain/graph_rag/schema.py +320 -0
- brain/graph_rag/sync.py +237 -0
- brain/graph_rag/tenancy.py +43 -0
- brain/graph_rag/themes.py +501 -0
- brain/graph_rag/weighting.py +202 -0
- brain/ingest/__init__.py +1926 -0
- brain/ingest/chunker.py +249 -0
- brain/ingest/docx.py +40 -0
- brain/ingest/gmail.py +621 -0
- brain/ingest/markdown.py +37 -0
- brain/ingest/pdf.py +61 -0
- brain/ingest/stdin.py +22 -0
- brain/ingest/sub_tokens.py +91 -0
- brain/ingest/text.py +16 -0
- brain/interactions.py +205 -0
- brain/maintenance.py +355 -0
- brain/mcp_server.py +3405 -0
- brain/migrations/001_init.sql +43 -0
- brain/migrations/002_qwen3_embedding.sql +17 -0
- brain/migrations/003_vault_model.sql +41 -0
- brain/migrations/004_relax_content_hash_uniqueness.sql +18 -0
- brain/migrations/005_derived_links.sql +67 -0
- brain/migrations/006_dedup_file_by_source_path.sql +25 -0
- brain/migrations/007_email_thread_and_draft.sql +15 -0
- brain/migrations/008_gmail_thread_unique.sql +11 -0
- brain/migrations/009_chunks_weighted_tsv.sql +28 -0
- brain/migrations/010_interactions.sql +30 -0
- brain/migrations/011_documents_summary.sql +23 -0
- brain/migrations/012_graphrag.sql +171 -0
- brain/migrations/013_graphrag_communities.sql +125 -0
- brain/migrations/014_graphrag_community_summary_hash.sql +33 -0
- brain/migrations/015_interactions_graph_targets.sql +89 -0
- brain/migrations/016_index_hygiene.sql +61 -0
- brain/migrations/017_elicit.sql +30 -0
- brain/migrations/018_review_gap_signal_kinds.sql +40 -0
- brain/migrations/019_search_queries.sql +35 -0
- brain/migrations/020_link_suggestions.sql +40 -0
- brain/migrations/021_timeline_doc_date.sql +34 -0
- brain/migrations/022_link_suggestions_undirected.sql +84 -0
- brain/migrations/023_search_queries_fts_count.sql +28 -0
- brain/quartz_overrides/__init__.py +8 -0
- brain/quartz_overrides/quartz/bootstrap-cli.mjs +65 -0
- brain/quartz_overrides/quartz/build.ts +568 -0
- brain/quartz_overrides/quartz/cli/args.js +152 -0
- brain/quartz_overrides/quartz/cli/build_partial_handler.js +544 -0
- brain/quartz_overrides/quartz/cli/handlers.js +636 -0
- brain/quartz_overrides/quartz/components/CommandPalette.tsx +172 -0
- brain/quartz_overrides/quartz/components/Explorer.tsx +198 -0
- brain/quartz_overrides/quartz/components/Footer.tsx +27 -0
- brain/quartz_overrides/quartz/components/Graph.tsx +468 -0
- brain/quartz_overrides/quartz/components/PageTitle.tsx +72 -0
- brain/quartz_overrides/quartz/components/RelatedDocs.tsx +38 -0
- brain/quartz_overrides/quartz/components/Search.tsx +161 -0
- brain/quartz_overrides/quartz/components/SummaryLede.tsx +72 -0
- brain/quartz_overrides/quartz/components/index.ts +92 -0
- brain/quartz_overrides/quartz/components/pages/TagContent.tsx +272 -0
- brain/quartz_overrides/quartz/components/scripts/commandPalette.inline.ts +665 -0
- brain/quartz_overrides/quartz/components/scripts/explorer.inline.ts +768 -0
- brain/quartz_overrides/quartz/components/scripts/graph.inline.ts +2302 -0
- brain/quartz_overrides/quartz/components/scripts/relatedDocs.inline.ts +163 -0
- brain/quartz_overrides/quartz/components/scripts/search.inline.ts +1011 -0
- brain/quartz_overrides/quartz/plugins/emitters/contentIndex.ts +546 -0
- brain/quartz_overrides/quartz/plugins/transformers/codeCopy.ts +94 -0
- brain/quartz_overrides/quartz/plugins/transformers/derivedFenceMark.ts +302 -0
- brain/quartz_overrides/quartz/plugins/transformers/emailThread.ts +148 -0
- brain/quartz_overrides/quartz/plugins/transformers/emptyDoorFilter.ts +213 -0
- brain/quartz_overrides/quartz/plugins/transformers/index.ts +114 -0
- brain/quartz_overrides/quartz/plugins/transformers/linkKindMark.ts +205 -0
- brain/quartz_overrides/quartz/plugins/transformers/linkSourceTag.ts +104 -0
- brain/quartz_overrides/quartz/plugins/transformers/relativeDate.ts +100 -0
- brain/quartz_overrides/quartz/plugins/transformers/reloadSignal.ts +131 -0
- brain/quartz_overrides/quartz/processors/parse.ts +371 -0
- brain/quartz_overrides/quartz/processors/parser_cache.ts +78 -0
- brain/quartz_overrides/quartz/static/brain-logo-dark.png +0 -0
- brain/quartz_overrides/quartz/static/brain-logo-light.png +0 -0
- brain/quartz_overrides/quartz/static/codeCopy.js +196 -0
- brain/quartz_overrides/quartz/static/emailThread.js +334 -0
- brain/quartz_overrides/quartz/static/favicon.ico +0 -0
- brain/quartz_overrides/quartz/static/icon.png +0 -0
- brain/quartz_overrides/quartz/static/linkSourceTag.js +104 -0
- brain/quartz_overrides/quartz/static/relativeDate.js +142 -0
- brain/quartz_overrides/quartz/static/reload.js +168 -0
- brain/quartz_overrides/quartz/styles/brain/_article.scss +252 -0
- brain/quartz_overrides/quartz/styles/brain/_atmosphere.scss +113 -0
- brain/quartz_overrides/quartz/styles/brain/_callouts.scss +180 -0
- brain/quartz_overrides/quartz/styles/brain/_cmdk.scss +7 -0
- brain/quartz_overrides/quartz/styles/brain/_code.scss +208 -0
- brain/quartz_overrides/quartz/styles/brain/_command_palette.scss +369 -0
- brain/quartz_overrides/quartz/styles/brain/_email_thread.scss +228 -0
- brain/quartz_overrides/quartz/styles/brain/_explorer.scss +142 -0
- brain/quartz_overrides/quartz/styles/brain/_home.scss +182 -0
- brain/quartz_overrides/quartz/styles/brain/_links.scss +322 -0
- brain/quartz_overrides/quartz/styles/brain/_marginalia.scss +117 -0
- brain/quartz_overrides/quartz/styles/brain/_motion.scss +175 -0
- brain/quartz_overrides/quartz/styles/brain/_people_hub.scss +100 -0
- brain/quartz_overrides/quartz/styles/brain/_related_docs.scss +137 -0
- brain/quartz_overrides/quartz/styles/brain/_search.scss +252 -0
- brain/quartz_overrides/quartz/styles/brain/_sidebar.scss +468 -0
- brain/quartz_overrides/quartz/styles/brain/_summary_lede.scss +56 -0
- brain/quartz_overrides/quartz/styles/brain/_surface.scss +43 -0
- brain/quartz_overrides/quartz/styles/brain/_tag_content.scss +118 -0
- brain/quartz_overrides/quartz/styles/brain/_tokens.scss +197 -0
- brain/quartz_overrides/quartz/styles/brain/_typography.scss +92 -0
- brain/quartz_overrides/quartz/styles/custom.scss +89 -0
- brain/quartz_overrides/quartz/styles/graph.scss +505 -0
- brain/quartz_overrides/quartz/util/ctx.ts +92 -0
- brain/quartz_overrides/quartz/util/fastpath_manifest.ts +608 -0
- brain/quartz_overrides/quartz/util/path.ts +358 -0
- brain/quartz_overrides/quartz/util/sourceIcons.ts +55 -0
- brain/quartz_overrides/quartz.config.ts +270 -0
- brain/quartz_overrides/quartz.layout.ts +314 -0
- brain/queries.py +1188 -0
- brain/rank_fusion.py +8 -0
- brain/resurface.py +210 -0
- brain/review/__init__.py +26 -0
- brain/review/emit.py +27 -0
- brain/review/queries.py +436 -0
- brain/review/render.py +196 -0
- brain/review/scans.py +355 -0
- brain/review/weekly.py +413 -0
- brain/search.py +704 -0
- brain/set_similarity.py +15 -0
- brain/setup.py +1205 -0
- brain/tags.py +56 -0
- brain/templates/Caddyfile.j2 +9 -0
- brain/templates/__init__.py +1 -0
- brain/templates/bin/__init__.py +1 -0
- brain/templates/bin/_brain-brief-fg.sh +25 -0
- brain/templates/bin/_brain-build-fg.sh +53 -0
- brain/templates/bin/_brain-watcher-fg.sh +65 -0
- brain/templates/bin/brain-down.sh +89 -0
- brain/templates/bin/brain-status.sh +83 -0
- brain/templates/bin/brain-up.sh +221 -0
- brain/templates/docker/age/Dockerfile +79 -0
- brain/templates/docker-compose.stock.yml.j2 +26 -0
- brain/templates/docker-compose.yml.j2 +34 -0
- brain/templates/env.example +190 -0
- brain/templates/launchd/__init__.py +1 -0
- brain/templates/launchd/com.brain.brief.plist.j2 +45 -0
- brain/templates/launchd/com.brain.build.plist.j2 +46 -0
- brain/templates/launchd/com.brain.watcher.plist.j2 +46 -0
- brain/templates/skill/SKILL.md +63 -0
- brain/templates/skill/__init__.py +1 -0
- brain/timeline.py +834 -0
- brain/todo.py +124 -0
- brain/uninstall.py +185 -0
- brain/vault/__init__.py +115 -0
- brain/vault/_atomic.py +25 -0
- brain/vault/daily_index.py +228 -0
- brain/vault/derived_links/__init__.py +50 -0
- brain/vault/derived_links/directory.py +683 -0
- brain/vault/derived_links/fence.py +408 -0
- brain/vault/derived_links/gws.py +64 -0
- brain/vault/derived_links/participants.py +143 -0
- brain/vault/derived_links/pass_runner.py +362 -0
- brain/vault/derived_links/rules.py +137 -0
- brain/vault/export.py +683 -0
- brain/vault/frontmatter.py +165 -0
- brain/vault/graph.py +620 -0
- brain/vault/graph_format.py +388 -0
- brain/vault/link_rewrite.py +235 -0
- brain/vault/links.py +260 -0
- brain/vault/note_builder.py +211 -0
- brain/vault/paths.py +55 -0
- brain/vault/quartz_overlay.py +236 -0
- brain/vault/rename.py +591 -0
- brain/vault/resolver.py +304 -0
- brain/vault/slug.py +127 -0
- brain/vault/sync.py +1513 -0
- brain/vault/sync_summaries.py +264 -0
- brain/vault/templates.py +145 -0
- brain/vault/watch.py +1052 -0
- brain/wiki/__init__.py +6 -0
- brain/wiki/_github_slugger.py +76 -0
- brain/wiki/_person_name.py +314 -0
- brain/wiki/build_homepage.py +541 -0
- brain/wiki/build_partial.py +273 -0
- brain/wiki/build_people.py +934 -0
- brain/wiki/build_related.py +758 -0
- brain/wiki/build_swap.py +585 -0
- brain/wiki/build_watcher.py +975 -0
- brain/wiki/edit_classifier.py +215 -0
- brain/wiki/errors.py +10 -0
- brain/wiki/fastpath_manifest.py +475 -0
- brain/wiki/fastpath_state.py +174 -0
- brain/wiki/install.py +296 -0
- brain/wiki/slug.py +111 -0
- secondbrain_py-0.2.1.dist-info/METADATA +195 -0
- secondbrain_py-0.2.1.dist-info/RECORD +273 -0
- secondbrain_py-0.2.1.dist-info/WHEEL +5 -0
- secondbrain_py-0.2.1.dist-info/entry_points.txt +11 -0
- secondbrain_py-0.2.1.dist-info/licenses/LICENSE +21 -0
- secondbrain_py-0.2.1.dist-info/top_level.txt +1 -0
brain/ingest/chunker.py
ADDED
|
@@ -0,0 +1,249 @@
|
|
|
1
|
+
"""Paragraph-aware text chunker with token budget and overlap."""
|
|
2
|
+
import logging
|
|
3
|
+
import re
|
|
4
|
+
from collections.abc import Callable
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
|
|
7
|
+
logger = logging.getLogger(__name__)
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
@dataclass(frozen=True)
|
|
11
|
+
class Chunk:
|
|
12
|
+
"""A single chunk of text produced by :func:`chunk_text`."""
|
|
13
|
+
|
|
14
|
+
index: int
|
|
15
|
+
content: str
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
_PARAGRAPH_SPLIT = re.compile(r"\n\s*\n")
|
|
19
|
+
_SENTENCE_SPLIT = re.compile(r"(?<=[.!?])\s+")
|
|
20
|
+
_LINE_SPLIT = re.compile(r"\n")
|
|
21
|
+
_WHITESPACE_SPLIT = re.compile(r"\s+")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def chunk_text(
|
|
25
|
+
text: str,
|
|
26
|
+
*,
|
|
27
|
+
target_tokens: int = 600,
|
|
28
|
+
overlap_tokens: int = 100,
|
|
29
|
+
count_tokens: Callable[[str], int],
|
|
30
|
+
) -> list[Chunk]:
|
|
31
|
+
"""Split text into paragraph-aware chunks under a token budget.
|
|
32
|
+
|
|
33
|
+
Strategy:
|
|
34
|
+
1. Split on blank lines (paragraphs).
|
|
35
|
+
2. Greedily pack paragraphs into a chunk until the next would exceed
|
|
36
|
+
``target_tokens``.
|
|
37
|
+
3. If a single paragraph exceeds ``target_tokens``, split it via a
|
|
38
|
+
fallback chain: sentence terminators → single newlines → whitespace
|
|
39
|
+
→ characters.
|
|
40
|
+
4. Add ``overlap_tokens`` worth of trailing content from chunk N onto
|
|
41
|
+
chunk N+1, capped so no chunk exceeds ``target_tokens + overlap_tokens``.
|
|
42
|
+
|
|
43
|
+
Every emitted chunk is guaranteed to satisfy
|
|
44
|
+
``count_tokens(content) <= target_tokens + overlap_tokens``.
|
|
45
|
+
"""
|
|
46
|
+
text = text.strip()
|
|
47
|
+
if not text:
|
|
48
|
+
return []
|
|
49
|
+
|
|
50
|
+
ceiling = target_tokens + overlap_tokens
|
|
51
|
+
paragraphs = [p.strip() for p in _PARAGRAPH_SPLIT.split(text) if p.strip()]
|
|
52
|
+
units: list[str] = []
|
|
53
|
+
for para in paragraphs:
|
|
54
|
+
if count_tokens(para) <= target_tokens:
|
|
55
|
+
units.append(para)
|
|
56
|
+
else:
|
|
57
|
+
units.extend(_split_long_paragraph(para, target_tokens, count_tokens))
|
|
58
|
+
|
|
59
|
+
chunks_text: list[str] = []
|
|
60
|
+
current: list[str] = []
|
|
61
|
+
current_tokens = 0
|
|
62
|
+
for unit in units:
|
|
63
|
+
unit_tokens = count_tokens(unit)
|
|
64
|
+
if current and current_tokens + unit_tokens > target_tokens:
|
|
65
|
+
chunks_text.append("\n\n".join(current))
|
|
66
|
+
current = []
|
|
67
|
+
current_tokens = 0
|
|
68
|
+
current.append(unit)
|
|
69
|
+
current_tokens += unit_tokens
|
|
70
|
+
if current:
|
|
71
|
+
chunks_text.append("\n\n".join(current))
|
|
72
|
+
|
|
73
|
+
if overlap_tokens > 0 and len(chunks_text) > 1:
|
|
74
|
+
chunks_text = _add_overlap(
|
|
75
|
+
chunks_text, overlap_tokens, count_tokens, ceiling=ceiling
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
# Defensive backstop: any chunk somehow over the ceiling gets hard-split.
|
|
79
|
+
# The fallback chain above should make this branch unreachable.
|
|
80
|
+
final: list[str] = []
|
|
81
|
+
for c in chunks_text:
|
|
82
|
+
if count_tokens(c) <= ceiling:
|
|
83
|
+
final.append(c)
|
|
84
|
+
else: # pragma: no cover - defensive
|
|
85
|
+
logger.warning(
|
|
86
|
+
"chunker backstop fired: chunk had %d tokens, ceiling=%d",
|
|
87
|
+
count_tokens(c),
|
|
88
|
+
ceiling,
|
|
89
|
+
)
|
|
90
|
+
final.extend(_split_long_paragraph(c, target_tokens, count_tokens))
|
|
91
|
+
|
|
92
|
+
return [Chunk(index=i, content=c) for i, c in enumerate(final)]
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _split_long_paragraph(
|
|
96
|
+
para: str, target_tokens: int, count_tokens: Callable[[str], int]
|
|
97
|
+
) -> list[str]:
|
|
98
|
+
"""Split an oversized paragraph into pieces each ``<= target_tokens``.
|
|
99
|
+
|
|
100
|
+
Cascades through progressively finer separators; each step only fires when
|
|
101
|
+
the previous step left a piece over budget, so well-formed prose flows
|
|
102
|
+
through the sentence-only fast path unchanged.
|
|
103
|
+
"""
|
|
104
|
+
pieces = _pack_split(para, _SENTENCE_SPLIT, target_tokens, count_tokens, joiner=" ")
|
|
105
|
+
pieces = _refine(pieces, _LINE_SPLIT, target_tokens, count_tokens, joiner="\n")
|
|
106
|
+
pieces = _refine(pieces, _WHITESPACE_SPLIT, target_tokens, count_tokens, joiner=" ")
|
|
107
|
+
out: list[str] = []
|
|
108
|
+
for piece in pieces:
|
|
109
|
+
if count_tokens(piece) <= target_tokens:
|
|
110
|
+
out.append(piece)
|
|
111
|
+
else:
|
|
112
|
+
out.extend(_split_by_chars(piece, target_tokens, count_tokens))
|
|
113
|
+
return out
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _pack_split(
|
|
117
|
+
text: str,
|
|
118
|
+
pattern: re.Pattern[str],
|
|
119
|
+
target_tokens: int,
|
|
120
|
+
count_tokens: Callable[[str], int],
|
|
121
|
+
*,
|
|
122
|
+
joiner: str,
|
|
123
|
+
) -> list[str]:
|
|
124
|
+
"""Split ``text`` by ``pattern``, then greedily pack parts into pieces.
|
|
125
|
+
|
|
126
|
+
Each emitted piece tries to stay ``<= target_tokens``. A part that is
|
|
127
|
+
individually larger than ``target_tokens`` is emitted alone; the caller is
|
|
128
|
+
expected to refine it with a finer split.
|
|
129
|
+
"""
|
|
130
|
+
parts = [p.strip() for p in pattern.split(text) if p.strip()]
|
|
131
|
+
pieces: list[str] = []
|
|
132
|
+
current: list[str] = []
|
|
133
|
+
current_tokens = 0
|
|
134
|
+
for part in parts:
|
|
135
|
+
part_tokens = count_tokens(part)
|
|
136
|
+
if current and current_tokens + part_tokens > target_tokens:
|
|
137
|
+
pieces.append(joiner.join(current))
|
|
138
|
+
current = []
|
|
139
|
+
current_tokens = 0
|
|
140
|
+
current.append(part)
|
|
141
|
+
current_tokens += part_tokens
|
|
142
|
+
if current:
|
|
143
|
+
pieces.append(joiner.join(current))
|
|
144
|
+
return pieces
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _refine(
|
|
148
|
+
pieces: list[str],
|
|
149
|
+
pattern: re.Pattern[str],
|
|
150
|
+
target_tokens: int,
|
|
151
|
+
count_tokens: Callable[[str], int],
|
|
152
|
+
*,
|
|
153
|
+
joiner: str,
|
|
154
|
+
) -> list[str]:
|
|
155
|
+
"""Re-split any piece that is still over budget using a finer pattern."""
|
|
156
|
+
out: list[str] = []
|
|
157
|
+
for piece in pieces:
|
|
158
|
+
if count_tokens(piece) <= target_tokens:
|
|
159
|
+
out.append(piece)
|
|
160
|
+
else:
|
|
161
|
+
out.extend(
|
|
162
|
+
_pack_split(piece, pattern, target_tokens, count_tokens, joiner=joiner)
|
|
163
|
+
)
|
|
164
|
+
return out
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _split_by_chars(
|
|
168
|
+
text: str, target_tokens: int, count_tokens: Callable[[str], int]
|
|
169
|
+
) -> list[str]:
|
|
170
|
+
"""Last-resort: split a single whitespace-free blob by character count.
|
|
171
|
+
|
|
172
|
+
Used when none of sentence/newline/whitespace splitting reduced a piece
|
|
173
|
+
below ``target_tokens`` — typical for base64 payloads or minified JSON.
|
|
174
|
+
Callers only invoke this when the input is already over budget.
|
|
175
|
+
"""
|
|
176
|
+
total_tokens = count_tokens(text)
|
|
177
|
+
avg_chars = max(1, len(text) // total_tokens)
|
|
178
|
+
char_budget = max(1, int(target_tokens * avg_chars * 0.9))
|
|
179
|
+
pieces: list[str] = []
|
|
180
|
+
i = 0
|
|
181
|
+
n = len(text)
|
|
182
|
+
while i < n:
|
|
183
|
+
end = min(n, i + char_budget)
|
|
184
|
+
piece = text[i:end]
|
|
185
|
+
# Token estimate may overshoot; iteratively shrink until under budget.
|
|
186
|
+
while count_tokens(piece) > target_tokens and end - i > 1:
|
|
187
|
+
shrink = max(1, (end - i) // 8)
|
|
188
|
+
end -= shrink
|
|
189
|
+
piece = text[i:end]
|
|
190
|
+
pieces.append(piece)
|
|
191
|
+
i = end
|
|
192
|
+
return pieces
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def _add_overlap(
|
|
196
|
+
chunks: list[str],
|
|
197
|
+
overlap_tokens: int,
|
|
198
|
+
count_tokens: Callable[[str], int],
|
|
199
|
+
*,
|
|
200
|
+
ceiling: int,
|
|
201
|
+
) -> list[str]:
|
|
202
|
+
"""Prepend a tail of the previous chunk onto each subsequent chunk.
|
|
203
|
+
|
|
204
|
+
The prepended tail is bounded so that
|
|
205
|
+
``count_tokens(combined) <= ceiling``. If the chunk is already at the
|
|
206
|
+
ceiling, no overlap is added.
|
|
207
|
+
"""
|
|
208
|
+
out = [chunks[0]]
|
|
209
|
+
for prev, cur in zip(chunks, chunks[1:], strict=False):
|
|
210
|
+
cur_tokens = count_tokens(cur)
|
|
211
|
+
room = ceiling - cur_tokens
|
|
212
|
+
if room <= 0: # pragma: no cover - defensive
|
|
213
|
+
out.append(cur)
|
|
214
|
+
continue
|
|
215
|
+
budget = min(overlap_tokens, room)
|
|
216
|
+
tail = _take_tail_tokens(prev, budget, count_tokens)
|
|
217
|
+
if not tail: # pragma: no cover - defensive
|
|
218
|
+
out.append(cur)
|
|
219
|
+
continue
|
|
220
|
+
combined = tail + "\n\n" + cur
|
|
221
|
+
if count_tokens(combined) > ceiling: # pragma: no cover - defensive
|
|
222
|
+
# Tail estimate overshot; drop overlap rather than violate ceiling.
|
|
223
|
+
out.append(cur)
|
|
224
|
+
else:
|
|
225
|
+
out.append(combined)
|
|
226
|
+
return out
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def _take_tail_tokens(
|
|
230
|
+
text: str, n_tokens: int, count_tokens: Callable[[str], int]
|
|
231
|
+
) -> str:
|
|
232
|
+
"""Return at most ~``n_tokens`` worth of trailing content.
|
|
233
|
+
|
|
234
|
+
Paragraph-aligned where possible; falls back to hard-splitting the last
|
|
235
|
+
paragraph if it alone is larger than ``n_tokens``.
|
|
236
|
+
"""
|
|
237
|
+
paragraphs = [p.strip() for p in _PARAGRAPH_SPLIT.split(text) if p.strip()]
|
|
238
|
+
selected: list[str] = []
|
|
239
|
+
total = 0
|
|
240
|
+
for para in reversed(paragraphs):
|
|
241
|
+
para_tokens = count_tokens(para)
|
|
242
|
+
if selected and total + para_tokens > n_tokens:
|
|
243
|
+
break
|
|
244
|
+
if not selected and para_tokens > n_tokens:
|
|
245
|
+
atoms = _split_long_paragraph(para, n_tokens, count_tokens)
|
|
246
|
+
return atoms[-1] if atoms else ""
|
|
247
|
+
selected.insert(0, para)
|
|
248
|
+
total += para_tokens
|
|
249
|
+
return "\n\n".join(selected)
|
brain/ingest/docx.py
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""DOCX extractor — paragraphs + tables."""
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
|
|
4
|
+
from docx import Document
|
|
5
|
+
|
|
6
|
+
from . import ExtractedDoc
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def extract_docx(path: Path) -> ExtractedDoc:
|
|
10
|
+
"""Extract an :class:`ExtractedDoc` from a ``.docx`` file on disk.
|
|
11
|
+
|
|
12
|
+
Paragraphs and table cell text are collected in document order.
|
|
13
|
+
The title is taken from the first Heading-styled paragraph; if none is
|
|
14
|
+
present, falls back to the file stem.
|
|
15
|
+
"""
|
|
16
|
+
d = Document(str(path))
|
|
17
|
+
parts: list[str] = []
|
|
18
|
+
title: str | None = None
|
|
19
|
+
|
|
20
|
+
for para in d.paragraphs:
|
|
21
|
+
text = para.text.strip()
|
|
22
|
+
if not text:
|
|
23
|
+
continue
|
|
24
|
+
style_name = (para.style.name or "") if para.style is not None else ""
|
|
25
|
+
if title is None and style_name.startswith("Heading"):
|
|
26
|
+
title = text
|
|
27
|
+
parts.append(text)
|
|
28
|
+
|
|
29
|
+
for table in d.tables:
|
|
30
|
+
for row in table.rows:
|
|
31
|
+
cells = [cell.text.strip() for cell in row.cells]
|
|
32
|
+
parts.append("\t".join(cells))
|
|
33
|
+
|
|
34
|
+
return ExtractedDoc(
|
|
35
|
+
title=title or Path(path).stem,
|
|
36
|
+
content="\n\n".join(parts),
|
|
37
|
+
content_type="docx",
|
|
38
|
+
source_path=str(Path(path).resolve()),
|
|
39
|
+
metadata={},
|
|
40
|
+
)
|