cortexdb-connectors 0.2.9__tar.gz → 0.2.11__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/PKG-INFO +4 -2
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/README.md +1 -1
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/base.py +277 -21
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/cli.py +7 -1
- cortexdb_connectors-0.2.11/cortexdb_connectors/jira/__init__.py +1099 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/webhooks.py +10 -2
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors.egg-info/PKG-INFO +4 -2
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors.egg-info/SOURCES.txt +2 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors.egg-info/requires.txt +2 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/pyproject.toml +92 -87
- cortexdb_connectors-0.2.11/tests/test_jira_attachments.py +265 -0
- cortexdb_connectors-0.2.11/tests/test_jira_findings_0_2_11.py +660 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/tests/test_jira_findings_0_2_9.py +341 -341
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/tests/test_webhooks.py +3 -1
- cortexdb_connectors-0.2.9/cortexdb_connectors/jira/__init__.py +0 -399
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/__init__.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/confluence/__init__.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/discord/__init__.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/freshdesk/__init__.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/github/__init__.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/gitlab/__init__.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/google_workspace/__init__.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/hubspot/__init__.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/insights/__init__.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/insights/api.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/insights/detectors.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/insights/engine.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/intercom/__init__.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/linear/__init__.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/notion/__init__.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/pagerduty/__init__.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/salesforce/__init__.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/servicenow/__init__.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/slack/__init__.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/state.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/teams/__init__.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/tldv/__init__.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/worker.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/zendesk/__init__.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors.egg-info/dependency_links.txt +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors.egg-info/entry_points.txt +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors.egg-info/top_level.txt +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/setup.cfg +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/tests/test_base.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/tests/test_connector_coverage.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/tests/test_github.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/tests/test_ingest_e2e.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/tests/test_slack.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/tests/test_streaming_cursor.py +0 -0
- {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/tests/test_webhooks_replay.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: cortexdb-connectors
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.11
|
|
4
4
|
Summary: Data connectors for CortexDB — ingest from Slack, GitHub, GitLab, Jira, Linear, Confluence, Notion, PagerDuty, Discord, Teams, Google Workspace, Salesforce, HubSpot, Zendesk, Intercom, and ServiceNow into the v1 memory API.
|
|
5
5
|
Author-email: CortexDB Team <team@cortexdb.ai>
|
|
6
6
|
License-Expression: Apache-2.0
|
|
@@ -32,6 +32,7 @@ Provides-Extra: pagerduty
|
|
|
32
32
|
Requires-Dist: pdpyras>=5.0; extra == "pagerduty"
|
|
33
33
|
Provides-Extra: jira
|
|
34
34
|
Requires-Dist: jira>=3.0; extra == "jira"
|
|
35
|
+
Requires-Dist: tzdata; extra == "jira"
|
|
35
36
|
Provides-Extra: confluence
|
|
36
37
|
Requires-Dist: atlassian-python-api>=3.0; extra == "confluence"
|
|
37
38
|
Provides-Extra: notion
|
|
@@ -62,6 +63,7 @@ Requires-Dist: slack-sdk>=3.0; extra == "all"
|
|
|
62
63
|
Requires-Dist: pygithub>=2.0; extra == "all"
|
|
63
64
|
Requires-Dist: pdpyras>=5.0; extra == "all"
|
|
64
65
|
Requires-Dist: jira>=3.0; extra == "all"
|
|
66
|
+
Requires-Dist: tzdata; extra == "all"
|
|
65
67
|
Requires-Dist: atlassian-python-api>=3.0; extra == "all"
|
|
66
68
|
Requires-Dist: notion-client>=2.0; extra == "all"
|
|
67
69
|
Requires-Dist: aiohttp>=3.9; extra == "all"
|
|
@@ -156,7 +158,7 @@ All 16 connectors are **🟢 Free** to self-host and **🔒 Starter+** ($29/mo)
|
|
|
156
158
|
slack SLACK_BOT_TOKEN
|
|
157
159
|
github GITHUB_TOKEN
|
|
158
160
|
gitlab GITLAB_TOKEN
|
|
159
|
-
jira JIRA_URL, JIRA_EMAIL, JIRA_API_TOKEN
|
|
161
|
+
jira JIRA_URL, JIRA_EMAIL, JIRA_API_TOKEN (incl. attachments → /v1/blobs multimodal extraction, 0.2.10+)
|
|
160
162
|
linear LINEAR_API_KEY
|
|
161
163
|
confluence CONFLUENCE_URL, CONFLUENCE_EMAIL, CONFLUENCE_API_TOKEN
|
|
162
164
|
notion NOTION_TOKEN
|
|
@@ -78,7 +78,7 @@ All 16 connectors are **🟢 Free** to self-host and **🔒 Starter+** ($29/mo)
|
|
|
78
78
|
slack SLACK_BOT_TOKEN
|
|
79
79
|
github GITHUB_TOKEN
|
|
80
80
|
gitlab GITLAB_TOKEN
|
|
81
|
-
jira JIRA_URL, JIRA_EMAIL, JIRA_API_TOKEN
|
|
81
|
+
jira JIRA_URL, JIRA_EMAIL, JIRA_API_TOKEN (incl. attachments → /v1/blobs multimodal extraction, 0.2.10+)
|
|
82
82
|
linear LINEAR_API_KEY
|
|
83
83
|
confluence CONFLUENCE_URL, CONFLUENCE_EMAIL, CONFLUENCE_API_TOKEN
|
|
84
84
|
notion NOTION_TOKEN
|
|
@@ -36,6 +36,11 @@ _ACTOR_LOCAL_ILLEGAL = re.compile(r"[^A-Za-z0-9_.@-]")
|
|
|
36
36
|
# never drop content silently.
|
|
37
37
|
_MAX_CONTENT_BYTES = 10 * 1024 * 1024 # 10 MB
|
|
38
38
|
|
|
39
|
+
# Server-enforced ceiling on `idempotency_key` (cortex-types
|
|
40
|
+
# `EnvelopeError::IdempotencyKeyTooLong`). Over it, the write is refused with
|
|
41
|
+
# 422 INVALID_BODY — which stalls a sync cursor — so keys are built to fit.
|
|
42
|
+
_MAX_IDEM_BYTES = 64
|
|
43
|
+
|
|
39
44
|
|
|
40
45
|
# ---------------------------------------------------------------------------
|
|
41
46
|
# Enums
|
|
@@ -102,7 +107,7 @@ class VisibilityLevel(str, Enum):
|
|
|
102
107
|
@dataclass(frozen=True)
|
|
103
108
|
class Source:
|
|
104
109
|
system: str
|
|
105
|
-
connector_version: str = "0.2.
|
|
110
|
+
connector_version: str = "0.2.11"
|
|
106
111
|
|
|
107
112
|
|
|
108
113
|
@dataclass(frozen=True)
|
|
@@ -126,6 +131,25 @@ class Visibility:
|
|
|
126
131
|
allowed_principals: tuple[str, ...] = ()
|
|
127
132
|
|
|
128
133
|
|
|
134
|
+
@dataclass
|
|
135
|
+
class Attachment:
|
|
136
|
+
"""A binary attachment destined for `POST /v1/blobs` + a `blob_ref`
|
|
137
|
+
experience, so CortexDB's multimodal pipeline (Tika/vision/ASR content
|
|
138
|
+
processors) extracts and indexes it.
|
|
139
|
+
|
|
140
|
+
`data` carries the raw bytes (downloaded by the connector's
|
|
141
|
+
`fetch_events`); when `data` is None the attachment is metadata-only
|
|
142
|
+
(e.g. oversized skip) and only its filename line survives in text.
|
|
143
|
+
"""
|
|
144
|
+
|
|
145
|
+
filename: str
|
|
146
|
+
mime_type: str
|
|
147
|
+
size_bytes: int = 0
|
|
148
|
+
data: bytes | None = None
|
|
149
|
+
source_url: str | None = None
|
|
150
|
+
attachment_id: str = ""
|
|
151
|
+
|
|
152
|
+
|
|
129
153
|
@dataclass
|
|
130
154
|
class RawEvent:
|
|
131
155
|
source: str
|
|
@@ -158,6 +182,11 @@ class Episode:
|
|
|
158
182
|
parent_id: str | None = None
|
|
159
183
|
idempotency_key: str | None = None
|
|
160
184
|
ttl_seconds: int | None = None
|
|
185
|
+
# When set, this episode is a BINARY attachment: `ingest_episode` uploads
|
|
186
|
+
# `blob.data` to `POST /v1/blobs` and sends the experience with
|
|
187
|
+
# `content.kind="blob_ref"` (the multimodal path) instead of text. The
|
|
188
|
+
# `content` field then serves only as the no-bytes fallback text.
|
|
189
|
+
blob: Attachment | None = None
|
|
161
190
|
|
|
162
191
|
|
|
163
192
|
@dataclass
|
|
@@ -234,6 +263,14 @@ class CortexConnector(ABC):
|
|
|
234
263
|
or self.scope
|
|
235
264
|
or self.actor
|
|
236
265
|
)
|
|
266
|
+
# idempotency_key → server blob_id for every blob uploaded this run,
|
|
267
|
+
# and → event_id for everything ingested this run. A connector that
|
|
268
|
+
# reconciles source deletions persists these so a LATER run can name
|
|
269
|
+
# the event and purge the bytes of an item deleted at the source;
|
|
270
|
+
# neither id is discoverable after the fact (the blob id is only
|
|
271
|
+
# visible at upload time, and idempotency records expire in 24h).
|
|
272
|
+
self._uploaded_blobs: dict[str, str] = {}
|
|
273
|
+
self._ingested_events: dict[str, str] = {}
|
|
237
274
|
|
|
238
275
|
def bind(self, *, actor: str | None = None, scope_template: str | None = None) -> "CortexConnector":
|
|
239
276
|
"""Set v1-only fields after construction. Returns self for chaining."""
|
|
@@ -338,8 +375,13 @@ class CortexConnector(ABC):
|
|
|
338
375
|
)
|
|
339
376
|
return tpl
|
|
340
377
|
|
|
341
|
-
def _envelope(self, episode: Episode) -> dict[str, Any]:
|
|
342
|
-
"""Build the v1 ExperienceItem JSON for this episode.
|
|
378
|
+
def _envelope(self, episode: Episode, blob_id: str | None = None) -> dict[str, Any]:
|
|
379
|
+
"""Build the v1 ExperienceItem JSON for this episode.
|
|
380
|
+
|
|
381
|
+
With `blob_id` (an id returned by `upload_blob`) the content becomes
|
|
382
|
+
`kind="blob_ref"` — the server fetches the stored bytes and runs the
|
|
383
|
+
per-modality content processor on them. Otherwise plain text.
|
|
384
|
+
"""
|
|
343
385
|
scope = self._resolve_scope(episode)
|
|
344
386
|
modality = (
|
|
345
387
|
episode.metadata.get("modality")
|
|
@@ -354,11 +396,19 @@ class CortexConnector(ABC):
|
|
|
354
396
|
for ent in episode.entities:
|
|
355
397
|
labels.append(f"{ent.entity_type}:{ent.entity_id}")
|
|
356
398
|
|
|
357
|
-
|
|
358
|
-
"kind": "
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
399
|
+
if blob_id is not None:
|
|
400
|
+
content: dict[str, Any] = {"kind": "blob_ref", "blob_id": blob_id}
|
|
401
|
+
# Filename/MIME ride labels — blob_ref content carries no text,
|
|
402
|
+
# and labels are the queryable metadata surface.
|
|
403
|
+
if episode.blob is not None:
|
|
404
|
+
labels.append(f"filename={episode.blob.filename}")
|
|
405
|
+
labels.append(f"mime={episode.blob.mime_type}")
|
|
406
|
+
else:
|
|
407
|
+
content = {
|
|
408
|
+
"kind": "message",
|
|
409
|
+
"role": "user" if modality == "conversation" else "system",
|
|
410
|
+
"text": self._guard_content(episode.content, episode),
|
|
411
|
+
}
|
|
362
412
|
# The v1 write envelope is STRICT (deny_unknown_fields) as of v0.8.0:
|
|
363
413
|
# thread_id / meta / preceded_by are NOT envelope context fields, and
|
|
364
414
|
# sending them returns 422 INVALID_BODY (which stalls the sync cursor
|
|
@@ -387,10 +437,7 @@ class CortexConnector(ABC):
|
|
|
387
437
|
"modality": modality,
|
|
388
438
|
"content": content,
|
|
389
439
|
"context": context,
|
|
390
|
-
"idempotency_key": (
|
|
391
|
-
episode.idempotency_key
|
|
392
|
-
or self._fallback_idem(episode, scope)
|
|
393
|
-
),
|
|
440
|
+
"idempotency_key": self._idem_key(episode, scope),
|
|
394
441
|
}
|
|
395
442
|
if episode.actor and episode.actor.id:
|
|
396
443
|
# Source actor ids are not guaranteed bare: Jira Cloud accountIds
|
|
@@ -414,6 +461,38 @@ class CortexConnector(ABC):
|
|
|
414
461
|
seed = f"{scope}|{src}|{episode.id}|{episode.content[:512]}"
|
|
415
462
|
return f"conn:{src}:{hashlib.sha256(seed.encode('utf-8')).hexdigest()[:24]}"
|
|
416
463
|
|
|
464
|
+
def _idem_key(self, episode: Episode, scope: str) -> str:
|
|
465
|
+
"""The key actually sent — the connector's key QUALIFIED BY SCOPE.
|
|
466
|
+
|
|
467
|
+
C4: the server namespaces idempotency records by the scope ROOT (the
|
|
468
|
+
first path segment) plus the caller, while the envelope it hashes
|
|
469
|
+
carries the FULL scope. So the same source item synced into two
|
|
470
|
+
sibling scopes under one root re-used one record with a different
|
|
471
|
+
body — `409 IDEMPOTENCY_CONFLICT`, and every event after the first
|
|
472
|
+
scope was silently dropped. Qualifying the key with the resolved
|
|
473
|
+
scope makes it mean what it always claimed: "this item, in this
|
|
474
|
+
scope".
|
|
475
|
+
|
|
476
|
+
The scope rides as an 8-hex-char digest because the wire field is
|
|
477
|
+
capped at `_MAX_IDEM_BYTES` and scope paths are unbounded; a key that
|
|
478
|
+
can't fit the suffix is replaced by a digest of (scope, key) rather
|
|
479
|
+
than truncated, since a truncated key would collide across items.
|
|
480
|
+
|
|
481
|
+
Same-scope semantics are unchanged: one item still maps to exactly
|
|
482
|
+
one key, so re-syncs replay and poll/webhook paths (which resolve to
|
|
483
|
+
the same scope) still share one key. `_fallback_idem` keys already
|
|
484
|
+
hash the scope in, so they pass through untouched.
|
|
485
|
+
"""
|
|
486
|
+
key = episode.idempotency_key
|
|
487
|
+
if not key:
|
|
488
|
+
return self._fallback_idem(episode, scope)
|
|
489
|
+
digest = hashlib.sha256(scope.encode("utf-8")).hexdigest()
|
|
490
|
+
suffix = f"@{digest[:8]}"
|
|
491
|
+
if len(key) + len(suffix) <= _MAX_IDEM_BYTES:
|
|
492
|
+
return f"{key}{suffix}"
|
|
493
|
+
seed = f"{scope}|{key}".encode("utf-8")
|
|
494
|
+
return f"cx:{hashlib.sha256(seed).hexdigest()[:40]}"
|
|
495
|
+
|
|
417
496
|
@staticmethod
|
|
418
497
|
def _guard_content(text: str, episode: Episode) -> str:
|
|
419
498
|
"""Hard 10 MB ceiling on content. Real items never hit this; if one
|
|
@@ -433,10 +512,168 @@ class CortexConnector(ABC):
|
|
|
433
512
|
)
|
|
434
513
|
return capped
|
|
435
514
|
|
|
515
|
+
def _headers(self, content_type: str = "application/json") -> dict[str, str]:
|
|
516
|
+
return {
|
|
517
|
+
"Authorization": f"Bearer {self.cortex_api_key}",
|
|
518
|
+
"X-Cortex-Actor": self.actor,
|
|
519
|
+
"Content-Type": content_type,
|
|
520
|
+
"User-Agent": f"cortexdb-connectors/{Source(system='base').connector_version}",
|
|
521
|
+
}
|
|
522
|
+
|
|
523
|
+
async def upload_blob(self, data: bytes, mime_type: str) -> str:
|
|
524
|
+
"""POST raw bytes to `/v1/blobs`; returns the server `blob_id`."""
|
|
525
|
+
import httpx
|
|
526
|
+
|
|
527
|
+
async with httpx.AsyncClient(timeout=120.0) as client:
|
|
528
|
+
resp = await client.post(
|
|
529
|
+
f"{self.cortex_url}/v1/blobs",
|
|
530
|
+
content=data,
|
|
531
|
+
headers=self._headers(mime_type or "application/octet-stream"),
|
|
532
|
+
)
|
|
533
|
+
if resp.status_code >= 400:
|
|
534
|
+
logger.error(
|
|
535
|
+
"blob upload REJECTED: status=%s bytes=%d mime=%s body=%s",
|
|
536
|
+
resp.status_code, len(data), mime_type, resp.text[:300],
|
|
537
|
+
)
|
|
538
|
+
resp.raise_for_status()
|
|
539
|
+
return resp.json()["blob_id"]
|
|
540
|
+
|
|
541
|
+
async def experience_status(
|
|
542
|
+
self, idempotency_key: str, scope: str
|
|
543
|
+
) -> dict[str, Any]:
|
|
544
|
+
"""`GET /v1/experience/status` for a prior write with this key.
|
|
545
|
+
|
|
546
|
+
Returns the server's status document (`{found, event_id, status,
|
|
547
|
+
…}`), or `{"found": False}` when the probe fails — callers treat an
|
|
548
|
+
unreachable probe as "unknown", never as proof of absence.
|
|
549
|
+
"""
|
|
550
|
+
import httpx
|
|
551
|
+
|
|
552
|
+
try:
|
|
553
|
+
async with httpx.AsyncClient(timeout=15.0) as client:
|
|
554
|
+
resp = await client.get(
|
|
555
|
+
f"{self.cortex_url}/v1/experience/status",
|
|
556
|
+
params={"idempotency_key": idempotency_key, "scope": scope},
|
|
557
|
+
headers=self._headers(),
|
|
558
|
+
)
|
|
559
|
+
if resp.status_code != 200:
|
|
560
|
+
return {"found": False}
|
|
561
|
+
body = resp.json()
|
|
562
|
+
return body if isinstance(body, dict) else {"found": False}
|
|
563
|
+
except Exception: # noqa: BLE001
|
|
564
|
+
return {"found": False}
|
|
565
|
+
|
|
566
|
+
async def experience_exists(self, idempotency_key: str, scope: str) -> bool:
|
|
567
|
+
"""Did a write with this key already land?
|
|
568
|
+
|
|
569
|
+
Used before attachment blob uploads so a re-sync doesn't store a
|
|
570
|
+
fresh (orphaned) blob copy for an experience that will just replay.
|
|
571
|
+
Best-effort: any probe failure returns False (upload proceeds — the
|
|
572
|
+
experience write itself still dedups via its idempotency key).
|
|
573
|
+
"""
|
|
574
|
+
status = await self.experience_status(idempotency_key, scope)
|
|
575
|
+
return bool(status.get("found"))
|
|
576
|
+
|
|
577
|
+
# -- v1 deletion propagation ---------------------------------------------
|
|
578
|
+
|
|
579
|
+
async def forget_events(
|
|
580
|
+
self, event_ids: list[str], scope: str, *, reason: str = ""
|
|
581
|
+
) -> None:
|
|
582
|
+
"""Redact specific events via `POST /v1/forget` (`memory_ids` selector).
|
|
583
|
+
|
|
584
|
+
The precise form of deletion propagation: it names exactly the events
|
|
585
|
+
to remove, so nothing else in the scope is touched. Used when a source
|
|
586
|
+
item that was ingested as its own event disappears at the source.
|
|
587
|
+
"""
|
|
588
|
+
import httpx
|
|
589
|
+
|
|
590
|
+
ids = [e for e in event_ids if e]
|
|
591
|
+
if not ids:
|
|
592
|
+
return
|
|
593
|
+
body = {
|
|
594
|
+
"scope": scope,
|
|
595
|
+
"selector": {"memory_ids": ids},
|
|
596
|
+
"cascade": "redact_events",
|
|
597
|
+
"reason": reason or "source deletion reconciled by connector",
|
|
598
|
+
}
|
|
599
|
+
async with httpx.AsyncClient(timeout=30.0) as client:
|
|
600
|
+
resp = await client.post(
|
|
601
|
+
f"{self.cortex_url}/v1/forget", json=body, headers=self._headers()
|
|
602
|
+
)
|
|
603
|
+
if resp.status_code >= 400:
|
|
604
|
+
logger.error(
|
|
605
|
+
"forget REJECTED by server: status=%s ids=%s body=%s",
|
|
606
|
+
resp.status_code, ids, resp.text[:300],
|
|
607
|
+
)
|
|
608
|
+
resp.raise_for_status()
|
|
609
|
+
|
|
610
|
+
async def forget_entity(
|
|
611
|
+
self, entity: str, scope: str, *, reason: str = ""
|
|
612
|
+
) -> None:
|
|
613
|
+
"""Redact everything about a deleted source entity (`about_entity`).
|
|
614
|
+
|
|
615
|
+
Used when the source hard-deletes a whole item (a Jira issue, a
|
|
616
|
+
Notion page) and every event mentioning it must go.
|
|
617
|
+
"""
|
|
618
|
+
import httpx
|
|
619
|
+
|
|
620
|
+
body = {
|
|
621
|
+
"scope": scope,
|
|
622
|
+
"selector": {"about_entity": entity},
|
|
623
|
+
"cascade": "redact_events",
|
|
624
|
+
"reason": reason or f"source deleted {entity}",
|
|
625
|
+
}
|
|
626
|
+
async with httpx.AsyncClient(timeout=30.0) as client:
|
|
627
|
+
resp = await client.post(
|
|
628
|
+
f"{self.cortex_url}/v1/forget", json=body, headers=self._headers()
|
|
629
|
+
)
|
|
630
|
+
if resp.status_code >= 400:
|
|
631
|
+
logger.error(
|
|
632
|
+
"forget REJECTED by server: status=%s entity=%s body=%s",
|
|
633
|
+
resp.status_code, entity, resp.text[:300],
|
|
634
|
+
)
|
|
635
|
+
resp.raise_for_status()
|
|
636
|
+
|
|
637
|
+
async def delete_blob(self, blob_id: str) -> bool:
|
|
638
|
+
"""`DELETE /v1/blobs/{id}` — purge uploaded bytes.
|
|
639
|
+
|
|
640
|
+
Redacting the blob_ref experience removes the memory; it does not
|
|
641
|
+
remove the stored object. A source-side file deletion has to purge
|
|
642
|
+
BOTH, or the deleted file stays byte-retrievable from CortexDB.
|
|
643
|
+
Returns True when the bytes are gone (deleted now, or already
|
|
644
|
+
absent — 404 is success for a delete). Raises on anything else so
|
|
645
|
+
the caller reports a failed purge instead of assuming one.
|
|
646
|
+
"""
|
|
647
|
+
import httpx
|
|
648
|
+
|
|
649
|
+
if not blob_id:
|
|
650
|
+
return False
|
|
651
|
+
async with httpx.AsyncClient(timeout=30.0) as client:
|
|
652
|
+
resp = await client.delete(
|
|
653
|
+
f"{self.cortex_url}/v1/blobs/{blob_id}", headers=self._headers()
|
|
654
|
+
)
|
|
655
|
+
if resp.status_code == 404:
|
|
656
|
+
return True
|
|
657
|
+
if resp.status_code >= 400:
|
|
658
|
+
logger.error(
|
|
659
|
+
"blob delete REJECTED by server: status=%s blob_id=%s body=%s",
|
|
660
|
+
resp.status_code, blob_id, resp.text[:300],
|
|
661
|
+
)
|
|
662
|
+
resp.raise_for_status()
|
|
663
|
+
return True
|
|
664
|
+
|
|
436
665
|
async def ingest_episode(self, episode: Episode) -> str:
|
|
437
666
|
"""POST `/v1/experience` with the v1 envelope.
|
|
438
667
|
|
|
439
|
-
|
|
668
|
+
Blob-carrying episodes (`episode.blob` set with bytes) first upload
|
|
669
|
+
the bytes to `/v1/blobs` and go in as `content.kind="blob_ref"` so the
|
|
670
|
+
server's content processors extract them. A status probe on the
|
|
671
|
+
episode's idempotency key skips the upload when the experience already
|
|
672
|
+
exists (re-sync), avoiding orphaned duplicate blobs. Bytes-less blob
|
|
673
|
+
episodes (oversized skips) fall back to their text content.
|
|
674
|
+
|
|
675
|
+
Returns the server-side event_id on success ("exists" when the
|
|
676
|
+
status probe short-circuited a replay).
|
|
440
677
|
"""
|
|
441
678
|
import httpx
|
|
442
679
|
|
|
@@ -452,13 +689,24 @@ class CortexConnector(ABC):
|
|
|
452
689
|
"explicitly. The actor must match the token's `sub` claim."
|
|
453
690
|
)
|
|
454
691
|
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
692
|
+
scope = self._resolve_scope(episode)
|
|
693
|
+
key = self._idem_key(episode, scope)
|
|
694
|
+
blob_id: str | None = None
|
|
695
|
+
if episode.blob is not None and episode.blob.data:
|
|
696
|
+
if await self.experience_exists(key, scope):
|
|
697
|
+
logger.info(
|
|
698
|
+
"attachment already ingested (idempotency_key=%s); "
|
|
699
|
+
"skipping blob upload",
|
|
700
|
+
key,
|
|
701
|
+
)
|
|
702
|
+
return "exists"
|
|
703
|
+
blob_id = await self.upload_blob(
|
|
704
|
+
episode.blob.data, episode.blob.mime_type
|
|
705
|
+
)
|
|
706
|
+
self._uploaded_blobs[key] = blob_id
|
|
707
|
+
|
|
708
|
+
envelope = self._envelope(episode, blob_id=blob_id)
|
|
709
|
+
headers = self._headers()
|
|
462
710
|
|
|
463
711
|
async with httpx.AsyncClient(timeout=30.0) as client:
|
|
464
712
|
resp = await client.post(
|
|
@@ -480,4 +728,12 @@ class CortexConnector(ABC):
|
|
|
480
728
|
)
|
|
481
729
|
resp.raise_for_status()
|
|
482
730
|
body = resp.json()
|
|
483
|
-
|
|
731
|
+
event_id = body.get("event_id") or body.get("id")
|
|
732
|
+
if event_id:
|
|
733
|
+
# Remembered so a connector that reconciles source deletions can
|
|
734
|
+
# name this exact event later. The server's idempotency records
|
|
735
|
+
# expire after 24h, so a status probe cannot resolve the id of
|
|
736
|
+
# something ingested days ago — which is exactly when a source
|
|
737
|
+
# deletion shows up.
|
|
738
|
+
self._ingested_events[key] = event_id
|
|
739
|
+
return event_id or episode.id
|
|
@@ -80,7 +80,12 @@ CONNECTORS: dict[str, dict[str, Any]] = {
|
|
|
80
80
|
("jira_email", "JIRA_EMAIL"),
|
|
81
81
|
("jira_api_token", "JIRA_API_TOKEN"),
|
|
82
82
|
],
|
|
83
|
-
"options": [
|
|
83
|
+
"options": [
|
|
84
|
+
("project_keys", "JIRA_PROJECT_KEYS", "list"),
|
|
85
|
+
("reconcile_deletions", "JIRA_RECONCILE_DELETIONS", "bool"),
|
|
86
|
+
("reconcile_interval_seconds", "JIRA_RECONCILE_INTERVAL_SECONDS", "int"),
|
|
87
|
+
("ledger_path", "JIRA_LEDGER_PATH", "str"),
|
|
88
|
+
],
|
|
84
89
|
},
|
|
85
90
|
"freshdesk": {
|
|
86
91
|
"import": "cortexdb_connectors.freshdesk:FreshdeskConnector",
|
|
@@ -260,6 +265,7 @@ def _load_cortex_creds(args: argparse.Namespace) -> dict[str, str]:
|
|
|
260
265
|
|
|
261
266
|
_OPTION_PARSERS: dict[str, Callable[[str], Any]] = {
|
|
262
267
|
"str": lambda v: v,
|
|
268
|
+
"int": int,
|
|
263
269
|
"list": lambda v: [x.strip() for x in v.split(",") if x.strip()],
|
|
264
270
|
"list-int": lambda v: [int(x.strip()) for x in v.split(",") if x.strip()],
|
|
265
271
|
"bool": lambda v: v.lower() in ("1", "true", "yes", "on"),
|