cortexdb-connectors 0.2.9__tar.gz → 0.2.11__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/PKG-INFO +4 -2
  2. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/README.md +1 -1
  3. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/base.py +277 -21
  4. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/cli.py +7 -1
  5. cortexdb_connectors-0.2.11/cortexdb_connectors/jira/__init__.py +1099 -0
  6. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/webhooks.py +10 -2
  7. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors.egg-info/PKG-INFO +4 -2
  8. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors.egg-info/SOURCES.txt +2 -0
  9. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors.egg-info/requires.txt +2 -0
  10. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/pyproject.toml +92 -87
  11. cortexdb_connectors-0.2.11/tests/test_jira_attachments.py +265 -0
  12. cortexdb_connectors-0.2.11/tests/test_jira_findings_0_2_11.py +660 -0
  13. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/tests/test_jira_findings_0_2_9.py +341 -341
  14. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/tests/test_webhooks.py +3 -1
  15. cortexdb_connectors-0.2.9/cortexdb_connectors/jira/__init__.py +0 -399
  16. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/__init__.py +0 -0
  17. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/confluence/__init__.py +0 -0
  18. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/discord/__init__.py +0 -0
  19. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/freshdesk/__init__.py +0 -0
  20. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/github/__init__.py +0 -0
  21. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/gitlab/__init__.py +0 -0
  22. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/google_workspace/__init__.py +0 -0
  23. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/hubspot/__init__.py +0 -0
  24. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/insights/__init__.py +0 -0
  25. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/insights/api.py +0 -0
  26. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/insights/detectors.py +0 -0
  27. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/insights/engine.py +0 -0
  28. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/intercom/__init__.py +0 -0
  29. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/linear/__init__.py +0 -0
  30. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/notion/__init__.py +0 -0
  31. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/pagerduty/__init__.py +0 -0
  32. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/salesforce/__init__.py +0 -0
  33. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/servicenow/__init__.py +0 -0
  34. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/slack/__init__.py +0 -0
  35. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/state.py +0 -0
  36. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/teams/__init__.py +0 -0
  37. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/tldv/__init__.py +0 -0
  38. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/worker.py +0 -0
  39. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors/zendesk/__init__.py +0 -0
  40. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors.egg-info/dependency_links.txt +0 -0
  41. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors.egg-info/entry_points.txt +0 -0
  42. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/cortexdb_connectors.egg-info/top_level.txt +0 -0
  43. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/setup.cfg +0 -0
  44. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/tests/test_base.py +0 -0
  45. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/tests/test_connector_coverage.py +0 -0
  46. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/tests/test_github.py +0 -0
  47. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/tests/test_ingest_e2e.py +0 -0
  48. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/tests/test_slack.py +0 -0
  49. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/tests/test_streaming_cursor.py +0 -0
  50. {cortexdb_connectors-0.2.9 → cortexdb_connectors-0.2.11}/tests/test_webhooks_replay.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: cortexdb-connectors
3
- Version: 0.2.9
3
+ Version: 0.2.11
4
4
  Summary: Data connectors for CortexDB — ingest from Slack, GitHub, GitLab, Jira, Linear, Confluence, Notion, PagerDuty, Discord, Teams, Google Workspace, Salesforce, HubSpot, Zendesk, Intercom, and ServiceNow into the v1 memory API.
5
5
  Author-email: CortexDB Team <team@cortexdb.ai>
6
6
  License-Expression: Apache-2.0
@@ -32,6 +32,7 @@ Provides-Extra: pagerduty
32
32
  Requires-Dist: pdpyras>=5.0; extra == "pagerduty"
33
33
  Provides-Extra: jira
34
34
  Requires-Dist: jira>=3.0; extra == "jira"
35
+ Requires-Dist: tzdata; extra == "jira"
35
36
  Provides-Extra: confluence
36
37
  Requires-Dist: atlassian-python-api>=3.0; extra == "confluence"
37
38
  Provides-Extra: notion
@@ -62,6 +63,7 @@ Requires-Dist: slack-sdk>=3.0; extra == "all"
62
63
  Requires-Dist: pygithub>=2.0; extra == "all"
63
64
  Requires-Dist: pdpyras>=5.0; extra == "all"
64
65
  Requires-Dist: jira>=3.0; extra == "all"
66
+ Requires-Dist: tzdata; extra == "all"
65
67
  Requires-Dist: atlassian-python-api>=3.0; extra == "all"
66
68
  Requires-Dist: notion-client>=2.0; extra == "all"
67
69
  Requires-Dist: aiohttp>=3.9; extra == "all"
@@ -156,7 +158,7 @@ All 16 connectors are **🟢 Free** to self-host and **🔒 Starter+** ($29/mo)
156
158
  slack SLACK_BOT_TOKEN
157
159
  github GITHUB_TOKEN
158
160
  gitlab GITLAB_TOKEN
159
- jira JIRA_URL, JIRA_EMAIL, JIRA_API_TOKEN
161
+ jira JIRA_URL, JIRA_EMAIL, JIRA_API_TOKEN (incl. attachments → /v1/blobs multimodal extraction, 0.2.10+)
160
162
  linear LINEAR_API_KEY
161
163
  confluence CONFLUENCE_URL, CONFLUENCE_EMAIL, CONFLUENCE_API_TOKEN
162
164
  notion NOTION_TOKEN
@@ -78,7 +78,7 @@ All 16 connectors are **🟢 Free** to self-host and **🔒 Starter+** ($29/mo)
78
78
  slack SLACK_BOT_TOKEN
79
79
  github GITHUB_TOKEN
80
80
  gitlab GITLAB_TOKEN
81
- jira JIRA_URL, JIRA_EMAIL, JIRA_API_TOKEN
81
+ jira JIRA_URL, JIRA_EMAIL, JIRA_API_TOKEN (incl. attachments → /v1/blobs multimodal extraction, 0.2.10+)
82
82
  linear LINEAR_API_KEY
83
83
  confluence CONFLUENCE_URL, CONFLUENCE_EMAIL, CONFLUENCE_API_TOKEN
84
84
  notion NOTION_TOKEN
@@ -36,6 +36,11 @@ _ACTOR_LOCAL_ILLEGAL = re.compile(r"[^A-Za-z0-9_.@-]")
36
36
  # never drop content silently.
37
37
  _MAX_CONTENT_BYTES = 10 * 1024 * 1024 # 10 MB
38
38
 
39
+ # Server-enforced ceiling on `idempotency_key` (cortex-types
40
+ # `EnvelopeError::IdempotencyKeyTooLong`). Over it, the write is refused with
41
+ # 422 INVALID_BODY — which stalls a sync cursor — so keys are built to fit.
42
+ _MAX_IDEM_BYTES = 64
43
+
39
44
 
40
45
  # ---------------------------------------------------------------------------
41
46
  # Enums
@@ -102,7 +107,7 @@ class VisibilityLevel(str, Enum):
102
107
  @dataclass(frozen=True)
103
108
  class Source:
104
109
  system: str
105
- connector_version: str = "0.2.9"
110
+ connector_version: str = "0.2.11"
106
111
 
107
112
 
108
113
  @dataclass(frozen=True)
@@ -126,6 +131,25 @@ class Visibility:
126
131
  allowed_principals: tuple[str, ...] = ()
127
132
 
128
133
 
134
+ @dataclass
135
+ class Attachment:
136
+ """A binary attachment destined for `POST /v1/blobs` + a `blob_ref`
137
+ experience, so CortexDB's multimodal pipeline (Tika/vision/ASR content
138
+ processors) extracts and indexes it.
139
+
140
+ `data` carries the raw bytes (downloaded by the connector's
141
+ `fetch_events`); when `data` is None the attachment is metadata-only
142
+ (e.g. oversized skip) and only its filename line survives in text.
143
+ """
144
+
145
+ filename: str
146
+ mime_type: str
147
+ size_bytes: int = 0
148
+ data: bytes | None = None
149
+ source_url: str | None = None
150
+ attachment_id: str = ""
151
+
152
+
129
153
  @dataclass
130
154
  class RawEvent:
131
155
  source: str
@@ -158,6 +182,11 @@ class Episode:
158
182
  parent_id: str | None = None
159
183
  idempotency_key: str | None = None
160
184
  ttl_seconds: int | None = None
185
+ # When set, this episode is a BINARY attachment: `ingest_episode` uploads
186
+ # `blob.data` to `POST /v1/blobs` and sends the experience with
187
+ # `content.kind="blob_ref"` (the multimodal path) instead of text. The
188
+ # `content` field then serves only as the no-bytes fallback text.
189
+ blob: Attachment | None = None
161
190
 
162
191
 
163
192
  @dataclass
@@ -234,6 +263,14 @@ class CortexConnector(ABC):
234
263
  or self.scope
235
264
  or self.actor
236
265
  )
266
+ # idempotency_key → server blob_id for every blob uploaded this run,
267
+ # and → event_id for everything ingested this run. A connector that
268
+ # reconciles source deletions persists these so a LATER run can name
269
+ # the event and purge the bytes of an item deleted at the source;
270
+ # neither id is discoverable after the fact (the blob id is only
271
+ # visible at upload time, and idempotency records expire in 24h).
272
+ self._uploaded_blobs: dict[str, str] = {}
273
+ self._ingested_events: dict[str, str] = {}
237
274
 
238
275
  def bind(self, *, actor: str | None = None, scope_template: str | None = None) -> "CortexConnector":
239
276
  """Set v1-only fields after construction. Returns self for chaining."""
@@ -338,8 +375,13 @@ class CortexConnector(ABC):
338
375
  )
339
376
  return tpl
340
377
 
341
- def _envelope(self, episode: Episode) -> dict[str, Any]:
342
- """Build the v1 ExperienceItem JSON for this episode."""
378
+ def _envelope(self, episode: Episode, blob_id: str | None = None) -> dict[str, Any]:
379
+ """Build the v1 ExperienceItem JSON for this episode.
380
+
381
+ With `blob_id` (an id returned by `upload_blob`) the content becomes
382
+ `kind="blob_ref"` — the server fetches the stored bytes and runs the
383
+ per-modality content processor on them. Otherwise plain text.
384
+ """
343
385
  scope = self._resolve_scope(episode)
344
386
  modality = (
345
387
  episode.metadata.get("modality")
@@ -354,11 +396,19 @@ class CortexConnector(ABC):
354
396
  for ent in episode.entities:
355
397
  labels.append(f"{ent.entity_type}:{ent.entity_id}")
356
398
 
357
- content: dict[str, Any] = {
358
- "kind": "message",
359
- "role": "user" if modality == "conversation" else "system",
360
- "text": self._guard_content(episode.content, episode),
361
- }
399
+ if blob_id is not None:
400
+ content: dict[str, Any] = {"kind": "blob_ref", "blob_id": blob_id}
401
+ # Filename/MIME ride labels — blob_ref content carries no text,
402
+ # and labels are the queryable metadata surface.
403
+ if episode.blob is not None:
404
+ labels.append(f"filename={episode.blob.filename}")
405
+ labels.append(f"mime={episode.blob.mime_type}")
406
+ else:
407
+ content = {
408
+ "kind": "message",
409
+ "role": "user" if modality == "conversation" else "system",
410
+ "text": self._guard_content(episode.content, episode),
411
+ }
362
412
  # The v1 write envelope is STRICT (deny_unknown_fields) as of v0.8.0:
363
413
  # thread_id / meta / preceded_by are NOT envelope context fields, and
364
414
  # sending them returns 422 INVALID_BODY (which stalls the sync cursor
@@ -387,10 +437,7 @@ class CortexConnector(ABC):
387
437
  "modality": modality,
388
438
  "content": content,
389
439
  "context": context,
390
- "idempotency_key": (
391
- episode.idempotency_key
392
- or self._fallback_idem(episode, scope)
393
- ),
440
+ "idempotency_key": self._idem_key(episode, scope),
394
441
  }
395
442
  if episode.actor and episode.actor.id:
396
443
  # Source actor ids are not guaranteed bare: Jira Cloud accountIds
@@ -414,6 +461,38 @@ class CortexConnector(ABC):
414
461
  seed = f"{scope}|{src}|{episode.id}|{episode.content[:512]}"
415
462
  return f"conn:{src}:{hashlib.sha256(seed.encode('utf-8')).hexdigest()[:24]}"
416
463
 
464
+ def _idem_key(self, episode: Episode, scope: str) -> str:
465
+ """The key actually sent — the connector's key QUALIFIED BY SCOPE.
466
+
467
+ C4: the server namespaces idempotency records by the scope ROOT (the
468
+ first path segment) plus the caller, while the envelope it hashes
469
+ carries the FULL scope. So the same source item synced into two
470
+ sibling scopes under one root re-used one record with a different
471
+ body — `409 IDEMPOTENCY_CONFLICT`, and every event after the first
472
+ scope was silently dropped. Qualifying the key with the resolved
473
+ scope makes it mean what it always claimed: "this item, in this
474
+ scope".
475
+
476
+ The scope rides as an 8-hex-char digest because the wire field is
477
+ capped at `_MAX_IDEM_BYTES` and scope paths are unbounded; a key that
478
+ can't fit the suffix is replaced by a digest of (scope, key) rather
479
+ than truncated, since a truncated key would collide across items.
480
+
481
+ Same-scope semantics are unchanged: one item still maps to exactly
482
+ one key, so re-syncs replay and poll/webhook paths (which resolve to
483
+ the same scope) still share one key. `_fallback_idem` keys already
484
+ hash the scope in, so they pass through untouched.
485
+ """
486
+ key = episode.idempotency_key
487
+ if not key:
488
+ return self._fallback_idem(episode, scope)
489
+ digest = hashlib.sha256(scope.encode("utf-8")).hexdigest()
490
+ suffix = f"@{digest[:8]}"
491
+ if len(key) + len(suffix) <= _MAX_IDEM_BYTES:
492
+ return f"{key}{suffix}"
493
+ seed = f"{scope}|{key}".encode("utf-8")
494
+ return f"cx:{hashlib.sha256(seed).hexdigest()[:40]}"
495
+
417
496
  @staticmethod
418
497
  def _guard_content(text: str, episode: Episode) -> str:
419
498
  """Hard 10 MB ceiling on content. Real items never hit this; if one
@@ -433,10 +512,168 @@ class CortexConnector(ABC):
433
512
  )
434
513
  return capped
435
514
 
515
+ def _headers(self, content_type: str = "application/json") -> dict[str, str]:
516
+ return {
517
+ "Authorization": f"Bearer {self.cortex_api_key}",
518
+ "X-Cortex-Actor": self.actor,
519
+ "Content-Type": content_type,
520
+ "User-Agent": f"cortexdb-connectors/{Source(system='base').connector_version}",
521
+ }
522
+
523
+ async def upload_blob(self, data: bytes, mime_type: str) -> str:
524
+ """POST raw bytes to `/v1/blobs`; returns the server `blob_id`."""
525
+ import httpx
526
+
527
+ async with httpx.AsyncClient(timeout=120.0) as client:
528
+ resp = await client.post(
529
+ f"{self.cortex_url}/v1/blobs",
530
+ content=data,
531
+ headers=self._headers(mime_type or "application/octet-stream"),
532
+ )
533
+ if resp.status_code >= 400:
534
+ logger.error(
535
+ "blob upload REJECTED: status=%s bytes=%d mime=%s body=%s",
536
+ resp.status_code, len(data), mime_type, resp.text[:300],
537
+ )
538
+ resp.raise_for_status()
539
+ return resp.json()["blob_id"]
540
+
541
+ async def experience_status(
542
+ self, idempotency_key: str, scope: str
543
+ ) -> dict[str, Any]:
544
+ """`GET /v1/experience/status` for a prior write with this key.
545
+
546
+ Returns the server's status document (`{found, event_id, status,
547
+ …}`), or `{"found": False}` when the probe fails — callers treat an
548
+ unreachable probe as "unknown", never as proof of absence.
549
+ """
550
+ import httpx
551
+
552
+ try:
553
+ async with httpx.AsyncClient(timeout=15.0) as client:
554
+ resp = await client.get(
555
+ f"{self.cortex_url}/v1/experience/status",
556
+ params={"idempotency_key": idempotency_key, "scope": scope},
557
+ headers=self._headers(),
558
+ )
559
+ if resp.status_code != 200:
560
+ return {"found": False}
561
+ body = resp.json()
562
+ return body if isinstance(body, dict) else {"found": False}
563
+ except Exception: # noqa: BLE001
564
+ return {"found": False}
565
+
566
+ async def experience_exists(self, idempotency_key: str, scope: str) -> bool:
567
+ """Did a write with this key already land?
568
+
569
+ Used before attachment blob uploads so a re-sync doesn't store a
570
+ fresh (orphaned) blob copy for an experience that will just replay.
571
+ Best-effort: any probe failure returns False (upload proceeds — the
572
+ experience write itself still dedups via its idempotency key).
573
+ """
574
+ status = await self.experience_status(idempotency_key, scope)
575
+ return bool(status.get("found"))
576
+
577
+ # -- v1 deletion propagation ---------------------------------------------
578
+
579
+ async def forget_events(
580
+ self, event_ids: list[str], scope: str, *, reason: str = ""
581
+ ) -> None:
582
+ """Redact specific events via `POST /v1/forget` (`memory_ids` selector).
583
+
584
+ The precise form of deletion propagation: it names exactly the events
585
+ to remove, so nothing else in the scope is touched. Used when a source
586
+ item that was ingested as its own event disappears at the source.
587
+ """
588
+ import httpx
589
+
590
+ ids = [e for e in event_ids if e]
591
+ if not ids:
592
+ return
593
+ body = {
594
+ "scope": scope,
595
+ "selector": {"memory_ids": ids},
596
+ "cascade": "redact_events",
597
+ "reason": reason or "source deletion reconciled by connector",
598
+ }
599
+ async with httpx.AsyncClient(timeout=30.0) as client:
600
+ resp = await client.post(
601
+ f"{self.cortex_url}/v1/forget", json=body, headers=self._headers()
602
+ )
603
+ if resp.status_code >= 400:
604
+ logger.error(
605
+ "forget REJECTED by server: status=%s ids=%s body=%s",
606
+ resp.status_code, ids, resp.text[:300],
607
+ )
608
+ resp.raise_for_status()
609
+
610
+ async def forget_entity(
611
+ self, entity: str, scope: str, *, reason: str = ""
612
+ ) -> None:
613
+ """Redact everything about a deleted source entity (`about_entity`).
614
+
615
+ Used when the source hard-deletes a whole item (a Jira issue, a
616
+ Notion page) and every event mentioning it must go.
617
+ """
618
+ import httpx
619
+
620
+ body = {
621
+ "scope": scope,
622
+ "selector": {"about_entity": entity},
623
+ "cascade": "redact_events",
624
+ "reason": reason or f"source deleted {entity}",
625
+ }
626
+ async with httpx.AsyncClient(timeout=30.0) as client:
627
+ resp = await client.post(
628
+ f"{self.cortex_url}/v1/forget", json=body, headers=self._headers()
629
+ )
630
+ if resp.status_code >= 400:
631
+ logger.error(
632
+ "forget REJECTED by server: status=%s entity=%s body=%s",
633
+ resp.status_code, entity, resp.text[:300],
634
+ )
635
+ resp.raise_for_status()
636
+
637
+ async def delete_blob(self, blob_id: str) -> bool:
638
+ """`DELETE /v1/blobs/{id}` — purge uploaded bytes.
639
+
640
+ Redacting the blob_ref experience removes the memory; it does not
641
+ remove the stored object. A source-side file deletion has to purge
642
+ BOTH, or the deleted file stays byte-retrievable from CortexDB.
643
+ Returns True when the bytes are gone (deleted now, or already
644
+ absent — 404 is success for a delete). Raises on anything else so
645
+ the caller reports a failed purge instead of assuming one.
646
+ """
647
+ import httpx
648
+
649
+ if not blob_id:
650
+ return False
651
+ async with httpx.AsyncClient(timeout=30.0) as client:
652
+ resp = await client.delete(
653
+ f"{self.cortex_url}/v1/blobs/{blob_id}", headers=self._headers()
654
+ )
655
+ if resp.status_code == 404:
656
+ return True
657
+ if resp.status_code >= 400:
658
+ logger.error(
659
+ "blob delete REJECTED by server: status=%s blob_id=%s body=%s",
660
+ resp.status_code, blob_id, resp.text[:300],
661
+ )
662
+ resp.raise_for_status()
663
+ return True
664
+
436
665
  async def ingest_episode(self, episode: Episode) -> str:
437
666
  """POST `/v1/experience` with the v1 envelope.
438
667
 
439
- Returns the server-side event_id on success.
668
+ Blob-carrying episodes (`episode.blob` set with bytes) first upload
669
+ the bytes to `/v1/blobs` and go in as `content.kind="blob_ref"` so the
670
+ server's content processors extract them. A status probe on the
671
+ episode's idempotency key skips the upload when the experience already
672
+ exists (re-sync), avoiding orphaned duplicate blobs. Bytes-less blob
673
+ episodes (oversized skips) fall back to their text content.
674
+
675
+ Returns the server-side event_id on success ("exists" when the
676
+ status probe short-circuited a replay).
440
677
  """
441
678
  import httpx
442
679
 
@@ -452,13 +689,24 @@ class CortexConnector(ABC):
452
689
  "explicitly. The actor must match the token's `sub` claim."
453
690
  )
454
691
 
455
- envelope = self._envelope(episode)
456
- headers = {
457
- "Authorization": f"Bearer {self.cortex_api_key}",
458
- "X-Cortex-Actor": self.actor,
459
- "Content-Type": "application/json",
460
- "User-Agent": f"cortexdb-connectors/{Source(system='base').connector_version}",
461
- }
692
+ scope = self._resolve_scope(episode)
693
+ key = self._idem_key(episode, scope)
694
+ blob_id: str | None = None
695
+ if episode.blob is not None and episode.blob.data:
696
+ if await self.experience_exists(key, scope):
697
+ logger.info(
698
+ "attachment already ingested (idempotency_key=%s); "
699
+ "skipping blob upload",
700
+ key,
701
+ )
702
+ return "exists"
703
+ blob_id = await self.upload_blob(
704
+ episode.blob.data, episode.blob.mime_type
705
+ )
706
+ self._uploaded_blobs[key] = blob_id
707
+
708
+ envelope = self._envelope(episode, blob_id=blob_id)
709
+ headers = self._headers()
462
710
 
463
711
  async with httpx.AsyncClient(timeout=30.0) as client:
464
712
  resp = await client.post(
@@ -480,4 +728,12 @@ class CortexConnector(ABC):
480
728
  )
481
729
  resp.raise_for_status()
482
730
  body = resp.json()
483
- return body.get("event_id") or body.get("id") or episode.id
731
+ event_id = body.get("event_id") or body.get("id")
732
+ if event_id:
733
+ # Remembered so a connector that reconciles source deletions can
734
+ # name this exact event later. The server's idempotency records
735
+ # expire after 24h, so a status probe cannot resolve the id of
736
+ # something ingested days ago — which is exactly when a source
737
+ # deletion shows up.
738
+ self._ingested_events[key] = event_id
739
+ return event_id or episode.id
@@ -80,7 +80,12 @@ CONNECTORS: dict[str, dict[str, Any]] = {
80
80
  ("jira_email", "JIRA_EMAIL"),
81
81
  ("jira_api_token", "JIRA_API_TOKEN"),
82
82
  ],
83
- "options": [("project_keys", "JIRA_PROJECT_KEYS", "list")],
83
+ "options": [
84
+ ("project_keys", "JIRA_PROJECT_KEYS", "list"),
85
+ ("reconcile_deletions", "JIRA_RECONCILE_DELETIONS", "bool"),
86
+ ("reconcile_interval_seconds", "JIRA_RECONCILE_INTERVAL_SECONDS", "int"),
87
+ ("ledger_path", "JIRA_LEDGER_PATH", "str"),
88
+ ],
84
89
  },
85
90
  "freshdesk": {
86
91
  "import": "cortexdb_connectors.freshdesk:FreshdeskConnector",
@@ -260,6 +265,7 @@ def _load_cortex_creds(args: argparse.Namespace) -> dict[str, str]:
260
265
 
261
266
  _OPTION_PARSERS: dict[str, Callable[[str], Any]] = {
262
267
  "str": lambda v: v,
268
+ "int": int,
263
269
  "list": lambda v: [x.strip() for x in v.split(",") if x.strip()],
264
270
  "list-int": lambda v: [int(x.strip()) for x in v.split(",") if x.strip()],
265
271
  "bool": lambda v: v.lower() in ("1", "true", "yes", "on"),