bridgekit 0.3.9__tar.gz → 0.3.10__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. {bridgekit-0.3.9 → bridgekit-0.3.10}/PKG-INFO +72 -4
  2. {bridgekit-0.3.9 → bridgekit-0.3.10}/README.md +71 -3
  3. {bridgekit-0.3.9 → bridgekit-0.3.10}/bridgekit/__init__.py +3 -2
  4. bridgekit-0.3.10/bridgekit/summarize.py +72 -0
  5. {bridgekit-0.3.9 → bridgekit-0.3.10}/bridgekit.egg-info/PKG-INFO +72 -4
  6. {bridgekit-0.3.9 → bridgekit-0.3.10}/bridgekit.egg-info/SOURCES.txt +3 -1
  7. {bridgekit-0.3.9 → bridgekit-0.3.10}/pyproject.toml +1 -1
  8. bridgekit-0.3.10/tests/test_summarize.py +212 -0
  9. {bridgekit-0.3.9 → bridgekit-0.3.10}/LICENSE +0 -0
  10. {bridgekit-0.3.9 → bridgekit-0.3.10}/bridgekit/cli.py +0 -0
  11. {bridgekit-0.3.9 → bridgekit-0.3.10}/bridgekit/compare.py +0 -0
  12. {bridgekit-0.3.9 → bridgekit-0.3.10}/bridgekit/config.py +0 -0
  13. {bridgekit-0.3.9 → bridgekit-0.3.10}/bridgekit/planner.py +0 -0
  14. {bridgekit-0.3.9 → bridgekit-0.3.10}/bridgekit/providers.py +0 -0
  15. {bridgekit-0.3.9 → bridgekit-0.3.10}/bridgekit/redteam.py +0 -0
  16. {bridgekit-0.3.9 → bridgekit-0.3.10}/bridgekit/reviewer.py +0 -0
  17. {bridgekit-0.3.9 → bridgekit-0.3.10}/bridgekit/search.py +0 -0
  18. {bridgekit-0.3.9 → bridgekit-0.3.10}/bridgekit.egg-info/dependency_links.txt +0 -0
  19. {bridgekit-0.3.9 → bridgekit-0.3.10}/bridgekit.egg-info/entry_points.txt +0 -0
  20. {bridgekit-0.3.9 → bridgekit-0.3.10}/bridgekit.egg-info/requires.txt +0 -0
  21. {bridgekit-0.3.9 → bridgekit-0.3.10}/bridgekit.egg-info/top_level.txt +0 -0
  22. {bridgekit-0.3.9 → bridgekit-0.3.10}/setup.cfg +0 -0
  23. {bridgekit-0.3.9 → bridgekit-0.3.10}/tests/test_cli.py +0 -0
  24. {bridgekit-0.3.9 → bridgekit-0.3.10}/tests/test_compare.py +0 -0
  25. {bridgekit-0.3.9 → bridgekit-0.3.10}/tests/test_config.py +0 -0
  26. {bridgekit-0.3.9 → bridgekit-0.3.10}/tests/test_planner.py +0 -0
  27. {bridgekit-0.3.9 → bridgekit-0.3.10}/tests/test_providers.py +0 -0
  28. {bridgekit-0.3.9 → bridgekit-0.3.10}/tests/test_redteam.py +0 -0
  29. {bridgekit-0.3.9 → bridgekit-0.3.10}/tests/test_reviewer.py +0 -0
  30. {bridgekit-0.3.9 → bridgekit-0.3.10}/tests/test_search.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: bridgekit
3
- Version: 0.3.9
3
+ Version: 0.3.10
4
4
  Summary: AI tools that make you a better data scientist, not a redundant one.
5
5
  License: MIT
6
6
  Project-URL: Homepage, https://usebridgekit.com
@@ -398,7 +398,7 @@ willing to commit to — and what's your confidence interval on that estimate?"
398
398
 
399
399
  ## Tool #5: Compare
400
400
 
401
- Run the same tool through two providers and see both outputs side by side. Useful for evaluating which model works best for your use case — as a one-liner.
401
+ Run the same tool through two providers and see both outputs side by side. A synthesis summary at the top highlights where the models agreed, where they differed in severity, and which gave more actionable feedback — so you get the key insight without reading both outputs in full. Useful for evaluating which model works best for your use case — as a one-liner.
402
402
 
403
403
  ```python
404
404
  from bridgekit import compare
@@ -437,6 +437,16 @@ print(compare(text, providers=["anthropic", "gemini"]))
437
437
  BRIDGEKIT COMPARE: EVALUATE
438
438
  ─────────────────────────────────────────
439
439
 
440
+ SUMMARY
441
+ ─────────────────────────────────────────
442
+ Both outputs rated Clarity as STRONG. They diverged on severity: Anthropic
443
+ rated Statistical Rigor as MISSING (harsher) while OpenAI called it NEEDS WORK.
444
+ Anthropic gave more specific feedback — naming the correlation-vs-causation
445
+ problem explicitly and suggesting concrete fixes. OpenAI's feedback stayed
446
+ more generic. Anthropic's bottom line targets the core analytical flaw;
447
+ OpenAI's restates the statistical point only.
448
+
449
+
440
450
  ANTHROPIC claude-opus-4-8
441
451
  ─────────────────────────────────────────
442
452
  BRIDGEKIT ANALYSIS REVIEW
@@ -458,7 +468,61 @@ BRIDGEKIT ANALYSIS REVIEW
458
468
  ...
459
469
  ```
460
470
 
461
- Both providers are called in parallel, so the total wait time is the slower of the two — not the sum.
471
+ Both providers are called in parallel, so the total wait time is the slower of the two — not the sum. The synthesis summary is a third sequential call made after both outputs are ready.
472
+
473
+ ---
474
+
475
+ ## Tool #6: Summarize
476
+
477
+ Turn a long analysis writeup, notebook, or report into a short executive summary — the gap between doing the analysis and communicating it to people who won't read the whole thing.
478
+
479
+ ```python
480
+ from bridgekit import summarize
481
+
482
+ text = """
483
+ I analyzed 90 days of user behavior data to understand what drives subscription
484
+ upgrades. Users who engaged with the reporting feature within their first week
485
+ were 3x more likely to upgrade within 30 days. Sample size was 1,200 users
486
+ across two acquisition channels, with consistent results in both. I recommend
487
+ we prioritize onboarding users to reporting as a growth lever.
488
+ """
489
+
490
+ # Default — general business audience
491
+ print(summarize(text))
492
+
493
+ # Or specify an audience
494
+ print(summarize(text, audience="VP of Marketing"))
495
+ print(summarize(text, audience="board"))
496
+
497
+ # Override for longer summaries
498
+ print(summarize(text, max_tokens=2048))
499
+ ```
500
+
501
+ **Output:**
502
+ ```
503
+ BRIDGEKIT SUMMARY
504
+ ─────────────────────────────────────────
505
+ AUDIENCE: VP of Marketing
506
+
507
+ KEY TAKEAWAY
508
+ Getting users into the reporting feature in their first week is our strongest
509
+ predictor of paid upgrades — users who engage with it are 3x more likely to
510
+ convert within 30 days.
511
+
512
+ WHAT WE FOUND
513
+ - Early reporting engagement (week 1) drives 3x higher upgrade rates within 30 days
514
+ - Finding holds across both acquisition channels, suggesting it's a genuine
515
+ behavior pattern, not channel-specific
516
+ - Analyzed 1,200 users over 90 days with sufficient scale to trust the result
517
+
518
+ SO WHAT
519
+ We should redesign onboarding to get users to reporting faster. This is a
520
+ high-confidence growth lever worth testing immediately.
521
+
522
+ ─────────────────────────────────────────
523
+ ```
524
+
525
+ `audience`, `provider`, `model`, `system_prompt`, and `max_tokens` are all optional — the more specific the audience, the more tailored the summary.
462
526
 
463
527
  ---
464
528
 
@@ -498,6 +562,7 @@ All tools support the same `provider` and `model` parameters:
498
562
  - `plan(question, provider=None, model=None, ..., system_prompt=None)`
499
563
  - `ask(question, provider=None, model=None, ..., system_prompt=None)`
500
564
  - `redteam(text, provider=None, model=None, ..., system_prompt=None)`
565
+ - `summarize(text, provider=None, model=None, ..., system_prompt=None)`
501
566
 
502
567
  ---
503
568
 
@@ -506,7 +571,7 @@ All tools support the same `provider` and `model` parameters:
506
571
  Every tool accepts an optional `system_prompt` parameter to override the default persona. Use this to adapt the tone or focus to a specific domain without changing anything else.
507
572
 
508
573
  ```python
509
- from bridgekit import evaluate, plan, ask, redteam
574
+ from bridgekit import evaluate, plan, ask, redteam, summarize
510
575
 
511
576
  # Narrow the reviewer to a specific domain
512
577
  print(evaluate("my analysis", system_prompt="You are a skeptical PhD statistician focused only on methodology"))
@@ -519,6 +584,9 @@ print(redteam("my analysis", system_prompt="You are a hostile regulator looking
519
584
 
520
585
  # Change the answering style for ask
521
586
  print(ask("my question", text="...", system_prompt="You are a financial analyst. Answer only in terms of revenue impact."))
587
+
588
+ # Replace the summarizer persona entirely
589
+ print(summarize("my analysis", system_prompt="You are a data journalist writing a one-paragraph news brief."))
522
590
  ```
523
591
 
524
592
  When `system_prompt` is not provided, each tool uses its built-in default — existing behavior is unchanged.
@@ -366,7 +366,7 @@ willing to commit to — and what's your confidence interval on that estimate?"
366
366
 
367
367
  ## Tool #5: Compare
368
368
 
369
- Run the same tool through two providers and see both outputs side by side. Useful for evaluating which model works best for your use case — as a one-liner.
369
+ Run the same tool through two providers and see both outputs side by side. A synthesis summary at the top highlights where the models agreed, where they differed in severity, and which gave more actionable feedback — so you get the key insight without reading both outputs in full. Useful for evaluating which model works best for your use case — as a one-liner.
370
370
 
371
371
  ```python
372
372
  from bridgekit import compare
@@ -405,6 +405,16 @@ print(compare(text, providers=["anthropic", "gemini"]))
405
405
  BRIDGEKIT COMPARE: EVALUATE
406
406
  ─────────────────────────────────────────
407
407
 
408
+ SUMMARY
409
+ ─────────────────────────────────────────
410
+ Both outputs rated Clarity as STRONG. They diverged on severity: Anthropic
411
+ rated Statistical Rigor as MISSING (harsher) while OpenAI called it NEEDS WORK.
412
+ Anthropic gave more specific feedback — naming the correlation-vs-causation
413
+ problem explicitly and suggesting concrete fixes. OpenAI's feedback stayed
414
+ more generic. Anthropic's bottom line targets the core analytical flaw;
415
+ OpenAI's restates the statistical point only.
416
+
417
+
408
418
  ANTHROPIC claude-opus-4-8
409
419
  ─────────────────────────────────────────
410
420
  BRIDGEKIT ANALYSIS REVIEW
@@ -426,7 +436,61 @@ BRIDGEKIT ANALYSIS REVIEW
426
436
  ...
427
437
  ```
428
438
 
429
- Both providers are called in parallel, so the total wait time is the slower of the two — not the sum.
439
+ Both providers are called in parallel, so the total wait time is the slower of the two — not the sum. The synthesis summary is a third sequential call made after both outputs are ready.
440
+
441
+ ---
442
+
443
+ ## Tool #6: Summarize
444
+
445
+ Turn a long analysis writeup, notebook, or report into a short executive summary — the gap between doing the analysis and communicating it to people who won't read the whole thing.
446
+
447
+ ```python
448
+ from bridgekit import summarize
449
+
450
+ text = """
451
+ I analyzed 90 days of user behavior data to understand what drives subscription
452
+ upgrades. Users who engaged with the reporting feature within their first week
453
+ were 3x more likely to upgrade within 30 days. Sample size was 1,200 users
454
+ across two acquisition channels, with consistent results in both. I recommend
455
+ we prioritize onboarding users to reporting as a growth lever.
456
+ """
457
+
458
+ # Default — general business audience
459
+ print(summarize(text))
460
+
461
+ # Or specify an audience
462
+ print(summarize(text, audience="VP of Marketing"))
463
+ print(summarize(text, audience="board"))
464
+
465
+ # Override for longer summaries
466
+ print(summarize(text, max_tokens=2048))
467
+ ```
468
+
469
+ **Output:**
470
+ ```
471
+ BRIDGEKIT SUMMARY
472
+ ─────────────────────────────────────────
473
+ AUDIENCE: VP of Marketing
474
+
475
+ KEY TAKEAWAY
476
+ Getting users into the reporting feature in their first week is our strongest
477
+ predictor of paid upgrades — users who engage with it are 3x more likely to
478
+ convert within 30 days.
479
+
480
+ WHAT WE FOUND
481
+ - Early reporting engagement (week 1) drives 3x higher upgrade rates within 30 days
482
+ - Finding holds across both acquisition channels, suggesting it's a genuine
483
+ behavior pattern, not channel-specific
484
+ - Analyzed 1,200 users over 90 days with sufficient scale to trust the result
485
+
486
+ SO WHAT
487
+ We should redesign onboarding to get users to reporting faster. This is a
488
+ high-confidence growth lever worth testing immediately.
489
+
490
+ ─────────────────────────────────────────
491
+ ```
492
+
493
+ `audience`, `provider`, `model`, `system_prompt`, and `max_tokens` are all optional — the more specific the audience, the more tailored the summary.
430
494
 
431
495
  ---
432
496
 
@@ -466,6 +530,7 @@ All tools support the same `provider` and `model` parameters:
466
530
  - `plan(question, provider=None, model=None, ..., system_prompt=None)`
467
531
  - `ask(question, provider=None, model=None, ..., system_prompt=None)`
468
532
  - `redteam(text, provider=None, model=None, ..., system_prompt=None)`
533
+ - `summarize(text, provider=None, model=None, ..., system_prompt=None)`
469
534
 
470
535
  ---
471
536
 
@@ -474,7 +539,7 @@ All tools support the same `provider` and `model` parameters:
474
539
  Every tool accepts an optional `system_prompt` parameter to override the default persona. Use this to adapt the tone or focus to a specific domain without changing anything else.
475
540
 
476
541
  ```python
477
- from bridgekit import evaluate, plan, ask, redteam
542
+ from bridgekit import evaluate, plan, ask, redteam, summarize
478
543
 
479
544
  # Narrow the reviewer to a specific domain
480
545
  print(evaluate("my analysis", system_prompt="You are a skeptical PhD statistician focused only on methodology"))
@@ -487,6 +552,9 @@ print(redteam("my analysis", system_prompt="You are a hostile regulator looking
487
552
 
488
553
  # Change the answering style for ask
489
554
  print(ask("my question", text="...", system_prompt="You are a financial analyst. Answer only in terms of revenue impact."))
555
+
556
+ # Replace the summarizer persona entirely
557
+ print(summarize("my analysis", system_prompt="You are a data journalist writing a one-paragraph news brief."))
490
558
  ```
491
559
 
492
560
  When `system_prompt` is not provided, each tool uses its built-in default — existing behavior is unchanged.
@@ -3,6 +3,7 @@ from .search import ask
3
3
  from .planner import plan
4
4
  from .redteam import redteam
5
5
  from .compare import compare
6
+ from .summarize import summarize
6
7
 
7
- __version__ = "0.3.9"
8
- __all__ = ["evaluate", "ask", "plan", "redteam", "compare"]
8
+ __version__ = "0.3.10"
9
+ __all__ = ["evaluate", "ask", "plan", "redteam", "compare", "summarize"]
@@ -0,0 +1,72 @@
1
+ from .config import DEFAULT_MODEL, parse_provider, get_default_model
2
+ from .providers import create_message
3
+
4
+ DEFAULT_AUDIENCE = "a general business audience with no technical or data science background"
5
+
6
+ SYSTEM_PROMPT_TEMPLATE = """You are a senior data scientist preparing an executive summary of a technical analysis for {audience}.
7
+
8
+ Your job is to translate technical detail into what matters for decision-making. Cut jargon, cut methodology detail unless it's essential to trust the conclusion, and lead with the takeaway. Write for someone who will skim this in 30 seconds.
9
+
10
+ Format your response exactly like this:
11
+
12
+ BRIDGEKIT SUMMARY
13
+ ─────────────────────────────────────────
14
+ AUDIENCE: {audience_label}
15
+
16
+ KEY TAKEAWAY
17
+ [1-2 sentences: the single most important finding or recommendation]
18
+
19
+ WHAT WE FOUND
20
+ [3-5 bullet points of the supporting findings, in plain language]
21
+
22
+ SO WHAT
23
+ [1-2 sentences on the business implication or recommended next step]
24
+
25
+ ─────────────────────────────────────────
26
+ """
27
+
28
+
29
+ def summarize(text: str, audience: str = None, provider: str = None, model: str = None, system_prompt: str = None, max_tokens: int = 1024) -> str:
30
+ """
31
+ Turn a long analysis, notebook, or report into a short executive summary.
32
+
33
+ Args:
34
+ text: The long-form analysis writeup, notebook output, or report as a plain string.
35
+ audience: Optional. Who the summary is for (e.g. "VP of Marketing", "board",
36
+ "engineering team"). Defaults to a general business audience.
37
+ provider: Optional. The AI provider to use ("anthropic", "openai", "gemini").
38
+ If not specified, defaults to "anthropic" or infers from model.
39
+ model: Optional. The specific model to use. If not specified, uses the provider's default.
40
+ system_prompt: Optional. A custom system prompt to fully override the default summarizer persona.
41
+ When provided, the audience parameter is ignored.
42
+ max_tokens: Optional. Maximum tokens in the response. Defaults to 1024.
43
+
44
+ Returns:
45
+ A short executive summary covering the key takeaway, supporting findings,
46
+ and the business implication.
47
+ """
48
+ if not text or not text.strip():
49
+ raise ValueError("Text cannot be empty.")
50
+
51
+ # Parse provider and determine model
52
+ provider_enum = parse_provider(provider, model)
53
+ if model is None:
54
+ model = get_default_model(provider_enum)
55
+
56
+ if system_prompt is None:
57
+ audience_label = audience if audience else "General Business Audience"
58
+ audience_desc = audience if audience else DEFAULT_AUDIENCE
59
+ system_prompt = SYSTEM_PROMPT_TEMPLATE.format(
60
+ audience=audience_desc,
61
+ audience_label=audience_label
62
+ )
63
+
64
+ user_message = f"Summarize this analysis:\n\n{text}"
65
+
66
+ return create_message(
67
+ provider=provider_enum,
68
+ system_prompt=system_prompt,
69
+ user_message=user_message,
70
+ model=model,
71
+ max_tokens=max_tokens
72
+ )
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: bridgekit
3
- Version: 0.3.9
3
+ Version: 0.3.10
4
4
  Summary: AI tools that make you a better data scientist, not a redundant one.
5
5
  License: MIT
6
6
  Project-URL: Homepage, https://usebridgekit.com
@@ -398,7 +398,7 @@ willing to commit to — and what's your confidence interval on that estimate?"
398
398
 
399
399
  ## Tool #5: Compare
400
400
 
401
- Run the same tool through two providers and see both outputs side by side. Useful for evaluating which model works best for your use case — as a one-liner.
401
+ Run the same tool through two providers and see both outputs side by side. A synthesis summary at the top highlights where the models agreed, where they differed in severity, and which gave more actionable feedback — so you get the key insight without reading both outputs in full. Useful for evaluating which model works best for your use case — as a one-liner.
402
402
 
403
403
  ```python
404
404
  from bridgekit import compare
@@ -437,6 +437,16 @@ print(compare(text, providers=["anthropic", "gemini"]))
437
437
  BRIDGEKIT COMPARE: EVALUATE
438
438
  ─────────────────────────────────────────
439
439
 
440
+ SUMMARY
441
+ ─────────────────────────────────────────
442
+ Both outputs rated Clarity as STRONG. They diverged on severity: Anthropic
443
+ rated Statistical Rigor as MISSING (harsher) while OpenAI called it NEEDS WORK.
444
+ Anthropic gave more specific feedback — naming the correlation-vs-causation
445
+ problem explicitly and suggesting concrete fixes. OpenAI's feedback stayed
446
+ more generic. Anthropic's bottom line targets the core analytical flaw;
447
+ OpenAI's restates the statistical point only.
448
+
449
+
440
450
  ANTHROPIC claude-opus-4-8
441
451
  ─────────────────────────────────────────
442
452
  BRIDGEKIT ANALYSIS REVIEW
@@ -458,7 +468,61 @@ BRIDGEKIT ANALYSIS REVIEW
458
468
  ...
459
469
  ```
460
470
 
461
- Both providers are called in parallel, so the total wait time is the slower of the two — not the sum.
471
+ Both providers are called in parallel, so the total wait time is the slower of the two — not the sum. The synthesis summary is a third sequential call made after both outputs are ready.
472
+
473
+ ---
474
+
475
+ ## Tool #6: Summarize
476
+
477
+ Turn a long analysis writeup, notebook, or report into a short executive summary — the gap between doing the analysis and communicating it to people who won't read the whole thing.
478
+
479
+ ```python
480
+ from bridgekit import summarize
481
+
482
+ text = """
483
+ I analyzed 90 days of user behavior data to understand what drives subscription
484
+ upgrades. Users who engaged with the reporting feature within their first week
485
+ were 3x more likely to upgrade within 30 days. Sample size was 1,200 users
486
+ across two acquisition channels, with consistent results in both. I recommend
487
+ we prioritize onboarding users to reporting as a growth lever.
488
+ """
489
+
490
+ # Default — general business audience
491
+ print(summarize(text))
492
+
493
+ # Or specify an audience
494
+ print(summarize(text, audience="VP of Marketing"))
495
+ print(summarize(text, audience="board"))
496
+
497
+ # Override for longer summaries
498
+ print(summarize(text, max_tokens=2048))
499
+ ```
500
+
501
+ **Output:**
502
+ ```
503
+ BRIDGEKIT SUMMARY
504
+ ─────────────────────────────────────────
505
+ AUDIENCE: VP of Marketing
506
+
507
+ KEY TAKEAWAY
508
+ Getting users into the reporting feature in their first week is our strongest
509
+ predictor of paid upgrades — users who engage with it are 3x more likely to
510
+ convert within 30 days.
511
+
512
+ WHAT WE FOUND
513
+ - Early reporting engagement (week 1) drives 3x higher upgrade rates within 30 days
514
+ - Finding holds across both acquisition channels, suggesting it's a genuine
515
+ behavior pattern, not channel-specific
516
+ - Analyzed 1,200 users over 90 days with sufficient scale to trust the result
517
+
518
+ SO WHAT
519
+ We should redesign onboarding to get users to reporting faster. This is a
520
+ high-confidence growth lever worth testing immediately.
521
+
522
+ ─────────────────────────────────────────
523
+ ```
524
+
525
+ `audience`, `provider`, `model`, `system_prompt`, and `max_tokens` are all optional — the more specific the audience, the more tailored the summary.
462
526
 
463
527
  ---
464
528
 
@@ -498,6 +562,7 @@ All tools support the same `provider` and `model` parameters:
498
562
  - `plan(question, provider=None, model=None, ..., system_prompt=None)`
499
563
  - `ask(question, provider=None, model=None, ..., system_prompt=None)`
500
564
  - `redteam(text, provider=None, model=None, ..., system_prompt=None)`
565
+ - `summarize(text, provider=None, model=None, ..., system_prompt=None)`
501
566
 
502
567
  ---
503
568
 
@@ -506,7 +571,7 @@ All tools support the same `provider` and `model` parameters:
506
571
  Every tool accepts an optional `system_prompt` parameter to override the default persona. Use this to adapt the tone or focus to a specific domain without changing anything else.
507
572
 
508
573
  ```python
509
- from bridgekit import evaluate, plan, ask, redteam
574
+ from bridgekit import evaluate, plan, ask, redteam, summarize
510
575
 
511
576
  # Narrow the reviewer to a specific domain
512
577
  print(evaluate("my analysis", system_prompt="You are a skeptical PhD statistician focused only on methodology"))
@@ -519,6 +584,9 @@ print(redteam("my analysis", system_prompt="You are a hostile regulator looking
519
584
 
520
585
  # Change the answering style for ask
521
586
  print(ask("my question", text="...", system_prompt="You are a financial analyst. Answer only in terms of revenue impact."))
587
+
588
+ # Replace the summarizer persona entirely
589
+ print(summarize("my analysis", system_prompt="You are a data journalist writing a one-paragraph news brief."))
522
590
  ```
523
591
 
524
592
  When `system_prompt` is not provided, each tool uses its built-in default — existing behavior is unchanged.
@@ -10,6 +10,7 @@ bridgekit/providers.py
10
10
  bridgekit/redteam.py
11
11
  bridgekit/reviewer.py
12
12
  bridgekit/search.py
13
+ bridgekit/summarize.py
13
14
  bridgekit.egg-info/PKG-INFO
14
15
  bridgekit.egg-info/SOURCES.txt
15
16
  bridgekit.egg-info/dependency_links.txt
@@ -23,4 +24,5 @@ tests/test_planner.py
23
24
  tests/test_providers.py
24
25
  tests/test_redteam.py
25
26
  tests/test_reviewer.py
26
- tests/test_search.py
27
+ tests/test_search.py
28
+ tests/test_summarize.py
@@ -7,7 +7,7 @@ include = ["bridgekit*"]
7
7
 
8
8
  [project]
9
9
  name = "bridgekit"
10
- version = "0.3.9"
10
+ version = "0.3.10"
11
11
  description = "AI tools that make you a better data scientist, not a redundant one."
12
12
  readme = "README.md"
13
13
  requires-python = ">=3.9"
@@ -0,0 +1,212 @@
1
+ import os
2
+ import pytest
3
+ from unittest.mock import MagicMock, patch
4
+
5
+
6
+ # ---------------------------------------------------------------------------
7
+ # Helpers
8
+ # ---------------------------------------------------------------------------
9
+
10
+ def _make_mock_message(text: str):
11
+ content_block = MagicMock()
12
+ content_block.text = text
13
+ message = MagicMock()
14
+ message.content = [content_block]
15
+ return message
16
+
17
+
18
+ FAKE_RESPONSE = (
19
+ "BRIDGEKIT SUMMARY\n"
20
+ "─────────────────────────────────────────\n"
21
+ "AUDIENCE: General Business Audience\n\n"
22
+ "KEY TAKEAWAY\n"
23
+ "The new onboarding flow increased upgrade rates by 3x within 30 days.\n\n"
24
+ "WHAT WE FOUND\n"
25
+ "- Users who engaged with reporting in week 1 were 3x more likely to upgrade\n"
26
+ "- The effect held across acquisition channels\n"
27
+ "- Sample size was 500 users\n\n"
28
+ "SO WHAT\n"
29
+ "Prioritize onboarding users to the reporting feature as a growth lever.\n"
30
+ "─────────────────────────────────────────\n"
31
+ )
32
+
33
+
34
+ # ---------------------------------------------------------------------------
35
+ # Tests
36
+ # ---------------------------------------------------------------------------
37
+
38
+ class TestSummarizeReturnsString:
39
+ """summarize() should return a non-empty string."""
40
+
41
+ def test_returns_string(self):
42
+ with patch.dict(os.environ, {"ANTHROPIC_API_KEY": "test-key"}):
43
+ with patch("anthropic.Anthropic") as MockAnthropic:
44
+ mock_client = MagicMock()
45
+ mock_client.messages.create.return_value = _make_mock_message(FAKE_RESPONSE)
46
+ MockAnthropic.return_value = mock_client
47
+
48
+ from bridgekit.summarize import summarize
49
+ result = summarize("We ran an A/B test on 500 users and saw a 3x lift in upgrades.")
50
+
51
+ assert isinstance(result, str)
52
+ assert len(result) > 0
53
+
54
+
55
+ class TestSummarizeOutputStructure:
56
+ """summarize() output should contain the required section headers."""
57
+
58
+ def test_output_contains_key_takeaway(self):
59
+ with patch.dict(os.environ, {"ANTHROPIC_API_KEY": "test-key"}):
60
+ with patch("anthropic.Anthropic") as MockAnthropic:
61
+ mock_client = MagicMock()
62
+ mock_client.messages.create.return_value = _make_mock_message(FAKE_RESPONSE)
63
+ MockAnthropic.return_value = mock_client
64
+
65
+ from bridgekit.summarize import summarize
66
+ result = summarize("Some analysis text.")
67
+
68
+ assert "KEY TAKEAWAY" in result
69
+
70
+ def test_output_contains_so_what(self):
71
+ with patch.dict(os.environ, {"ANTHROPIC_API_KEY": "test-key"}):
72
+ with patch("anthropic.Anthropic") as MockAnthropic:
73
+ mock_client = MagicMock()
74
+ mock_client.messages.create.return_value = _make_mock_message(FAKE_RESPONSE)
75
+ MockAnthropic.return_value = mock_client
76
+
77
+ from bridgekit.summarize import summarize
78
+ result = summarize("Some analysis text.")
79
+
80
+ assert "SO WHAT" in result
81
+
82
+
83
+ class TestSummarizeMissingApiKey:
84
+ """summarize() should raise EnvironmentError when the API key is absent."""
85
+
86
+ def test_raises_environment_error_when_key_missing(self):
87
+ env = {k: v for k, v in os.environ.items() if k != "ANTHROPIC_API_KEY"}
88
+ with patch.dict(os.environ, env, clear=True):
89
+ from bridgekit.summarize import summarize
90
+ with pytest.raises(EnvironmentError):
91
+ summarize("Some analysis text.")
92
+
93
+ def test_error_message_mentions_key(self):
94
+ env = {k: v for k, v in os.environ.items() if k != "ANTHROPIC_API_KEY"}
95
+ with patch.dict(os.environ, env, clear=True):
96
+ from bridgekit.summarize import summarize
97
+ with pytest.raises(EnvironmentError, match="ANTHROPIC_API_KEY"):
98
+ summarize("Some analysis text.")
99
+
100
+
101
+ class TestSummarizeEmptyInput:
102
+ """summarize() should raise ValueError for empty or whitespace-only input."""
103
+
104
+ def test_empty_string_raises_value_error(self):
105
+ with patch.dict(os.environ, {"ANTHROPIC_API_KEY": "test-key"}):
106
+ from bridgekit.summarize import summarize
107
+ with pytest.raises(ValueError, match="empty"):
108
+ summarize("")
109
+
110
+ def test_whitespace_only_raises_value_error(self):
111
+ with patch.dict(os.environ, {"ANTHROPIC_API_KEY": "test-key"}):
112
+ from bridgekit.summarize import summarize
113
+ with pytest.raises(ValueError, match="empty"):
114
+ summarize(" ")
115
+
116
+
117
+ class TestSummarizeAudience:
118
+ """summarize() should include a custom audience in the system prompt."""
119
+
120
+ def test_custom_audience_reaches_system_prompt(self):
121
+ with patch.dict(os.environ, {"ANTHROPIC_API_KEY": "test-key"}):
122
+ with patch("anthropic.Anthropic") as MockAnthropic:
123
+ mock_client = MagicMock()
124
+ mock_client.messages.create.return_value = _make_mock_message(FAKE_RESPONSE)
125
+ MockAnthropic.return_value = mock_client
126
+
127
+ from bridgekit.summarize import summarize
128
+ summarize("Some analysis text.", audience="VP of Marketing")
129
+
130
+ call_kwargs = mock_client.messages.create.call_args
131
+ assert "VP of Marketing" in call_kwargs.kwargs.get("system", "")
132
+
133
+ def test_default_audience_used_when_not_specified(self):
134
+ with patch.dict(os.environ, {"ANTHROPIC_API_KEY": "test-key"}):
135
+ with patch("anthropic.Anthropic") as MockAnthropic:
136
+ mock_client = MagicMock()
137
+ mock_client.messages.create.return_value = _make_mock_message(FAKE_RESPONSE)
138
+ MockAnthropic.return_value = mock_client
139
+
140
+ from bridgekit.summarize import summarize
141
+ summarize("Some analysis text.")
142
+
143
+ call_kwargs = mock_client.messages.create.call_args
144
+ assert "General Business Audience" in call_kwargs.kwargs.get("system", "")
145
+
146
+
147
+ class TestSummarizeCustomSystemPrompt:
148
+ """summarize() should forward a custom system_prompt to the API, ignoring audience."""
149
+
150
+ def test_custom_system_prompt_reaches_api(self):
151
+ custom_prompt = "You are a data journalist writing a one-paragraph news brief."
152
+ with patch.dict(os.environ, {"ANTHROPIC_API_KEY": "test-key"}):
153
+ with patch("anthropic.Anthropic") as MockAnthropic:
154
+ mock_client = MagicMock()
155
+ mock_client.messages.create.return_value = _make_mock_message(FAKE_RESPONSE)
156
+ MockAnthropic.return_value = mock_client
157
+
158
+ from bridgekit.summarize import summarize
159
+ summarize("Some analysis text.", system_prompt=custom_prompt)
160
+
161
+ call_kwargs = mock_client.messages.create.call_args
162
+ assert call_kwargs.kwargs.get("system") == custom_prompt
163
+
164
+
165
+ class TestSummarizeApiCallShape:
166
+ """summarize() should pass the user text through to the Anthropic API."""
167
+
168
+ def test_api_called_with_user_text(self):
169
+ user_text = "Our conversion rate improved after the campaign."
170
+ with patch.dict(os.environ, {"ANTHROPIC_API_KEY": "test-key"}):
171
+ with patch("anthropic.Anthropic") as MockAnthropic:
172
+ mock_client = MagicMock()
173
+ mock_client.messages.create.return_value = _make_mock_message(FAKE_RESPONSE)
174
+ MockAnthropic.return_value = mock_client
175
+
176
+ from bridgekit.summarize import summarize
177
+ summarize(user_text)
178
+
179
+ call_kwargs = mock_client.messages.create.call_args
180
+ messages_arg = call_kwargs.kwargs.get("messages") or call_kwargs.args[0]
181
+ content = str(messages_arg)
182
+ assert user_text in content
183
+
184
+
185
+ class TestSummarizeMaxTokens:
186
+ """summarize() should pass max_tokens through to the API."""
187
+
188
+ def test_default_max_tokens_is_1024(self):
189
+ with patch.dict(os.environ, {"ANTHROPIC_API_KEY": "test-key"}):
190
+ with patch("anthropic.Anthropic") as MockAnthropic:
191
+ mock_client = MagicMock()
192
+ mock_client.messages.create.return_value = _make_mock_message(FAKE_RESPONSE)
193
+ MockAnthropic.return_value = mock_client
194
+
195
+ from bridgekit.summarize import summarize
196
+ summarize("Some analysis text.")
197
+
198
+ call_kwargs = mock_client.messages.create.call_args
199
+ assert call_kwargs.kwargs.get("max_tokens") == 1024
200
+
201
+ def test_custom_max_tokens_reaches_api(self):
202
+ with patch.dict(os.environ, {"ANTHROPIC_API_KEY": "test-key"}):
203
+ with patch("anthropic.Anthropic") as MockAnthropic:
204
+ mock_client = MagicMock()
205
+ mock_client.messages.create.return_value = _make_mock_message(FAKE_RESPONSE)
206
+ MockAnthropic.return_value = mock_client
207
+
208
+ from bridgekit.summarize import summarize
209
+ summarize("Some analysis text.", max_tokens=2048)
210
+
211
+ call_kwargs = mock_client.messages.create.call_args
212
+ assert call_kwargs.kwargs.get("max_tokens") == 2048
File without changes
File without changes
File without changes
File without changes