bridgekit 0.3.8__tar.gz → 0.3.10__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {bridgekit-0.3.8 → bridgekit-0.3.10}/PKG-INFO +136 -2
- {bridgekit-0.3.8 → bridgekit-0.3.10}/README.md +135 -1
- bridgekit-0.3.10/bridgekit/__init__.py +9 -0
- bridgekit-0.3.10/bridgekit/compare.py +128 -0
- bridgekit-0.3.10/bridgekit/summarize.py +72 -0
- {bridgekit-0.3.8 → bridgekit-0.3.10}/bridgekit.egg-info/PKG-INFO +136 -2
- {bridgekit-0.3.8 → bridgekit-0.3.10}/bridgekit.egg-info/SOURCES.txt +5 -1
- {bridgekit-0.3.8 → bridgekit-0.3.10}/pyproject.toml +1 -1
- bridgekit-0.3.10/tests/test_compare.py +275 -0
- bridgekit-0.3.10/tests/test_summarize.py +212 -0
- bridgekit-0.3.8/bridgekit/__init__.py +0 -7
- {bridgekit-0.3.8 → bridgekit-0.3.10}/LICENSE +0 -0
- {bridgekit-0.3.8 → bridgekit-0.3.10}/bridgekit/cli.py +0 -0
- {bridgekit-0.3.8 → bridgekit-0.3.10}/bridgekit/config.py +0 -0
- {bridgekit-0.3.8 → bridgekit-0.3.10}/bridgekit/planner.py +0 -0
- {bridgekit-0.3.8 → bridgekit-0.3.10}/bridgekit/providers.py +0 -0
- {bridgekit-0.3.8 → bridgekit-0.3.10}/bridgekit/redteam.py +0 -0
- {bridgekit-0.3.8 → bridgekit-0.3.10}/bridgekit/reviewer.py +0 -0
- {bridgekit-0.3.8 → bridgekit-0.3.10}/bridgekit/search.py +0 -0
- {bridgekit-0.3.8 → bridgekit-0.3.10}/bridgekit.egg-info/dependency_links.txt +0 -0
- {bridgekit-0.3.8 → bridgekit-0.3.10}/bridgekit.egg-info/entry_points.txt +0 -0
- {bridgekit-0.3.8 → bridgekit-0.3.10}/bridgekit.egg-info/requires.txt +0 -0
- {bridgekit-0.3.8 → bridgekit-0.3.10}/bridgekit.egg-info/top_level.txt +0 -0
- {bridgekit-0.3.8 → bridgekit-0.3.10}/setup.cfg +0 -0
- {bridgekit-0.3.8 → bridgekit-0.3.10}/tests/test_cli.py +0 -0
- {bridgekit-0.3.8 → bridgekit-0.3.10}/tests/test_config.py +0 -0
- {bridgekit-0.3.8 → bridgekit-0.3.10}/tests/test_planner.py +0 -0
- {bridgekit-0.3.8 → bridgekit-0.3.10}/tests/test_providers.py +0 -0
- {bridgekit-0.3.8 → bridgekit-0.3.10}/tests/test_redteam.py +0 -0
- {bridgekit-0.3.8 → bridgekit-0.3.10}/tests/test_reviewer.py +0 -0
- {bridgekit-0.3.8 → bridgekit-0.3.10}/tests/test_search.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: bridgekit
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.10
|
|
4
4
|
Summary: AI tools that make you a better data scientist, not a redundant one.
|
|
5
5
|
License: MIT
|
|
6
6
|
Project-URL: Homepage, https://usebridgekit.com
|
|
@@ -396,6 +396,136 @@ willing to commit to — and what's your confidence interval on that estimate?"
|
|
|
396
396
|
|
|
397
397
|
---
|
|
398
398
|
|
|
399
|
+
## Tool #5: Compare
|
|
400
|
+
|
|
401
|
+
Run the same tool through two providers and see both outputs side by side. A synthesis summary at the top highlights where the models agreed, where they differed in severity, and which gave more actionable feedback — so you get the key insight without reading both outputs in full. Useful for evaluating which model works best for your use case — as a one-liner.
|
|
402
|
+
|
|
403
|
+
```python
|
|
404
|
+
from bridgekit import compare
|
|
405
|
+
|
|
406
|
+
text = """
|
|
407
|
+
I analyzed 90 days of user behavior data to understand what drives subscription
|
|
408
|
+
upgrades. Users who engaged with the reporting feature within their first week
|
|
409
|
+
were 3x more likely to upgrade within 30 days. I recommend we prioritize
|
|
410
|
+
onboarding users to reporting as a growth lever.
|
|
411
|
+
"""
|
|
412
|
+
|
|
413
|
+
# Compare evaluate across Anthropic and OpenAI (default)
|
|
414
|
+
print(compare(text, tool="evaluate"))
|
|
415
|
+
|
|
416
|
+
# Compare plan across two providers
|
|
417
|
+
print(compare("Did our onboarding flow reduce churn?", tool="plan"))
|
|
418
|
+
|
|
419
|
+
# Compare redteam with a specific stakeholder
|
|
420
|
+
print(compare(text, tool="redteam", stakeholder="VP of Finance"))
|
|
421
|
+
|
|
422
|
+
# Override models for each provider
|
|
423
|
+
print(compare(text, model_a="claude-haiku-4-5-20251001", model_b="gpt-4-turbo"))
|
|
424
|
+
|
|
425
|
+
# Compare two specific providers
|
|
426
|
+
print(compare(text, providers=["anthropic", "gemini"]))
|
|
427
|
+
```
|
|
428
|
+
|
|
429
|
+
**Parameters:**
|
|
430
|
+
- `tool` - which tool to run: `"evaluate"`, `"plan"`, or `"redteam"` (defaults to `"evaluate"`)
|
|
431
|
+
- `providers` - list of exactly two providers to compare (defaults to `["anthropic", "openai"]`)
|
|
432
|
+
- `model_a`, `model_b` - optional model overrides for the first and second provider
|
|
433
|
+
- `**kwargs` - additional arguments passed through to the underlying tool (e.g. `max_tokens`, `stakeholder`, `data_description`)
|
|
434
|
+
|
|
435
|
+
**Output:**
|
|
436
|
+
```
|
|
437
|
+
BRIDGEKIT COMPARE: EVALUATE
|
|
438
|
+
─────────────────────────────────────────
|
|
439
|
+
|
|
440
|
+
SUMMARY
|
|
441
|
+
─────────────────────────────────────────
|
|
442
|
+
Both outputs rated Clarity as STRONG. They diverged on severity: Anthropic
|
|
443
|
+
rated Statistical Rigor as MISSING (harsher) while OpenAI called it NEEDS WORK.
|
|
444
|
+
Anthropic gave more specific feedback — naming the correlation-vs-causation
|
|
445
|
+
problem explicitly and suggesting concrete fixes. OpenAI's feedback stayed
|
|
446
|
+
more generic. Anthropic's bottom line targets the core analytical flaw;
|
|
447
|
+
OpenAI's restates the statistical point only.
|
|
448
|
+
|
|
449
|
+
|
|
450
|
+
ANTHROPIC claude-opus-4-8
|
|
451
|
+
─────────────────────────────────────────
|
|
452
|
+
BRIDGEKIT ANALYSIS REVIEW
|
|
453
|
+
─────────────────────────────────────────
|
|
454
|
+
|
|
455
|
+
1. CLARITY
|
|
456
|
+
✅ STRONG — Clean and jargon-free.
|
|
457
|
+
|
|
458
|
+
...
|
|
459
|
+
|
|
460
|
+
OPENAI gpt-4o
|
|
461
|
+
─────────────────────────────────────────
|
|
462
|
+
BRIDGEKIT ANALYSIS REVIEW
|
|
463
|
+
─────────────────────────────────────────
|
|
464
|
+
|
|
465
|
+
1. CLARITY
|
|
466
|
+
⚠️ NEEDS WORK — The phrase "engagement feature" needs more context.
|
|
467
|
+
|
|
468
|
+
...
|
|
469
|
+
```
|
|
470
|
+
|
|
471
|
+
Both providers are called in parallel, so the total wait time is the slower of the two — not the sum. The synthesis summary is a third sequential call made after both outputs are ready.
|
|
472
|
+
|
|
473
|
+
---
|
|
474
|
+
|
|
475
|
+
## Tool #6: Summarize
|
|
476
|
+
|
|
477
|
+
Turn a long analysis writeup, notebook, or report into a short executive summary — the gap between doing the analysis and communicating it to people who won't read the whole thing.
|
|
478
|
+
|
|
479
|
+
```python
|
|
480
|
+
from bridgekit import summarize
|
|
481
|
+
|
|
482
|
+
text = """
|
|
483
|
+
I analyzed 90 days of user behavior data to understand what drives subscription
|
|
484
|
+
upgrades. Users who engaged with the reporting feature within their first week
|
|
485
|
+
were 3x more likely to upgrade within 30 days. Sample size was 1,200 users
|
|
486
|
+
across two acquisition channels, with consistent results in both. I recommend
|
|
487
|
+
we prioritize onboarding users to reporting as a growth lever.
|
|
488
|
+
"""
|
|
489
|
+
|
|
490
|
+
# Default — general business audience
|
|
491
|
+
print(summarize(text))
|
|
492
|
+
|
|
493
|
+
# Or specify an audience
|
|
494
|
+
print(summarize(text, audience="VP of Marketing"))
|
|
495
|
+
print(summarize(text, audience="board"))
|
|
496
|
+
|
|
497
|
+
# Override for longer summaries
|
|
498
|
+
print(summarize(text, max_tokens=2048))
|
|
499
|
+
```
|
|
500
|
+
|
|
501
|
+
**Output:**
|
|
502
|
+
```
|
|
503
|
+
BRIDGEKIT SUMMARY
|
|
504
|
+
─────────────────────────────────────────
|
|
505
|
+
AUDIENCE: VP of Marketing
|
|
506
|
+
|
|
507
|
+
KEY TAKEAWAY
|
|
508
|
+
Getting users into the reporting feature in their first week is our strongest
|
|
509
|
+
predictor of paid upgrades — users who engage with it are 3x more likely to
|
|
510
|
+
convert within 30 days.
|
|
511
|
+
|
|
512
|
+
WHAT WE FOUND
|
|
513
|
+
- Early reporting engagement (week 1) drives 3x higher upgrade rates within 30 days
|
|
514
|
+
- Finding holds across both acquisition channels, suggesting it's a genuine
|
|
515
|
+
behavior pattern, not channel-specific
|
|
516
|
+
- Analyzed 1,200 users over 90 days with sufficient scale to trust the result
|
|
517
|
+
|
|
518
|
+
SO WHAT
|
|
519
|
+
We should redesign onboarding to get users to reporting faster. This is a
|
|
520
|
+
high-confidence growth lever worth testing immediately.
|
|
521
|
+
|
|
522
|
+
─────────────────────────────────────────
|
|
523
|
+
```
|
|
524
|
+
|
|
525
|
+
`audience`, `provider`, `model`, `system_prompt`, and `max_tokens` are all optional — the more specific the audience, the more tailored the summary.
|
|
526
|
+
|
|
527
|
+
---
|
|
528
|
+
|
|
399
529
|
## Multi-Provider Support
|
|
400
530
|
|
|
401
531
|
Bridgekit now supports multiple AI providers so you're not locked into one API. You can use Anthropic, OpenAI, or Google Gemini models with any tool.
|
|
@@ -432,6 +562,7 @@ All tools support the same `provider` and `model` parameters:
|
|
|
432
562
|
- `plan(question, provider=None, model=None, ..., system_prompt=None)`
|
|
433
563
|
- `ask(question, provider=None, model=None, ..., system_prompt=None)`
|
|
434
564
|
- `redteam(text, provider=None, model=None, ..., system_prompt=None)`
|
|
565
|
+
- `summarize(text, provider=None, model=None, ..., system_prompt=None)`
|
|
435
566
|
|
|
436
567
|
---
|
|
437
568
|
|
|
@@ -440,7 +571,7 @@ All tools support the same `provider` and `model` parameters:
|
|
|
440
571
|
Every tool accepts an optional `system_prompt` parameter to override the default persona. Use this to adapt the tone or focus to a specific domain without changing anything else.
|
|
441
572
|
|
|
442
573
|
```python
|
|
443
|
-
from bridgekit import evaluate, plan, ask, redteam
|
|
574
|
+
from bridgekit import evaluate, plan, ask, redteam, summarize
|
|
444
575
|
|
|
445
576
|
# Narrow the reviewer to a specific domain
|
|
446
577
|
print(evaluate("my analysis", system_prompt="You are a skeptical PhD statistician focused only on methodology"))
|
|
@@ -453,6 +584,9 @@ print(redteam("my analysis", system_prompt="You are a hostile regulator looking
|
|
|
453
584
|
|
|
454
585
|
# Change the answering style for ask
|
|
455
586
|
print(ask("my question", text="...", system_prompt="You are a financial analyst. Answer only in terms of revenue impact."))
|
|
587
|
+
|
|
588
|
+
# Replace the summarizer persona entirely
|
|
589
|
+
print(summarize("my analysis", system_prompt="You are a data journalist writing a one-paragraph news brief."))
|
|
456
590
|
```
|
|
457
591
|
|
|
458
592
|
When `system_prompt` is not provided, each tool uses its built-in default — existing behavior is unchanged.
|
|
@@ -364,6 +364,136 @@ willing to commit to — and what's your confidence interval on that estimate?"
|
|
|
364
364
|
|
|
365
365
|
---
|
|
366
366
|
|
|
367
|
+
## Tool #5: Compare
|
|
368
|
+
|
|
369
|
+
Run the same tool through two providers and see both outputs side by side. A synthesis summary at the top highlights where the models agreed, where they differed in severity, and which gave more actionable feedback — so you get the key insight without reading both outputs in full. Useful for evaluating which model works best for your use case — as a one-liner.
|
|
370
|
+
|
|
371
|
+
```python
|
|
372
|
+
from bridgekit import compare
|
|
373
|
+
|
|
374
|
+
text = """
|
|
375
|
+
I analyzed 90 days of user behavior data to understand what drives subscription
|
|
376
|
+
upgrades. Users who engaged with the reporting feature within their first week
|
|
377
|
+
were 3x more likely to upgrade within 30 days. I recommend we prioritize
|
|
378
|
+
onboarding users to reporting as a growth lever.
|
|
379
|
+
"""
|
|
380
|
+
|
|
381
|
+
# Compare evaluate across Anthropic and OpenAI (default)
|
|
382
|
+
print(compare(text, tool="evaluate"))
|
|
383
|
+
|
|
384
|
+
# Compare plan across two providers
|
|
385
|
+
print(compare("Did our onboarding flow reduce churn?", tool="plan"))
|
|
386
|
+
|
|
387
|
+
# Compare redteam with a specific stakeholder
|
|
388
|
+
print(compare(text, tool="redteam", stakeholder="VP of Finance"))
|
|
389
|
+
|
|
390
|
+
# Override models for each provider
|
|
391
|
+
print(compare(text, model_a="claude-haiku-4-5-20251001", model_b="gpt-4-turbo"))
|
|
392
|
+
|
|
393
|
+
# Compare two specific providers
|
|
394
|
+
print(compare(text, providers=["anthropic", "gemini"]))
|
|
395
|
+
```
|
|
396
|
+
|
|
397
|
+
**Parameters:**
|
|
398
|
+
- `tool` - which tool to run: `"evaluate"`, `"plan"`, or `"redteam"` (defaults to `"evaluate"`)
|
|
399
|
+
- `providers` - list of exactly two providers to compare (defaults to `["anthropic", "openai"]`)
|
|
400
|
+
- `model_a`, `model_b` - optional model overrides for the first and second provider
|
|
401
|
+
- `**kwargs` - additional arguments passed through to the underlying tool (e.g. `max_tokens`, `stakeholder`, `data_description`)
|
|
402
|
+
|
|
403
|
+
**Output:**
|
|
404
|
+
```
|
|
405
|
+
BRIDGEKIT COMPARE: EVALUATE
|
|
406
|
+
─────────────────────────────────────────
|
|
407
|
+
|
|
408
|
+
SUMMARY
|
|
409
|
+
─────────────────────────────────────────
|
|
410
|
+
Both outputs rated Clarity as STRONG. They diverged on severity: Anthropic
|
|
411
|
+
rated Statistical Rigor as MISSING (harsher) while OpenAI called it NEEDS WORK.
|
|
412
|
+
Anthropic gave more specific feedback — naming the correlation-vs-causation
|
|
413
|
+
problem explicitly and suggesting concrete fixes. OpenAI's feedback stayed
|
|
414
|
+
more generic. Anthropic's bottom line targets the core analytical flaw;
|
|
415
|
+
OpenAI's restates the statistical point only.
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
ANTHROPIC claude-opus-4-8
|
|
419
|
+
─────────────────────────────────────────
|
|
420
|
+
BRIDGEKIT ANALYSIS REVIEW
|
|
421
|
+
─────────────────────────────────────────
|
|
422
|
+
|
|
423
|
+
1. CLARITY
|
|
424
|
+
✅ STRONG — Clean and jargon-free.
|
|
425
|
+
|
|
426
|
+
...
|
|
427
|
+
|
|
428
|
+
OPENAI gpt-4o
|
|
429
|
+
─────────────────────────────────────────
|
|
430
|
+
BRIDGEKIT ANALYSIS REVIEW
|
|
431
|
+
─────────────────────────────────────────
|
|
432
|
+
|
|
433
|
+
1. CLARITY
|
|
434
|
+
⚠️ NEEDS WORK — The phrase "engagement feature" needs more context.
|
|
435
|
+
|
|
436
|
+
...
|
|
437
|
+
```
|
|
438
|
+
|
|
439
|
+
Both providers are called in parallel, so the total wait time is the slower of the two — not the sum. The synthesis summary is a third sequential call made after both outputs are ready.
|
|
440
|
+
|
|
441
|
+
---
|
|
442
|
+
|
|
443
|
+
## Tool #6: Summarize
|
|
444
|
+
|
|
445
|
+
Turn a long analysis writeup, notebook, or report into a short executive summary — the gap between doing the analysis and communicating it to people who won't read the whole thing.
|
|
446
|
+
|
|
447
|
+
```python
|
|
448
|
+
from bridgekit import summarize
|
|
449
|
+
|
|
450
|
+
text = """
|
|
451
|
+
I analyzed 90 days of user behavior data to understand what drives subscription
|
|
452
|
+
upgrades. Users who engaged with the reporting feature within their first week
|
|
453
|
+
were 3x more likely to upgrade within 30 days. Sample size was 1,200 users
|
|
454
|
+
across two acquisition channels, with consistent results in both. I recommend
|
|
455
|
+
we prioritize onboarding users to reporting as a growth lever.
|
|
456
|
+
"""
|
|
457
|
+
|
|
458
|
+
# Default — general business audience
|
|
459
|
+
print(summarize(text))
|
|
460
|
+
|
|
461
|
+
# Or specify an audience
|
|
462
|
+
print(summarize(text, audience="VP of Marketing"))
|
|
463
|
+
print(summarize(text, audience="board"))
|
|
464
|
+
|
|
465
|
+
# Override for longer summaries
|
|
466
|
+
print(summarize(text, max_tokens=2048))
|
|
467
|
+
```
|
|
468
|
+
|
|
469
|
+
**Output:**
|
|
470
|
+
```
|
|
471
|
+
BRIDGEKIT SUMMARY
|
|
472
|
+
─────────────────────────────────────────
|
|
473
|
+
AUDIENCE: VP of Marketing
|
|
474
|
+
|
|
475
|
+
KEY TAKEAWAY
|
|
476
|
+
Getting users into the reporting feature in their first week is our strongest
|
|
477
|
+
predictor of paid upgrades — users who engage with it are 3x more likely to
|
|
478
|
+
convert within 30 days.
|
|
479
|
+
|
|
480
|
+
WHAT WE FOUND
|
|
481
|
+
- Early reporting engagement (week 1) drives 3x higher upgrade rates within 30 days
|
|
482
|
+
- Finding holds across both acquisition channels, suggesting it's a genuine
|
|
483
|
+
behavior pattern, not channel-specific
|
|
484
|
+
- Analyzed 1,200 users over 90 days with sufficient scale to trust the result
|
|
485
|
+
|
|
486
|
+
SO WHAT
|
|
487
|
+
We should redesign onboarding to get users to reporting faster. This is a
|
|
488
|
+
high-confidence growth lever worth testing immediately.
|
|
489
|
+
|
|
490
|
+
─────────────────────────────────────────
|
|
491
|
+
```
|
|
492
|
+
|
|
493
|
+
`audience`, `provider`, `model`, `system_prompt`, and `max_tokens` are all optional — the more specific the audience, the more tailored the summary.
|
|
494
|
+
|
|
495
|
+
---
|
|
496
|
+
|
|
367
497
|
## Multi-Provider Support
|
|
368
498
|
|
|
369
499
|
Bridgekit now supports multiple AI providers so you're not locked into one API. You can use Anthropic, OpenAI, or Google Gemini models with any tool.
|
|
@@ -400,6 +530,7 @@ All tools support the same `provider` and `model` parameters:
|
|
|
400
530
|
- `plan(question, provider=None, model=None, ..., system_prompt=None)`
|
|
401
531
|
- `ask(question, provider=None, model=None, ..., system_prompt=None)`
|
|
402
532
|
- `redteam(text, provider=None, model=None, ..., system_prompt=None)`
|
|
533
|
+
- `summarize(text, provider=None, model=None, ..., system_prompt=None)`
|
|
403
534
|
|
|
404
535
|
---
|
|
405
536
|
|
|
@@ -408,7 +539,7 @@ All tools support the same `provider` and `model` parameters:
|
|
|
408
539
|
Every tool accepts an optional `system_prompt` parameter to override the default persona. Use this to adapt the tone or focus to a specific domain without changing anything else.
|
|
409
540
|
|
|
410
541
|
```python
|
|
411
|
-
from bridgekit import evaluate, plan, ask, redteam
|
|
542
|
+
from bridgekit import evaluate, plan, ask, redteam, summarize
|
|
412
543
|
|
|
413
544
|
# Narrow the reviewer to a specific domain
|
|
414
545
|
print(evaluate("my analysis", system_prompt="You are a skeptical PhD statistician focused only on methodology"))
|
|
@@ -421,6 +552,9 @@ print(redteam("my analysis", system_prompt="You are a hostile regulator looking
|
|
|
421
552
|
|
|
422
553
|
# Change the answering style for ask
|
|
423
554
|
print(ask("my question", text="...", system_prompt="You are a financial analyst. Answer only in terms of revenue impact."))
|
|
555
|
+
|
|
556
|
+
# Replace the summarizer persona entirely
|
|
557
|
+
print(summarize("my analysis", system_prompt="You are a data journalist writing a one-paragraph news brief."))
|
|
424
558
|
```
|
|
425
559
|
|
|
426
560
|
When `system_prompt` is not provided, each tool uses its built-in default — existing behavior is unchanged.
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
from .reviewer import evaluate
|
|
2
|
+
from .search import ask
|
|
3
|
+
from .planner import plan
|
|
4
|
+
from .redteam import redteam
|
|
5
|
+
from .compare import compare
|
|
6
|
+
from .summarize import summarize
|
|
7
|
+
|
|
8
|
+
__version__ = "0.3.10"
|
|
9
|
+
__all__ = ["evaluate", "ask", "plan", "redteam", "compare", "summarize"]
|
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
2
|
+
from .config import parse_provider, get_default_model, Provider
|
|
3
|
+
from .providers import create_message
|
|
4
|
+
|
|
5
|
+
SUPPORTED_TOOLS = ["evaluate", "plan", "redteam"]
|
|
6
|
+
|
|
7
|
+
SYNTHESIS_PROMPT = """You are comparing two AI-generated outputs for the same analysis task.
|
|
8
|
+
|
|
9
|
+
Your job is to write a short, direct summary (4-6 sentences) covering:
|
|
10
|
+
- Where both outputs agreed
|
|
11
|
+
- Where they differed — including any cases where one rated a dimension more harshly than the other
|
|
12
|
+
- Which output gave more specific or actionable feedback, and why
|
|
13
|
+
|
|
14
|
+
Be concrete. Reference specific dimensions or findings. No fluff."""
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _get_tool_fn(tool_name: str):
|
|
18
|
+
if tool_name == "evaluate":
|
|
19
|
+
from .reviewer import evaluate
|
|
20
|
+
return evaluate
|
|
21
|
+
elif tool_name == "plan":
|
|
22
|
+
from .planner import plan
|
|
23
|
+
return plan
|
|
24
|
+
elif tool_name == "redteam":
|
|
25
|
+
from .redteam import redteam
|
|
26
|
+
return redteam
|
|
27
|
+
raise ValueError(f"Unknown tool: {tool_name!r}. Supported tools: {SUPPORTED_TOOLS}")
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _call_tool(tool_name: str, text: str, provider: str, model, kwargs: dict) -> str:
|
|
31
|
+
fn = _get_tool_fn(tool_name)
|
|
32
|
+
if tool_name == "plan":
|
|
33
|
+
return fn(question=text, provider=provider, model=model, **kwargs)
|
|
34
|
+
return fn(text=text, provider=provider, model=model, **kwargs)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _synthesize(results: list) -> str:
|
|
38
|
+
(provider_a, model_a, output_a), (provider_b, model_b, output_b) = results
|
|
39
|
+
user_message = (
|
|
40
|
+
f"OUTPUT 1 ({provider_a.upper()} / {model_a}):\n{output_a}\n\n"
|
|
41
|
+
f"OUTPUT 2 ({provider_b.upper()} / {model_b}):\n{output_b}"
|
|
42
|
+
)
|
|
43
|
+
return create_message(
|
|
44
|
+
provider=Provider.ANTHROPIC,
|
|
45
|
+
system_prompt=SYNTHESIS_PROMPT,
|
|
46
|
+
user_message=user_message,
|
|
47
|
+
max_tokens=512,
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _format_output(tool: str, summary: str, results: list) -> str:
|
|
52
|
+
divider = "─" * 41
|
|
53
|
+
lines = [
|
|
54
|
+
f"BRIDGEKIT COMPARE: {tool.upper()}",
|
|
55
|
+
divider,
|
|
56
|
+
"",
|
|
57
|
+
"SUMMARY",
|
|
58
|
+
divider,
|
|
59
|
+
summary,
|
|
60
|
+
"",
|
|
61
|
+
"",
|
|
62
|
+
]
|
|
63
|
+
for i, (provider_name, model, output) in enumerate(results):
|
|
64
|
+
lines.append(f"{provider_name.upper()} {model}")
|
|
65
|
+
lines.append(divider)
|
|
66
|
+
lines.append(output)
|
|
67
|
+
if i < len(results) - 1:
|
|
68
|
+
lines.append("")
|
|
69
|
+
lines.append("")
|
|
70
|
+
return "\n".join(lines)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def compare(
|
|
74
|
+
text: str,
|
|
75
|
+
tool: str = "evaluate",
|
|
76
|
+
providers: list = None,
|
|
77
|
+
model_a: str = None,
|
|
78
|
+
model_b: str = None,
|
|
79
|
+
**kwargs
|
|
80
|
+
) -> str:
|
|
81
|
+
"""
|
|
82
|
+
Run the same tool through two providers and return both outputs with a summary.
|
|
83
|
+
|
|
84
|
+
Args:
|
|
85
|
+
text: The input text or question to analyze.
|
|
86
|
+
tool: Which tool to run: "evaluate", "plan", or "redteam". Defaults to "evaluate".
|
|
87
|
+
providers: List of exactly two providers to compare. Defaults to ["anthropic", "openai"].
|
|
88
|
+
model_a: Optional model override for the first provider.
|
|
89
|
+
model_b: Optional model override for the second provider.
|
|
90
|
+
**kwargs: Additional arguments passed through to the underlying tool
|
|
91
|
+
(e.g. max_tokens, stakeholder for redteam, data_description/goal for plan).
|
|
92
|
+
|
|
93
|
+
Returns:
|
|
94
|
+
A summary of key differences followed by both full outputs with provider headers.
|
|
95
|
+
"""
|
|
96
|
+
if not text or not text.strip():
|
|
97
|
+
raise ValueError("Text cannot be empty.")
|
|
98
|
+
|
|
99
|
+
if tool not in SUPPORTED_TOOLS:
|
|
100
|
+
raise ValueError(f"Unknown tool: {tool!r}. Supported tools: {SUPPORTED_TOOLS}")
|
|
101
|
+
|
|
102
|
+
if providers is None:
|
|
103
|
+
providers = ["anthropic", "openai"]
|
|
104
|
+
|
|
105
|
+
if len(providers) != 2:
|
|
106
|
+
raise ValueError(f"providers must contain exactly 2 providers, got {len(providers)}.")
|
|
107
|
+
|
|
108
|
+
models = [model_a, model_b]
|
|
109
|
+
resolved_models = []
|
|
110
|
+
for provider, model in zip(providers, models):
|
|
111
|
+
provider_enum = parse_provider(provider)
|
|
112
|
+
resolved_models.append(model if model else get_default_model(provider_enum))
|
|
113
|
+
|
|
114
|
+
results_map = {}
|
|
115
|
+
|
|
116
|
+
def run_one(idx):
|
|
117
|
+
output = _call_tool(tool, text, providers[idx], models[idx], kwargs)
|
|
118
|
+
return idx, providers[idx], resolved_models[idx], output
|
|
119
|
+
|
|
120
|
+
with ThreadPoolExecutor(max_workers=2) as executor:
|
|
121
|
+
futures = {executor.submit(run_one, i): i for i in range(2)}
|
|
122
|
+
for future in as_completed(futures):
|
|
123
|
+
idx, provider, model, output = future.result()
|
|
124
|
+
results_map[idx] = (provider, model, output)
|
|
125
|
+
|
|
126
|
+
results = [results_map[0], results_map[1]]
|
|
127
|
+
summary = _synthesize(results)
|
|
128
|
+
return _format_output(tool, summary, results)
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
from .config import DEFAULT_MODEL, parse_provider, get_default_model
|
|
2
|
+
from .providers import create_message
|
|
3
|
+
|
|
4
|
+
DEFAULT_AUDIENCE = "a general business audience with no technical or data science background"
|
|
5
|
+
|
|
6
|
+
SYSTEM_PROMPT_TEMPLATE = """You are a senior data scientist preparing an executive summary of a technical analysis for {audience}.
|
|
7
|
+
|
|
8
|
+
Your job is to translate technical detail into what matters for decision-making. Cut jargon, cut methodology detail unless it's essential to trust the conclusion, and lead with the takeaway. Write for someone who will skim this in 30 seconds.
|
|
9
|
+
|
|
10
|
+
Format your response exactly like this:
|
|
11
|
+
|
|
12
|
+
BRIDGEKIT SUMMARY
|
|
13
|
+
─────────────────────────────────────────
|
|
14
|
+
AUDIENCE: {audience_label}
|
|
15
|
+
|
|
16
|
+
KEY TAKEAWAY
|
|
17
|
+
[1-2 sentences: the single most important finding or recommendation]
|
|
18
|
+
|
|
19
|
+
WHAT WE FOUND
|
|
20
|
+
[3-5 bullet points of the supporting findings, in plain language]
|
|
21
|
+
|
|
22
|
+
SO WHAT
|
|
23
|
+
[1-2 sentences on the business implication or recommended next step]
|
|
24
|
+
|
|
25
|
+
─────────────────────────────────────────
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def summarize(text: str, audience: str = None, provider: str = None, model: str = None, system_prompt: str = None, max_tokens: int = 1024) -> str:
|
|
30
|
+
"""
|
|
31
|
+
Turn a long analysis, notebook, or report into a short executive summary.
|
|
32
|
+
|
|
33
|
+
Args:
|
|
34
|
+
text: The long-form analysis writeup, notebook output, or report as a plain string.
|
|
35
|
+
audience: Optional. Who the summary is for (e.g. "VP of Marketing", "board",
|
|
36
|
+
"engineering team"). Defaults to a general business audience.
|
|
37
|
+
provider: Optional. The AI provider to use ("anthropic", "openai", "gemini").
|
|
38
|
+
If not specified, defaults to "anthropic" or infers from model.
|
|
39
|
+
model: Optional. The specific model to use. If not specified, uses the provider's default.
|
|
40
|
+
system_prompt: Optional. A custom system prompt to fully override the default summarizer persona.
|
|
41
|
+
When provided, the audience parameter is ignored.
|
|
42
|
+
max_tokens: Optional. Maximum tokens in the response. Defaults to 1024.
|
|
43
|
+
|
|
44
|
+
Returns:
|
|
45
|
+
A short executive summary covering the key takeaway, supporting findings,
|
|
46
|
+
and the business implication.
|
|
47
|
+
"""
|
|
48
|
+
if not text or not text.strip():
|
|
49
|
+
raise ValueError("Text cannot be empty.")
|
|
50
|
+
|
|
51
|
+
# Parse provider and determine model
|
|
52
|
+
provider_enum = parse_provider(provider, model)
|
|
53
|
+
if model is None:
|
|
54
|
+
model = get_default_model(provider_enum)
|
|
55
|
+
|
|
56
|
+
if system_prompt is None:
|
|
57
|
+
audience_label = audience if audience else "General Business Audience"
|
|
58
|
+
audience_desc = audience if audience else DEFAULT_AUDIENCE
|
|
59
|
+
system_prompt = SYSTEM_PROMPT_TEMPLATE.format(
|
|
60
|
+
audience=audience_desc,
|
|
61
|
+
audience_label=audience_label
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
user_message = f"Summarize this analysis:\n\n{text}"
|
|
65
|
+
|
|
66
|
+
return create_message(
|
|
67
|
+
provider=provider_enum,
|
|
68
|
+
system_prompt=system_prompt,
|
|
69
|
+
user_message=user_message,
|
|
70
|
+
model=model,
|
|
71
|
+
max_tokens=max_tokens
|
|
72
|
+
)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: bridgekit
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.10
|
|
4
4
|
Summary: AI tools that make you a better data scientist, not a redundant one.
|
|
5
5
|
License: MIT
|
|
6
6
|
Project-URL: Homepage, https://usebridgekit.com
|
|
@@ -396,6 +396,136 @@ willing to commit to — and what's your confidence interval on that estimate?"
|
|
|
396
396
|
|
|
397
397
|
---
|
|
398
398
|
|
|
399
|
+
## Tool #5: Compare
|
|
400
|
+
|
|
401
|
+
Run the same tool through two providers and see both outputs side by side. A synthesis summary at the top highlights where the models agreed, where they differed in severity, and which gave more actionable feedback — so you get the key insight without reading both outputs in full. Useful for evaluating which model works best for your use case — as a one-liner.
|
|
402
|
+
|
|
403
|
+
```python
|
|
404
|
+
from bridgekit import compare
|
|
405
|
+
|
|
406
|
+
text = """
|
|
407
|
+
I analyzed 90 days of user behavior data to understand what drives subscription
|
|
408
|
+
upgrades. Users who engaged with the reporting feature within their first week
|
|
409
|
+
were 3x more likely to upgrade within 30 days. I recommend we prioritize
|
|
410
|
+
onboarding users to reporting as a growth lever.
|
|
411
|
+
"""
|
|
412
|
+
|
|
413
|
+
# Compare evaluate across Anthropic and OpenAI (default)
|
|
414
|
+
print(compare(text, tool="evaluate"))
|
|
415
|
+
|
|
416
|
+
# Compare plan across two providers
|
|
417
|
+
print(compare("Did our onboarding flow reduce churn?", tool="plan"))
|
|
418
|
+
|
|
419
|
+
# Compare redteam with a specific stakeholder
|
|
420
|
+
print(compare(text, tool="redteam", stakeholder="VP of Finance"))
|
|
421
|
+
|
|
422
|
+
# Override models for each provider
|
|
423
|
+
print(compare(text, model_a="claude-haiku-4-5-20251001", model_b="gpt-4-turbo"))
|
|
424
|
+
|
|
425
|
+
# Compare two specific providers
|
|
426
|
+
print(compare(text, providers=["anthropic", "gemini"]))
|
|
427
|
+
```
|
|
428
|
+
|
|
429
|
+
**Parameters:**
|
|
430
|
+
- `tool` - which tool to run: `"evaluate"`, `"plan"`, or `"redteam"` (defaults to `"evaluate"`)
|
|
431
|
+
- `providers` - list of exactly two providers to compare (defaults to `["anthropic", "openai"]`)
|
|
432
|
+
- `model_a`, `model_b` - optional model overrides for the first and second provider
|
|
433
|
+
- `**kwargs` - additional arguments passed through to the underlying tool (e.g. `max_tokens`, `stakeholder`, `data_description`)
|
|
434
|
+
|
|
435
|
+
**Output:**
|
|
436
|
+
```
|
|
437
|
+
BRIDGEKIT COMPARE: EVALUATE
|
|
438
|
+
─────────────────────────────────────────
|
|
439
|
+
|
|
440
|
+
SUMMARY
|
|
441
|
+
─────────────────────────────────────────
|
|
442
|
+
Both outputs rated Clarity as STRONG. They diverged on severity: Anthropic
|
|
443
|
+
rated Statistical Rigor as MISSING (harsher) while OpenAI called it NEEDS WORK.
|
|
444
|
+
Anthropic gave more specific feedback — naming the correlation-vs-causation
|
|
445
|
+
problem explicitly and suggesting concrete fixes. OpenAI's feedback stayed
|
|
446
|
+
more generic. Anthropic's bottom line targets the core analytical flaw;
|
|
447
|
+
OpenAI's restates the statistical point only.
|
|
448
|
+
|
|
449
|
+
|
|
450
|
+
ANTHROPIC claude-opus-4-8
|
|
451
|
+
─────────────────────────────────────────
|
|
452
|
+
BRIDGEKIT ANALYSIS REVIEW
|
|
453
|
+
─────────────────────────────────────────
|
|
454
|
+
|
|
455
|
+
1. CLARITY
|
|
456
|
+
✅ STRONG — Clean and jargon-free.
|
|
457
|
+
|
|
458
|
+
...
|
|
459
|
+
|
|
460
|
+
OPENAI gpt-4o
|
|
461
|
+
─────────────────────────────────────────
|
|
462
|
+
BRIDGEKIT ANALYSIS REVIEW
|
|
463
|
+
─────────────────────────────────────────
|
|
464
|
+
|
|
465
|
+
1. CLARITY
|
|
466
|
+
⚠️ NEEDS WORK — The phrase "engagement feature" needs more context.
|
|
467
|
+
|
|
468
|
+
...
|
|
469
|
+
```
|
|
470
|
+
|
|
471
|
+
Both providers are called in parallel, so the total wait time is the slower of the two — not the sum. The synthesis summary is a third sequential call made after both outputs are ready.
|
|
472
|
+
|
|
473
|
+
---
|
|
474
|
+
|
|
475
|
+
## Tool #6: Summarize
|
|
476
|
+
|
|
477
|
+
Turn a long analysis writeup, notebook, or report into a short executive summary — the gap between doing the analysis and communicating it to people who won't read the whole thing.
|
|
478
|
+
|
|
479
|
+
```python
|
|
480
|
+
from bridgekit import summarize
|
|
481
|
+
|
|
482
|
+
text = """
|
|
483
|
+
I analyzed 90 days of user behavior data to understand what drives subscription
|
|
484
|
+
upgrades. Users who engaged with the reporting feature within their first week
|
|
485
|
+
were 3x more likely to upgrade within 30 days. Sample size was 1,200 users
|
|
486
|
+
across two acquisition channels, with consistent results in both. I recommend
|
|
487
|
+
we prioritize onboarding users to reporting as a growth lever.
|
|
488
|
+
"""
|
|
489
|
+
|
|
490
|
+
# Default — general business audience
|
|
491
|
+
print(summarize(text))
|
|
492
|
+
|
|
493
|
+
# Or specify an audience
|
|
494
|
+
print(summarize(text, audience="VP of Marketing"))
|
|
495
|
+
print(summarize(text, audience="board"))
|
|
496
|
+
|
|
497
|
+
# Override for longer summaries
|
|
498
|
+
print(summarize(text, max_tokens=2048))
|
|
499
|
+
```
|
|
500
|
+
|
|
501
|
+
**Output:**
|
|
502
|
+
```
|
|
503
|
+
BRIDGEKIT SUMMARY
|
|
504
|
+
─────────────────────────────────────────
|
|
505
|
+
AUDIENCE: VP of Marketing
|
|
506
|
+
|
|
507
|
+
KEY TAKEAWAY
|
|
508
|
+
Getting users into the reporting feature in their first week is our strongest
|
|
509
|
+
predictor of paid upgrades — users who engage with it are 3x more likely to
|
|
510
|
+
convert within 30 days.
|
|
511
|
+
|
|
512
|
+
WHAT WE FOUND
|
|
513
|
+
- Early reporting engagement (week 1) drives 3x higher upgrade rates within 30 days
|
|
514
|
+
- Finding holds across both acquisition channels, suggesting it's a genuine
|
|
515
|
+
behavior pattern, not channel-specific
|
|
516
|
+
- Analyzed 1,200 users over 90 days with sufficient scale to trust the result
|
|
517
|
+
|
|
518
|
+
SO WHAT
|
|
519
|
+
We should redesign onboarding to get users to reporting faster. This is a
|
|
520
|
+
high-confidence growth lever worth testing immediately.
|
|
521
|
+
|
|
522
|
+
─────────────────────────────────────────
|
|
523
|
+
```
|
|
524
|
+
|
|
525
|
+
`audience`, `provider`, `model`, `system_prompt`, and `max_tokens` are all optional — the more specific the audience, the more tailored the summary.
|
|
526
|
+
|
|
527
|
+
---
|
|
528
|
+
|
|
399
529
|
## Multi-Provider Support
|
|
400
530
|
|
|
401
531
|
Bridgekit now supports multiple AI providers so you're not locked into one API. You can use Anthropic, OpenAI, or Google Gemini models with any tool.
|
|
@@ -432,6 +562,7 @@ All tools support the same `provider` and `model` parameters:
|
|
|
432
562
|
- `plan(question, provider=None, model=None, ..., system_prompt=None)`
|
|
433
563
|
- `ask(question, provider=None, model=None, ..., system_prompt=None)`
|
|
434
564
|
- `redteam(text, provider=None, model=None, ..., system_prompt=None)`
|
|
565
|
+
- `summarize(text, provider=None, model=None, ..., system_prompt=None)`
|
|
435
566
|
|
|
436
567
|
---
|
|
437
568
|
|
|
@@ -440,7 +571,7 @@ All tools support the same `provider` and `model` parameters:
|
|
|
440
571
|
Every tool accepts an optional `system_prompt` parameter to override the default persona. Use this to adapt the tone or focus to a specific domain without changing anything else.
|
|
441
572
|
|
|
442
573
|
```python
|
|
443
|
-
from bridgekit import evaluate, plan, ask, redteam
|
|
574
|
+
from bridgekit import evaluate, plan, ask, redteam, summarize
|
|
444
575
|
|
|
445
576
|
# Narrow the reviewer to a specific domain
|
|
446
577
|
print(evaluate("my analysis", system_prompt="You are a skeptical PhD statistician focused only on methodology"))
|
|
@@ -453,6 +584,9 @@ print(redteam("my analysis", system_prompt="You are a hostile regulator looking
|
|
|
453
584
|
|
|
454
585
|
# Change the answering style for ask
|
|
455
586
|
print(ask("my question", text="...", system_prompt="You are a financial analyst. Answer only in terms of revenue impact."))
|
|
587
|
+
|
|
588
|
+
# Replace the summarizer persona entirely
|
|
589
|
+
print(summarize("my analysis", system_prompt="You are a data journalist writing a one-paragraph news brief."))
|
|
456
590
|
```
|
|
457
591
|
|
|
458
592
|
When `system_prompt` is not provided, each tool uses its built-in default — existing behavior is unchanged.
|
|
@@ -3,12 +3,14 @@ README.md
|
|
|
3
3
|
pyproject.toml
|
|
4
4
|
bridgekit/__init__.py
|
|
5
5
|
bridgekit/cli.py
|
|
6
|
+
bridgekit/compare.py
|
|
6
7
|
bridgekit/config.py
|
|
7
8
|
bridgekit/planner.py
|
|
8
9
|
bridgekit/providers.py
|
|
9
10
|
bridgekit/redteam.py
|
|
10
11
|
bridgekit/reviewer.py
|
|
11
12
|
bridgekit/search.py
|
|
13
|
+
bridgekit/summarize.py
|
|
12
14
|
bridgekit.egg-info/PKG-INFO
|
|
13
15
|
bridgekit.egg-info/SOURCES.txt
|
|
14
16
|
bridgekit.egg-info/dependency_links.txt
|
|
@@ -16,9 +18,11 @@ bridgekit.egg-info/entry_points.txt
|
|
|
16
18
|
bridgekit.egg-info/requires.txt
|
|
17
19
|
bridgekit.egg-info/top_level.txt
|
|
18
20
|
tests/test_cli.py
|
|
21
|
+
tests/test_compare.py
|
|
19
22
|
tests/test_config.py
|
|
20
23
|
tests/test_planner.py
|
|
21
24
|
tests/test_providers.py
|
|
22
25
|
tests/test_redteam.py
|
|
23
26
|
tests/test_reviewer.py
|
|
24
|
-
tests/test_search.py
|
|
27
|
+
tests/test_search.py
|
|
28
|
+
tests/test_summarize.py
|
|
@@ -0,0 +1,275 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import pytest
|
|
3
|
+
from unittest.mock import patch
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
# ---------------------------------------------------------------------------
|
|
7
|
+
# Helpers
|
|
8
|
+
# ---------------------------------------------------------------------------
|
|
9
|
+
|
|
10
|
+
FAKE_ANTHROPIC = (
|
|
11
|
+
"BRIDGEKIT ANALYSIS REVIEW\n"
|
|
12
|
+
"─────────────────────────────────────────\n\n"
|
|
13
|
+
"1. CLARITY\n"
|
|
14
|
+
"✅ STRONG — Clear and jargon-free.\n\n"
|
|
15
|
+
"─────────────────────────────────────────\n"
|
|
16
|
+
"BOTTOM LINE\n"
|
|
17
|
+
"Add quantified business impact."
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
FAKE_OPENAI = (
|
|
21
|
+
"BRIDGEKIT ANALYSIS REVIEW\n"
|
|
22
|
+
"─────────────────────────────────────────\n\n"
|
|
23
|
+
"1. CLARITY\n"
|
|
24
|
+
"⚠️ NEEDS WORK — Too much jargon.\n\n"
|
|
25
|
+
"─────────────────────────────────────────\n"
|
|
26
|
+
"BOTTOM LINE\n"
|
|
27
|
+
"Define your methodology more clearly."
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
FAKE_SUMMARY = "Both agreed on Clarity. Anthropic was harsher on Statistical Rigor."
|
|
31
|
+
|
|
32
|
+
DEFAULT_RESPONSES = {
|
|
33
|
+
"anthropic": FAKE_ANTHROPIC,
|
|
34
|
+
"openai": FAKE_OPENAI,
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _make_call_tool_side_effect(responses: dict):
|
|
39
|
+
def side_effect(tool_name, text, provider, model, kwargs):
|
|
40
|
+
return responses.get(provider, "fallback output")
|
|
41
|
+
return side_effect
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _patches(responses=None):
|
|
45
|
+
"""Context manager stacking _call_tool and _synthesize patches."""
|
|
46
|
+
if responses is None:
|
|
47
|
+
responses = DEFAULT_RESPONSES
|
|
48
|
+
|
|
49
|
+
class _Ctx:
|
|
50
|
+
def __enter__(self):
|
|
51
|
+
self._p1 = patch("bridgekit.compare._call_tool", side_effect=_make_call_tool_side_effect(responses))
|
|
52
|
+
self._p2 = patch("bridgekit.compare._synthesize", return_value=FAKE_SUMMARY)
|
|
53
|
+
self._p1.__enter__()
|
|
54
|
+
self._p2.__enter__()
|
|
55
|
+
return self
|
|
56
|
+
|
|
57
|
+
def __exit__(self, *args):
|
|
58
|
+
self._p2.__exit__(*args)
|
|
59
|
+
self._p1.__exit__(*args)
|
|
60
|
+
|
|
61
|
+
return _Ctx()
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
# ---------------------------------------------------------------------------
|
|
65
|
+
# Tests
|
|
66
|
+
# ---------------------------------------------------------------------------
|
|
67
|
+
|
|
68
|
+
class TestCompareReturnsString:
|
|
69
|
+
def test_returns_string(self):
|
|
70
|
+
with _patches():
|
|
71
|
+
from bridgekit.compare import compare
|
|
72
|
+
result = compare("Some analysis.")
|
|
73
|
+
assert isinstance(result, str)
|
|
74
|
+
|
|
75
|
+
def test_returns_non_empty_string(self):
|
|
76
|
+
with _patches():
|
|
77
|
+
from bridgekit.compare import compare
|
|
78
|
+
result = compare("Some analysis.")
|
|
79
|
+
assert len(result) > 0
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
class TestCompareOutputStructure:
|
|
83
|
+
def test_output_contains_tool_header(self):
|
|
84
|
+
with _patches():
|
|
85
|
+
from bridgekit.compare import compare
|
|
86
|
+
result = compare("Some analysis.")
|
|
87
|
+
assert "BRIDGEKIT COMPARE: EVALUATE" in result
|
|
88
|
+
|
|
89
|
+
def test_output_contains_summary_section(self):
|
|
90
|
+
with _patches():
|
|
91
|
+
from bridgekit.compare import compare
|
|
92
|
+
result = compare("Some analysis.")
|
|
93
|
+
assert "SUMMARY" in result
|
|
94
|
+
assert FAKE_SUMMARY in result
|
|
95
|
+
|
|
96
|
+
def test_summary_appears_before_provider_outputs(self):
|
|
97
|
+
with _patches():
|
|
98
|
+
from bridgekit.compare import compare
|
|
99
|
+
result = compare("Some analysis.")
|
|
100
|
+
assert result.index("SUMMARY") < result.index(FAKE_ANTHROPIC)
|
|
101
|
+
assert result.index("SUMMARY") < result.index(FAKE_OPENAI)
|
|
102
|
+
|
|
103
|
+
def test_output_contains_both_provider_labels(self):
|
|
104
|
+
with _patches():
|
|
105
|
+
from bridgekit.compare import compare
|
|
106
|
+
result = compare("Some analysis.")
|
|
107
|
+
assert "ANTHROPIC" in result
|
|
108
|
+
assert "OPENAI" in result
|
|
109
|
+
|
|
110
|
+
def test_output_contains_both_responses(self):
|
|
111
|
+
with _patches():
|
|
112
|
+
from bridgekit.compare import compare
|
|
113
|
+
result = compare("Some analysis.")
|
|
114
|
+
assert FAKE_ANTHROPIC in result
|
|
115
|
+
assert FAKE_OPENAI in result
|
|
116
|
+
|
|
117
|
+
def test_anthropic_appears_before_openai(self):
|
|
118
|
+
with _patches():
|
|
119
|
+
from bridgekit.compare import compare
|
|
120
|
+
result = compare("Some analysis.")
|
|
121
|
+
assert result.index(FAKE_ANTHROPIC) < result.index(FAKE_OPENAI)
|
|
122
|
+
|
|
123
|
+
def test_plan_tool_header(self):
|
|
124
|
+
with _patches():
|
|
125
|
+
from bridgekit.compare import compare
|
|
126
|
+
result = compare("What caused churn?", tool="plan")
|
|
127
|
+
assert "BRIDGEKIT COMPARE: PLAN" in result
|
|
128
|
+
|
|
129
|
+
def test_redteam_tool_header(self):
|
|
130
|
+
with _patches():
|
|
131
|
+
from bridgekit.compare import compare
|
|
132
|
+
result = compare("Some analysis.", tool="redteam")
|
|
133
|
+
assert "BRIDGEKIT COMPARE: REDTEAM" in result
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
class TestCompareValidation:
|
|
137
|
+
def test_empty_text_raises_value_error(self):
|
|
138
|
+
from bridgekit.compare import compare
|
|
139
|
+
with pytest.raises(ValueError, match="empty"):
|
|
140
|
+
compare("")
|
|
141
|
+
|
|
142
|
+
def test_whitespace_text_raises_value_error(self):
|
|
143
|
+
from bridgekit.compare import compare
|
|
144
|
+
with pytest.raises(ValueError, match="empty"):
|
|
145
|
+
compare(" ")
|
|
146
|
+
|
|
147
|
+
def test_invalid_tool_raises_value_error(self):
|
|
148
|
+
from bridgekit.compare import compare
|
|
149
|
+
with pytest.raises(ValueError, match="Unknown tool"):
|
|
150
|
+
compare("Some analysis.", tool="summarize")
|
|
151
|
+
|
|
152
|
+
def test_too_few_providers_raises_value_error(self):
|
|
153
|
+
from bridgekit.compare import compare
|
|
154
|
+
with pytest.raises(ValueError, match="exactly 2"):
|
|
155
|
+
compare("Some analysis.", providers=["anthropic"])
|
|
156
|
+
|
|
157
|
+
def test_too_many_providers_raises_value_error(self):
|
|
158
|
+
from bridgekit.compare import compare
|
|
159
|
+
with pytest.raises(ValueError, match="exactly 2"):
|
|
160
|
+
compare("Some analysis.", providers=["anthropic", "openai", "gemini"])
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
class TestCompareModelOverrides:
|
|
164
|
+
def test_model_a_appears_in_output(self):
|
|
165
|
+
with _patches():
|
|
166
|
+
from bridgekit.compare import compare
|
|
167
|
+
result = compare("Some analysis.", model_a="claude-haiku-4-5-20251001")
|
|
168
|
+
assert "claude-haiku-4-5-20251001" in result
|
|
169
|
+
|
|
170
|
+
def test_model_b_appears_in_output(self):
|
|
171
|
+
with _patches():
|
|
172
|
+
from bridgekit.compare import compare
|
|
173
|
+
result = compare("Some analysis.", model_b="gpt-4-turbo")
|
|
174
|
+
assert "gpt-4-turbo" in result
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
class TestCompareDefaultProviders:
|
|
178
|
+
def test_defaults_to_anthropic_and_openai(self):
|
|
179
|
+
with _patches():
|
|
180
|
+
from bridgekit.compare import compare
|
|
181
|
+
result = compare("Some analysis.")
|
|
182
|
+
assert "ANTHROPIC" in result
|
|
183
|
+
assert "OPENAI" in result
|
|
184
|
+
|
|
185
|
+
def test_default_models_shown(self):
|
|
186
|
+
with _patches():
|
|
187
|
+
from bridgekit.compare import compare
|
|
188
|
+
result = compare("Some analysis.")
|
|
189
|
+
assert "claude-opus-4-8" in result
|
|
190
|
+
assert "gpt-4o" in result
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
class TestCompareSynthesis:
|
|
194
|
+
def test_synthesize_called_once(self):
|
|
195
|
+
with patch("bridgekit.compare._call_tool", side_effect=_make_call_tool_side_effect(DEFAULT_RESPONSES)):
|
|
196
|
+
with patch("bridgekit.compare._synthesize", return_value=FAKE_SUMMARY) as mock_syn:
|
|
197
|
+
from bridgekit.compare import compare
|
|
198
|
+
compare("Some analysis.")
|
|
199
|
+
mock_syn.assert_called_once()
|
|
200
|
+
|
|
201
|
+
def test_synthesize_receives_both_outputs(self):
|
|
202
|
+
captured = []
|
|
203
|
+
|
|
204
|
+
def capture_synth(results):
|
|
205
|
+
captured.extend(results)
|
|
206
|
+
return FAKE_SUMMARY
|
|
207
|
+
|
|
208
|
+
with patch("bridgekit.compare._call_tool", side_effect=_make_call_tool_side_effect(DEFAULT_RESPONSES)):
|
|
209
|
+
with patch("bridgekit.compare._synthesize", side_effect=capture_synth):
|
|
210
|
+
from bridgekit.compare import compare
|
|
211
|
+
compare("Some analysis.")
|
|
212
|
+
|
|
213
|
+
providers_seen = {r[0] for r in captured}
|
|
214
|
+
assert "anthropic" in providers_seen
|
|
215
|
+
assert "openai" in providers_seen
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
class TestCompareApiCallShape:
|
|
219
|
+
def test_both_providers_called(self):
|
|
220
|
+
calls = []
|
|
221
|
+
|
|
222
|
+
def capture(tool_name, text, provider, model, kwargs):
|
|
223
|
+
calls.append(provider)
|
|
224
|
+
return "output"
|
|
225
|
+
|
|
226
|
+
with patch("bridgekit.compare._call_tool", side_effect=capture):
|
|
227
|
+
with patch("bridgekit.compare._synthesize", return_value=FAKE_SUMMARY):
|
|
228
|
+
from bridgekit.compare import compare
|
|
229
|
+
compare("Some analysis.")
|
|
230
|
+
|
|
231
|
+
assert sorted(calls) == ["anthropic", "openai"]
|
|
232
|
+
|
|
233
|
+
def test_user_text_passed_to_both_calls(self):
|
|
234
|
+
user_text = "Our conversion rate improved after the campaign."
|
|
235
|
+
seen_texts = []
|
|
236
|
+
|
|
237
|
+
def capture(tool_name, text, provider, model, kwargs):
|
|
238
|
+
seen_texts.append(text)
|
|
239
|
+
return "output"
|
|
240
|
+
|
|
241
|
+
with patch("bridgekit.compare._call_tool", side_effect=capture):
|
|
242
|
+
with patch("bridgekit.compare._synthesize", return_value=FAKE_SUMMARY):
|
|
243
|
+
from bridgekit.compare import compare
|
|
244
|
+
compare(user_text)
|
|
245
|
+
|
|
246
|
+
assert len(seen_texts) == 2
|
|
247
|
+
assert all(t == user_text for t in seen_texts)
|
|
248
|
+
|
|
249
|
+
def test_tool_name_passed_to_both_calls(self):
|
|
250
|
+
tool_names = []
|
|
251
|
+
|
|
252
|
+
def capture(tool_name, text, provider, model, kwargs):
|
|
253
|
+
tool_names.append(tool_name)
|
|
254
|
+
return "output"
|
|
255
|
+
|
|
256
|
+
with patch("bridgekit.compare._call_tool", side_effect=capture):
|
|
257
|
+
with patch("bridgekit.compare._synthesize", return_value=FAKE_SUMMARY):
|
|
258
|
+
from bridgekit.compare import compare
|
|
259
|
+
compare("Some analysis.", tool="redteam")
|
|
260
|
+
|
|
261
|
+
assert all(n == "redteam" for n in tool_names)
|
|
262
|
+
|
|
263
|
+
def test_kwargs_forwarded_to_tool(self):
|
|
264
|
+
received_kwargs = []
|
|
265
|
+
|
|
266
|
+
def capture(tool_name, text, provider, model, kwargs):
|
|
267
|
+
received_kwargs.append(kwargs)
|
|
268
|
+
return "output"
|
|
269
|
+
|
|
270
|
+
with patch("bridgekit.compare._call_tool", side_effect=capture):
|
|
271
|
+
with patch("bridgekit.compare._synthesize", return_value=FAKE_SUMMARY):
|
|
272
|
+
from bridgekit.compare import compare
|
|
273
|
+
compare("Some analysis.", max_tokens=2048)
|
|
274
|
+
|
|
275
|
+
assert all(kw.get("max_tokens") == 2048 for kw in received_kwargs)
|
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import pytest
|
|
3
|
+
from unittest.mock import MagicMock, patch
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
# ---------------------------------------------------------------------------
|
|
7
|
+
# Helpers
|
|
8
|
+
# ---------------------------------------------------------------------------
|
|
9
|
+
|
|
10
|
+
def _make_mock_message(text: str):
|
|
11
|
+
content_block = MagicMock()
|
|
12
|
+
content_block.text = text
|
|
13
|
+
message = MagicMock()
|
|
14
|
+
message.content = [content_block]
|
|
15
|
+
return message
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
FAKE_RESPONSE = (
|
|
19
|
+
"BRIDGEKIT SUMMARY\n"
|
|
20
|
+
"─────────────────────────────────────────\n"
|
|
21
|
+
"AUDIENCE: General Business Audience\n\n"
|
|
22
|
+
"KEY TAKEAWAY\n"
|
|
23
|
+
"The new onboarding flow increased upgrade rates by 3x within 30 days.\n\n"
|
|
24
|
+
"WHAT WE FOUND\n"
|
|
25
|
+
"- Users who engaged with reporting in week 1 were 3x more likely to upgrade\n"
|
|
26
|
+
"- The effect held across acquisition channels\n"
|
|
27
|
+
"- Sample size was 500 users\n\n"
|
|
28
|
+
"SO WHAT\n"
|
|
29
|
+
"Prioritize onboarding users to the reporting feature as a growth lever.\n"
|
|
30
|
+
"─────────────────────────────────────────\n"
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
# ---------------------------------------------------------------------------
|
|
35
|
+
# Tests
|
|
36
|
+
# ---------------------------------------------------------------------------
|
|
37
|
+
|
|
38
|
+
class TestSummarizeReturnsString:
|
|
39
|
+
"""summarize() should return a non-empty string."""
|
|
40
|
+
|
|
41
|
+
def test_returns_string(self):
|
|
42
|
+
with patch.dict(os.environ, {"ANTHROPIC_API_KEY": "test-key"}):
|
|
43
|
+
with patch("anthropic.Anthropic") as MockAnthropic:
|
|
44
|
+
mock_client = MagicMock()
|
|
45
|
+
mock_client.messages.create.return_value = _make_mock_message(FAKE_RESPONSE)
|
|
46
|
+
MockAnthropic.return_value = mock_client
|
|
47
|
+
|
|
48
|
+
from bridgekit.summarize import summarize
|
|
49
|
+
result = summarize("We ran an A/B test on 500 users and saw a 3x lift in upgrades.")
|
|
50
|
+
|
|
51
|
+
assert isinstance(result, str)
|
|
52
|
+
assert len(result) > 0
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class TestSummarizeOutputStructure:
|
|
56
|
+
"""summarize() output should contain the required section headers."""
|
|
57
|
+
|
|
58
|
+
def test_output_contains_key_takeaway(self):
|
|
59
|
+
with patch.dict(os.environ, {"ANTHROPIC_API_KEY": "test-key"}):
|
|
60
|
+
with patch("anthropic.Anthropic") as MockAnthropic:
|
|
61
|
+
mock_client = MagicMock()
|
|
62
|
+
mock_client.messages.create.return_value = _make_mock_message(FAKE_RESPONSE)
|
|
63
|
+
MockAnthropic.return_value = mock_client
|
|
64
|
+
|
|
65
|
+
from bridgekit.summarize import summarize
|
|
66
|
+
result = summarize("Some analysis text.")
|
|
67
|
+
|
|
68
|
+
assert "KEY TAKEAWAY" in result
|
|
69
|
+
|
|
70
|
+
def test_output_contains_so_what(self):
|
|
71
|
+
with patch.dict(os.environ, {"ANTHROPIC_API_KEY": "test-key"}):
|
|
72
|
+
with patch("anthropic.Anthropic") as MockAnthropic:
|
|
73
|
+
mock_client = MagicMock()
|
|
74
|
+
mock_client.messages.create.return_value = _make_mock_message(FAKE_RESPONSE)
|
|
75
|
+
MockAnthropic.return_value = mock_client
|
|
76
|
+
|
|
77
|
+
from bridgekit.summarize import summarize
|
|
78
|
+
result = summarize("Some analysis text.")
|
|
79
|
+
|
|
80
|
+
assert "SO WHAT" in result
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
class TestSummarizeMissingApiKey:
|
|
84
|
+
"""summarize() should raise EnvironmentError when the API key is absent."""
|
|
85
|
+
|
|
86
|
+
def test_raises_environment_error_when_key_missing(self):
|
|
87
|
+
env = {k: v for k, v in os.environ.items() if k != "ANTHROPIC_API_KEY"}
|
|
88
|
+
with patch.dict(os.environ, env, clear=True):
|
|
89
|
+
from bridgekit.summarize import summarize
|
|
90
|
+
with pytest.raises(EnvironmentError):
|
|
91
|
+
summarize("Some analysis text.")
|
|
92
|
+
|
|
93
|
+
def test_error_message_mentions_key(self):
|
|
94
|
+
env = {k: v for k, v in os.environ.items() if k != "ANTHROPIC_API_KEY"}
|
|
95
|
+
with patch.dict(os.environ, env, clear=True):
|
|
96
|
+
from bridgekit.summarize import summarize
|
|
97
|
+
with pytest.raises(EnvironmentError, match="ANTHROPIC_API_KEY"):
|
|
98
|
+
summarize("Some analysis text.")
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
class TestSummarizeEmptyInput:
|
|
102
|
+
"""summarize() should raise ValueError for empty or whitespace-only input."""
|
|
103
|
+
|
|
104
|
+
def test_empty_string_raises_value_error(self):
|
|
105
|
+
with patch.dict(os.environ, {"ANTHROPIC_API_KEY": "test-key"}):
|
|
106
|
+
from bridgekit.summarize import summarize
|
|
107
|
+
with pytest.raises(ValueError, match="empty"):
|
|
108
|
+
summarize("")
|
|
109
|
+
|
|
110
|
+
def test_whitespace_only_raises_value_error(self):
|
|
111
|
+
with patch.dict(os.environ, {"ANTHROPIC_API_KEY": "test-key"}):
|
|
112
|
+
from bridgekit.summarize import summarize
|
|
113
|
+
with pytest.raises(ValueError, match="empty"):
|
|
114
|
+
summarize(" ")
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
class TestSummarizeAudience:
|
|
118
|
+
"""summarize() should include a custom audience in the system prompt."""
|
|
119
|
+
|
|
120
|
+
def test_custom_audience_reaches_system_prompt(self):
|
|
121
|
+
with patch.dict(os.environ, {"ANTHROPIC_API_KEY": "test-key"}):
|
|
122
|
+
with patch("anthropic.Anthropic") as MockAnthropic:
|
|
123
|
+
mock_client = MagicMock()
|
|
124
|
+
mock_client.messages.create.return_value = _make_mock_message(FAKE_RESPONSE)
|
|
125
|
+
MockAnthropic.return_value = mock_client
|
|
126
|
+
|
|
127
|
+
from bridgekit.summarize import summarize
|
|
128
|
+
summarize("Some analysis text.", audience="VP of Marketing")
|
|
129
|
+
|
|
130
|
+
call_kwargs = mock_client.messages.create.call_args
|
|
131
|
+
assert "VP of Marketing" in call_kwargs.kwargs.get("system", "")
|
|
132
|
+
|
|
133
|
+
def test_default_audience_used_when_not_specified(self):
|
|
134
|
+
with patch.dict(os.environ, {"ANTHROPIC_API_KEY": "test-key"}):
|
|
135
|
+
with patch("anthropic.Anthropic") as MockAnthropic:
|
|
136
|
+
mock_client = MagicMock()
|
|
137
|
+
mock_client.messages.create.return_value = _make_mock_message(FAKE_RESPONSE)
|
|
138
|
+
MockAnthropic.return_value = mock_client
|
|
139
|
+
|
|
140
|
+
from bridgekit.summarize import summarize
|
|
141
|
+
summarize("Some analysis text.")
|
|
142
|
+
|
|
143
|
+
call_kwargs = mock_client.messages.create.call_args
|
|
144
|
+
assert "General Business Audience" in call_kwargs.kwargs.get("system", "")
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
class TestSummarizeCustomSystemPrompt:
|
|
148
|
+
"""summarize() should forward a custom system_prompt to the API, ignoring audience."""
|
|
149
|
+
|
|
150
|
+
def test_custom_system_prompt_reaches_api(self):
|
|
151
|
+
custom_prompt = "You are a data journalist writing a one-paragraph news brief."
|
|
152
|
+
with patch.dict(os.environ, {"ANTHROPIC_API_KEY": "test-key"}):
|
|
153
|
+
with patch("anthropic.Anthropic") as MockAnthropic:
|
|
154
|
+
mock_client = MagicMock()
|
|
155
|
+
mock_client.messages.create.return_value = _make_mock_message(FAKE_RESPONSE)
|
|
156
|
+
MockAnthropic.return_value = mock_client
|
|
157
|
+
|
|
158
|
+
from bridgekit.summarize import summarize
|
|
159
|
+
summarize("Some analysis text.", system_prompt=custom_prompt)
|
|
160
|
+
|
|
161
|
+
call_kwargs = mock_client.messages.create.call_args
|
|
162
|
+
assert call_kwargs.kwargs.get("system") == custom_prompt
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
class TestSummarizeApiCallShape:
|
|
166
|
+
"""summarize() should pass the user text through to the Anthropic API."""
|
|
167
|
+
|
|
168
|
+
def test_api_called_with_user_text(self):
|
|
169
|
+
user_text = "Our conversion rate improved after the campaign."
|
|
170
|
+
with patch.dict(os.environ, {"ANTHROPIC_API_KEY": "test-key"}):
|
|
171
|
+
with patch("anthropic.Anthropic") as MockAnthropic:
|
|
172
|
+
mock_client = MagicMock()
|
|
173
|
+
mock_client.messages.create.return_value = _make_mock_message(FAKE_RESPONSE)
|
|
174
|
+
MockAnthropic.return_value = mock_client
|
|
175
|
+
|
|
176
|
+
from bridgekit.summarize import summarize
|
|
177
|
+
summarize(user_text)
|
|
178
|
+
|
|
179
|
+
call_kwargs = mock_client.messages.create.call_args
|
|
180
|
+
messages_arg = call_kwargs.kwargs.get("messages") or call_kwargs.args[0]
|
|
181
|
+
content = str(messages_arg)
|
|
182
|
+
assert user_text in content
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
class TestSummarizeMaxTokens:
|
|
186
|
+
"""summarize() should pass max_tokens through to the API."""
|
|
187
|
+
|
|
188
|
+
def test_default_max_tokens_is_1024(self):
|
|
189
|
+
with patch.dict(os.environ, {"ANTHROPIC_API_KEY": "test-key"}):
|
|
190
|
+
with patch("anthropic.Anthropic") as MockAnthropic:
|
|
191
|
+
mock_client = MagicMock()
|
|
192
|
+
mock_client.messages.create.return_value = _make_mock_message(FAKE_RESPONSE)
|
|
193
|
+
MockAnthropic.return_value = mock_client
|
|
194
|
+
|
|
195
|
+
from bridgekit.summarize import summarize
|
|
196
|
+
summarize("Some analysis text.")
|
|
197
|
+
|
|
198
|
+
call_kwargs = mock_client.messages.create.call_args
|
|
199
|
+
assert call_kwargs.kwargs.get("max_tokens") == 1024
|
|
200
|
+
|
|
201
|
+
def test_custom_max_tokens_reaches_api(self):
|
|
202
|
+
with patch.dict(os.environ, {"ANTHROPIC_API_KEY": "test-key"}):
|
|
203
|
+
with patch("anthropic.Anthropic") as MockAnthropic:
|
|
204
|
+
mock_client = MagicMock()
|
|
205
|
+
mock_client.messages.create.return_value = _make_mock_message(FAKE_RESPONSE)
|
|
206
|
+
MockAnthropic.return_value = mock_client
|
|
207
|
+
|
|
208
|
+
from bridgekit.summarize import summarize
|
|
209
|
+
summarize("Some analysis text.", max_tokens=2048)
|
|
210
|
+
|
|
211
|
+
call_kwargs = mock_client.messages.create.call_args
|
|
212
|
+
assert call_kwargs.kwargs.get("max_tokens") == 2048
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|