interp-engine 0.0.24__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,3125 @@
1
+ # Edited examples - adds top_positive_logits to the examples
2
+ # Few-shot examples for generating and simulating neuron explanations.
3
+
4
+ from __future__ import annotations
5
+
6
+ from dataclasses import dataclass
7
+ from enum import Enum
8
+ from typing import List, Optional
9
+
10
+ from neuron_explainer.activations.activations import ActivationRecord
11
+ from neuron_explainer.fast_dataclasses import FastDataclass
12
+
13
+
14
+ @dataclass
15
+ class Example(FastDataclass):
16
+ activation_records: List[ActivationRecord]
17
+ explanation: str
18
+ first_revealed_activation_indices: List[int]
19
+ """
20
+ For each activation record, the index of the first token for which the activation value in the
21
+ prompt should be an actual number rather than "unknown".
22
+
23
+ Examples all start with the activations rendered as "unknown", then transition to revealing
24
+ specific normalized activation values. The goal is to lead the model to predict that activation
25
+ sequences will eventually transition to predicting specific activation values instead of just
26
+ "unknown". This lets us cheat and get predictions of activation values for every token in a
27
+ single round of inference by having the activations in the sequence we're predicting always be
28
+ "unknown" in the prompt: the model will always think that maybe the next token will be a real
29
+ activation.
30
+ """
31
+ token_index_to_score: Optional[int] = None
32
+ """
33
+ If the prompt is used as an example for one-token-at-a-time scoring, this is the index of the
34
+ token to score.
35
+ """
36
+ top_positive_logits: Optional[List[str]] = None
37
+
38
+
39
+ class FewShotExampleSet(Enum):
40
+ """Determines which few-shot examples to use when sampling explanations."""
41
+
42
+ ORIGINAL = "original"
43
+ NEWER = "newer"
44
+ TEST = "test"
45
+ LOGITS = "logits"
46
+ ACTIVATIONS = "activations"
47
+ JL_FINE_TUNED = "jl_fine_tuned"
48
+
49
+ @classmethod
50
+ def from_string(cls, string: str) -> FewShotExampleSet:
51
+ for example_set in FewShotExampleSet:
52
+ if example_set.value == string:
53
+ return example_set
54
+ raise ValueError(f"Unrecognized example set: {string}")
55
+
56
+ def get_examples(self) -> list[Example]:
57
+ """Returns regular examples for use in a few-shot prompt."""
58
+ if self is FewShotExampleSet.ORIGINAL:
59
+ return ORIGINAL_EXAMPLES
60
+ elif self is FewShotExampleSet.LOGITS:
61
+ return LOGITS_EXAMPLES
62
+ elif self is FewShotExampleSet.ACTIVATIONS:
63
+ return ACTIVATIONS_EXAMPLES
64
+ elif self is FewShotExampleSet.NEWER:
65
+ return NEWER_EXAMPLES
66
+ elif self is FewShotExampleSet.TEST:
67
+ return TEST_EXAMPLES
68
+ elif self is FewShotExampleSet.JL_FINE_TUNED:
69
+ return JL_FINE_TUNED_EXAMPLES
70
+ else:
71
+ raise ValueError(f"Unhandled example set: {self}")
72
+
73
+ def get_single_token_prediction_example(self) -> Example:
74
+ """
75
+ Returns an example suitable for use in a subprompt for predicting a single token's
76
+ normalized activation, for use with the "one token at a time" scoring approach.
77
+ """
78
+ if self is FewShotExampleSet.NEWER:
79
+ return NEWER_SINGLE_TOKEN_EXAMPLE
80
+ elif self is FewShotExampleSet.TEST:
81
+ return TEST_SINGLE_TOKEN_EXAMPLE
82
+ else:
83
+ raise ValueError(f"Unhandled example set: {self}")
84
+
85
+
86
+ TEST_EXAMPLES = [
87
+ Example(
88
+ activation_records=[
89
+ ActivationRecord(
90
+ tokens=["a", "b", "c"],
91
+ activations=[1.0, 0.0, 0.0],
92
+ ),
93
+ ActivationRecord(
94
+ tokens=["d", "e", "f"],
95
+ activations=[0.0, 1.0, 0.0],
96
+ ),
97
+ ],
98
+ explanation="vowels",
99
+ first_revealed_activation_indices=[0, 1],
100
+ ),
101
+ ]
102
+
103
+ TEST_SINGLE_TOKEN_EXAMPLE = Example(
104
+ activation_records=[
105
+ ActivationRecord(
106
+ activations=[0.0, 0.0, 1.0],
107
+ tokens=["g", "h", "i"],
108
+ ),
109
+ ],
110
+ first_revealed_activation_indices=[],
111
+ token_index_to_score=2,
112
+ explanation="test explanation",
113
+ )
114
+
115
+
116
+ ORIGINAL_EXAMPLES = [
117
+ Example(
118
+ activation_records=[
119
+ ActivationRecord(
120
+ tokens=[
121
+ "t",
122
+ "urt",
123
+ "ur",
124
+ "ro",
125
+ " is",
126
+ " fab",
127
+ "ulously",
128
+ " funny",
129
+ " and",
130
+ " over",
131
+ " the",
132
+ " top",
133
+ " as",
134
+ " a",
135
+ " '",
136
+ "very",
137
+ " sneaky",
138
+ "'",
139
+ " but",
140
+ "ler",
141
+ " who",
142
+ " excel",
143
+ "s",
144
+ " in",
145
+ " the",
146
+ " art",
147
+ " of",
148
+ " impossible",
149
+ " disappearing",
150
+ "/",
151
+ "re",
152
+ "app",
153
+ "earing",
154
+ " acts",
155
+ ],
156
+ activations=[
157
+ -0.71,
158
+ -1.85,
159
+ -2.39,
160
+ -2.58,
161
+ -1.34,
162
+ -1.92,
163
+ -1.69,
164
+ -0.84,
165
+ -1.25,
166
+ -1.75,
167
+ -1.42,
168
+ -1.47,
169
+ -1.51,
170
+ -0.8,
171
+ -1.89,
172
+ -1.56,
173
+ -1.63,
174
+ 0.44,
175
+ -1.87,
176
+ -2.55,
177
+ -2.09,
178
+ -1.76,
179
+ -1.33,
180
+ -0.88,
181
+ -1.63,
182
+ -2.39,
183
+ -2.63,
184
+ -0.99,
185
+ 2.83,
186
+ -1.11,
187
+ -1.19,
188
+ -1.33,
189
+ 4.24,
190
+ -1.51,
191
+ ],
192
+ ),
193
+ ActivationRecord(
194
+ tokens=[
195
+ "esc",
196
+ "aping",
197
+ " the",
198
+ " studio",
199
+ " ,",
200
+ " pic",
201
+ "col",
202
+ "i",
203
+ " is",
204
+ " warm",
205
+ "ly",
206
+ " affecting",
207
+ " and",
208
+ " so",
209
+ " is",
210
+ " this",
211
+ " ad",
212
+ "roit",
213
+ "ly",
214
+ " minimalist",
215
+ " movie",
216
+ " .",
217
+ ],
218
+ activations=[
219
+ -0.69,
220
+ 4.12,
221
+ 1.83,
222
+ -2.28,
223
+ -0.28,
224
+ -0.79,
225
+ -2.2,
226
+ -2.03,
227
+ -1.77,
228
+ -1.71,
229
+ -2.44,
230
+ 1.6,
231
+ -1,
232
+ -0.38,
233
+ -1.93,
234
+ -2.09,
235
+ -1.63,
236
+ -1.94,
237
+ -1.82,
238
+ -1.64,
239
+ -1.32,
240
+ -1.92,
241
+ ],
242
+ ),
243
+ ],
244
+ first_revealed_activation_indices=[10, 3],
245
+ explanation="present tense verbs ending in 'ing'",
246
+ top_positive_logits=[
247
+ " calling",
248
+ " sleeping",
249
+ " laughing",
250
+ " dancing",
251
+ " singing",
252
+ " frying",
253
+ " grounding",
254
+ " laying",
255
+ " eating",
256
+ " meeting",
257
+ ],
258
+ ),
259
+ Example(
260
+ activation_records=[
261
+ ActivationRecord(
262
+ tokens=[
263
+ "as",
264
+ " sac",
265
+ "char",
266
+ "ine",
267
+ " movies",
268
+ " go",
269
+ " ,",
270
+ " this",
271
+ " is",
272
+ " likely",
273
+ " to",
274
+ " cause",
275
+ " massive",
276
+ " cardiac",
277
+ " arrest",
278
+ " if",
279
+ " taken",
280
+ " in",
281
+ " large",
282
+ " doses",
283
+ " .",
284
+ ],
285
+ activations=[
286
+ -0.14,
287
+ -1.37,
288
+ -0.68,
289
+ -2.27,
290
+ -1.46,
291
+ -1.11,
292
+ -0.9,
293
+ -2.48,
294
+ -2.07,
295
+ -3.49,
296
+ -2.16,
297
+ -1.79,
298
+ -0.23,
299
+ -0.04,
300
+ 4.46,
301
+ -1.02,
302
+ -2.26,
303
+ -2.95,
304
+ -1.49,
305
+ -1.46,
306
+ -0.6,
307
+ ],
308
+ ),
309
+ ActivationRecord(
310
+ tokens=[
311
+ "shot",
312
+ " perhaps",
313
+ " '",
314
+ "art",
315
+ "istically",
316
+ "'",
317
+ " with",
318
+ " handheld",
319
+ " cameras",
320
+ " and",
321
+ " apparently",
322
+ " no",
323
+ " movie",
324
+ " lights",
325
+ " by",
326
+ " jo",
327
+ "aquin",
328
+ " b",
329
+ "aca",
330
+ "-",
331
+ "as",
332
+ "ay",
333
+ " ,",
334
+ " the",
335
+ " low",
336
+ "-",
337
+ "budget",
338
+ " production",
339
+ " swings",
340
+ " annoy",
341
+ "ingly",
342
+ " between",
343
+ " vert",
344
+ "igo",
345
+ " and",
346
+ " opacity",
347
+ " .",
348
+ ],
349
+ activations=[
350
+ -0.09,
351
+ -3.53,
352
+ -0.72,
353
+ -2.36,
354
+ -1.05,
355
+ -1.12,
356
+ -2.49,
357
+ -2.14,
358
+ -1.98,
359
+ -1.59,
360
+ -2.62,
361
+ -2,
362
+ -2.73,
363
+ -2.87,
364
+ -3.23,
365
+ -1.11,
366
+ -2.23,
367
+ -0.97,
368
+ -2.28,
369
+ -2.37,
370
+ -1.5,
371
+ -2.81,
372
+ -1.73,
373
+ -3.14,
374
+ -2.61,
375
+ -1.7,
376
+ -3.08,
377
+ -4,
378
+ -0.71,
379
+ -2.48,
380
+ -1.39,
381
+ -1.96,
382
+ -1.09,
383
+ 4.37,
384
+ -0.74,
385
+ -0.5,
386
+ -0.62,
387
+ ],
388
+ ),
389
+ ],
390
+ first_revealed_activation_indices=[5, 20],
391
+ explanation="words related to physical medical conditions",
392
+ top_positive_logits=[
393
+ " fainting",
394
+ " bleed",
395
+ " stroke",
396
+ " hernia",
397
+ " disease",
398
+ " broken",
399
+ " cast",
400
+ " emergency",
401
+ " sprain",
402
+ " pain",
403
+ ],
404
+ ),
405
+ Example(
406
+ activation_records=[
407
+ ActivationRecord(
408
+ tokens=[
409
+ "the",
410
+ " sense",
411
+ " of",
412
+ " together",
413
+ "ness",
414
+ " in",
415
+ " our",
416
+ " town",
417
+ " is",
418
+ " strong",
419
+ " .",
420
+ ],
421
+ activations=[
422
+ 0,
423
+ 0,
424
+ 0,
425
+ 1,
426
+ 2,
427
+ 0,
428
+ 0.23,
429
+ 0.5,
430
+ 0,
431
+ 0,
432
+ 0,
433
+ ],
434
+ ),
435
+ ActivationRecord(
436
+ tokens=[
437
+ "a",
438
+ " buoy",
439
+ "ant",
440
+ " romantic",
441
+ " comedy",
442
+ " about",
443
+ " friendship",
444
+ " ,",
445
+ " love",
446
+ " ,",
447
+ " and",
448
+ " the",
449
+ " truth",
450
+ " that",
451
+ " we",
452
+ "'re",
453
+ " all",
454
+ " in",
455
+ " this",
456
+ " together",
457
+ " .",
458
+ ],
459
+ activations=[
460
+ -0.15,
461
+ -2.33,
462
+ -1.4,
463
+ -2.17,
464
+ -2.53,
465
+ -0.85,
466
+ 0.23,
467
+ -1.89,
468
+ 0.09,
469
+ -0.47,
470
+ -0.5,
471
+ -0.58,
472
+ -0.87,
473
+ 0.22,
474
+ 0.58,
475
+ 1.34,
476
+ 0.98,
477
+ 2.21,
478
+ 2.84,
479
+ 1.7,
480
+ -0.89,
481
+ ],
482
+ ),
483
+ ],
484
+ first_revealed_activation_indices=[0, 10],
485
+ explanation="phrases related to community",
486
+ top_positive_logits=[
487
+ " community",
488
+ " together",
489
+ " love",
490
+ " friendship",
491
+ " friendly",
492
+ " joint",
493
+ " peace",
494
+ " companion",
495
+ " team",
496
+ " sympathy",
497
+ ],
498
+ ),
499
+ ]
500
+
501
+
502
+ LOGITS_EXAMPLES = [
503
+ Example(
504
+ activation_records=[
505
+ ActivationRecord(
506
+ tokens=[
507
+ "She",
508
+ " was",
509
+ " taking",
510
+ " a",
511
+ " nap",
512
+ " when",
513
+ " her",
514
+ " phone",
515
+ " started",
516
+ " ringing",
517
+ ".",
518
+ ],
519
+ activations=[8, 0, 0, 0, 7, 0, 0, 0, 0, 0, 0],
520
+ ),
521
+ ActivationRecord(
522
+ tokens=[
523
+ "I",
524
+ " enjoy",
525
+ " watching",
526
+ " movies",
527
+ " with",
528
+ " my",
529
+ " family",
530
+ ".",
531
+ ],
532
+ activations=[0, 9, 0, 7.5, 0, 0, 0, 0],
533
+ ),
534
+ ],
535
+ first_revealed_activation_indices=[0, 1],
536
+ explanation='Method 1 fails: MAX_ACTIVATING_TOKENS (She, enjoy) are not similar tokens.\nMethod 2 succeeds: All TOKENS_AFTER_MAX_ACTIVATING_TOKEN have a pattern in common: they all start with "w".\nExplanation: say "w" words',
537
+ top_positive_logits=[
538
+ " walking",
539
+ " WA",
540
+ " waiting",
541
+ " was",
542
+ " we",
543
+ " WHAM",
544
+ " wish",
545
+ " win",
546
+ " wake",
547
+ " whisper",
548
+ ],
549
+ ),
550
+ Example(
551
+ activation_records=[
552
+ ActivationRecord(
553
+ tokens=[
554
+ "The",
555
+ " apple",
556
+ " and",
557
+ " banana",
558
+ " are",
559
+ " delicious",
560
+ " foods",
561
+ " that",
562
+ " provide",
563
+ " essential",
564
+ " vitamins",
565
+ " and",
566
+ " nutrients",
567
+ ".",
568
+ ],
569
+ activations=[0, 20, 0, 30, 0, 0, 1, 0, 0, 0, 0, 0, 0, 0],
570
+ ),
571
+ ActivationRecord(
572
+ tokens=[
573
+ "I",
574
+ " enjoy",
575
+ " eating",
576
+ " fresh",
577
+ " strawberries",
578
+ ",",
579
+ " blueberries",
580
+ ",",
581
+ " and",
582
+ " mangoes",
583
+ " during",
584
+ " the",
585
+ " summer",
586
+ " months",
587
+ ".",
588
+ ],
589
+ activations=[0, 0, 0, 0, 25, 0, 35, 0, 0, 15, 0, 0, 0, 0, 0],
590
+ ),
591
+ ],
592
+ first_revealed_activation_indices=[1, 4],
593
+ explanation="Method 1 succeeds: All MAX_ACTIVATING_TOKENS (banana, blueberries) are fruits.\nExplanation: fruits",
594
+ top_positive_logits=[
595
+ " apple",
596
+ " orange",
597
+ " pineapple",
598
+ " watermelon",
599
+ " kiwi",
600
+ " peach",
601
+ " pear",
602
+ " grape",
603
+ " cherry",
604
+ " plum",
605
+ ],
606
+ ),
607
+ Example(
608
+ activation_records=[
609
+ ActivationRecord(
610
+ tokens=[
611
+ "It",
612
+ " was",
613
+ " a",
614
+ " beautiful",
615
+ " day",
616
+ " outside",
617
+ " with",
618
+ " clear",
619
+ " skies",
620
+ " and",
621
+ " warm",
622
+ " sunshine",
623
+ ".",
624
+ ],
625
+ activations=[0, 0, 0, 0, 0, 0, 0, 0, 0, 8, 0, 0, 0],
626
+ ),
627
+ ActivationRecord(
628
+ tokens=[
629
+ "And",
630
+ " the",
631
+ " garden",
632
+ " has",
633
+ " roses",
634
+ " and",
635
+ " tulips",
636
+ " and",
637
+ " daisies",
638
+ " and",
639
+ " sunflowers",
640
+ " blooming",
641
+ " together",
642
+ ".",
643
+ ],
644
+ activations=[9, 0, 0, 0, 0, 6, 0, 5, 0, 5, 0, 0, 0, 0],
645
+ ),
646
+ ],
647
+ first_revealed_activation_indices=[1, 4],
648
+ explanation='Method 1 succeeds: All MAX_ACTIVATING_TOKENS are the word "and".\nExplanation: and',
649
+ top_positive_logits=[
650
+ " elephant",
651
+ " guitar",
652
+ " mountain",
653
+ " bicycle",
654
+ " ocean",
655
+ " telescope",
656
+ " candle",
657
+ " umbrella",
658
+ " tornado",
659
+ " butterfly",
660
+ ],
661
+ ),
662
+ Example(
663
+ activation_records=[
664
+ ActivationRecord(
665
+ tokens=[
666
+ "the",
667
+ " civil",
668
+ " war",
669
+ " was",
670
+ " a",
671
+ " major",
672
+ " topic",
673
+ " in",
674
+ " history",
675
+ " class",
676
+ " .",
677
+ ],
678
+ activations=[
679
+ 0,
680
+ 0,
681
+ 15,
682
+ 0,
683
+ 0,
684
+ 0,
685
+ 0,
686
+ 0,
687
+ 0,
688
+ 0,
689
+ 0,
690
+ ],
691
+ ),
692
+ ActivationRecord(
693
+ tokens=[
694
+ " seasons",
695
+ " of",
696
+ " the",
697
+ " year",
698
+ " are",
699
+ " winter",
700
+ " ,",
701
+ " spring",
702
+ " ,",
703
+ " summer",
704
+ " ,",
705
+ " and",
706
+ " fall",
707
+ " or",
708
+ " autumn",
709
+ " in",
710
+ " some",
711
+ " places",
712
+ " .",
713
+ ],
714
+ activations=[
715
+ 0,
716
+ 0,
717
+ 0,
718
+ 0,
719
+ 0,
720
+ 0,
721
+ 0,
722
+ 0,
723
+ 0,
724
+ 0,
725
+ 0,
726
+ 0,
727
+ 0,
728
+ 0,
729
+ 0,
730
+ 0,
731
+ 10,
732
+ 0,
733
+ 0,
734
+ ],
735
+ ),
736
+ ],
737
+ first_revealed_activation_indices=[2, 1],
738
+ explanation="Method 1 fails: MAX_ACTIVATING_TOKENS (war, some) are not all the same token.\nMethod 2 fails: TOKENS_AFTER_MAX_ACTIVATING_TOKEN (was, places) are not all similar tokens and don't have a text pattern in common.\nMethod 3 succeeds: All TOP_POSITIVE_LOGITS are the number 4.\nExplanation: 4",
739
+ top_positive_logits=[
740
+ " 4",
741
+ " four",
742
+ " fourth",
743
+ " 4th",
744
+ " IV",
745
+ " Four",
746
+ " FOUR",
747
+ " ~4",
748
+ " 4.0",
749
+ " quartet",
750
+ ],
751
+ ),
752
+ ]
753
+
754
+
755
+ ACTIVATIONS_EXAMPLES = [
756
+ Example(
757
+ activation_records=[
758
+ ActivationRecord(
759
+ tokens=[
760
+ "The",
761
+ " apple",
762
+ " and",
763
+ " banana",
764
+ " are",
765
+ " delicious",
766
+ " foods",
767
+ " that",
768
+ " provide",
769
+ " essential",
770
+ " vitamins",
771
+ " and",
772
+ " nutrients",
773
+ ".",
774
+ ],
775
+ activations=[0, 20, 0, 30, 0, 0, 1, 0, 0, 0, 0, 0, 0, 0],
776
+ ),
777
+ ActivationRecord(
778
+ tokens=[
779
+ "I",
780
+ " enjoy",
781
+ " eating",
782
+ " fresh",
783
+ " strawberries",
784
+ ",",
785
+ " blueberries",
786
+ ",",
787
+ " and",
788
+ " mangoes",
789
+ " during",
790
+ " the",
791
+ " summer",
792
+ " months",
793
+ ".",
794
+ ],
795
+ activations=[0, 0, 0, 0, 25, 0, 35, 0, 0, 15, 0, 0, 0, 0, 0],
796
+ ),
797
+ ],
798
+ first_revealed_activation_indices=[1, 4],
799
+ explanation="Method 1 succeeds: All MAX_ACTIVATING_TOKENS (banana, blueberries) are fruits.\nExplanation: fruits",
800
+ ),
801
+ Example(
802
+ activation_records=[
803
+ ActivationRecord(
804
+ tokens=[
805
+ "It",
806
+ " was",
807
+ " a",
808
+ " beautiful",
809
+ " day",
810
+ " outside",
811
+ " with",
812
+ " clear",
813
+ " skies",
814
+ " and",
815
+ " warm",
816
+ " sunshine",
817
+ ".",
818
+ ],
819
+ activations=[0, 0, 0, 0, 0, 0, 0, 0, 0, 8, 0, 0, 0],
820
+ ),
821
+ ActivationRecord(
822
+ tokens=[
823
+ "And",
824
+ " the",
825
+ " garden",
826
+ " has",
827
+ " roses",
828
+ " and",
829
+ " tulips",
830
+ " and",
831
+ " daisies",
832
+ " and",
833
+ " sunflowers",
834
+ " blooming",
835
+ " together",
836
+ ".",
837
+ ],
838
+ activations=[9, 0, 0, 0, 0, 6, 0, 5, 0, 5, 0, 0, 0, 0],
839
+ ),
840
+ ],
841
+ first_revealed_activation_indices=[1, 4],
842
+ explanation='Method 1 succeeds: All MAX_ACTIVATING_TOKENS are the word "and".\nExplanation: and',
843
+ ),
844
+ Example(
845
+ activation_records=[
846
+ ActivationRecord(
847
+ tokens=[
848
+ "the",
849
+ " civil",
850
+ " war",
851
+ " was",
852
+ " a",
853
+ " major",
854
+ " topic",
855
+ " in",
856
+ " history",
857
+ " class",
858
+ " .",
859
+ ],
860
+ activations=[
861
+ 0,
862
+ 5,
863
+ 10,
864
+ 3,
865
+ 2,
866
+ 0,
867
+ 3,
868
+ 4,
869
+ 3,
870
+ 20,
871
+ 2,
872
+ ],
873
+ ),
874
+ ActivationRecord(
875
+ tokens=[
876
+ " her",
877
+ " professor",
878
+ " teaches",
879
+ " psychology",
880
+ " courses",
881
+ " and",
882
+ " is",
883
+ " a",
884
+ " tough",
885
+ " grader",
886
+ " .",
887
+ ],
888
+ activations=[0, 8, 15, 5, 10, 0, 0, 0, 2, 10, 2],
889
+ ),
890
+ ],
891
+ first_revealed_activation_indices=[2, 1],
892
+ explanation="Method 1 fails: MAX_ACTIVATING_TOKENS (class, teaches) are not all the same token.\nMethod 2 succeeds: The TOP_ACTIVATING_TEXTS are broadly about education.\nExplanation: education",
893
+ ),
894
+ # # It doesn't seem to work as well when you give a 'fallback' example like this one - the model is too tempted to always use the fallback. So we comment it out.
895
+ # Example(
896
+ # activation_records=[
897
+ # ActivationRecord(
898
+ # tokens=[
899
+ # "the",
900
+ # " bright",
901
+ # " red",
902
+ # " apple",
903
+ # " fell",
904
+ # " from",
905
+ # " the",
906
+ # " tree",
907
+ # " and",
908
+ # " rolled",
909
+ # " down",
910
+ # " the",
911
+ # " hill",
912
+ # ".",
913
+ # ],
914
+ # activations=[
915
+ # 0,
916
+ # 2,
917
+ # 1,
918
+ # 15,
919
+ # 3,
920
+ # 0,
921
+ # 0,
922
+ # 5,
923
+ # 2,
924
+ # 8,
925
+ # 4,
926
+ # 0,
927
+ # 1,
928
+ # 0,
929
+ # ],
930
+ # ),
931
+ # ActivationRecord(
932
+ # tokens=[
933
+ # " the",
934
+ # " rocket",
935
+ # " launch",
936
+ # " was",
937
+ # " scheduled",
938
+ # " for",
939
+ # " midnight",
940
+ # " but",
941
+ # " weather",
942
+ # " conditions",
943
+ # " caused",
944
+ # " a",
945
+ # " delay",
946
+ # ".",
947
+ # ],
948
+ # activations=[0, 3, 12, 1, 2, 0, 8, 1, 4, 6, 2, 0, 5, 0],
949
+ # ),
950
+ # ],
951
+ # first_revealed_activation_indices=[3, 2],
952
+ # explanation="Method 1 fails: MAX_ACTIVATING_TOKENS (apple, launch) are not all the same token.\nMethod 2 fails: The TOP_ACTIVATING_TEXTS are completely unrelated with no overlap whatsoever (fruit falling vs space launch).\nSince both fail, I'll return the first token in MAX_ACTIVATING_TOKENS.\nExplanation: apple",
953
+ # ),
954
+ ]
955
+
956
+ NEWER_EXAMPLES = [
957
+ Example(
958
+ activation_records=[
959
+ ActivationRecord(
960
+ tokens=[
961
+ "The",
962
+ " editors",
963
+ " of",
964
+ " Bi",
965
+ "opol",
966
+ "ym",
967
+ "ers",
968
+ " are",
969
+ " delighted",
970
+ " to",
971
+ " present",
972
+ " the",
973
+ " ",
974
+ "201",
975
+ "8",
976
+ " Murray",
977
+ " Goodman",
978
+ " Memorial",
979
+ " Prize",
980
+ " to",
981
+ " Professor",
982
+ " David",
983
+ " N",
984
+ ".",
985
+ " Ber",
986
+ "atan",
987
+ " in",
988
+ " recognition",
989
+ " of",
990
+ " his",
991
+ " seminal",
992
+ " contributions",
993
+ " to",
994
+ " bi",
995
+ "oph",
996
+ "ysics",
997
+ " and",
998
+ " their",
999
+ " impact",
1000
+ " on",
1001
+ " our",
1002
+ " understanding",
1003
+ " of",
1004
+ " charge",
1005
+ " transport",
1006
+ " in",
1007
+ " biom",
1008
+ "olecules",
1009
+ ".\n\n",
1010
+ "In",
1011
+ "aug",
1012
+ "ur",
1013
+ "ated",
1014
+ " in",
1015
+ " ",
1016
+ "200",
1017
+ "7",
1018
+ " in",
1019
+ " honor",
1020
+ " of",
1021
+ " the",
1022
+ " Bi",
1023
+ "opol",
1024
+ "ym",
1025
+ "ers",
1026
+ " Found",
1027
+ "ing",
1028
+ " Editor",
1029
+ ",",
1030
+ " the",
1031
+ " prize",
1032
+ " is",
1033
+ " awarded",
1034
+ " for",
1035
+ " outstanding",
1036
+ " accomplishments",
1037
+ ],
1038
+ activations=[
1039
+ 0,
1040
+ 0.01,
1041
+ 0.01,
1042
+ 0,
1043
+ 0,
1044
+ 0,
1045
+ -0.01,
1046
+ 0,
1047
+ -0.01,
1048
+ 0,
1049
+ 0,
1050
+ 0,
1051
+ 0,
1052
+ 0,
1053
+ 0.04,
1054
+ 0,
1055
+ 0,
1056
+ 0,
1057
+ 0,
1058
+ 0,
1059
+ 0,
1060
+ 0,
1061
+ 0,
1062
+ 0,
1063
+ 0,
1064
+ 0,
1065
+ 0,
1066
+ 0,
1067
+ 0,
1068
+ 0,
1069
+ 3.39,
1070
+ 0.12,
1071
+ 0,
1072
+ -0.01,
1073
+ 0,
1074
+ 0,
1075
+ 0,
1076
+ 0,
1077
+ -0,
1078
+ 0,
1079
+ -0,
1080
+ 0,
1081
+ 0,
1082
+ -0,
1083
+ 0,
1084
+ 0,
1085
+ 0,
1086
+ 0,
1087
+ 0,
1088
+ 0,
1089
+ 0,
1090
+ 0,
1091
+ 0,
1092
+ 0,
1093
+ 0,
1094
+ 0,
1095
+ 0,
1096
+ 0,
1097
+ 0,
1098
+ 0,
1099
+ 0,
1100
+ -0,
1101
+ 0,
1102
+ 0,
1103
+ -0.01,
1104
+ 0,
1105
+ 0.41,
1106
+ 0,
1107
+ 0,
1108
+ 0,
1109
+ -0.01,
1110
+ 0,
1111
+ 0,
1112
+ 0,
1113
+ 0,
1114
+ 0,
1115
+ ],
1116
+ ),
1117
+ # We sometimes exceed the max context size when this is included :(
1118
+ # ActivationRecord(
1119
+ # tokens=[
1120
+ # " We",
1121
+ # " are",
1122
+ # " proud",
1123
+ # " of",
1124
+ # " our",
1125
+ # " national",
1126
+ # " achievements",
1127
+ # " in",
1128
+ # " mastering",
1129
+ # " all",
1130
+ # " aspects",
1131
+ # " of",
1132
+ # " the",
1133
+ # " fuel",
1134
+ # " cycle",
1135
+ # ".",
1136
+ # " The",
1137
+ # " current",
1138
+ # " international",
1139
+ # " interest",
1140
+ # " in",
1141
+ # " closing",
1142
+ # " the",
1143
+ # " fuel",
1144
+ # " cycle",
1145
+ # " is",
1146
+ # " a",
1147
+ # " vind",
1148
+ # "ication",
1149
+ # " of",
1150
+ # " Dr",
1151
+ # ".",
1152
+ # " B",
1153
+ # "hab",
1154
+ # "ha",
1155
+ # "’s",
1156
+ # " pioneering",
1157
+ # " vision",
1158
+ # " and",
1159
+ # " genius",
1160
+ # ],
1161
+ # activations=[
1162
+ # -0,
1163
+ # -0,
1164
+ # 0,
1165
+ # -0,
1166
+ # -0,
1167
+ # 0,
1168
+ # 0,
1169
+ # 0,
1170
+ # -0,
1171
+ # 0,
1172
+ # 0,
1173
+ # -0,
1174
+ # 0,
1175
+ # -0.01,
1176
+ # 0,
1177
+ # 0,
1178
+ # -0,
1179
+ # -0,
1180
+ # 0,
1181
+ # 0,
1182
+ # 0,
1183
+ # -0,
1184
+ # -0,
1185
+ # -0.01,
1186
+ # 0,
1187
+ # 0,
1188
+ # -0,
1189
+ # 0,
1190
+ # 0,
1191
+ # 0,
1192
+ # 0,
1193
+ # 0,
1194
+ # -0,
1195
+ # 0,
1196
+ # 0,
1197
+ # 0,
1198
+ # 2.15,
1199
+ # 0,
1200
+ # 0,
1201
+ # 0.03,
1202
+ # ],
1203
+ # ),
1204
+ ],
1205
+ first_revealed_activation_indices=[7], # , 19],
1206
+ explanation="language related to something being groundbreaking",
1207
+ ),
1208
+ Example(
1209
+ activation_records=[
1210
+ ActivationRecord(
1211
+ tokens=[
1212
+ '{"',
1213
+ "widget",
1214
+ "Class",
1215
+ '":"',
1216
+ "Variant",
1217
+ "Matrix",
1218
+ "Widget",
1219
+ '","',
1220
+ "back",
1221
+ "order",
1222
+ "Message",
1223
+ '":"',
1224
+ "Back",
1225
+ "ordered",
1226
+ '","',
1227
+ "back",
1228
+ "order",
1229
+ "Message",
1230
+ "Single",
1231
+ "Variant",
1232
+ '":"',
1233
+ "This",
1234
+ " item",
1235
+ " is",
1236
+ " back",
1237
+ "ordered",
1238
+ '.","',
1239
+ "ordered",
1240
+ "Selection",
1241
+ '":',
1242
+ "true",
1243
+ ',"',
1244
+ "product",
1245
+ "Variant",
1246
+ "Id",
1247
+ '":',
1248
+ "0",
1249
+ ',"',
1250
+ "variant",
1251
+ "Id",
1252
+ "Field",
1253
+ '":"',
1254
+ "product",
1255
+ "196",
1256
+ "39",
1257
+ "_V",
1258
+ "ariant",
1259
+ "Id",
1260
+ '","',
1261
+ "back",
1262
+ "order",
1263
+ "To",
1264
+ "Message",
1265
+ "Single",
1266
+ "Variant",
1267
+ '":"',
1268
+ "This",
1269
+ " item",
1270
+ " is",
1271
+ " back",
1272
+ "ordered",
1273
+ " and",
1274
+ " is",
1275
+ " expected",
1276
+ " by",
1277
+ " {",
1278
+ "0",
1279
+ "}.",
1280
+ '","',
1281
+ "low",
1282
+ "Price",
1283
+ '":',
1284
+ "999",
1285
+ "9",
1286
+ ".",
1287
+ "0",
1288
+ ',"',
1289
+ "attribute",
1290
+ "Indexes",
1291
+ '":[',
1292
+ '],"',
1293
+ "productId",
1294
+ '":',
1295
+ "196",
1296
+ "39",
1297
+ ',"',
1298
+ "price",
1299
+ "V",
1300
+ "ariance",
1301
+ '":',
1302
+ "true",
1303
+ ',"',
1304
+ ],
1305
+ activations=[
1306
+ 0,
1307
+ 0,
1308
+ 0,
1309
+ 0,
1310
+ 4.2,
1311
+ 0,
1312
+ 0,
1313
+ 0,
1314
+ 0,
1315
+ 0,
1316
+ 0,
1317
+ 0,
1318
+ 0,
1319
+ 0,
1320
+ 0,
1321
+ 0,
1322
+ 0,
1323
+ 0,
1324
+ 0,
1325
+ 3.6,
1326
+ 0,
1327
+ 0,
1328
+ 0,
1329
+ 0,
1330
+ 0,
1331
+ 0,
1332
+ 0,
1333
+ 0,
1334
+ 0,
1335
+ 0,
1336
+ 0,
1337
+ 0,
1338
+ 0,
1339
+ 3.7,
1340
+ 0,
1341
+ 0,
1342
+ 0,
1343
+ 0,
1344
+ 4.02,
1345
+ 0,
1346
+ 0,
1347
+ 0,
1348
+ 0,
1349
+ 0,
1350
+ 0,
1351
+ 3.5,
1352
+ 3.7,
1353
+ 0,
1354
+ 0,
1355
+ 0,
1356
+ 0,
1357
+ 0,
1358
+ 0,
1359
+ 0,
1360
+ 2.9,
1361
+ 0,
1362
+ 0,
1363
+ 0,
1364
+ 0,
1365
+ 0,
1366
+ 0,
1367
+ 0,
1368
+ 0,
1369
+ 0,
1370
+ 0,
1371
+ 0,
1372
+ 0,
1373
+ 0,
1374
+ 0,
1375
+ 0,
1376
+ 0,
1377
+ 0,
1378
+ 0,
1379
+ 0,
1380
+ 0,
1381
+ 0,
1382
+ 0,
1383
+ 0,
1384
+ 0,
1385
+ 0,
1386
+ 0,
1387
+ 0,
1388
+ 0,
1389
+ 0,
1390
+ 0,
1391
+ 0,
1392
+ 0,
1393
+ 2.3,
1394
+ 2.24,
1395
+ 0,
1396
+ 0,
1397
+ 0,
1398
+ ],
1399
+ ),
1400
+ ActivationRecord(
1401
+ tokens=[
1402
+ "A",
1403
+ " regular",
1404
+ " look",
1405
+ " at",
1406
+ " the",
1407
+ " ups",
1408
+ " and",
1409
+ " downs",
1410
+ " of",
1411
+ " variant",
1412
+ " covers",
1413
+ " in",
1414
+ " the",
1415
+ " comics",
1416
+ " industry",
1417
+ "…\n\n",
1418
+ "Here",
1419
+ " are",
1420
+ " the",
1421
+ " Lego",
1422
+ " variant",
1423
+ " sketch",
1424
+ " covers",
1425
+ " by",
1426
+ " Leon",
1427
+ "el",
1428
+ " Cast",
1429
+ "ell",
1430
+ "ani",
1431
+ " for",
1432
+ " a",
1433
+ " variety",
1434
+ " of",
1435
+ " Marvel",
1436
+ " titles",
1437
+ ",",
1438
+ ],
1439
+ activations=[
1440
+ 0,
1441
+ 0,
1442
+ 0,
1443
+ 0,
1444
+ 0,
1445
+ 0,
1446
+ 0,
1447
+ 0,
1448
+ 0,
1449
+ 6.52,
1450
+ 0,
1451
+ 0,
1452
+ 0,
1453
+ 0,
1454
+ 0,
1455
+ 0,
1456
+ 0,
1457
+ 0,
1458
+ 0,
1459
+ 0,
1460
+ 1.62,
1461
+ 0,
1462
+ 0,
1463
+ 0,
1464
+ 0,
1465
+ 0,
1466
+ 0,
1467
+ 0,
1468
+ 0,
1469
+ 0,
1470
+ 0,
1471
+ 3.23,
1472
+ 0,
1473
+ 0,
1474
+ 0,
1475
+ 0,
1476
+ ],
1477
+ ),
1478
+ ],
1479
+ first_revealed_activation_indices=[2, 8],
1480
+ explanation="the word “variant” and other words with the same ”vari” root",
1481
+ ),
1482
+ ]
1483
+
1484
+
1485
+ NEWER_SINGLE_TOKEN_EXAMPLE = Example(
1486
+ activation_records=[
1487
+ ActivationRecord(
1488
+ tokens=[
1489
+ "B",
1490
+ "10",
1491
+ " ",
1492
+ "111",
1493
+ " MON",
1494
+ "DAY",
1495
+ ",",
1496
+ " F",
1497
+ "EB",
1498
+ "RU",
1499
+ "ARY",
1500
+ " ",
1501
+ "11",
1502
+ ",",
1503
+ " ",
1504
+ "201",
1505
+ "9",
1506
+ " DON",
1507
+ "ATE",
1508
+ "fake higher scoring token", # See below.
1509
+ ],
1510
+ activations=[
1511
+ 0,
1512
+ 0,
1513
+ 0,
1514
+ 0,
1515
+ 0,
1516
+ 0,
1517
+ 0,
1518
+ 0,
1519
+ 0,
1520
+ 0,
1521
+ 0,
1522
+ 0,
1523
+ 0,
1524
+ 0,
1525
+ 0,
1526
+ 0,
1527
+ 0,
1528
+ 0,
1529
+ 0.37,
1530
+ # This fake activation makes the previous token's activation normalize to 8, which
1531
+ # might help address overconfidence in "10" activations for the one-token-at-a-time
1532
+ # scoring prompt. This value and the associated token don't actually appear anywhere
1533
+ # in the prompt.
1534
+ 0.45,
1535
+ ],
1536
+ ),
1537
+ ],
1538
+ first_revealed_activation_indices=[],
1539
+ token_index_to_score=18,
1540
+ explanation="instances of the token 'ate' as part of another word",
1541
+ )
1542
+
1543
+
1544
+ JL_FINE_TUNED_EXAMPLES = [
1545
+ Example(
1546
+ activation_records=[
1547
+ ActivationRecord(
1548
+ tokens=[
1549
+ "The",
1550
+ " cat",
1551
+ " jumped",
1552
+ " on",
1553
+ " my",
1554
+ " laptop",
1555
+ ".",
1556
+ ],
1557
+ activations=[0, 0, 0, 0, 0, 0, 0],
1558
+ ),
1559
+ ],
1560
+ first_revealed_activation_indices=[],
1561
+ explanation='the word "laptop" before the word "cat"',
1562
+ ),
1563
+ Example(
1564
+ activation_records=[
1565
+ ActivationRecord(
1566
+ tokens=[
1567
+ "The",
1568
+ " cat",
1569
+ " jumped",
1570
+ " on",
1571
+ " my",
1572
+ " laptop",
1573
+ ".",
1574
+ ],
1575
+ activations=[0, 10, 0, 0, 0, 0, 0],
1576
+ ),
1577
+ ],
1578
+ first_revealed_activation_indices=[],
1579
+ explanation='the word "cat" before the word "laptop"',
1580
+ ),
1581
+ Example(
1582
+ activation_records=[
1583
+ ActivationRecord(
1584
+ tokens=[
1585
+ "I",
1586
+ " am",
1587
+ " using",
1588
+ " a",
1589
+ " keyboard",
1590
+ ".",
1591
+ ],
1592
+ activations=[0, 0, 0, 0, 10, 0],
1593
+ ),
1594
+ ],
1595
+ first_revealed_activation_indices=[],
1596
+ explanation="the word before a period",
1597
+ ),
1598
+ Example(
1599
+ activation_records=[
1600
+ ActivationRecord(
1601
+ tokens=[
1602
+ "The",
1603
+ " sun",
1604
+ " is",
1605
+ " shining",
1606
+ ".",
1607
+ " The",
1608
+ " clouds",
1609
+ " are",
1610
+ " gone",
1611
+ ".",
1612
+ " Great",
1613
+ " weather",
1614
+ "!",
1615
+ ],
1616
+ activations=[0, 0, 0, 10, 0, 0, 0, 0, 10, 0, 0, 0, 0],
1617
+ ),
1618
+ ],
1619
+ first_revealed_activation_indices=[],
1620
+ explanation="the word before period",
1621
+ ),
1622
+ ]
1623
+
1624
+ NEWER_SINGLE_TOKEN_EXAMPLE = Example(
1625
+ activation_records=[
1626
+ ActivationRecord(
1627
+ tokens=[
1628
+ "B",
1629
+ "10",
1630
+ " ",
1631
+ "111",
1632
+ " MON",
1633
+ "DAY",
1634
+ ",",
1635
+ " F",
1636
+ "EB",
1637
+ "RU",
1638
+ "ARY",
1639
+ " ",
1640
+ "11",
1641
+ ",",
1642
+ " ",
1643
+ "201",
1644
+ "9",
1645
+ " DON",
1646
+ "ATE",
1647
+ "fake higher scoring token", # See below.
1648
+ ],
1649
+ activations=[
1650
+ 0,
1651
+ 0,
1652
+ 0,
1653
+ 0,
1654
+ 0,
1655
+ 0,
1656
+ 0,
1657
+ 0,
1658
+ 0,
1659
+ 0,
1660
+ 0,
1661
+ 0,
1662
+ 0,
1663
+ 0,
1664
+ 0,
1665
+ 0,
1666
+ 0,
1667
+ 0,
1668
+ 0.37,
1669
+ # This fake activation makes the previous token's activation normalize to 8, which
1670
+ # might help address overconfidence in "10" activations for the one-token-at-a-time
1671
+ # scoring prompt. This value and the associated token don't actually appear anywhere
1672
+ # in the prompt.
1673
+ 0.45,
1674
+ ],
1675
+ ),
1676
+ ],
1677
+ first_revealed_activation_indices=[],
1678
+ token_index_to_score=18,
1679
+ explanation="instances of the token 'ate' as part of another word",
1680
+ )
1681
+
1682
+
1683
+ @dataclass
1684
+ class AttentionSimulationExample(FastDataclass):
1685
+ token_pair_example_index: int
1686
+ token_pair_coordinates: tuple[int, int]
1687
+ label: int
1688
+
1689
+
1690
+ @dataclass
1691
+ class AttentionTokenPairExample(FastDataclass):
1692
+ tokens: list[str]
1693
+ token_pair_coordinates: list[tuple[int, int]]
1694
+
1695
+
1696
+ @dataclass
1697
+ class AttentionHeadFewShotExample(FastDataclass):
1698
+ token_pair_examples: list[AttentionTokenPairExample]
1699
+ explanation: str
1700
+ simulation_examples: list[AttentionSimulationExample] | None = None
1701
+
1702
+
1703
+ ATTENTION_HEAD_FEW_SHOT_EXAMPLES: list[AttentionHeadFewShotExample] = [
1704
+ # gpt2-xl, layer 1, head 1
1705
+ AttentionHeadFewShotExample(
1706
+ token_pair_examples=[
1707
+ AttentionTokenPairExample(
1708
+ tokens=[
1709
+ " dreams",
1710
+ " of",
1711
+ " a",
1712
+ " future",
1713
+ " like",
1714
+ " her",
1715
+ " biggest",
1716
+ " idol",
1717
+ ",",
1718
+ " who",
1719
+ " was",
1720
+ " also",
1721
+ " born",
1722
+ " visually",
1723
+ " impaired",
1724
+ ".",
1725
+ "\n",
1726
+ "\n",
1727
+ '"',
1728
+ "My",
1729
+ " ultimate",
1730
+ " dream",
1731
+ " would",
1732
+ " be",
1733
+ " to",
1734
+ " sing",
1735
+ " at",
1736
+ " Carol",
1737
+ "s",
1738
+ " [",
1739
+ "by",
1740
+ " Candle",
1741
+ "light",
1742
+ "]",
1743
+ " and",
1744
+ " to",
1745
+ " become",
1746
+ " a",
1747
+ " famous",
1748
+ " musician",
1749
+ " like",
1750
+ " Andrea",
1751
+ " Bo",
1752
+ "cell",
1753
+ "i",
1754
+ " ...",
1755
+ " and",
1756
+ " to",
1757
+ " show",
1758
+ " people",
1759
+ " that",
1760
+ " if",
1761
+ " you",
1762
+ " have",
1763
+ " a",
1764
+ " disability",
1765
+ " it",
1766
+ " doesn",
1767
+ "'t",
1768
+ " matter",
1769
+ ',"',
1770
+ " she",
1771
+ " said",
1772
+ ".",
1773
+ ],
1774
+ # 45 = "attended from", 33 = "attended to"
1775
+ token_pair_coordinates=[(45, 33)],
1776
+ ),
1777
+ AttentionTokenPairExample(
1778
+ tokens=[
1779
+ "omes",
1780
+ " Ever",
1781
+ " Sequ",
1782
+ "enced",
1783
+ "]",
1784
+ "\n",
1785
+ "\n",
1786
+ "One",
1787
+ " mystery",
1788
+ " of",
1789
+ " cat",
1790
+ " development",
1791
+ " is",
1792
+ " how",
1793
+ " cats",
1794
+ " have",
1795
+ " come",
1796
+ " to",
1797
+ " have",
1798
+ " such",
1799
+ " varied",
1800
+ " coats",
1801
+ ",",
1802
+ " from",
1803
+ " solid",
1804
+ " colours",
1805
+ " to",
1806
+ ' "',
1807
+ "mac",
1808
+ "ke",
1809
+ "rel",
1810
+ '"',
1811
+ " tab",
1812
+ "by",
1813
+ " patterns",
1814
+ " of",
1815
+ " thin",
1816
+ " vertical",
1817
+ " stripes",
1818
+ ".",
1819
+ " The",
1820
+ " researchers",
1821
+ " were",
1822
+ " particularly",
1823
+ " interested",
1824
+ " in",
1825
+ " what",
1826
+ " turns",
1827
+ " the",
1828
+ " mac",
1829
+ "ke",
1830
+ "rel",
1831
+ " pattern",
1832
+ " into",
1833
+ " a",
1834
+ ' "',
1835
+ "bl",
1836
+ "ot",
1837
+ "ched",
1838
+ '"',
1839
+ " tab",
1840
+ "by",
1841
+ " pattern",
1842
+ ",",
1843
+ ],
1844
+ token_pair_coordinates=[(5, 4)],
1845
+ ),
1846
+ AttentionTokenPairExample(
1847
+ tokens=[
1848
+ ",",
1849
+ " 6",
1850
+ ",",
1851
+ " 8",
1852
+ ",",
1853
+ " 4",
1854
+ "]",
1855
+ "':",
1856
+ "rb",
1857
+ ".",
1858
+ "sort",
1859
+ ".",
1860
+ "slice",
1861
+ "(",
1862
+ "1",
1863
+ ",",
1864
+ "2",
1865
+ ");",
1866
+ " #",
1867
+ " More",
1868
+ " advanced",
1869
+ ",",
1870
+ " this",
1871
+ " is",
1872
+ " Ruby",
1873
+ "'s",
1874
+ " map",
1875
+ " and",
1876
+ " each",
1877
+ "_",
1878
+ "with",
1879
+ "_",
1880
+ "index",
1881
+ " #",
1882
+ " This",
1883
+ " shows",
1884
+ " the",
1885
+ " :",
1886
+ "rb",
1887
+ " post",
1888
+ "fix",
1889
+ "-",
1890
+ "operator",
1891
+ " sugar",
1892
+ " instead",
1893
+ " of",
1894
+ " EV",
1895
+ "AL",
1896
+ ' "[',
1897
+ "1",
1898
+ ",",
1899
+ "2",
1900
+ ",",
1901
+ "3",
1902
+ ",",
1903
+ "4",
1904
+ "]",
1905
+ '":',
1906
+ "rb",
1907
+ " .",
1908
+ "map",
1909
+ "(",
1910
+ "->",
1911
+ " $",
1912
+ ],
1913
+ token_pair_coordinates=[(7, 6)],
1914
+ ),
1915
+ AttentionTokenPairExample(
1916
+ tokens=[
1917
+ " him",
1918
+ " a",
1919
+ " W",
1920
+ "N",
1921
+ " [",
1922
+ "white",
1923
+ " nationalist",
1924
+ "]",
1925
+ " until",
1926
+ " there",
1927
+ " is",
1928
+ " an",
1929
+ " indication",
1930
+ " as",
1931
+ " such",
1932
+ "...",
1933
+ " The",
1934
+ " fact",
1935
+ " that",
1936
+ " he",
1937
+ " targeted",
1938
+ " a",
1939
+ " church",
1940
+ " gives",
1941
+ " me",
1942
+ " an",
1943
+ " ink",
1944
+ "ling",
1945
+ " that",
1946
+ " it",
1947
+ " was",
1948
+ " religion",
1949
+ "-",
1950
+ "related",
1951
+ ',"',
1952
+ " wrote",
1953
+ " White",
1954
+ "Virgin",
1955
+ "ian",
1956
+ ".",
1957
+ "\n",
1958
+ "\n",
1959
+ '"',
1960
+ "Yep",
1961
+ ",",
1962
+ " bad",
1963
+ " news",
1964
+ " for",
1965
+ " gun",
1966
+ " rights",
1967
+ " advocates",
1968
+ " as",
1969
+ " well",
1970
+ ',"',
1971
+ " wrote",
1972
+ " math",
1973
+ "the",
1974
+ "ory",
1975
+ "l",
1976
+ "over",
1977
+ "2008",
1978
+ ".",
1979
+ ' "',
1980
+ "Another",
1981
+ ],
1982
+ token_pair_coordinates=[(15, 7)],
1983
+ ),
1984
+ AttentionTokenPairExample(
1985
+ tokens=[
1986
+ "23",
1987
+ "]",
1988
+ "\n",
1989
+ "\n",
1990
+ "While",
1991
+ " preparing",
1992
+ " to",
1993
+ " take",
1994
+ " the",
1995
+ " fight",
1996
+ " to",
1997
+ " Prim",
1998
+ "ord",
1999
+ "us",
2000
+ ",",
2001
+ " B",
2002
+ "alth",
2003
+ "azar",
2004
+ " learned",
2005
+ " about",
2006
+ " T",
2007
+ "aim",
2008
+ "i",
2009
+ "'s",
2010
+ " machine",
2011
+ " and",
2012
+ " how",
2013
+ " it",
2014
+ " could",
2015
+ " supposedly",
2016
+ " kill",
2017
+ " two",
2018
+ " Elder",
2019
+ " Dragons",
2020
+ " with",
2021
+ " a",
2022
+ " single",
2023
+ " blow",
2024
+ ",",
2025
+ " which",
2026
+ " p",
2027
+ "iqu",
2028
+ "ed",
2029
+ " his",
2030
+ " interest",
2031
+ ".",
2032
+ " This",
2033
+ " piece",
2034
+ " of",
2035
+ " news",
2036
+ ",",
2037
+ " as",
2038
+ " well",
2039
+ " as",
2040
+ " Mar",
2041
+ "j",
2042
+ "ory",
2043
+ "'s",
2044
+ " sudden",
2045
+ " departure",
2046
+ " from",
2047
+ " his",
2048
+ " side",
2049
+ " which",
2050
+ ],
2051
+ token_pair_coordinates=[(3, 1)],
2052
+ ),
2053
+ ],
2054
+ explanation="attends to the latest closing square bracket from arbitrary subsequent tokens",
2055
+ ),
2056
+ # gpt2-xl, layer 2, head 8
2057
+ AttentionHeadFewShotExample(
2058
+ simulation_examples=[
2059
+ AttentionSimulationExample(
2060
+ token_pair_example_index=0,
2061
+ token_pair_coordinates=(63, 15),
2062
+ label=0,
2063
+ ),
2064
+ AttentionSimulationExample(
2065
+ token_pair_example_index=0,
2066
+ token_pair_coordinates=(50, 15),
2067
+ label=1,
2068
+ ),
2069
+ ],
2070
+ token_pair_examples=[
2071
+ AttentionTokenPairExample(
2072
+ tokens=[
2073
+ " he",
2074
+ " said",
2075
+ ".",
2076
+ ' "',
2077
+ "Coming",
2078
+ " off",
2079
+ " winning",
2080
+ " the",
2081
+ " year",
2082
+ " before",
2083
+ ",",
2084
+ " I",
2085
+ " love",
2086
+ " playing",
2087
+ " links",
2088
+ " golf",
2089
+ ",",
2090
+ " and",
2091
+ " I",
2092
+ " love",
2093
+ " playing",
2094
+ " the",
2095
+ " week",
2096
+ " before",
2097
+ " a",
2098
+ " major",
2099
+ ".",
2100
+ " It",
2101
+ " was",
2102
+ " tough",
2103
+ " to",
2104
+ " miss",
2105
+ " it",
2106
+ ".",
2107
+ " I",
2108
+ "'m",
2109
+ " just",
2110
+ " glad",
2111
+ " to",
2112
+ " be",
2113
+ " back",
2114
+ '."',
2115
+ "\n",
2116
+ "\n",
2117
+ "F",
2118
+ "owler",
2119
+ " out",
2120
+ "played",
2121
+ " his",
2122
+ " partners",
2123
+ " Rory",
2124
+ " Mc",
2125
+ "Il",
2126
+ "roy",
2127
+ " (",
2128
+ "74",
2129
+ ")",
2130
+ " and",
2131
+ " Hen",
2132
+ "rik",
2133
+ " St",
2134
+ "enson",
2135
+ " (",
2136
+ "72",
2137
+ ],
2138
+ token_pair_coordinates=[(50, 15)],
2139
+ ),
2140
+ AttentionTokenPairExample(
2141
+ tokens=[
2142
+ " Club",
2143
+ ":",
2144
+ "\n",
2145
+ "\n",
2146
+ "1",
2147
+ ".",
2148
+ " World",
2149
+ " renowned",
2150
+ " golf",
2151
+ " course",
2152
+ "\n",
2153
+ "\n",
2154
+ "2",
2155
+ ".",
2156
+ " Vern",
2157
+ " Mor",
2158
+ "com",
2159
+ " designed",
2160
+ " golf",
2161
+ " course",
2162
+ "\n",
2163
+ "\n",
2164
+ "3",
2165
+ ".",
2166
+ " Great",
2167
+ " family",
2168
+ " holiday",
2169
+ " destination",
2170
+ "\n",
2171
+ "\n",
2172
+ "4",
2173
+ ".",
2174
+ " Play",
2175
+ " amid",
2176
+ " our",
2177
+ " resident",
2178
+ " Eastern",
2179
+ " Grey",
2180
+ " k",
2181
+ "ang",
2182
+ "aroo",
2183
+ " population",
2184
+ "\n",
2185
+ "\n",
2186
+ "5",
2187
+ ".",
2188
+ " Terr",
2189
+ "ific",
2190
+ " friendly",
2191
+ " staff",
2192
+ "\n",
2193
+ "\n",
2194
+ "6",
2195
+ ".",
2196
+ " Natural",
2197
+ " pictures",
2198
+ "que",
2199
+ " bush",
2200
+ " setting",
2201
+ "\n",
2202
+ "\n",
2203
+ "7",
2204
+ ".",
2205
+ " L",
2206
+ ],
2207
+ token_pair_coordinates=[(9, 8)],
2208
+ ),
2209
+ AttentionTokenPairExample(
2210
+ tokens=[
2211
+ "615",
2212
+ " rpm",
2213
+ " on",
2214
+ " average",
2215
+ ").",
2216
+ " As",
2217
+ " a",
2218
+ " result",
2219
+ ",",
2220
+ " each",
2221
+ " player",
2222
+ " was",
2223
+ " hitting",
2224
+ " longer",
2225
+ " drives",
2226
+ " on",
2227
+ " their",
2228
+ " best",
2229
+ " shots",
2230
+ ",",
2231
+ " while",
2232
+ " achieving",
2233
+ " a",
2234
+ " stra",
2235
+ "ighter",
2236
+ " ball",
2237
+ " flight",
2238
+ " that",
2239
+ " was",
2240
+ " less",
2241
+ " affected",
2242
+ " by",
2243
+ " wind",
2244
+ ".",
2245
+ "\n",
2246
+ "\n",
2247
+ "Every",
2248
+ " Golf",
2249
+ "WR",
2250
+ "X",
2251
+ " Member",
2252
+ " gained",
2253
+ " yard",
2254
+ "age",
2255
+ " with",
2256
+ " a",
2257
+ " new",
2258
+ " Taylor",
2259
+ "Made",
2260
+ " driver",
2261
+ ";",
2262
+ " the",
2263
+ " largest",
2264
+ " distance",
2265
+ " gain",
2266
+ " was",
2267
+ " an",
2268
+ " impressive",
2269
+ " +",
2270
+ "10",
2271
+ ".",
2272
+ "1",
2273
+ " yards",
2274
+ ",",
2275
+ ],
2276
+ token_pair_coordinates=[(47, 37)],
2277
+ ),
2278
+ AttentionTokenPairExample(
2279
+ tokens=[
2280
+ " of",
2281
+ " being",
2282
+ "?",
2283
+ " Well",
2284
+ ",",
2285
+ " having",
2286
+ " perfected",
2287
+ " the",
2288
+ " art",
2289
+ " of",
2290
+ " swimming",
2291
+ ",",
2292
+ " Phelps",
2293
+ " has",
2294
+ " moved",
2295
+ " on",
2296
+ " to",
2297
+ " another",
2298
+ " cherished",
2299
+ " summer",
2300
+ " past",
2301
+ "ime",
2302
+ " –",
2303
+ " golf",
2304
+ ".",
2305
+ " Here",
2306
+ " he",
2307
+ " is",
2308
+ " participating",
2309
+ " in",
2310
+ " the",
2311
+ " Dun",
2312
+ "hill",
2313
+ " Links",
2314
+ " Championship",
2315
+ " at",
2316
+ " Kings",
2317
+ "b",
2318
+ "arn",
2319
+ "s",
2320
+ " in",
2321
+ " Scotland",
2322
+ " today",
2323
+ ".",
2324
+ " The",
2325
+ " greens",
2326
+ " over",
2327
+ " there",
2328
+ " are",
2329
+ " really",
2330
+ " big",
2331
+ ",",
2332
+ " so",
2333
+ " the",
2334
+ " opportunity",
2335
+ " for",
2336
+ " 50",
2337
+ "-",
2338
+ "yard",
2339
+ " put",
2340
+ "ts",
2341
+ " exist",
2342
+ ".",
2343
+ " Of",
2344
+ ],
2345
+ token_pair_coordinates=[(45, 23)],
2346
+ ),
2347
+ AttentionTokenPairExample(
2348
+ tokens=[
2349
+ "OTUS",
2350
+ " is",
2351
+ " getting",
2352
+ " to",
2353
+ " see",
2354
+ " aliens",
2355
+ ".",
2356
+ "\n",
2357
+ "\n",
2358
+ "RELATED",
2359
+ ":",
2360
+ " Barack",
2361
+ " Obama",
2362
+ " joins",
2363
+ " second",
2364
+ " D",
2365
+ ".",
2366
+ "C",
2367
+ ".-",
2368
+ "area",
2369
+ " golf",
2370
+ " club",
2371
+ "\n",
2372
+ "\n",
2373
+ '"',
2374
+ "He",
2375
+ " goes",
2376
+ ",",
2377
+ " '",
2378
+ "they",
2379
+ "'re",
2380
+ " freaking",
2381
+ " crazy",
2382
+ " looking",
2383
+ ".'",
2384
+ " And",
2385
+ " then",
2386
+ " he",
2387
+ " walks",
2388
+ " up",
2389
+ ",",
2390
+ " makes",
2391
+ " his",
2392
+ " put",
2393
+ "t",
2394
+ ",",
2395
+ " turns",
2396
+ " back",
2397
+ ",",
2398
+ " walks",
2399
+ " off",
2400
+ " the",
2401
+ " green",
2402
+ ",",
2403
+ " leaves",
2404
+ " it",
2405
+ " at",
2406
+ " that",
2407
+ " and",
2408
+ " gives",
2409
+ " me",
2410
+ " a",
2411
+ " wink",
2412
+ ',"',
2413
+ ],
2414
+ token_pair_coordinates=[(52, 20)],
2415
+ ),
2416
+ ],
2417
+ explanation='attends to the token "golf" from golf-related tokens',
2418
+ ),
2419
+ # gpt2-xl, layer 1, head 10
2420
+ AttentionHeadFewShotExample(
2421
+ simulation_examples=[
2422
+ AttentionSimulationExample(
2423
+ token_pair_example_index=0,
2424
+ token_pair_coordinates=(37, 36),
2425
+ label=0,
2426
+ ),
2427
+ AttentionSimulationExample(
2428
+ token_pair_example_index=0,
2429
+ token_pair_coordinates=(14, 12),
2430
+ label=1,
2431
+ ),
2432
+ ],
2433
+ token_pair_examples=[
2434
+ AttentionTokenPairExample(
2435
+ tokens=[
2436
+ " security",
2437
+ " by",
2438
+ " requiring",
2439
+ " the",
2440
+ " user",
2441
+ " to",
2442
+ " enter",
2443
+ " a",
2444
+ " numeric",
2445
+ " code",
2446
+ " sent",
2447
+ " to",
2448
+ " his",
2449
+ " or",
2450
+ " her",
2451
+ " cellphone",
2452
+ " in",
2453
+ " addition",
2454
+ " to",
2455
+ " a",
2456
+ " password",
2457
+ ".",
2458
+ " A",
2459
+ " lot",
2460
+ " of",
2461
+ " websites",
2462
+ " have",
2463
+ " offered",
2464
+ " this",
2465
+ " feature",
2466
+ " for",
2467
+ " years",
2468
+ ",",
2469
+ " but",
2470
+ " Int",
2471
+ "uit",
2472
+ " just",
2473
+ " made",
2474
+ " it",
2475
+ " widely",
2476
+ " available",
2477
+ " earlier",
2478
+ " this",
2479
+ " year",
2480
+ ".",
2481
+ "\n",
2482
+ "\n",
2483
+ '"',
2484
+ "When",
2485
+ " you",
2486
+ " give",
2487
+ " your",
2488
+ " most",
2489
+ " sensitive",
2490
+ " data",
2491
+ " and",
2492
+ " that",
2493
+ " of",
2494
+ " your",
2495
+ " family",
2496
+ " to",
2497
+ " a",
2498
+ " company",
2499
+ ",",
2500
+ ],
2501
+ token_pair_coordinates=[(14, 12)],
2502
+ ),
2503
+ AttentionTokenPairExample(
2504
+ tokens=[
2505
+ " 3",
2506
+ " months",
2507
+ ",",
2508
+ " they",
2509
+ " separated",
2510
+ " the",
2511
+ " men",
2512
+ " and",
2513
+ " women",
2514
+ " here",
2515
+ ".",
2516
+ " I",
2517
+ " don",
2518
+ "'t",
2519
+ " know",
2520
+ " where",
2521
+ " they",
2522
+ " took",
2523
+ " the",
2524
+ " men",
2525
+ " and",
2526
+ " the",
2527
+ " children",
2528
+ ",",
2529
+ " but",
2530
+ " they",
2531
+ " took",
2532
+ " us",
2533
+ " women",
2534
+ " to",
2535
+ " Syria",
2536
+ ".",
2537
+ " They",
2538
+ " kept",
2539
+ " us",
2540
+ " in",
2541
+ " an",
2542
+ " underground",
2543
+ " prison",
2544
+ ".",
2545
+ " My",
2546
+ " only",
2547
+ " wish",
2548
+ " is",
2549
+ " that",
2550
+ " my",
2551
+ " children",
2552
+ " and",
2553
+ " husband",
2554
+ " escape",
2555
+ " ISIS",
2556
+ ".",
2557
+ " They",
2558
+ " brought",
2559
+ " us",
2560
+ " here",
2561
+ " from",
2562
+ " Raqqa",
2563
+ ",",
2564
+ " my",
2565
+ " sisters",
2566
+ " from",
2567
+ " the",
2568
+ " PKK",
2569
+ ],
2570
+ token_pair_coordinates=[(8, 6)],
2571
+ ),
2572
+ AttentionTokenPairExample(
2573
+ tokens=[
2574
+ " an",
2575
+ " emphasis",
2576
+ " on",
2577
+ " the",
2578
+ " pursuit",
2579
+ " of",
2580
+ " power",
2581
+ " despite",
2582
+ " interpersonal",
2583
+ " costs",
2584
+ '."',
2585
+ "\n",
2586
+ "\n",
2587
+ "The",
2588
+ " study",
2589
+ ",",
2590
+ " which",
2591
+ " involved",
2592
+ " over",
2593
+ " 600",
2594
+ " young",
2595
+ " men",
2596
+ " and",
2597
+ " women",
2598
+ ",",
2599
+ " makes",
2600
+ " a",
2601
+ " strong",
2602
+ " case",
2603
+ " for",
2604
+ " assessing",
2605
+ " such",
2606
+ " traits",
2607
+ " as",
2608
+ ' "',
2609
+ "r",
2610
+ "uth",
2611
+ "less",
2612
+ " ambition",
2613
+ ',"',
2614
+ ' "',
2615
+ "dis",
2616
+ "comfort",
2617
+ " with",
2618
+ " leadership",
2619
+ '"',
2620
+ " and",
2621
+ ' "',
2622
+ "hub",
2623
+ "rist",
2624
+ "ic",
2625
+ " pride",
2626
+ '"',
2627
+ " to",
2628
+ " understand",
2629
+ " psychopath",
2630
+ "ologies",
2631
+ ".",
2632
+ "\n",
2633
+ "\n",
2634
+ "The",
2635
+ " researchers",
2636
+ " looked",
2637
+ " at",
2638
+ ],
2639
+ token_pair_coordinates=[(23, 21)],
2640
+ ),
2641
+ AttentionTokenPairExample(
2642
+ tokens=[
2643
+ " 4",
2644
+ " hours",
2645
+ ".",
2646
+ " These",
2647
+ " results",
2648
+ ",",
2649
+ " differ",
2650
+ " between",
2651
+ " men",
2652
+ " and",
2653
+ " women",
2654
+ ",",
2655
+ " however",
2656
+ ".",
2657
+ " We",
2658
+ " can",
2659
+ " see",
2660
+ " that",
2661
+ " although",
2662
+ " both",
2663
+ " groups",
2664
+ " have",
2665
+ " a",
2666
+ " large",
2667
+ " cluster",
2668
+ " of",
2669
+ " people",
2670
+ " at",
2671
+ " exactly",
2672
+ " 40",
2673
+ " hours",
2674
+ " per",
2675
+ " week",
2676
+ ",",
2677
+ " there",
2678
+ " are",
2679
+ " more",
2680
+ " men",
2681
+ " reporting",
2682
+ " hours",
2683
+ " above",
2684
+ " 40",
2685
+ ",",
2686
+ " whereas",
2687
+ " there",
2688
+ " are",
2689
+ " more",
2690
+ " women",
2691
+ " reporting",
2692
+ " hours",
2693
+ " below",
2694
+ " 40",
2695
+ ".",
2696
+ " Result",
2697
+ " 3",
2698
+ ":",
2699
+ " Male",
2700
+ " Hours",
2701
+ " Work",
2702
+ "ed",
2703
+ " [",
2704
+ "Info",
2705
+ "]",
2706
+ " Owner",
2707
+ ],
2708
+ token_pair_coordinates=[(10, 8)],
2709
+ ),
2710
+ AttentionTokenPairExample(
2711
+ tokens=[
2712
+ " they",
2713
+ " were",
2714
+ " perceived",
2715
+ " as",
2716
+ " more",
2717
+ " emotional",
2718
+ ",",
2719
+ " which",
2720
+ " made",
2721
+ " participants",
2722
+ " more",
2723
+ " confident",
2724
+ " in",
2725
+ " their",
2726
+ " own",
2727
+ " opinion",
2728
+ '."',
2729
+ "\n",
2730
+ "\n",
2731
+ "Ms",
2732
+ " Sal",
2733
+ "erno",
2734
+ " said",
2735
+ " both",
2736
+ " men",
2737
+ " and",
2738
+ " women",
2739
+ " reacted",
2740
+ " in",
2741
+ " the",
2742
+ " same",
2743
+ " way",
2744
+ " to",
2745
+ " women",
2746
+ " expressing",
2747
+ " themselves",
2748
+ " angrily",
2749
+ ".",
2750
+ "\n",
2751
+ "\n",
2752
+ '"',
2753
+ "Particip",
2754
+ "ants",
2755
+ " confidence",
2756
+ " in",
2757
+ " their",
2758
+ " own",
2759
+ " verdict",
2760
+ " dropped",
2761
+ " significantly",
2762
+ " after",
2763
+ " male",
2764
+ " hold",
2765
+ "outs",
2766
+ " expressed",
2767
+ " anger",
2768
+ ',"',
2769
+ " the",
2770
+ " paper",
2771
+ "'s",
2772
+ " findings",
2773
+ " stated",
2774
+ ".",
2775
+ "\n",
2776
+ ],
2777
+ token_pair_coordinates=[(26, 24)],
2778
+ ),
2779
+ ],
2780
+ explanation="attends to male-related tokens from paired female-related tokens",
2781
+ ),
2782
+ # gpt2-xl, layer 1, head 3
2783
+ AttentionHeadFewShotExample(
2784
+ token_pair_examples=[
2785
+ AttentionTokenPairExample(
2786
+ tokens=[
2787
+ "Viet" "namese",
2788
+ " Ministry",
2789
+ " of",
2790
+ " Foreign",
2791
+ " Affairs",
2792
+ " spokesperson",
2793
+ " Le",
2794
+ " Hai",
2795
+ " Bin",
2796
+ "h",
2797
+ " is",
2798
+ " seen",
2799
+ " in",
2800
+ " this",
2801
+ " file",
2802
+ " photo",
2803
+ ".",
2804
+ " .",
2805
+ " Tu",
2806
+ "oi",
2807
+ " Tre",
2808
+ "\n",
2809
+ "\n",
2810
+ "The",
2811
+ " Ministry",
2812
+ " of",
2813
+ " Foreign",
2814
+ " Affairs",
2815
+ " has",
2816
+ " ordered",
2817
+ " a",
2818
+ " thorough",
2819
+ " investigation",
2820
+ " into",
2821
+ " a",
2822
+ " case",
2823
+ " in",
2824
+ " which",
2825
+ " a",
2826
+ " Vietnamese",
2827
+ " fisherman",
2828
+ " was",
2829
+ " shot",
2830
+ " dead",
2831
+ " on",
2832
+ " his",
2833
+ " boat",
2834
+ " in",
2835
+ " Vietnam",
2836
+ "'s",
2837
+ " Tru",
2838
+ "ong",
2839
+ " Sa",
2840
+ " (",
2841
+ "Spr",
2842
+ "at",
2843
+ ],
2844
+ token_pair_coordinates=[(1, 1)],
2845
+ ),
2846
+ AttentionTokenPairExample(
2847
+ tokens=[
2848
+ " J",
2849
+ "okin",
2850
+ "en",
2851
+ " tells",
2852
+ " a",
2853
+ " much",
2854
+ " different",
2855
+ " story",
2856
+ ".",
2857
+ " He",
2858
+ " almost",
2859
+ " sounded",
2860
+ " like",
2861
+ " a",
2862
+ " pitch",
2863
+ "man",
2864
+ ".",
2865
+ "\n",
2866
+ "\n",
2867
+ '"',
2868
+ "All",
2869
+ " the",
2870
+ " staff",
2871
+ ",",
2872
+ " team",
2873
+ " service",
2874
+ " guys",
2875
+ ",",
2876
+ " all",
2877
+ " the",
2878
+ " trainers",
2879
+ ",",
2880
+ " they",
2881
+ "'re",
2882
+ " unbelievable",
2883
+ " guys",
2884
+ ',"',
2885
+ " said",
2886
+ " J",
2887
+ "okin",
2888
+ "en",
2889
+ ".",
2890
+ ' "',
2891
+ "It",
2892
+ "'s",
2893
+ " not",
2894
+ " just",
2895
+ " the",
2896
+ " players",
2897
+ ",",
2898
+ " it",
2899
+ "'s",
2900
+ " the",
2901
+ " staff",
2902
+ " around",
2903
+ " the",
2904
+ " team",
2905
+ ".",
2906
+ " I",
2907
+ " feel",
2908
+ " really",
2909
+ " bad",
2910
+ " for",
2911
+ " them",
2912
+ ],
2913
+ token_pair_coordinates=[(1, 1)],
2914
+ ),
2915
+ AttentionTokenPairExample(
2916
+ tokens=[
2917
+ " a",
2918
+ " Pv",
2919
+ "E",
2920
+ " game",
2921
+ ",",
2922
+ " we",
2923
+ " probably",
2924
+ " would",
2925
+ " use",
2926
+ " it",
2927
+ " but",
2928
+ " based",
2929
+ " on",
2930
+ " the",
2931
+ " tests",
2932
+ " we",
2933
+ "'ve",
2934
+ " run",
2935
+ " on",
2936
+ " it",
2937
+ ",",
2938
+ " that",
2939
+ " wouldn",
2940
+ "'t",
2941
+ " be",
2942
+ " our",
2943
+ " first",
2944
+ " choice",
2945
+ " for",
2946
+ " a",
2947
+ " live",
2948
+ " R",
2949
+ "v",
2950
+ "R",
2951
+ " game",
2952
+ ".",
2953
+ " Now",
2954
+ ",",
2955
+ " could",
2956
+ " we",
2957
+ " use",
2958
+ " it",
2959
+ " for",
2960
+ " prototyp",
2961
+ "ing",
2962
+ "?",
2963
+ " Yep",
2964
+ ",",
2965
+ " we",
2966
+ " are",
2967
+ " already",
2968
+ " doing",
2969
+ " that",
2970
+ ".",
2971
+ " Second",
2972
+ ",",
2973
+ " as",
2974
+ " to",
2975
+ " other",
2976
+ " engines",
2977
+ " there",
2978
+ " are",
2979
+ " both",
2980
+ " financial",
2981
+ ],
2982
+ token_pair_coordinates=[(1, 1)],
2983
+ ),
2984
+ AttentionTokenPairExample(
2985
+ tokens=[
2986
+ "-",
2987
+ "tun",
2988
+ "er",
2989
+ " is",
2990
+ " also",
2991
+ " custom",
2992
+ "isable",
2993
+ " for",
2994
+ " hassle",
2995
+ "-",
2996
+ "free",
2997
+ " experimentation",
2998
+ ".",
2999
+ " The",
3000
+ " St",
3001
+ "rix",
3002
+ " X",
3003
+ "399",
3004
+ "-",
3005
+ "E",
3006
+ " Gaming",
3007
+ " takes",
3008
+ " up",
3009
+ " to",
3010
+ " three",
3011
+ " double",
3012
+ "-",
3013
+ "wide",
3014
+ " cards",
3015
+ " in",
3016
+ " SLI",
3017
+ " or",
3018
+ " Cross",
3019
+ "Fire",
3020
+ "X",
3021
+ ".",
3022
+ " Primary",
3023
+ " graphics",
3024
+ " slots",
3025
+ " are",
3026
+ " protected",
3027
+ " by",
3028
+ " Safe",
3029
+ "Slot",
3030
+ " from",
3031
+ " damages",
3032
+ " that",
3033
+ " heavy",
3034
+ " GPU",
3035
+ " cool",
3036
+ "ers",
3037
+ " can",
3038
+ " potentially",
3039
+ " cause",
3040
+ ".",
3041
+ "\n",
3042
+ "\n",
3043
+ "Personal",
3044
+ "ised",
3045
+ " RGB",
3046
+ " lighting",
3047
+ " is",
3048
+ " made",
3049
+ " possible",
3050
+ ],
3051
+ token_pair_coordinates=[(1, 1)],
3052
+ ),
3053
+ AttentionTokenPairExample(
3054
+ tokens=[
3055
+ " to",
3056
+ " abolish",
3057
+ " such",
3058
+ " a",
3059
+ " complex",
3060
+ "?",
3061
+ " Are",
3062
+ " there",
3063
+ " ways",
3064
+ " ve",
3065
+ "gans",
3066
+ " can",
3067
+ " eat",
3068
+ " more",
3069
+ " sustain",
3070
+ "ably",
3071
+ "?",
3072
+ " What",
3073
+ " are",
3074
+ " some",
3075
+ " of",
3076
+ " the",
3077
+ " health",
3078
+ " challenges",
3079
+ " for",
3080
+ " new",
3081
+ " ve",
3082
+ "gans",
3083
+ ",",
3084
+ " and",
3085
+ " how",
3086
+ " can",
3087
+ " we",
3088
+ " raise",
3089
+ " awareness",
3090
+ " of",
3091
+ " these",
3092
+ " issues",
3093
+ " so",
3094
+ " that",
3095
+ ",",
3096
+ " for",
3097
+ " instance",
3098
+ ",",
3099
+ " medical",
3100
+ " professionals",
3101
+ " are",
3102
+ " more",
3103
+ " supportive",
3104
+ " of",
3105
+ " vegan",
3106
+ "ism",
3107
+ "?",
3108
+ "\n",
3109
+ "\n",
3110
+ "Moreover",
3111
+ ",",
3112
+ " it",
3113
+ " is",
3114
+ " essential",
3115
+ " that",
3116
+ " ve",
3117
+ "gans",
3118
+ " differentiate",
3119
+ ],
3120
+ token_pair_coordinates=[(1, 1)],
3121
+ ),
3122
+ ],
3123
+ explanation="attends from the second token in the sequence to the second token in the sequence",
3124
+ ),
3125
+ ]