openpond 0.0.34 → 0.0.36

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. package/dist/chunks/{app-layer-UQQNF5VZ.js → app-layer-4AA4YYGK.js} +3 -3
  2. package/dist/chunks/{apps-7NSC56K5.js → apps-V2JTB5MD.js} +3 -3
  3. package/dist/chunks/{chunk-EBCLEJRO.js → chunk-3DOOLZZM.js} +1 -1
  4. package/dist/chunks/{chunk-ZRNFOHFH.js → chunk-3JJBNVNX.js} +3 -3
  5. package/dist/chunks/{chunk-UASXM3RR.js → chunk-65BVQIHR.js} +4 -4
  6. package/dist/chunks/{chunk-E3RLJ5XT.js → chunk-6L5WJF7T.js} +2 -2
  7. package/dist/chunks/{chunk-G7XOSECC.js → chunk-7IPF4X3K.js} +1033 -529
  8. package/dist/chunks/{chunk-HGZ3O4TS.js → chunk-CJSD5ADV.js} +2 -2
  9. package/dist/chunks/{chunk-2GCNPJMD.js → chunk-G6WW2656.js} +1 -1
  10. package/dist/chunks/{chunk-5MM3P5KY.js → chunk-GALAW74P.js} +2 -2
  11. package/dist/chunks/{chunk-TF234YN4.js → chunk-JAS6QZ2W.js} +1 -1
  12. package/dist/chunks/{chunk-ZDEBKTWR.js → chunk-MJT7EW5Q.js} +1 -1
  13. package/dist/chunks/{chunk-MMDKFTJB.js → chunk-MSX2J653.js} +1 -1
  14. package/dist/chunks/{chunk-U4TUOBXO.js → chunk-NVTC3XOA.js} +1 -1
  15. package/dist/chunks/{chunk-B5CFIWWU.js → chunk-NWQTF6O4.js} +1 -1
  16. package/dist/chunks/{chunk-F6FFL3AA.js → chunk-NWVIUBJ3.js} +2 -2
  17. package/dist/chunks/{chunk-JVTACGKD.js → chunk-TYYR7KWY.js} +3 -3
  18. package/dist/chunks/{chunk-V7SHAMUN.js → chunk-YLULZMWE.js} +33 -33
  19. package/dist/chunks/{chunk-BZPAW4JY.js → chunk-YRFQKUZD.js} +958 -528
  20. package/dist/chunks/{cli-VEDK2B42.js → cli-4JA3SIDE.js} +6 -6
  21. package/dist/chunks/{cli-4ETE5RS2.js → cli-RJHEE4FA.js} +4 -4
  22. package/dist/chunks/{core-commands-J3IQGQOL.js → core-commands-CQSPCVC4.js} +4 -4
  23. package/dist/chunks/{extend-3F3LLDOF.js → extend-4MZ75SHV.js} +9 -9
  24. package/dist/chunks/{harness-JER7P6MJ.js → harness-YVZU42S7.js} +3 -3
  25. package/dist/chunks/{help-75NFCYJ4.js → help-6SULYXVS.js} +1 -1
  26. package/dist/chunks/{opchat-UT5V7MR2.js → opchat-XGS4AXHA.js} +5 -5
  27. package/dist/chunks/{organizations-ERKRLVNQ.js → organizations-4GFQ3NT2.js} +4 -4
  28. package/dist/chunks/{profile-A3CUSDIF.js → profile-VWMBIHIE.js} +7 -7
  29. package/dist/chunks/{project-agent-FCV5HPYT.js → project-agent-SQ6ODALG.js} +6 -6
  30. package/dist/chunks/{sandbox-command-L2QPIWZ5.js → sandbox-command-34S2V7BN.js} +4 -4
  31. package/dist/chunks/{sandbox-template-XOIWROEK.js → sandbox-template-VP5HMGQI.js} +5 -5
  32. package/dist/chunks/{src-V4FUQNHR.js → src-X3R3XIJT.js} +27795 -24249
  33. package/dist/chunks/{src-ENVQ7ARM.js → src-YHUMH6W6.js} +2 -2
  34. package/dist/chunks/{teams-bot-OR3ZJHQ5.js → teams-bot-EIVRNO5S.js} +3 -3
  35. package/dist/chunks/{workspaces-L73SRLPA.js → workspaces-2N52V3BI.js} +2 -2
  36. package/dist/cli.js +10 -10
  37. package/dist/web/assets/AppDialog-j4N-gIi3.js +1 -0
  38. package/dist/web/assets/{AppsView-CLjjVkjr.js → AppsView-YiK3pqNb.js} +1 -1
  39. package/dist/web/assets/{BrowserSidebar-CkNO8723.js → BrowserSidebar-BWFLEQ8j.js} +1 -1
  40. package/dist/web/assets/CloudWorkView-3sS_1bhN.js +1 -0
  41. package/dist/web/assets/CommandMenu-B11OrTCq.js +1 -0
  42. package/dist/web/assets/CommunityView-DXyXTWhP.js +1 -0
  43. package/dist/web/assets/ComposerCreateImproveStrip-3NoRiJJq.js +1 -0
  44. package/dist/web/assets/GetStartedView-CPV5lgVH.js +1 -0
  45. package/dist/web/assets/GetStartedView-DBLdduwr.css +1 -0
  46. package/dist/web/assets/LabsRoute-B_nsEZXy.js +3 -0
  47. package/dist/web/assets/LabsRoute-dNN-odU0.css +1 -0
  48. package/dist/web/assets/MainChatThread-Dg7p3R2O.js +2 -0
  49. package/dist/web/assets/MakeAgentTutorialLearningPanel-C6F-kJQd.js +2 -0
  50. package/dist/web/assets/MarkdownText-CkVkr3P1.js +7 -0
  51. package/dist/web/assets/Messages-C7uHGoFw.js +10 -0
  52. package/dist/web/assets/NativeSkillSidebar-BPpoHjEW.js +1 -0
  53. package/dist/web/assets/NativeSkillSidebar-T8KXjnUH.css +1 -0
  54. package/dist/web/assets/{NewProjectDialog-PrbA4fL6.js → NewProjectDialog-DBMhkTvD.js} +1 -1
  55. package/dist/web/assets/PostTrainingLearningPanel-B1Rp7iIL.js +2 -0
  56. package/dist/web/assets/PostTrainingStatusPill-BGZ9h5eW.js +1 -0
  57. package/dist/web/assets/ProfileSettingsSection-2NOHEz2b.js +1 -0
  58. package/dist/web/assets/{ProfileSettingsSection-CqnfjiCl.css → ProfileSettingsSection-Cg4tKO-C.css} +1 -1
  59. package/dist/web/assets/RightChatPanelStack-CLK7_zKe.js +1 -0
  60. package/dist/web/assets/SettingsView-BAqea_KX.js +6 -0
  61. package/dist/web/assets/SettingsView-uHhQnazW.css +1 -0
  62. package/dist/web/assets/TeamChatView-qVscSedQ.js +3 -0
  63. package/dist/web/assets/{TerminalOverlay-DaGfEprz.js → TerminalOverlay-Bkw7XqMy.js} +11 -11
  64. package/dist/web/assets/TrainingCreationPanel-COg1wvw8.js +1 -0
  65. package/dist/web/assets/{TrainingDraftPanel-BX5CCqlO.js → TrainingDraftPanel-BNK2znmI.js} +1 -1
  66. package/dist/web/assets/UsageSettingsSection-CUxBaohC.js +59 -0
  67. package/dist/web/assets/UsageSettingsSection-O8VK0qBL.css +1 -0
  68. package/dist/web/assets/WorkspaceDiffPanel-qfFj8IRq.js +14 -0
  69. package/dist/web/assets/WorkspaceEnvironmentMenu-C0-mH7oF.js +32 -0
  70. package/dist/web/assets/WorkspaceGitDialogs-VvC2px0X.js +1 -0
  71. package/dist/web/assets/{WorkspaceMonacoEditor-C0fDue2d.js → WorkspaceMonacoEditor-CUIwoHv1.js} +3 -3
  72. package/dist/web/assets/activity-BhqHgkfJ.js +1 -0
  73. package/dist/web/assets/{boxes-CXLqRF98.js → boxes-BpqUKFIm.js} +1 -1
  74. package/dist/web/assets/chevron-up-BTn9DVRP.js +1 -0
  75. package/dist/web/assets/circle-x-B9Qa8o5a.js +1 -0
  76. package/dist/web/assets/copy-BsYs6wBI.js +1 -0
  77. package/dist/web/assets/{create-pipeline-request-C26NOc7v.js → create-pipeline-request-vxgKoAvv.js} +3 -3
  78. package/dist/web/assets/credit-card-BCePAwMI.js +1 -0
  79. package/dist/web/assets/{cssMode-DmtwKUVm.js → cssMode-mw8RI2fR.js} +1 -1
  80. package/dist/web/assets/git-branch-C7qQYFIs.js +1 -0
  81. package/dist/web/assets/{git-commit-horizontal-CDDZl4iA.js → git-commit-horizontal-CbX4pTOA.js} +1 -1
  82. package/dist/web/assets/{htmlMode-Box0EOm1.js → htmlMode-Cc4Czh_M.js} +1 -1
  83. package/dist/web/assets/index-9tbdqUau.css +1 -0
  84. package/dist/web/assets/index-B_3z_lxq.js +1 -0
  85. package/dist/web/assets/index-Cr_RZ7GD.js +177 -0
  86. package/dist/web/assets/{info-4Iy6BKvT.js → info-BNGbwgtq.js} +1 -1
  87. package/dist/web/assets/{jsonMode-g7t_3HFv.js → jsonMode-CbqbqGVz.js} +1 -1
  88. package/dist/web/assets/{lspLanguageFeatures-CwhFOYc9.js → lspLanguageFeatures-DKiKr5tH.js} +1 -1
  89. package/dist/web/assets/make-agent-tutorial-KKauJpB9.js +2 -0
  90. package/dist/web/assets/{monaco.contribution-B-QC_ls9.js → monaco.contribution-CNgwz_pw.js} +2 -2
  91. package/dist/web/assets/{monaco.contribution-DGZAePHI.js → monaco.contribution-CdUAX47Q.js} +2 -2
  92. package/dist/web/assets/{monaco.contribution-CwMt3eHi.js → monaco.contribution-D_-wu578.js} +2 -2
  93. package/dist/web/assets/{monaco.contribution-CW2ctrTm.js → monaco.contribution-Dxc4IxFM.js} +2 -2
  94. package/dist/web/assets/{monitor-DOVjBLkr.js → monitor-CNdx_GTg.js} +1 -1
  95. package/dist/web/assets/{play-DLCuESeG.js → play-BLRANWaV.js} +1 -1
  96. package/dist/web/assets/{python-DRNmHcYd.js → python-DYgDMu2R.js} +1 -1
  97. package/dist/web/assets/{refresh-cw-JNUovwk4.js → refresh-cw-BFjjwYNf.js} +1 -1
  98. package/dist/web/assets/{reply-D4QvJyzE.js → reply-D012cuZl.js} +1 -1
  99. package/dist/web/assets/rotate-cw-CTfjlR4G.js +1 -0
  100. package/dist/web/assets/{save-BR4O4UDP.js → save-B1zYlRyN.js} +1 -1
  101. package/dist/web/assets/{toggleHighContrast-CrfdWIsV.js → toggleHighContrast-D9XFH7EA.js} +1 -1
  102. package/dist/web/assets/training-BN8CALIc.css +1 -0
  103. package/dist/web/assets/{tsMode-ZuV9go1B.js → tsMode-BU-tuKsY.js} +1 -1
  104. package/dist/web/assets/upload-DU_81jvL.js +1 -0
  105. package/dist/web/assets/{wifi-off-XFVqsf4c.js → wifi-off-TPsBnbGW.js} +1 -1
  106. package/dist/web/assets/{workers-CJpO1Wzj.js → workers-NhWySRdT.js} +1 -1
  107. package/dist/web/assets/{yaml-AQ4ok5ED.js → yaml-DWXa_3ZL.js} +1 -1
  108. package/dist/web/courses/post-training/01-how-post-training-works-poster.webp +0 -0
  109. package/dist/web/courses/post-training/01-how-post-training-works.vtt +25 -0
  110. package/dist/web/courses/post-training/02-definitions-poster.webp +0 -0
  111. package/dist/web/courses/post-training/02-definitions.vtt +117 -0
  112. package/dist/web/courses/post-training/03-on-policy-off-policy-poster.webp +0 -0
  113. package/dist/web/courses/post-training/03-on-policy-off-policy.vtt +29 -0
  114. package/dist/web/courses/post-training/04-rewards-credit-assignment-poster.webp +0 -0
  115. package/dist/web/courses/post-training/04-rewards-credit-assignment.vtt +65 -0
  116. package/dist/web/courses/post-training/05-verifiable-rewards-rlvr-poster.webp +0 -0
  117. package/dist/web/courses/post-training/05-verifiable-rewards-rlvr.vtt +69 -0
  118. package/dist/web/courses/post-training/06-ppo-grpo-poster.webp +0 -0
  119. package/dist/web/courses/post-training/06-ppo-grpo.vtt +65 -0
  120. package/dist/web/courses/post-training/07-distillation-poster.webp +0 -0
  121. package/dist/web/courses/post-training/07-distillation.vtt +53 -0
  122. package/dist/web/courses/post-training/08-opsd-sdft-sdpo-poster.webp +0 -0
  123. package/dist/web/courses/post-training/08-opsd-sdft-sdpo.vtt +61 -0
  124. package/dist/web/courses/post-training/09-credible-experiments-poster.webp +0 -0
  125. package/dist/web/courses/post-training/09-credible-experiments.vtt +69 -0
  126. package/dist/web/courses/post-training/10-technical-appendix-poster.webp +0 -0
  127. package/dist/web/courses/post-training/10-technical-appendix.vtt +69 -0
  128. package/dist/web/courses/post-training/full-course-poster.webp +0 -0
  129. package/dist/web/courses/post-training/full-course.vtt +623 -0
  130. package/dist/web/courses/post-training/scripts/script_01.md +31 -0
  131. package/dist/web/courses/post-training/scripts/script_02.md +49 -0
  132. package/dist/web/courses/post-training/scripts/script_03.md +35 -0
  133. package/dist/web/courses/post-training/scripts/script_04.md +43 -0
  134. package/dist/web/courses/post-training/scripts/script_05.md +43 -0
  135. package/dist/web/courses/post-training/scripts/script_06.md +41 -0
  136. package/dist/web/courses/post-training/scripts/script_07.md +37 -0
  137. package/dist/web/courses/post-training/scripts/script_08.md +37 -0
  138. package/dist/web/courses/post-training/scripts/script_09.md +45 -0
  139. package/dist/web/courses/post-training/scripts/script_10.md +47 -0
  140. package/dist/web/index.html +2 -2
  141. package/dist/web/tutorials/how-to-make-an-agent-create-poster.png +0 -0
  142. package/dist/web/tutorials/how-to-make-an-agent-create.vtt +49 -0
  143. package/dist/web/tutorials/how-to-make-an-agent-improve-poster.png +0 -0
  144. package/dist/web/tutorials/how-to-make-an-agent-improve.vtt +45 -0
  145. package/dist/web/tutorials/how-to-make-an-agent-poster.png +0 -0
  146. package/dist/web/tutorials/how-to-make-an-agent-use-poster.png +0 -0
  147. package/dist/web/tutorials/how-to-make-an-agent-use.vtt +33 -0
  148. package/dist/web/tutorials/how-to-make-an-agent.vtt +125 -0
  149. package/dist/web/tutorials/what-is-an-openpond-agent-poster.png +0 -0
  150. package/dist/web/tutorials/what-is-an-openpond-agent.vtt +25 -0
  151. package/package.json +1 -1
  152. package/dist/web/assets/CloudWorkView-Cu_VqkaD.js +0 -1
  153. package/dist/web/assets/CommandMenu-DWtdkvmq.js +0 -1
  154. package/dist/web/assets/CommunityView-f03qOTER.js +0 -1
  155. package/dist/web/assets/ComposerCreateImproveStrip-CZpiQJsJ.js +0 -1
  156. package/dist/web/assets/GetStartedView-C3ZrnEz5.css +0 -1
  157. package/dist/web/assets/GetStartedView-CLMmryTG.js +0 -1
  158. package/dist/web/assets/LabsRoute-BwR2g71k.js +0 -4
  159. package/dist/web/assets/LabsRoute-CgKtVYTP.css +0 -1
  160. package/dist/web/assets/ProfileSettingsSection-CD_x_V8d.js +0 -1
  161. package/dist/web/assets/SettingsView-BMmlKvlB.js +0 -63
  162. package/dist/web/assets/SettingsView-CAPf81v3.css +0 -1
  163. package/dist/web/assets/TeamChatView-B5SlhDso.js +0 -3
  164. package/dist/web/assets/TrainingComputeDialog-BFY6E0yU.js +0 -1
  165. package/dist/web/assets/TrainingCreationPanel-Cp66gMed.js +0 -1
  166. package/dist/web/assets/WorkspaceDiffPanel-D7ZEbryF.js +0 -14
  167. package/dist/web/assets/WorkspaceEnvironmentMenu-D4DmAJaS.js +0 -32
  168. package/dist/web/assets/WorkspaceGitDialogs-DHOfJxbL.js +0 -1
  169. package/dist/web/assets/index-8H-GDlM5.css +0 -1
  170. package/dist/web/assets/index-BUyeKpY6.js +0 -1
  171. package/dist/web/assets/index-CKmte5ok.js +0 -192
  172. package/dist/web/assets/training-DeVTd0m8.css +0 -1
  173. package/dist/web/assets/useComputeSettings-DUNS6uOl.js +0 -1
  174. package/dist/web/assets/useComputeSettings-R2Sw621I.css +0 -1
@@ -0,0 +1,53 @@
1
+ WEBVTT
2
+
3
+ 1
4
+ 00:00:00.700 --> 00:00:10.606
5
+ Outcome rewards compress an attempt to a scalar. Distillation supplies richer guidance by specifying a probability distribution over plausible next tokens.
6
+
7
+ 2
8
+ 00:00:13.965 --> 00:00:28.970
9
+ A one-hot target says “Cancelled Error” and assigns every alternative zero. A teacher distribution can say that “Cancelled Error” is most likely, “Timeout Error” is also plausible, “return” is weak, and “retry” is unlikely.
10
+
11
+ 3
12
+ 00:00:32.328 --> 00:00:36.101
13
+ That structure contains more information than the one sampled token.
14
+
15
+ 4
16
+ 00:00:39.459 --> 00:00:47.830
17
+ Teacher probability q weights the log of student probability p over the vocabulary. Minimizing that cross-entropy moves the student toward the teacher.
18
+
19
+ 5
20
+ 00:00:51.188 --> 00:00:59.599
21
+ In our example, the student initially favors “return,” while the teacher favors “Cancelled Error.” After an update, the two distributions move closer.
22
+
23
+ 6
24
+ 00:01:02.957 --> 00:01:14.229
25
+ Teacher temperature controls how much of that structure is visible. Low temperature makes the target almost one-hot. Higher temperature reveals alternatives in the tail; too much can magnify noise.
26
+
27
+ 7
28
+ 00:01:17.587 --> 00:01:22.495
29
+ Teacher-target temperature is separate from the rollout temperature used to sample behavior.
30
+
31
+ 8
32
+ 00:01:25.853 --> 00:01:39.854
33
+ KL direction changes the lesson too. Forward KL penalizes the student for missing teacher-supported alternatives. Reverse KL strongly penalizes student probability where the teacher assigns little mass, often favoring a narrower mode.
34
+
35
+ 9
36
+ 00:01:43.213 --> 00:01:46.986
37
+ “We used KL” is incomplete without direction and temperature.
38
+
39
+ 10
40
+ 00:01:50.344 --> 00:02:04.024
41
+ Prefix provenance determines which states receive teacher targets. Offline distillation uses fixed teacher trajectories. On-policy distillation lets the student reach its own strange cancellation state, then asks the teacher what should come next there.
42
+
43
+ 11
44
+ 00:02:07.383 --> 00:02:19.748
45
+ The teacher does not need larger weights. The same frozen model can see the student's failed prefix plus extra training-only evidence: a verified repair, an expert demonstration, or a test explanation.
46
+
47
+ 12
48
+ 00:02:23.107 --> 00:02:26.558
49
+ That evidence changes the teacher distribution.
50
+
51
+ 13
52
+ 00:02:29.917 --> 00:02:40.967
53
+ Training transfers the useful part of that privileged view into student weights. Deployment removes the evidence. A verified solution, demonstration, or failure explanation produces a different teacher target.
@@ -0,0 +1,61 @@
1
+ WEBVTT
2
+
3
+ 1
4
+ 00:00:00.700 --> 00:00:11.660
5
+ OPSD, SDFT, and SDPO share one mechanism. A frozen teacher scores the student's prefix while receiving additional evidence. The evidence source defines the method.
6
+
7
+ 2
8
+ 00:00:13.090 --> 00:00:22.997
9
+ The student distribution is p theta given the issue and its prefix. Teacher target q scores the same next token at the same prefix, but also conditions on evidence e.
10
+
11
+ 3
12
+ 00:00:24.427 --> 00:00:29.425
13
+ Training matches q into p; deployment removes e.
14
+
15
+ 4
16
+ 00:00:30.855 --> 00:00:40.180
17
+ On-Policy Self-Distillation, or OPSD, gives the teacher a trusted solution. Here that is the verified repair: preserve cleanup and raise Cancelled Error.
18
+
19
+ 5
20
+ 00:00:41.610 --> 00:00:48.294
21
+ The teacher uses it to guide the student's own prefix without placing the privileged patch in the deployed student's prompt.
22
+
23
+ 6
24
+ 00:00:49.724 --> 00:00:57.904
25
+ The OPSD paper evaluated this mechanism on reasoning tasks; the animation maps the information flow onto our code case.
26
+
27
+ 7
28
+ 00:00:59.335 --> 00:01:11.018
29
+ Self-Distillation Fine-Tuning, or SDFT, gives the teacher an expert demonstration. A related repair shows the sequence inspect the flag, run cleanup, then raise the exception.
30
+
31
+ 8
32
+ 00:01:12.448 --> 00:01:20.628
33
+ Unlike offline imitation, SDFT scores the current student's return-None prefix, so the demonstration guides the state the student actually reached.
34
+
35
+ 9
36
+ 00:01:22.058 --> 00:01:36.471
37
+ Self-Distillation Policy Optimization, or SDPO, gives the teacher feedback about the current failure. The scalar reward is zero, but the test trace says test-cancel failed, expected Cancelled Error, and return skipped cancellation.
38
+
39
+ 10
40
+ 00:01:37.901 --> 00:01:43.451
41
+ The teacher converts that explanation into dense token probabilities along the failed prefix.
42
+
43
+ 11
44
+ 00:01:44.881 --> 00:01:56.383
45
+ The choice follows the evidence. A trusted solution supports OPSD. A demonstration supports SDFT. An explanatory failure trace supports SDPO.
46
+
47
+ 12
48
+ 00:01:57.814 --> 00:02:06.817
49
+ If all you have is a binary outcome, GRPO may be the honest baseline; inventing a generic “feedback” field would hide these different trust boundaries.
50
+
51
+ 13
52
+ 00:02:08.247 --> 00:02:18.064
53
+ None of the dense methods is automatically safe. A privileged patch can leak. A demonstration can be irrelevant. A teacher can misread feedback and densify an error.
54
+
55
+ 14
56
+ 00:02:19.494 --> 00:02:27.955
57
+ New capability can overwrite old behavior, entropy can collapse, and teacher forward passes can erase claimed efficiency savings.
58
+
59
+ 15
60
+ 00:02:29.385 --> 00:02:42.434
61
+ So the durable comparison is not acronym against acronym. Hold the student prefix fixed, vary the evidence, record the cost, and measure both the cancellation repair and unrelated retained capability.
@@ -0,0 +1,69 @@
1
+ WEBVTT
2
+
3
+ 1
4
+ 00:00:00.700 --> 00:00:08.157
5
+ A research result requires many versioned tasks, protected information boundaries, matched baselines, and held-out evaluation.
6
+
7
+ 2
8
+ 00:00:09.546 --> 00:00:21.550
9
+ Policy-visible input includes the issue, repository revision, allowed tools, and public tests. Training-only evidence may contain an expert repair, privileged patch, or public diagnostic.
10
+
11
+ 3
12
+ 00:00:22.939 --> 00:00:29.533
13
+ Evaluator-only hidden tests, anti-exploit checks, and held-out repository clusters must remain outside both.
14
+
15
+ 4
16
+ 00:00:30.921 --> 00:00:44.512
17
+ The Taskset preserves those boundaries together with source revision, license, split, verifier version, and content hash. Related issues from the same repository should be clustered before splitting; otherwise memorization can look like generalization.
18
+
19
+ 5
20
+ 00:00:45.901 --> 00:00:56.770
21
+ Hugging Face support is worthwhile when it behaves as a reproducible importer. Pin the repository revision. Inspect the dataset card and license. Choose configuration and split explicitly.
22
+
23
+ 6
24
+ 00:00:58.159 --> 00:01:09.430
25
+ Map columns through a reviewed transform, log rejected rows, preserve original identifiers, and materialize an immutable snapshot. A training run should never silently follow a changing main branch.
26
+
27
+ 7
28
+ 00:01:10.819 --> 00:01:23.637
29
+ Every proposed method needs a simple credible baseline. Begin with the base model and the most direct token-imitation objective supported by the data. Add offline distillation or preference learning when those signals are relevant.
30
+
31
+ 8
32
+ 00:01:25.026 --> 00:01:33.578
33
+ Compare with GRPO or another RLVR baseline before claiming value from OPSD, SDFT, or SDPO.
34
+
35
+ 9
36
+ 00:01:34.966 --> 00:01:45.786
37
+ Equal optimizer tokens and equal total compute answer different questions. A distillation method may use fewer rollouts while spending additional teacher forward passes. Report both controls.
38
+
39
+ 10
40
+ 00:01:47.175 --> 00:02:04.540
41
+ Evaluation should span four dimensions. Capability measures task success. Diversity measures pass-at-k, entropy, and distinct strategies. Retention tests old domains and distribution drift. Integrity audits reward exploits, leakage, and disagreement with a shadow verifier.
42
+
43
+ 11
44
+ 00:02:05.929 --> 00:02:09.200
45
+ A single average can hide serious regressions.
46
+
47
+ 12
48
+ 00:02:10.589 --> 00:02:21.368
49
+ Compute accounting includes student rollout tokens, teacher-scored tokens, backward tokens, verifier time, external calls, failed jobs, memory, storage, and wall clock.
50
+
51
+ 13
52
+ 00:02:22.757 --> 00:02:39.168
53
+ The primary study compares GRPO, SDPO, and a routed success-and-failure method on code repair while separating public diagnostics from hidden tests. Verified math can replicate the GRPO and OPSD claims with exact graders.
54
+
55
+ 14
56
+ 00:02:40.557 --> 00:02:52.330
57
+ A tool protocol can replicate token imitation and SDFT while measuring retention. The extra domains test whether a conclusion transfers; they do not replace the main story halfway through it.
58
+
59
+ 15
60
+ 00:02:53.719 --> 00:03:04.398
61
+ The paper should bind every claim to a dataset revision, metric, run set, seed, confidence interval, and ablation. That structure keeps a local result from turning into a universal claim.
62
+
63
+ 16
64
+ 00:03:05.787 --> 00:03:19.839
65
+ The final map is simple. Demonstrations produce imitation targets. Preferences produce relative targets. Verifiable outcomes produce rewards. Privileged solutions and failure explanations produce context-conditioned teacher distributions.
66
+
67
+ 17
68
+ 00:03:21.228 --> 00:03:31.234
69
+ Evidence determines the target. The target determines the update. The update changes future behavior. Evaluation measures what improved—and what changed elsewhere.
@@ -0,0 +1,69 @@
1
+ WEBVTT
2
+
3
+ 1
4
+ 00:00:00.700 --> 00:00:07.565
5
+ The first appendix isolates two GRPO evaluation choices that matter after the main mechanism is understood.
6
+
7
+ 2
8
+ 00:00:12.059 --> 00:00:22.337
9
+ Length normalization changes gradient weight. A fourteen-hundred-token correct answer and a one-hundred-eighty-token correct answer may receive the same reward but contribute very different numbers of token terms.
10
+
11
+ 3
12
+ 00:00:26.831 --> 00:00:34.831
13
+ Sequence averaging, token averaging, overlong penalties, and group normalization therefore change the effective objective.
14
+
15
+ 4
16
+ 00:00:39.326 --> 00:00:48.329
17
+ Pass-at-one measures reliability from one sample. Pass-at-k asks whether any of k samples succeeds and exposes remaining search diversity.
18
+
19
+ 5
20
+ 00:00:52.824 --> 00:01:00.100
21
+ A model can improve pass-at-one while collapsing onto one strategy, so report both metrics with the sampling configuration.
22
+
23
+ 6
24
+ 00:01:01.500 --> 00:01:09.590
25
+ The systems problem is storage. A large vocabulary creates one teacher logit for every token at every trajectory position.
26
+
27
+ 7
28
+ 00:01:13.011 --> 00:01:22.055
29
+ Top-k compression stores the most likely teacher logits and approximates the remaining tail. This reduces memory and bandwidth but changes the target distribution.
30
+
31
+ 8
32
+ 00:01:25.476 --> 00:01:32.200
33
+ Measure divergence from full logits on a validation sample, and count teacher-forward cost when comparing efficiency.
34
+
35
+ 9
36
+ 00:01:33.600 --> 00:01:40.143
37
+ The reported graphs belong in an appendix because they come from different models, datasets, and experimental budgets.
38
+
39
+ 10
40
+ 00:01:42.001 --> 00:01:47.822
41
+ The OPSD study reports its largest displayed aggregate gain at the smallest Qwen3 model.
42
+
43
+ 11
44
+ 00:01:49.680 --> 00:02:02.137
45
+ At one point seven billion parameters, the displayed base score is thirty-seven point one, GRPO is thirty-seven point seven, and OPSD is forty-three point four. The gaps are smaller at four and eight billion parameters.
46
+
47
+ 12
48
+ 00:02:03.994 --> 00:02:09.133
49
+ That is a hypothesis for replication, not a universal model-size law.
50
+
51
+ 13
52
+ 00:02:10.991 --> 00:02:21.720
53
+ The SDFT study emphasizes knowledge acquisition and retention. Its displayed strict aggregate moves from eighty to eighty-nine, while its out-of-distribution aggregate moves from eighty to ninety-eight.
54
+
55
+ 14
56
+ 00:02:23.578 --> 00:02:35.713
57
+ The SDPO study reports forty-one point two for GRPO and forty-eight point eight for SDPO on its code benchmark. Their bars answer different questions and should not be read as a controlled head-to-head comparison.
58
+
59
+ 15
60
+ 00:02:37.571 --> 00:02:50.208
61
+ SRPO means Sample-Routed Policy Optimization. It sends verified successes toward a GRPO-style outcome update and failures with explanations toward an SDPO-style corrective target.
62
+
63
+ 16
64
+ 00:02:52.066 --> 00:03:03.839
65
+ The useful principle is signal routing. A successful rollout already shows what worked and can support sparse selection. A failed rollout with diagnostics may reveal how to recover and can support dense correction.
66
+
67
+ 17
68
+ 00:03:05.697 --> 00:03:11.067
69
+ The router preserves that asymmetry instead of forcing every sample through one uniform loss.