openpond 0.0.34 → 0.0.35

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (172) hide show
  1. package/dist/chunks/{app-layer-UQQNF5VZ.js → app-layer-GUF2CKYH.js} +3 -3
  2. package/dist/chunks/{apps-7NSC56K5.js → apps-DIVHAUOV.js} +3 -3
  3. package/dist/chunks/{chunk-MMDKFTJB.js → chunk-2GV3QBBZ.js} +1 -1
  4. package/dist/chunks/{chunk-BZPAW4JY.js → chunk-2I5TZW66.js} +918 -527
  5. package/dist/chunks/{chunk-TF234YN4.js → chunk-4Y25BACE.js} +1 -1
  6. package/dist/chunks/{chunk-ZRNFOHFH.js → chunk-CRFB72X2.js} +3 -3
  7. package/dist/chunks/{chunk-ZDEBKTWR.js → chunk-EY52SLOQ.js} +1 -1
  8. package/dist/chunks/{chunk-EBCLEJRO.js → chunk-FRPL5XPX.js} +1 -1
  9. package/dist/chunks/{chunk-UASXM3RR.js → chunk-GUGFYXFS.js} +4 -4
  10. package/dist/chunks/{chunk-5MM3P5KY.js → chunk-II2ERA2H.js} +2 -2
  11. package/dist/chunks/{chunk-V7SHAMUN.js → chunk-LUA5QGQP.js} +33 -33
  12. package/dist/chunks/{chunk-E3RLJ5XT.js → chunk-NTRTK6C6.js} +2 -2
  13. package/dist/chunks/{chunk-F6FFL3AA.js → chunk-RLQGMSW7.js} +2 -2
  14. package/dist/chunks/{chunk-JVTACGKD.js → chunk-S3VA47DG.js} +3 -3
  15. package/dist/chunks/{chunk-G7XOSECC.js → chunk-SECNDLPQ.js} +966 -528
  16. package/dist/chunks/{chunk-2GCNPJMD.js → chunk-TXJ4TIDO.js} +1 -1
  17. package/dist/chunks/{chunk-B5CFIWWU.js → chunk-XIF53VR3.js} +1 -1
  18. package/dist/chunks/{chunk-U4TUOBXO.js → chunk-YPK2W34Z.js} +1 -1
  19. package/dist/chunks/{chunk-HGZ3O4TS.js → chunk-ZZOTKXRD.js} +2 -2
  20. package/dist/chunks/{cli-VEDK2B42.js → cli-3AEIYSMD.js} +6 -6
  21. package/dist/chunks/{cli-4ETE5RS2.js → cli-HUVI6ELM.js} +4 -4
  22. package/dist/chunks/{core-commands-J3IQGQOL.js → core-commands-2HUTKFHQ.js} +4 -4
  23. package/dist/chunks/{extend-3F3LLDOF.js → extend-PAJMXYKV.js} +9 -9
  24. package/dist/chunks/{harness-JER7P6MJ.js → harness-UWAS2VAN.js} +3 -3
  25. package/dist/chunks/{help-75NFCYJ4.js → help-M7D2XAOX.js} +1 -1
  26. package/dist/chunks/{opchat-UT5V7MR2.js → opchat-FQ2CJVM6.js} +5 -5
  27. package/dist/chunks/{organizations-ERKRLVNQ.js → organizations-L5N6LJ7O.js} +4 -4
  28. package/dist/chunks/{profile-A3CUSDIF.js → profile-KH5IKVNY.js} +7 -7
  29. package/dist/chunks/{project-agent-FCV5HPYT.js → project-agent-F4NHVRGY.js} +6 -6
  30. package/dist/chunks/{sandbox-command-L2QPIWZ5.js → sandbox-command-DDXRQNB6.js} +4 -4
  31. package/dist/chunks/{sandbox-template-XOIWROEK.js → sandbox-template-5EEEQKR5.js} +5 -5
  32. package/dist/chunks/{src-V4FUQNHR.js → src-34FV42HE.js} +27227 -24056
  33. package/dist/chunks/{src-ENVQ7ARM.js → src-LHZWUSD7.js} +2 -2
  34. package/dist/chunks/{teams-bot-OR3ZJHQ5.js → teams-bot-HFWO5GOV.js} +3 -3
  35. package/dist/chunks/{workspaces-L73SRLPA.js → workspaces-LO44JAC2.js} +2 -2
  36. package/dist/cli.js +10 -10
  37. package/dist/web/assets/AppDialog-T3j6O3yg.js +1 -0
  38. package/dist/web/assets/{AppsView-CLjjVkjr.js → AppsView-70Qqf9eJ.js} +1 -1
  39. package/dist/web/assets/{BrowserSidebar-CkNO8723.js → BrowserSidebar-CdBLHdNW.js} +1 -1
  40. package/dist/web/assets/CloudWorkView-C5-jvlOo.js +1 -0
  41. package/dist/web/assets/{CommandMenu-DWtdkvmq.js → CommandMenu-B3Y4MvmT.js} +1 -1
  42. package/dist/web/assets/CommunityView-BHUJ9Hun.js +1 -0
  43. package/dist/web/assets/ComposerCreateImproveStrip-DeRI5XEO.js +1 -0
  44. package/dist/web/assets/GetStartedView-CIXScviw.js +1 -0
  45. package/dist/web/assets/GetStartedView-DBLdduwr.css +1 -0
  46. package/dist/web/assets/LabsRoute-Cs_R1fuu.js +3 -0
  47. package/dist/web/assets/MainChatThread-h7VSA_Rc.js +2 -0
  48. package/dist/web/assets/MakeAgentTutorialLearningPanel-r7jyZVc0.js +2 -0
  49. package/dist/web/assets/MarkdownText-E35C404I.js +7 -0
  50. package/dist/web/assets/Messages-BANZuAnA.js +10 -0
  51. package/dist/web/assets/NativeSkillSidebar-Qgccncuc.js +1 -0
  52. package/dist/web/assets/NativeSkillSidebar-T8KXjnUH.css +1 -0
  53. package/dist/web/assets/{NewProjectDialog-PrbA4fL6.js → NewProjectDialog-BowacUmA.js} +1 -1
  54. package/dist/web/assets/PostTrainingLearningPanel-oJu5qaHK.js +2 -0
  55. package/dist/web/assets/PostTrainingStatusPill-B9Eoc4s0.js +1 -0
  56. package/dist/web/assets/{ProfileSettingsSection-CqnfjiCl.css → ProfileSettingsSection-Cg4tKO-C.css} +1 -1
  57. package/dist/web/assets/ProfileSettingsSection-Dnh92gDH.js +1 -0
  58. package/dist/web/assets/RightChatPanelStack-pGwbt6Wy.js +1 -0
  59. package/dist/web/assets/SettingsView-Bjehxwt1.js +6 -0
  60. package/dist/web/assets/SettingsView-uHhQnazW.css +1 -0
  61. package/dist/web/assets/TeamChatView-Tn4X5cuY.js +3 -0
  62. package/dist/web/assets/{TerminalOverlay-DaGfEprz.js → TerminalOverlay-DoX5LYzK.js} +11 -11
  63. package/dist/web/assets/TrainingCreationPanel-BLSf1Bnp.js +1 -0
  64. package/dist/web/assets/{TrainingDraftPanel-BX5CCqlO.js → TrainingDraftPanel-Da6uGu9y.js} +1 -1
  65. package/dist/web/assets/UsageSettingsSection-DKky0p47.js +59 -0
  66. package/dist/web/assets/UsageSettingsSection-O8VK0qBL.css +1 -0
  67. package/dist/web/assets/WorkspaceDiffPanel-DTRtW9n8.js +14 -0
  68. package/dist/web/assets/WorkspaceEnvironmentMenu-Ba4M8z1N.js +32 -0
  69. package/dist/web/assets/WorkspaceGitDialogs-CSepNOb8.js +1 -0
  70. package/dist/web/assets/{WorkspaceMonacoEditor-C0fDue2d.js → WorkspaceMonacoEditor-LhFdzCRs.js} +3 -3
  71. package/dist/web/assets/activity-bW02ZsBD.js +1 -0
  72. package/dist/web/assets/{boxes-CXLqRF98.js → boxes-B1dRA1pi.js} +1 -1
  73. package/dist/web/assets/chevron-up-CySy6WNq.js +1 -0
  74. package/dist/web/assets/circle-alert-DMHz5tj_.js +1 -0
  75. package/dist/web/assets/circle-x-DY8vKAKi.js +1 -0
  76. package/dist/web/assets/copy-DuHI1UZ_.js +1 -0
  77. package/dist/web/assets/{create-pipeline-request-C26NOc7v.js → create-pipeline-request-CGBcdK9r.js} +2 -2
  78. package/dist/web/assets/credit-card-Bot0OuTl.js +1 -0
  79. package/dist/web/assets/{cssMode-DmtwKUVm.js → cssMode-BIjAothW.js} +1 -1
  80. package/dist/web/assets/git-branch-Cm02aNWq.js +1 -0
  81. package/dist/web/assets/{git-commit-horizontal-CDDZl4iA.js → git-commit-horizontal-FYILXW0v.js} +1 -1
  82. package/dist/web/assets/{htmlMode-Box0EOm1.js → htmlMode-zO2xLLgL.js} +1 -1
  83. package/dist/web/assets/index-B5AN29EO.js +177 -0
  84. package/dist/web/assets/index-BnwYyUgT.js +1 -0
  85. package/dist/web/assets/index-COGlKIJ4.css +1 -0
  86. package/dist/web/assets/{info-4Iy6BKvT.js → info-DeuEU_RO.js} +1 -1
  87. package/dist/web/assets/{jsonMode-g7t_3HFv.js → jsonMode-D2A3mj4n.js} +1 -1
  88. package/dist/web/assets/{lspLanguageFeatures-CwhFOYc9.js → lspLanguageFeatures-aPSqvscU.js} +1 -1
  89. package/dist/web/assets/make-agent-tutorial-A2W9Ajwv.js +2 -0
  90. package/dist/web/assets/{monaco.contribution-CW2ctrTm.js → monaco.contribution-CHVe9CQ9.js} +2 -2
  91. package/dist/web/assets/{monaco.contribution-DGZAePHI.js → monaco.contribution-DAq6Jihl.js} +2 -2
  92. package/dist/web/assets/{monaco.contribution-CwMt3eHi.js → monaco.contribution-Do3Jjslo.js} +2 -2
  93. package/dist/web/assets/{monaco.contribution-B-QC_ls9.js → monaco.contribution-KqCt1n9-.js} +2 -2
  94. package/dist/web/assets/{monitor-DOVjBLkr.js → monitor-DXvHWFeW.js} +1 -1
  95. package/dist/web/assets/{play-DLCuESeG.js → play-BIHyH-Yu.js} +1 -1
  96. package/dist/web/assets/{python-DRNmHcYd.js → python-g3mtmU2J.js} +1 -1
  97. package/dist/web/assets/{refresh-cw-JNUovwk4.js → refresh-cw-Daz0oqwK.js} +1 -1
  98. package/dist/web/assets/{reply-D4QvJyzE.js → reply-wu9DIBuK.js} +1 -1
  99. package/dist/web/assets/rotate-cw-D7q9GgNc.js +1 -0
  100. package/dist/web/assets/{save-BR4O4UDP.js → save-BVFY6VTI.js} +1 -1
  101. package/dist/web/assets/{toggleHighContrast-CrfdWIsV.js → toggleHighContrast-D9C_WUDu.js} +1 -1
  102. package/dist/web/assets/training-1du5QD-P.css +1 -0
  103. package/dist/web/assets/{tsMode-ZuV9go1B.js → tsMode-B4lEUN5X.js} +1 -1
  104. package/dist/web/assets/upload-DNbYB4Rl.js +1 -0
  105. package/dist/web/assets/{wifi-off-XFVqsf4c.js → wifi-off-C5QLifIJ.js} +1 -1
  106. package/dist/web/assets/{workers-CJpO1Wzj.js → workers-CV4BZb_x.js} +1 -1
  107. package/dist/web/assets/{yaml-AQ4ok5ED.js → yaml-3ah8oX43.js} +1 -1
  108. package/dist/web/courses/post-training/01-how-post-training-works-poster.webp +0 -0
  109. package/dist/web/courses/post-training/01-how-post-training-works.vtt +25 -0
  110. package/dist/web/courses/post-training/02-definitions-poster.webp +0 -0
  111. package/dist/web/courses/post-training/02-definitions.vtt +117 -0
  112. package/dist/web/courses/post-training/03-on-policy-off-policy-poster.webp +0 -0
  113. package/dist/web/courses/post-training/03-on-policy-off-policy.vtt +29 -0
  114. package/dist/web/courses/post-training/04-rewards-credit-assignment-poster.webp +0 -0
  115. package/dist/web/courses/post-training/04-rewards-credit-assignment.vtt +65 -0
  116. package/dist/web/courses/post-training/05-verifiable-rewards-rlvr-poster.webp +0 -0
  117. package/dist/web/courses/post-training/05-verifiable-rewards-rlvr.vtt +69 -0
  118. package/dist/web/courses/post-training/06-ppo-grpo-poster.webp +0 -0
  119. package/dist/web/courses/post-training/06-ppo-grpo.vtt +65 -0
  120. package/dist/web/courses/post-training/07-distillation-poster.webp +0 -0
  121. package/dist/web/courses/post-training/07-distillation.vtt +53 -0
  122. package/dist/web/courses/post-training/08-opsd-sdft-sdpo-poster.webp +0 -0
  123. package/dist/web/courses/post-training/08-opsd-sdft-sdpo.vtt +61 -0
  124. package/dist/web/courses/post-training/09-credible-experiments-poster.webp +0 -0
  125. package/dist/web/courses/post-training/09-credible-experiments.vtt +69 -0
  126. package/dist/web/courses/post-training/10-technical-appendix-poster.webp +0 -0
  127. package/dist/web/courses/post-training/10-technical-appendix.vtt +69 -0
  128. package/dist/web/courses/post-training/full-course-poster.webp +0 -0
  129. package/dist/web/courses/post-training/full-course.vtt +623 -0
  130. package/dist/web/courses/post-training/scripts/script_01.md +31 -0
  131. package/dist/web/courses/post-training/scripts/script_02.md +49 -0
  132. package/dist/web/courses/post-training/scripts/script_03.md +35 -0
  133. package/dist/web/courses/post-training/scripts/script_04.md +43 -0
  134. package/dist/web/courses/post-training/scripts/script_05.md +43 -0
  135. package/dist/web/courses/post-training/scripts/script_06.md +41 -0
  136. package/dist/web/courses/post-training/scripts/script_07.md +37 -0
  137. package/dist/web/courses/post-training/scripts/script_08.md +37 -0
  138. package/dist/web/courses/post-training/scripts/script_09.md +45 -0
  139. package/dist/web/courses/post-training/scripts/script_10.md +47 -0
  140. package/dist/web/index.html +2 -2
  141. package/dist/web/tutorials/how-to-make-an-agent-create-poster.png +0 -0
  142. package/dist/web/tutorials/how-to-make-an-agent-create.vtt +49 -0
  143. package/dist/web/tutorials/how-to-make-an-agent-improve-poster.png +0 -0
  144. package/dist/web/tutorials/how-to-make-an-agent-improve.vtt +45 -0
  145. package/dist/web/tutorials/how-to-make-an-agent-poster.png +0 -0
  146. package/dist/web/tutorials/how-to-make-an-agent-use-poster.png +0 -0
  147. package/dist/web/tutorials/how-to-make-an-agent-use.vtt +33 -0
  148. package/dist/web/tutorials/how-to-make-an-agent.vtt +125 -0
  149. package/dist/web/tutorials/what-is-an-openpond-agent-poster.png +0 -0
  150. package/dist/web/tutorials/what-is-an-openpond-agent.vtt +25 -0
  151. package/package.json +1 -1
  152. package/dist/web/assets/CloudWorkView-Cu_VqkaD.js +0 -1
  153. package/dist/web/assets/CommunityView-f03qOTER.js +0 -1
  154. package/dist/web/assets/ComposerCreateImproveStrip-CZpiQJsJ.js +0 -1
  155. package/dist/web/assets/GetStartedView-C3ZrnEz5.css +0 -1
  156. package/dist/web/assets/GetStartedView-CLMmryTG.js +0 -1
  157. package/dist/web/assets/LabsRoute-BwR2g71k.js +0 -4
  158. package/dist/web/assets/ProfileSettingsSection-CD_x_V8d.js +0 -1
  159. package/dist/web/assets/SettingsView-BMmlKvlB.js +0 -63
  160. package/dist/web/assets/SettingsView-CAPf81v3.css +0 -1
  161. package/dist/web/assets/TeamChatView-B5SlhDso.js +0 -3
  162. package/dist/web/assets/TrainingComputeDialog-BFY6E0yU.js +0 -1
  163. package/dist/web/assets/TrainingCreationPanel-Cp66gMed.js +0 -1
  164. package/dist/web/assets/WorkspaceDiffPanel-D7ZEbryF.js +0 -14
  165. package/dist/web/assets/WorkspaceEnvironmentMenu-D4DmAJaS.js +0 -32
  166. package/dist/web/assets/WorkspaceGitDialogs-DHOfJxbL.js +0 -1
  167. package/dist/web/assets/index-8H-GDlM5.css +0 -1
  168. package/dist/web/assets/index-BUyeKpY6.js +0 -1
  169. package/dist/web/assets/index-CKmte5ok.js +0 -192
  170. package/dist/web/assets/training-DeVTd0m8.css +0 -1
  171. package/dist/web/assets/useComputeSettings-DUNS6uOl.js +0 -1
  172. package/dist/web/assets/useComputeSettings-R2Sw621I.css +0 -1
@@ -0,0 +1,623 @@
1
+ WEBVTT
2
+
3
+ 1
4
+ 00:00:00.700 --> 00:00:10.887
5
+ Post-training changes what a model is likely to do. An input creates a distribution over possible actions, evaluation supplies evidence, and an optimizer changes that distribution.
6
+
7
+ 2
8
+ 00:00:12.598 --> 00:00:24.693
9
+ Here a cancellation test is failing. The policy assigns chances to return None, raise the cancellation error, or retry. Those probabilities describe the available behavior; sampling produces one actual patch.
10
+
11
+ 3
12
+ 00:00:26.404 --> 00:00:34.313
13
+ Loss is a number saying how bad behavior looks under the training rule. Useful training lowers both training loss and held-out error.
14
+
15
+ 4
16
+ 00:00:36.024 --> 00:00:42.025
17
+ If training loss falls while held-out error rises, the model is learning the wrong thing.
18
+
19
+ 5
20
+ 00:00:43.736 --> 00:00:56.875
21
+ The sampled patch passes. Training makes the exception-raising action slightly easier to choose in similar states. Ordinary backpropagation implements the change; reinforcement learning determines which sampled behavior receives credit.
22
+
23
+ 6
24
+ 00:00:58.587 --> 00:01:04.267
25
+ A usable training record must also identify who generated the attempt and what evidence accompanied it.
26
+
27
+
28
+ 7
29
+ 00:01:05.667 --> 00:01:17.209
30
+ The expression pi theta of a given state is a compact description of model behavior. Pi names the policy: a probability distribution, not one completed answer. Theta means all adjustable model parameters.
31
+
32
+ 8
33
+ 00:01:18.341 --> 00:01:25.205
34
+ The state is the context available now, and an action can be one token, one tool call, or another sampled decision.
35
+
36
+ 9
37
+ 00:01:26.336 --> 00:01:34.837
38
+ Before sampling, the network produces raw scores called logits. A logit is not a percentage and logits do not need to add to one. They can be positive or negative.
39
+
40
+ 10
41
+ 00:01:35.968 --> 00:01:43.968
42
+ Only their relative differences matter: a larger logit means the model prefers that action compared with the alternatives available at that position.
43
+
44
+ 11
45
+ 00:01:45.099 --> 00:01:56.460
46
+ Softmax converts logits into probabilities. It exponentiates each logit, then divides by the sum of all exponentiated logits. This makes every result positive and makes the distribution sum to one.
47
+
48
+ 12
49
+ 00:01:57.592 --> 00:02:11.864
50
+ In the example, logits three, one, and point two become probabilities of roughly eighty-four, eleven, and five percent. Temperature divides the logits before softmax. Lower temperature sharpens the distribution; higher temperature flattens it.
51
+
52
+ 13
53
+ 00:02:12.995 --> 00:02:22.590
54
+ A rollout, also called a trajectory and written tau, is the interaction record used for learning. The behavior policy mu generated each sampled action.
55
+
56
+ 14
57
+ 00:02:23.721 --> 00:02:37.452
58
+ Its log-probability is stored beside the action so training can later compare the old behavior distribution with the current policy. Environment observations affect later states, but only the model's sampled actions receive policy-gradient terms.
59
+
60
+ 15
61
+ 00:02:38.583 --> 00:02:50.899
62
+ Reward is the scalar number emitted by an evaluator. A math checker might return one for an exact answer and zero otherwise. A code environment might return one only when every hidden test passes.
63
+
64
+ 16
65
+ 00:02:52.030 --> 00:03:05.710
66
+ A tool task might inspect whether the final database state matches a target. The rule can be binary or graded, but the optimizer sees the number, not the evaluator's full explanation. That richer explanation is feedback.
67
+
68
+ 17
69
+ 00:03:06.841 --> 00:03:19.067
70
+ Reward, return, and advantage are related but not interchangeable. Reward is the scalar emitted at a step or at the end. Return, G at time t, combines future rewards and can discount distant outcomes with gamma.
71
+
72
+ 18
73
+ 00:03:20.198 --> 00:03:34.149
74
+ Advantage subtracts a baseline from that return. Positive advantage means the action did better than expected; negative advantage means it did worse. PPO learns a value baseline, while GRPO estimates one from sibling responses.
75
+
76
+ 19
77
+ 00:03:35.280 --> 00:03:50.506
78
+ PPO means Proximal Policy Optimization. It is a policy-gradient method that trains a second network, called a critic or value function, to estimate expected return. The observed return minus that expectation becomes advantage.
79
+
80
+ 20
81
+ 00:03:51.638 --> 00:03:58.954
82
+ PPO also clips the probability-ratio incentive so one sampled action cannot drive an arbitrarily large update.
83
+
84
+ 21
85
+ 00:04:00.085 --> 00:04:12.591
86
+ GRPO means Group Relative Policy Optimization. It removes the learned critic and samples several sibling responses for the same prompt. Each response is compared with the group's reward mean and standard deviation.
87
+
88
+ 22
89
+ 00:04:13.723 --> 00:04:23.177
90
+ Better-than-group responses receive positive advantage; worse responses receive negative advantage. GRPO keeps policy ratios and clipping.
91
+
92
+ 23
93
+ 00:04:24.308 --> 00:04:38.540
94
+ Both methods need a baseline because raw reward lacks context. PPO's baseline comes from the critic, and Generalized Advantage Estimation combines value predictions across time. GRPO's baseline comes from sibling rewards.
95
+
96
+ 24
97
+ 00:04:39.671 --> 00:04:46.676
98
+ If every sibling receives the same reward, the group standard deviation is zero and there is no relative advantage to learn from.
99
+
100
+ 25
101
+ 00:04:47.807 --> 00:04:59.762
102
+ An objective, often written L of theta, turns the training goal into one number. The gradient is the local slope of that objective with respect to every adjustable parameter. Backpropagation computes those slopes.
103
+
104
+ 26
105
+ 00:05:00.893 --> 00:05:13.801
106
+ The optimizer applies a learning rate alpha to make a small update from theta to theta prime. In policy-gradient training, the RL method determines the credit weights inside the loss; ordinary differentiation performs the parameter update.
107
+
108
+ 27
109
+ 00:05:14.932 --> 00:05:26.434
110
+ PPO and GRPO use a probability ratio. The numerator is the current policy's probability for the stored action. The denominator is the behavior probability recorded during rollout.
111
+
112
+ 28
113
+ 00:05:27.565 --> 00:05:35.745
114
+ A ratio above one means that action became more likely. Clipping with epsilon caps the incentive once the ratio moves too far in one update.
115
+
116
+ 29
117
+ 00:05:36.876 --> 00:05:41.965
118
+ It limits this sampled term, but does not prove that the entire model stayed close.
119
+
120
+ 30
121
+ 00:05:43.096 --> 00:05:55.512
122
+ Entropy measures how spread out one policy distribution is. KL divergence compares two distributions and is commonly used to measure movement from a reference model or mismatch between teacher and student. Its direction matters.
123
+
124
+ 31
125
+ 00:05:56.643 --> 00:06:07.784
126
+ Cross-entropy uses target probabilities q to weight the log of student probabilities p. Distillation usually minimizes this quantity so the student assigns probability where the teacher does.
127
+
128
+ 32
129
+ 00:06:08.915 --> 00:06:24.965
130
+ RL means reinforcement learning. RFT means reinforcement fine-tuning: applying an RL-style objective to an already pretrained model. RLVR adds verifiable rewards, such as executable tests. PPO and GRPO are policy update rules.
131
+
132
+ 33
133
+ 00:06:26.096 --> 00:06:45.096
134
+ PPO is Proximal Policy Optimization. GRPO is Group Relative Policy Optimization. GAE is Generalized Advantage Estimation. Pass-at-k asks whether at least one of k sampled attempts succeeds. On-policy and off-policy describe who generated the data.
135
+
136
+ 34
137
+ 00:06:46.227 --> 00:07:01.453
138
+ The teacher-guided methods use dense distributions instead of only scalar outcomes. OPSD is On-Policy Self-Distillation and conditions the teacher on a trusted solution. SDFT is Self-Distillation Fine-Tuning and supplies a demonstration.
139
+
140
+ 35
141
+ 00:07:02.584 --> 00:07:18.634
142
+ SDPO is Self-Distillation Policy Optimization and supplies failure feedback. In each case the student creates the prefix and the frozen teacher receives extra evidence. SRPO, Self-Rewarding Policy Optimization, routes successful and unsuccessful samples through different signals.
143
+
144
+
145
+ 36
146
+ 00:07:20.034 --> 00:07:26.808
147
+ On-policy data comes from the learner being updated. Policy version twelve generates an attempt that will update version twelve.
148
+
149
+ 37
150
+ 00:07:28.399 --> 00:07:35.716
151
+ Off-policy data comes from a teacher, human, or older checkpoint. These labels describe source, not quality.
152
+
153
+ 38
154
+ 00:07:37.307 --> 00:07:45.898
155
+ An RL rollout stores the problem, sampled actions, generating checkpoint, behavior probabilities, environment observations, and reward.
156
+
157
+ 39
158
+ 00:07:47.489 --> 00:07:54.173
159
+ Text alone cannot connect evaluated behavior to an update. The missing record matters as much as the final response.
160
+
161
+ 40
162
+ 00:07:55.764 --> 00:08:04.135
163
+ Other sources carry different signals: teacher targets, chosen and rejected responses, or old actions with original probabilities and rewards.
164
+
165
+ 41
166
+ 00:08:05.726 --> 00:08:16.867
167
+ PPO and GRPO usually score learner attempts. Offline distillation uses teacher examples. On-policy distillation labels the student's current prefix.
168
+
169
+ 42
170
+ 00:08:18.458 --> 00:08:25.001
171
+ Inside any RL rollout, the final patch is only one part of a sequence of actions and observations.
172
+
173
+
174
+ 43
175
+ 00:08:26.401 --> 00:08:35.354
176
+ An RL rollout is a sequence, not only a final response. Credit assignment asks which decisions inside that sequence should become more or less likely.
177
+
178
+ 44
179
+ 00:08:37.082 --> 00:08:49.990
180
+ The issue and repository form state zero. The behavior policy samples “inspect files.” File contents are an environment observation, not another sampled action. The next state contains those contents.
181
+
182
+ 45
183
+ 00:08:51.717 --> 00:08:57.579
184
+ The policy then writes raise Cancelled Error, and test output closes the trajectory.
185
+
186
+ 46
187
+ 00:08:59.306 --> 00:09:12.124
188
+ The tuple makes those roles explicit. Tau contains states, sampled actions, behavior log-probabilities, observations, and terminal reward. Gradients apply to action positions controlled by the policy.
189
+
190
+ 47
191
+ 00:09:13.851 --> 00:09:18.538
192
+ Observations condition later actions but are masked out of the policy-gradient loss.
193
+
194
+ 48
195
+ 00:09:20.266 --> 00:09:31.316
196
+ Reward and feedback carry different information. A patch can receive reward zero while the environment still knows that eight tests passed, cancellation failed, and the expected exception was Cancelled Error.
197
+
198
+ 49
199
+ 00:09:33.044 --> 00:09:38.634
200
+ Reward is the scalar consumed by an objective. Feedback is the richer evidence the environment produced.
201
+
202
+ 50
203
+ 00:09:40.362 --> 00:09:51.633
204
+ A terminal score also leaves the cause ambiguous. Was the wrong file inspected? Was the timeout branch incorrect? Was “None” the decisive token? Response-level credit applies one outcome broadly.
205
+
206
+ 51
207
+ 00:09:53.360 --> 00:09:58.640
208
+ Step- or token-level signals can be denser, but density alone does not prove causal accuracy.
209
+
210
+ 52
211
+ 00:10:00.367 --> 00:10:13.867
212
+ For multi-step problems, return carries future rewards backward. G at time t is the discounted sum of later rewards. With gamma point nine and success two steps later, the earlier action receives return point eight one.
213
+
214
+ 53
215
+ 00:10:15.595 --> 00:10:21.185
216
+ PPO critics and Generalized Advantage Estimation, or GAE, build on this idea.
217
+
218
+ 54
219
+ 00:10:22.912 --> 00:10:34.686
220
+ Advantage then asks: better than what was expected? If observed return is point eight and the baseline is point five, advantage is positive point three. A return of point two gives negative point three.
221
+
222
+ 55
223
+ 00:10:36.413 --> 00:10:49.010
224
+ Positive advantage raises the sampled behavior’s probability; negative advantage lowers it. PPO learns a critic for the baseline. GRPO replaces that critic with statistics from sibling responses.
225
+
226
+ 56
227
+ 00:10:50.738 --> 00:11:01.878
228
+ Learning also changes exploration. Pass-at-one can rise while entropy and pass-at-k fall if the model collapses onto one strategy. Reliability and breadth therefore need separate metrics.
229
+
230
+ 57
231
+ 00:11:03.606 --> 00:11:18.149
232
+ Finally, a frozen reference policy anchors behavior outside the optimized task. Kullback–Leibler divergence, shortened to KL, measures distribution shift. Too little constraint permits drift and reward exploits; too much prevents useful learning.
233
+
234
+ 58
235
+ 00:11:19.877 --> 00:11:24.835
236
+ A test-based reward is useful only when the test measures the intended outcome reliably.
237
+
238
+
239
+ 59
240
+ 00:11:26.235 --> 00:11:32.828
241
+ Every reward-based update depends on whether its evaluator measures the right thing. This is the verifier problem.
242
+
243
+ 60
244
+ 00:11:33.989 --> 00:11:44.487
245
+ RLVR means Reinforcement Learning with Verifiable Rewards. For our repair, the verifier applies the patch, runs the cancellation tests, and returns a reproducible outcome.
246
+
247
+ 61
248
+ 00:11:45.647 --> 00:11:50.696
249
+ The same idea applies to checked math answers, tool states, and agent environments.
250
+
251
+ 62
252
+ 00:11:51.856 --> 00:11:58.399
253
+ A preference such as “patch A is cleaner” depends on a judge. A verifiable reward applies an explicit rule.
254
+
255
+ 63
256
+ 00:11:59.560 --> 00:12:07.468
257
+ Verifiable does not mean infallible; it means the same patch should receive the same result under the same evaluator state.
258
+
259
+ 64
260
+ 00:12:08.629 --> 00:12:20.081
261
+ Write the rule as r equals V of x, y, and e. X is the public task. Y is the sampled attempt. E is evaluator-only state such as hidden tests.
262
+
263
+ 65
264
+ 00:12:21.241 --> 00:12:25.426
265
+ For a binary verifier, reward is zero or one.
266
+
267
+ 66
268
+ 00:12:26.586 --> 00:12:39.364
269
+ This equation separates the setting from the optimizer. RLVR supplies reward. PPO, GRPO, or another policy optimizer supplies the update. DeepSeek-R1 is an important example of reasoning training built from this combination.
270
+
271
+ 67
272
+ 00:12:40.524 --> 00:12:51.796
273
+ The full loop is prompt, policy attempt, execution, verification, reward, and policy update. Automatic checks scale because a person does not need to score every trajectory.
274
+
275
+ 68
276
+ 00:12:52.956 --> 00:12:57.733
277
+ The verifier also becomes both the task specification and an attack surface.
278
+
279
+ 69
280
+ 00:12:58.894 --> 00:13:12.213
281
+ RLVR applies beyond short math answers. Code can compile and run tests. Tool agents can be checked against a target database state. Search can verify retrieved evidence. Scientific tasks can use simulators and constraints.
282
+
283
+ 70
284
+ 00:13:13.374 --> 00:13:20.509
285
+ It is a poor fit when quality is inherently subjective, outcomes cannot be reproduced, or the checker is easy to game.
286
+
287
+ 71
288
+ 00:13:21.670 --> 00:13:33.262
289
+ Verifier errors have two directions. A false negative rejects a valid solution. A false positive is more dangerous under optimization because a wrong solution receives positive gradient and can become an exploit.
290
+
291
+ 72
292
+ 00:13:34.423 --> 00:13:43.788
293
+ Suppose visible tests cover only three examples. Hard-coding those outputs and implementing the general rule both receive reward one. The optimizer cannot distinguish them.
294
+
295
+ 73
296
+ 00:13:44.948 --> 00:13:52.405
297
+ Hidden tests, randomized cases, invariants, and shadow verifiers make the measured objective closer to the intended task.
298
+
299
+ 74
300
+ 00:13:53.565 --> 00:14:07.567
301
+ That requires an information boundary. The policy sees the public task and allowed observations. The evaluator alone sees hidden checks and anti-cheat state. Training-only teacher evidence belongs in a third protected channel.
302
+
303
+ 75
304
+ 00:14:08.727 --> 00:14:17.369
305
+ So “RLVR with GRPO” is precise: the first term names how outcomes are labeled, and the second names how those labels become an update.
306
+
307
+
308
+ 76
309
+ 00:14:18.769 --> 00:14:27.321
310
+ A trustworthy reward still has to become a learning signal. PPO and GRPO differ mainly in how they decide whether an outcome was better or worse than expected.
311
+
312
+ 77
313
+ 00:14:28.868 --> 00:14:43.281
314
+ PPO means Proximal Policy Optimization. For the successful inspect, patch, and test trajectory, observed return is point eight. A learned critic expected point five. Their difference, positive point three, is the advantage.
315
+
316
+ 78
317
+ 00:14:44.829 --> 00:14:49.877
318
+ PPO combines that advantage with a clipped probability ratio and updates the policy.
319
+
320
+ 79
321
+ 00:14:51.425 --> 00:15:06.249
322
+ GRPO means Group Relative Policy Optimization. It keeps rollouts, rewards, probability ratios, and clipping, but removes the critic. Instead, it samples several patches for the same cancellation bug and compares their verifier scores.
323
+
324
+ 80
325
+ 00:15:07.797 --> 00:15:16.710
326
+ Our six sibling patches produce rewards one, zero, zero, one, one, zero. The mean is one half and the standard deviation is one half.
327
+
328
+ 81
329
+ 00:15:18.257 --> 00:15:24.119
330
+ Normalizing gives every passing patch advantage positive one and every failing patch negative one.
331
+
332
+ 82
333
+ 00:15:25.666 --> 00:15:35.031
334
+ That comparison is useful but coarse. Raise Timeout Error and return None both failed, so binary GRPO treats them alike even if one was closer.
335
+
336
+ 83
337
+ 00:15:36.579 --> 00:15:40.804
338
+ The group says which siblings did better; it does not explain why.
339
+
340
+ 84
341
+ 00:15:42.351 --> 00:15:52.538
342
+ The clipping graph controls how much one sampled action can influence a single update. If its probability moves from point one zero to point one three, the ratio is one point three.
343
+
344
+ 85
345
+ 00:15:54.086 --> 00:16:03.902
346
+ With a cap at one point two, additional positive credit stops growing. Clipping limits a local incentive; it does not guarantee that the whole policy stayed close.
347
+
348
+ 86
349
+ 00:16:05.450 --> 00:16:08.681
350
+ A group teaches only when it contains contrast.
351
+
352
+ 87
353
+ 00:16:10.228 --> 00:16:22.684
354
+ If one patch succeeds with probability p, a group of size G is mixed with probability one minus p to the G minus one minus p, all to the G. That subtracts all-success and all-failure groups.
355
+
356
+ 88
357
+ 00:16:24.232 --> 00:16:38.735
358
+ Near zero success, almost every group fails. Near one, almost every group passes. Either case produces little relative signal. Larger groups widen the useful middle but require more patch generation and test execution.
359
+
360
+ 89
361
+ 00:16:40.283 --> 00:16:50.690
362
+ This is why prompts need to sit near the learning frontier: difficult enough to fail, possible enough to solve. If every cancellation patch receives the same reward, normalized group advantage is zero.
363
+
364
+ 90
365
+ 00:16:52.238 --> 00:17:04.283
366
+ Both PPO and GRPO need trajectories, rewards, behavior log-probabilities, clipped ratios, and gradients. PPO learns expected value with a critic. GRPO estimates a baseline from sibling patches.
367
+
368
+ 91
369
+ 00:17:05.830 --> 00:17:10.969
370
+ The cancellation environment stays the same; only the baseline estimator changes.
371
+
372
+
373
+ 92
374
+ 00:17:12.369 --> 00:17:22.275
375
+ Outcome rewards compress an attempt to a scalar. Distillation supplies richer guidance by specifying a probability distribution over plausible next tokens.
376
+
377
+ 93
378
+ 00:17:25.634 --> 00:17:40.639
379
+ A one-hot target says “Cancelled Error” and assigns every alternative zero. A teacher distribution can say that “Cancelled Error” is most likely, “Timeout Error” is also plausible, “return” is weak, and “retry” is unlikely.
380
+
381
+ 94
382
+ 00:17:43.997 --> 00:17:47.770
383
+ That structure contains more information than the one sampled token.
384
+
385
+ 95
386
+ 00:17:51.128 --> 00:17:59.499
387
+ Teacher probability q weights the log of student probability p over the vocabulary. Minimizing that cross-entropy moves the student toward the teacher.
388
+
389
+ 96
390
+ 00:18:02.857 --> 00:18:11.268
391
+ In our example, the student initially favors “return,” while the teacher favors “Cancelled Error.” After an update, the two distributions move closer.
392
+
393
+ 97
394
+ 00:18:14.626 --> 00:18:25.898
395
+ Teacher temperature controls how much of that structure is visible. Low temperature makes the target almost one-hot. Higher temperature reveals alternatives in the tail; too much can magnify noise.
396
+
397
+ 98
398
+ 00:18:29.256 --> 00:18:34.164
399
+ Teacher-target temperature is separate from the rollout temperature used to sample behavior.
400
+
401
+ 99
402
+ 00:18:37.522 --> 00:18:51.523
403
+ KL direction changes the lesson too. Forward KL penalizes the student for missing teacher-supported alternatives. Reverse KL strongly penalizes student probability where the teacher assigns little mass, often favoring a narrower mode.
404
+
405
+ 100
406
+ 00:18:54.882 --> 00:18:58.655
407
+ “We used KL” is incomplete without direction and temperature.
408
+
409
+ 101
410
+ 00:19:02.013 --> 00:19:15.693
411
+ Prefix provenance determines which states receive teacher targets. Offline distillation uses fixed teacher trajectories. On-policy distillation lets the student reach its own strange cancellation state, then asks the teacher what should come next there.
412
+
413
+ 102
414
+ 00:19:19.052 --> 00:19:31.417
415
+ The teacher does not need larger weights. The same frozen model can see the student's failed prefix plus extra training-only evidence: a verified repair, an expert demonstration, or a test explanation.
416
+
417
+ 103
418
+ 00:19:34.776 --> 00:19:38.227
419
+ That evidence changes the teacher distribution.
420
+
421
+ 104
422
+ 00:19:41.586 --> 00:19:52.636
423
+ Training transfers the useful part of that privileged view into student weights. Deployment removes the evidence. A verified solution, demonstration, or failure explanation produces a different teacher target.
424
+
425
+
426
+ 105
427
+ 00:19:54.036 --> 00:20:04.996
428
+ OPSD, SDFT, and SDPO share one mechanism. A frozen teacher scores the student's prefix while receiving additional evidence. The evidence source defines the method.
429
+
430
+ 106
431
+ 00:20:06.426 --> 00:20:16.333
432
+ The student distribution is p theta given the issue and its prefix. Teacher target q scores the same next token at the same prefix, but also conditions on evidence e.
433
+
434
+ 107
435
+ 00:20:17.763 --> 00:20:22.761
436
+ Training matches q into p; deployment removes e.
437
+
438
+ 108
439
+ 00:20:24.191 --> 00:20:33.516
440
+ On-Policy Self-Distillation, or OPSD, gives the teacher a trusted solution. Here that is the verified repair: preserve cleanup and raise Cancelled Error.
441
+
442
+ 109
443
+ 00:20:34.946 --> 00:20:41.630
444
+ The teacher uses it to guide the student's own prefix without placing the privileged patch in the deployed student's prompt.
445
+
446
+ 110
447
+ 00:20:43.060 --> 00:20:51.240
448
+ The OPSD paper evaluated this mechanism on reasoning tasks; the animation maps the information flow onto our code case.
449
+
450
+ 111
451
+ 00:20:52.671 --> 00:21:04.354
452
+ Self-Distillation Fine-Tuning, or SDFT, gives the teacher an expert demonstration. A related repair shows the sequence inspect the flag, run cleanup, then raise the exception.
453
+
454
+ 112
455
+ 00:21:05.784 --> 00:21:13.964
456
+ Unlike offline imitation, SDFT scores the current student's return-None prefix, so the demonstration guides the state the student actually reached.
457
+
458
+ 113
459
+ 00:21:15.394 --> 00:21:29.807
460
+ Self-Distillation Policy Optimization, or SDPO, gives the teacher feedback about the current failure. The scalar reward is zero, but the test trace says test-cancel failed, expected Cancelled Error, and return skipped cancellation.
461
+
462
+ 114
463
+ 00:21:31.237 --> 00:21:36.787
464
+ The teacher converts that explanation into dense token probabilities along the failed prefix.
465
+
466
+ 115
467
+ 00:21:38.217 --> 00:21:49.719
468
+ The choice follows the evidence. A trusted solution supports OPSD. A demonstration supports SDFT. An explanatory failure trace supports SDPO.
469
+
470
+ 116
471
+ 00:21:51.150 --> 00:22:00.153
472
+ If all you have is a binary outcome, GRPO may be the honest baseline; inventing a generic “feedback” field would hide these different trust boundaries.
473
+
474
+ 117
475
+ 00:22:01.583 --> 00:22:11.400
476
+ None of the dense methods is automatically safe. A privileged patch can leak. A demonstration can be irrelevant. A teacher can misread feedback and densify an error.
477
+
478
+ 118
479
+ 00:22:12.830 --> 00:22:21.291
480
+ New capability can overwrite old behavior, entropy can collapse, and teacher forward passes can erase claimed efficiency savings.
481
+
482
+ 119
483
+ 00:22:22.721 --> 00:22:35.770
484
+ So the durable comparison is not acronym against acronym. Hold the student prefix fixed, vary the evidence, record the cost, and measure both the cancellation repair and unrelated retained capability.
485
+
486
+
487
+ 120
488
+ 00:22:37.170 --> 00:22:44.627
489
+ A research result requires many versioned tasks, protected information boundaries, matched baselines, and held-out evaluation.
490
+
491
+ 121
492
+ 00:22:46.016 --> 00:22:58.020
493
+ Policy-visible input includes the issue, repository revision, allowed tools, and public tests. Training-only evidence may contain an expert repair, privileged patch, or public diagnostic.
494
+
495
+ 122
496
+ 00:22:59.409 --> 00:23:06.003
497
+ Evaluator-only hidden tests, anti-exploit checks, and held-out repository clusters must remain outside both.
498
+
499
+ 123
500
+ 00:23:07.391 --> 00:23:20.982
501
+ The Taskset preserves those boundaries together with source revision, license, split, verifier version, and content hash. Related issues from the same repository should be clustered before splitting; otherwise memorization can look like generalization.
502
+
503
+ 124
504
+ 00:23:22.371 --> 00:23:33.240
505
+ Hugging Face support is worthwhile when it behaves as a reproducible importer. Pin the repository revision. Inspect the dataset card and license. Choose configuration and split explicitly.
506
+
507
+ 125
508
+ 00:23:34.629 --> 00:23:45.900
509
+ Map columns through a reviewed transform, log rejected rows, preserve original identifiers, and materialize an immutable snapshot. A training run should never silently follow a changing main branch.
510
+
511
+ 126
512
+ 00:23:47.289 --> 00:24:00.107
513
+ Every proposed method needs a simple credible baseline. Begin with the base model and the most direct token-imitation objective supported by the data. Add offline distillation or preference learning when those signals are relevant.
514
+
515
+ 127
516
+ 00:24:01.496 --> 00:24:10.048
517
+ Compare with GRPO or another RLVR baseline before claiming value from OPSD, SDFT, or SDPO.
518
+
519
+ 128
520
+ 00:24:11.436 --> 00:24:22.256
521
+ Equal optimizer tokens and equal total compute answer different questions. A distillation method may use fewer rollouts while spending additional teacher forward passes. Report both controls.
522
+
523
+ 129
524
+ 00:24:23.645 --> 00:24:41.010
525
+ Evaluation should span four dimensions. Capability measures task success. Diversity measures pass-at-k, entropy, and distinct strategies. Retention tests old domains and distribution drift. Integrity audits reward exploits, leakage, and disagreement with a shadow verifier.
526
+
527
+ 130
528
+ 00:24:42.399 --> 00:24:45.670
529
+ A single average can hide serious regressions.
530
+
531
+ 131
532
+ 00:24:47.059 --> 00:24:57.838
533
+ Compute accounting includes student rollout tokens, teacher-scored tokens, backward tokens, verifier time, external calls, failed jobs, memory, storage, and wall clock.
534
+
535
+ 132
536
+ 00:24:59.227 --> 00:25:15.638
537
+ The primary study compares GRPO, SDPO, and a routed success-and-failure method on code repair while separating public diagnostics from hidden tests. Verified math can replicate the GRPO and OPSD claims with exact graders.
538
+
539
+ 133
540
+ 00:25:17.027 --> 00:25:28.800
541
+ A tool protocol can replicate token imitation and SDFT while measuring retention. The extra domains test whether a conclusion transfers; they do not replace the main story halfway through it.
542
+
543
+ 134
544
+ 00:25:30.189 --> 00:25:40.868
545
+ The paper should bind every claim to a dataset revision, metric, run set, seed, confidence interval, and ablation. That structure keeps a local result from turning into a universal claim.
546
+
547
+ 135
548
+ 00:25:42.257 --> 00:25:56.309
549
+ The final map is simple. Demonstrations produce imitation targets. Preferences produce relative targets. Verifiable outcomes produce rewards. Privileged solutions and failure explanations produce context-conditioned teacher distributions.
550
+
551
+ 136
552
+ 00:25:57.698 --> 00:26:07.704
553
+ Evidence determines the target. The target determines the update. The update changes future behavior. Evaluation measures what improved—and what changed elsewhere.
554
+
555
+
556
+ 137
557
+ 00:26:09.104 --> 00:26:15.969
558
+ The first appendix isolates two GRPO evaluation choices that matter after the main mechanism is understood.
559
+
560
+ 138
561
+ 00:26:20.463 --> 00:26:30.741
562
+ Length normalization changes gradient weight. A fourteen-hundred-token correct answer and a one-hundred-eighty-token correct answer may receive the same reward but contribute very different numbers of token terms.
563
+
564
+ 139
565
+ 00:26:35.235 --> 00:26:43.235
566
+ Sequence averaging, token averaging, overlong penalties, and group normalization therefore change the effective objective.
567
+
568
+ 140
569
+ 00:26:47.730 --> 00:26:56.733
570
+ Pass-at-one measures reliability from one sample. Pass-at-k asks whether any of k samples succeeds and exposes remaining search diversity.
571
+
572
+ 141
573
+ 00:27:01.228 --> 00:27:08.504
574
+ A model can improve pass-at-one while collapsing onto one strategy, so report both metrics with the sampling configuration.
575
+
576
+ 142
577
+ 00:27:09.904 --> 00:27:17.994
578
+ The systems problem is storage. A large vocabulary creates one teacher logit for every token at every trajectory position.
579
+
580
+ 143
581
+ 00:27:21.415 --> 00:27:30.459
582
+ Top-k compression stores the most likely teacher logits and approximates the remaining tail. This reduces memory and bandwidth but changes the target distribution.
583
+
584
+ 144
585
+ 00:27:33.880 --> 00:27:40.604
586
+ Measure divergence from full logits on a validation sample, and count teacher-forward cost when comparing efficiency.
587
+
588
+ 145
589
+ 00:27:42.004 --> 00:27:48.547
590
+ The reported graphs belong in an appendix because they come from different models, datasets, and experimental budgets.
591
+
592
+ 146
593
+ 00:27:50.405 --> 00:27:56.226
594
+ The OPSD study reports its largest displayed aggregate gain at the smallest Qwen3 model.
595
+
596
+ 147
597
+ 00:27:58.084 --> 00:28:10.541
598
+ At one point seven billion parameters, the displayed base score is thirty-seven point one, GRPO is thirty-seven point seven, and OPSD is forty-three point four. The gaps are smaller at four and eight billion parameters.
599
+
600
+ 148
601
+ 00:28:12.398 --> 00:28:17.537
602
+ That is a hypothesis for replication, not a universal model-size law.
603
+
604
+ 149
605
+ 00:28:19.395 --> 00:28:30.124
606
+ The SDFT study emphasizes knowledge acquisition and retention. Its displayed strict aggregate moves from eighty to eighty-nine, while its out-of-distribution aggregate moves from eighty to ninety-eight.
607
+
608
+ 150
609
+ 00:28:31.982 --> 00:28:44.117
610
+ The SDPO study reports forty-one point two for GRPO and forty-eight point eight for SDPO on its code benchmark. Their bars answer different questions and should not be read as a controlled head-to-head comparison.
611
+
612
+ 151
613
+ 00:28:45.975 --> 00:28:58.612
614
+ SRPO means Sample-Routed Policy Optimization. It sends verified successes toward a GRPO-style outcome update and failures with explanations toward an SDPO-style corrective target.
615
+
616
+ 152
617
+ 00:29:00.470 --> 00:29:12.243
618
+ The useful principle is signal routing. A successful rollout already shows what worked and can support sparse selection. A failed rollout with diagnostics may reveal how to recover and can support dense correction.
619
+
620
+ 153
621
+ 00:29:14.101 --> 00:29:19.471
622
+ The router preserves that asymmetry instead of forcing every sample through one uniform loss.
623
+
@@ -0,0 +1,31 @@
1
+ # How post-training works
2
+
3
+ Post-training from first principles · Lesson 1 of 10 · 1:05
4
+
5
+ ## Learning objective
6
+
7
+ The choose, judge, and update loop behind every method in this series.
8
+
9
+ ## Using this with an LLM
10
+
11
+ Use this script as lesson source material. Preserve its distinctions and caveats when summarizing, making flash cards, proposing experiments, or answering questions. The narration is the source of truth; ask for missing experimental details instead of inventing them.
12
+
13
+ ## Visual context
14
+
15
+ OpenPond lesson intro, full chapter timestamp map, one cancellation-policy choice, useful versus misleading loss curves, and a plain-language choose-test-update loop.
16
+
17
+ ## Narration transcript
18
+
19
+ Post-training changes what a model is likely to do. An input creates a distribution over possible actions, evaluation supplies evidence, and an optimizer changes that distribution.
20
+
21
+ Here a cancellation test is failing. The policy assigns chances to return None, raise the cancellation error, or retry. Those probabilities describe the available behavior; sampling produces one actual patch.
22
+
23
+ Loss is a number saying how bad behavior looks under the training rule. Useful training lowers both training loss and held-out error. If training loss falls while held-out error rises, the model is learning the wrong thing.
24
+
25
+ The sampled patch passes. Training makes the exception-raising action slightly easier to choose in similar states. Ordinary backpropagation implements the change; reinforcement learning determines which sampled behavior receives credit.
26
+
27
+ A usable training record must also identify who generated the attempt and what evidence accompanied it.
28
+
29
+ ## Provenance
30
+
31
+ This is the production narration for the OpenPond learning series, generated from the canonical course script. Equations, diagrams, and cited paper results remain in the accompanying video and research document.