openpond 0.0.34 → 0.0.35

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (172) hide show
  1. package/dist/chunks/{app-layer-UQQNF5VZ.js → app-layer-GUF2CKYH.js} +3 -3
  2. package/dist/chunks/{apps-7NSC56K5.js → apps-DIVHAUOV.js} +3 -3
  3. package/dist/chunks/{chunk-MMDKFTJB.js → chunk-2GV3QBBZ.js} +1 -1
  4. package/dist/chunks/{chunk-BZPAW4JY.js → chunk-2I5TZW66.js} +918 -527
  5. package/dist/chunks/{chunk-TF234YN4.js → chunk-4Y25BACE.js} +1 -1
  6. package/dist/chunks/{chunk-ZRNFOHFH.js → chunk-CRFB72X2.js} +3 -3
  7. package/dist/chunks/{chunk-ZDEBKTWR.js → chunk-EY52SLOQ.js} +1 -1
  8. package/dist/chunks/{chunk-EBCLEJRO.js → chunk-FRPL5XPX.js} +1 -1
  9. package/dist/chunks/{chunk-UASXM3RR.js → chunk-GUGFYXFS.js} +4 -4
  10. package/dist/chunks/{chunk-5MM3P5KY.js → chunk-II2ERA2H.js} +2 -2
  11. package/dist/chunks/{chunk-V7SHAMUN.js → chunk-LUA5QGQP.js} +33 -33
  12. package/dist/chunks/{chunk-E3RLJ5XT.js → chunk-NTRTK6C6.js} +2 -2
  13. package/dist/chunks/{chunk-F6FFL3AA.js → chunk-RLQGMSW7.js} +2 -2
  14. package/dist/chunks/{chunk-JVTACGKD.js → chunk-S3VA47DG.js} +3 -3
  15. package/dist/chunks/{chunk-G7XOSECC.js → chunk-SECNDLPQ.js} +966 -528
  16. package/dist/chunks/{chunk-2GCNPJMD.js → chunk-TXJ4TIDO.js} +1 -1
  17. package/dist/chunks/{chunk-B5CFIWWU.js → chunk-XIF53VR3.js} +1 -1
  18. package/dist/chunks/{chunk-U4TUOBXO.js → chunk-YPK2W34Z.js} +1 -1
  19. package/dist/chunks/{chunk-HGZ3O4TS.js → chunk-ZZOTKXRD.js} +2 -2
  20. package/dist/chunks/{cli-VEDK2B42.js → cli-3AEIYSMD.js} +6 -6
  21. package/dist/chunks/{cli-4ETE5RS2.js → cli-HUVI6ELM.js} +4 -4
  22. package/dist/chunks/{core-commands-J3IQGQOL.js → core-commands-2HUTKFHQ.js} +4 -4
  23. package/dist/chunks/{extend-3F3LLDOF.js → extend-PAJMXYKV.js} +9 -9
  24. package/dist/chunks/{harness-JER7P6MJ.js → harness-UWAS2VAN.js} +3 -3
  25. package/dist/chunks/{help-75NFCYJ4.js → help-M7D2XAOX.js} +1 -1
  26. package/dist/chunks/{opchat-UT5V7MR2.js → opchat-FQ2CJVM6.js} +5 -5
  27. package/dist/chunks/{organizations-ERKRLVNQ.js → organizations-L5N6LJ7O.js} +4 -4
  28. package/dist/chunks/{profile-A3CUSDIF.js → profile-KH5IKVNY.js} +7 -7
  29. package/dist/chunks/{project-agent-FCV5HPYT.js → project-agent-F4NHVRGY.js} +6 -6
  30. package/dist/chunks/{sandbox-command-L2QPIWZ5.js → sandbox-command-DDXRQNB6.js} +4 -4
  31. package/dist/chunks/{sandbox-template-XOIWROEK.js → sandbox-template-5EEEQKR5.js} +5 -5
  32. package/dist/chunks/{src-V4FUQNHR.js → src-34FV42HE.js} +27227 -24056
  33. package/dist/chunks/{src-ENVQ7ARM.js → src-LHZWUSD7.js} +2 -2
  34. package/dist/chunks/{teams-bot-OR3ZJHQ5.js → teams-bot-HFWO5GOV.js} +3 -3
  35. package/dist/chunks/{workspaces-L73SRLPA.js → workspaces-LO44JAC2.js} +2 -2
  36. package/dist/cli.js +10 -10
  37. package/dist/web/assets/AppDialog-T3j6O3yg.js +1 -0
  38. package/dist/web/assets/{AppsView-CLjjVkjr.js → AppsView-70Qqf9eJ.js} +1 -1
  39. package/dist/web/assets/{BrowserSidebar-CkNO8723.js → BrowserSidebar-CdBLHdNW.js} +1 -1
  40. package/dist/web/assets/CloudWorkView-C5-jvlOo.js +1 -0
  41. package/dist/web/assets/{CommandMenu-DWtdkvmq.js → CommandMenu-B3Y4MvmT.js} +1 -1
  42. package/dist/web/assets/CommunityView-BHUJ9Hun.js +1 -0
  43. package/dist/web/assets/ComposerCreateImproveStrip-DeRI5XEO.js +1 -0
  44. package/dist/web/assets/GetStartedView-CIXScviw.js +1 -0
  45. package/dist/web/assets/GetStartedView-DBLdduwr.css +1 -0
  46. package/dist/web/assets/LabsRoute-Cs_R1fuu.js +3 -0
  47. package/dist/web/assets/MainChatThread-h7VSA_Rc.js +2 -0
  48. package/dist/web/assets/MakeAgentTutorialLearningPanel-r7jyZVc0.js +2 -0
  49. package/dist/web/assets/MarkdownText-E35C404I.js +7 -0
  50. package/dist/web/assets/Messages-BANZuAnA.js +10 -0
  51. package/dist/web/assets/NativeSkillSidebar-Qgccncuc.js +1 -0
  52. package/dist/web/assets/NativeSkillSidebar-T8KXjnUH.css +1 -0
  53. package/dist/web/assets/{NewProjectDialog-PrbA4fL6.js → NewProjectDialog-BowacUmA.js} +1 -1
  54. package/dist/web/assets/PostTrainingLearningPanel-oJu5qaHK.js +2 -0
  55. package/dist/web/assets/PostTrainingStatusPill-B9Eoc4s0.js +1 -0
  56. package/dist/web/assets/{ProfileSettingsSection-CqnfjiCl.css → ProfileSettingsSection-Cg4tKO-C.css} +1 -1
  57. package/dist/web/assets/ProfileSettingsSection-Dnh92gDH.js +1 -0
  58. package/dist/web/assets/RightChatPanelStack-pGwbt6Wy.js +1 -0
  59. package/dist/web/assets/SettingsView-Bjehxwt1.js +6 -0
  60. package/dist/web/assets/SettingsView-uHhQnazW.css +1 -0
  61. package/dist/web/assets/TeamChatView-Tn4X5cuY.js +3 -0
  62. package/dist/web/assets/{TerminalOverlay-DaGfEprz.js → TerminalOverlay-DoX5LYzK.js} +11 -11
  63. package/dist/web/assets/TrainingCreationPanel-BLSf1Bnp.js +1 -0
  64. package/dist/web/assets/{TrainingDraftPanel-BX5CCqlO.js → TrainingDraftPanel-Da6uGu9y.js} +1 -1
  65. package/dist/web/assets/UsageSettingsSection-DKky0p47.js +59 -0
  66. package/dist/web/assets/UsageSettingsSection-O8VK0qBL.css +1 -0
  67. package/dist/web/assets/WorkspaceDiffPanel-DTRtW9n8.js +14 -0
  68. package/dist/web/assets/WorkspaceEnvironmentMenu-Ba4M8z1N.js +32 -0
  69. package/dist/web/assets/WorkspaceGitDialogs-CSepNOb8.js +1 -0
  70. package/dist/web/assets/{WorkspaceMonacoEditor-C0fDue2d.js → WorkspaceMonacoEditor-LhFdzCRs.js} +3 -3
  71. package/dist/web/assets/activity-bW02ZsBD.js +1 -0
  72. package/dist/web/assets/{boxes-CXLqRF98.js → boxes-B1dRA1pi.js} +1 -1
  73. package/dist/web/assets/chevron-up-CySy6WNq.js +1 -0
  74. package/dist/web/assets/circle-alert-DMHz5tj_.js +1 -0
  75. package/dist/web/assets/circle-x-DY8vKAKi.js +1 -0
  76. package/dist/web/assets/copy-DuHI1UZ_.js +1 -0
  77. package/dist/web/assets/{create-pipeline-request-C26NOc7v.js → create-pipeline-request-CGBcdK9r.js} +2 -2
  78. package/dist/web/assets/credit-card-Bot0OuTl.js +1 -0
  79. package/dist/web/assets/{cssMode-DmtwKUVm.js → cssMode-BIjAothW.js} +1 -1
  80. package/dist/web/assets/git-branch-Cm02aNWq.js +1 -0
  81. package/dist/web/assets/{git-commit-horizontal-CDDZl4iA.js → git-commit-horizontal-FYILXW0v.js} +1 -1
  82. package/dist/web/assets/{htmlMode-Box0EOm1.js → htmlMode-zO2xLLgL.js} +1 -1
  83. package/dist/web/assets/index-B5AN29EO.js +177 -0
  84. package/dist/web/assets/index-BnwYyUgT.js +1 -0
  85. package/dist/web/assets/index-COGlKIJ4.css +1 -0
  86. package/dist/web/assets/{info-4Iy6BKvT.js → info-DeuEU_RO.js} +1 -1
  87. package/dist/web/assets/{jsonMode-g7t_3HFv.js → jsonMode-D2A3mj4n.js} +1 -1
  88. package/dist/web/assets/{lspLanguageFeatures-CwhFOYc9.js → lspLanguageFeatures-aPSqvscU.js} +1 -1
  89. package/dist/web/assets/make-agent-tutorial-A2W9Ajwv.js +2 -0
  90. package/dist/web/assets/{monaco.contribution-CW2ctrTm.js → monaco.contribution-CHVe9CQ9.js} +2 -2
  91. package/dist/web/assets/{monaco.contribution-DGZAePHI.js → monaco.contribution-DAq6Jihl.js} +2 -2
  92. package/dist/web/assets/{monaco.contribution-CwMt3eHi.js → monaco.contribution-Do3Jjslo.js} +2 -2
  93. package/dist/web/assets/{monaco.contribution-B-QC_ls9.js → monaco.contribution-KqCt1n9-.js} +2 -2
  94. package/dist/web/assets/{monitor-DOVjBLkr.js → monitor-DXvHWFeW.js} +1 -1
  95. package/dist/web/assets/{play-DLCuESeG.js → play-BIHyH-Yu.js} +1 -1
  96. package/dist/web/assets/{python-DRNmHcYd.js → python-g3mtmU2J.js} +1 -1
  97. package/dist/web/assets/{refresh-cw-JNUovwk4.js → refresh-cw-Daz0oqwK.js} +1 -1
  98. package/dist/web/assets/{reply-D4QvJyzE.js → reply-wu9DIBuK.js} +1 -1
  99. package/dist/web/assets/rotate-cw-D7q9GgNc.js +1 -0
  100. package/dist/web/assets/{save-BR4O4UDP.js → save-BVFY6VTI.js} +1 -1
  101. package/dist/web/assets/{toggleHighContrast-CrfdWIsV.js → toggleHighContrast-D9C_WUDu.js} +1 -1
  102. package/dist/web/assets/training-1du5QD-P.css +1 -0
  103. package/dist/web/assets/{tsMode-ZuV9go1B.js → tsMode-B4lEUN5X.js} +1 -1
  104. package/dist/web/assets/upload-DNbYB4Rl.js +1 -0
  105. package/dist/web/assets/{wifi-off-XFVqsf4c.js → wifi-off-C5QLifIJ.js} +1 -1
  106. package/dist/web/assets/{workers-CJpO1Wzj.js → workers-CV4BZb_x.js} +1 -1
  107. package/dist/web/assets/{yaml-AQ4ok5ED.js → yaml-3ah8oX43.js} +1 -1
  108. package/dist/web/courses/post-training/01-how-post-training-works-poster.webp +0 -0
  109. package/dist/web/courses/post-training/01-how-post-training-works.vtt +25 -0
  110. package/dist/web/courses/post-training/02-definitions-poster.webp +0 -0
  111. package/dist/web/courses/post-training/02-definitions.vtt +117 -0
  112. package/dist/web/courses/post-training/03-on-policy-off-policy-poster.webp +0 -0
  113. package/dist/web/courses/post-training/03-on-policy-off-policy.vtt +29 -0
  114. package/dist/web/courses/post-training/04-rewards-credit-assignment-poster.webp +0 -0
  115. package/dist/web/courses/post-training/04-rewards-credit-assignment.vtt +65 -0
  116. package/dist/web/courses/post-training/05-verifiable-rewards-rlvr-poster.webp +0 -0
  117. package/dist/web/courses/post-training/05-verifiable-rewards-rlvr.vtt +69 -0
  118. package/dist/web/courses/post-training/06-ppo-grpo-poster.webp +0 -0
  119. package/dist/web/courses/post-training/06-ppo-grpo.vtt +65 -0
  120. package/dist/web/courses/post-training/07-distillation-poster.webp +0 -0
  121. package/dist/web/courses/post-training/07-distillation.vtt +53 -0
  122. package/dist/web/courses/post-training/08-opsd-sdft-sdpo-poster.webp +0 -0
  123. package/dist/web/courses/post-training/08-opsd-sdft-sdpo.vtt +61 -0
  124. package/dist/web/courses/post-training/09-credible-experiments-poster.webp +0 -0
  125. package/dist/web/courses/post-training/09-credible-experiments.vtt +69 -0
  126. package/dist/web/courses/post-training/10-technical-appendix-poster.webp +0 -0
  127. package/dist/web/courses/post-training/10-technical-appendix.vtt +69 -0
  128. package/dist/web/courses/post-training/full-course-poster.webp +0 -0
  129. package/dist/web/courses/post-training/full-course.vtt +623 -0
  130. package/dist/web/courses/post-training/scripts/script_01.md +31 -0
  131. package/dist/web/courses/post-training/scripts/script_02.md +49 -0
  132. package/dist/web/courses/post-training/scripts/script_03.md +35 -0
  133. package/dist/web/courses/post-training/scripts/script_04.md +43 -0
  134. package/dist/web/courses/post-training/scripts/script_05.md +43 -0
  135. package/dist/web/courses/post-training/scripts/script_06.md +41 -0
  136. package/dist/web/courses/post-training/scripts/script_07.md +37 -0
  137. package/dist/web/courses/post-training/scripts/script_08.md +37 -0
  138. package/dist/web/courses/post-training/scripts/script_09.md +45 -0
  139. package/dist/web/courses/post-training/scripts/script_10.md +47 -0
  140. package/dist/web/index.html +2 -2
  141. package/dist/web/tutorials/how-to-make-an-agent-create-poster.png +0 -0
  142. package/dist/web/tutorials/how-to-make-an-agent-create.vtt +49 -0
  143. package/dist/web/tutorials/how-to-make-an-agent-improve-poster.png +0 -0
  144. package/dist/web/tutorials/how-to-make-an-agent-improve.vtt +45 -0
  145. package/dist/web/tutorials/how-to-make-an-agent-poster.png +0 -0
  146. package/dist/web/tutorials/how-to-make-an-agent-use-poster.png +0 -0
  147. package/dist/web/tutorials/how-to-make-an-agent-use.vtt +33 -0
  148. package/dist/web/tutorials/how-to-make-an-agent.vtt +125 -0
  149. package/dist/web/tutorials/what-is-an-openpond-agent-poster.png +0 -0
  150. package/dist/web/tutorials/what-is-an-openpond-agent.vtt +25 -0
  151. package/package.json +1 -1
  152. package/dist/web/assets/CloudWorkView-Cu_VqkaD.js +0 -1
  153. package/dist/web/assets/CommunityView-f03qOTER.js +0 -1
  154. package/dist/web/assets/ComposerCreateImproveStrip-CZpiQJsJ.js +0 -1
  155. package/dist/web/assets/GetStartedView-C3ZrnEz5.css +0 -1
  156. package/dist/web/assets/GetStartedView-CLMmryTG.js +0 -1
  157. package/dist/web/assets/LabsRoute-BwR2g71k.js +0 -4
  158. package/dist/web/assets/ProfileSettingsSection-CD_x_V8d.js +0 -1
  159. package/dist/web/assets/SettingsView-BMmlKvlB.js +0 -63
  160. package/dist/web/assets/SettingsView-CAPf81v3.css +0 -1
  161. package/dist/web/assets/TeamChatView-B5SlhDso.js +0 -3
  162. package/dist/web/assets/TrainingComputeDialog-BFY6E0yU.js +0 -1
  163. package/dist/web/assets/TrainingCreationPanel-Cp66gMed.js +0 -1
  164. package/dist/web/assets/WorkspaceDiffPanel-D7ZEbryF.js +0 -14
  165. package/dist/web/assets/WorkspaceEnvironmentMenu-D4DmAJaS.js +0 -32
  166. package/dist/web/assets/WorkspaceGitDialogs-DHOfJxbL.js +0 -1
  167. package/dist/web/assets/index-8H-GDlM5.css +0 -1
  168. package/dist/web/assets/index-BUyeKpY6.js +0 -1
  169. package/dist/web/assets/index-CKmte5ok.js +0 -192
  170. package/dist/web/assets/training-DeVTd0m8.css +0 -1
  171. package/dist/web/assets/useComputeSettings-DUNS6uOl.js +0 -1
  172. package/dist/web/assets/useComputeSettings-R2Sw621I.css +0 -1
@@ -0,0 +1,49 @@
1
+ # Definitions
2
+
3
+ Post-training from first principles · Lesson 2 of 10 · 6:14
4
+
5
+ ## Learning objective
6
+
7
+ Decode the notation, objectives, estimators, and acronyms used throughout modern reinforcement fine-tuning.
8
+
9
+ ## Using this with an LLM
10
+
11
+ Use this script as lesson source material. Preserve its distinctions and caveats when summarizing, making flash cards, proposing experiments, or answering questions. The narration is the source of truth; ask for missing experimental details instead of inventing them.
12
+
13
+ ## Visual context
14
+
15
+ OpenPond lesson intro, annotated policy notation, separate logits and softmax explainers, a rollout tuple, concrete reward examples, reward-to-advantage relationships, PPO and GRPO definition cards, baseline estimators, a gradient step, probability ratios and clipping, distribution metrics, and full acronym maps.
16
+
17
+ ## Narration transcript
18
+
19
+ The expression pi theta of a given state is a compact description of model behavior. Pi names the policy: a probability distribution, not one completed answer. Theta means all adjustable model parameters. The state is the context available now, and an action can be one token, one tool call, or another sampled decision.
20
+
21
+ Before sampling, the network produces raw scores called logits. A logit is not a percentage and logits do not need to add to one. They can be positive or negative. Only their relative differences matter: a larger logit means the model prefers that action compared with the alternatives available at that position.
22
+
23
+ Softmax converts logits into probabilities. It exponentiates each logit, then divides by the sum of all exponentiated logits. This makes every result positive and makes the distribution sum to one. In the example, logits three, one, and point two become probabilities of roughly eighty-four, eleven, and five percent. Temperature divides the logits before softmax. Lower temperature sharpens the distribution; higher temperature flattens it.
24
+
25
+ A rollout, also called a trajectory and written tau, is the interaction record used for learning. The behavior policy mu generated each sampled action. Its log-probability is stored beside the action so training can later compare the old behavior distribution with the current policy. Environment observations affect later states, but only the model's sampled actions receive policy-gradient terms.
26
+
27
+ Reward is the scalar number emitted by an evaluator. A math checker might return one for an exact answer and zero otherwise. A code environment might return one only when every hidden test passes. A tool task might inspect whether the final database state matches a target. The rule can be binary or graded, but the optimizer sees the number, not the evaluator's full explanation. That richer explanation is feedback.
28
+
29
+ Reward, return, and advantage are related but not interchangeable. Reward is the scalar emitted at a step or at the end. Return, G at time t, combines future rewards and can discount distant outcomes with gamma. Advantage subtracts a baseline from that return. Positive advantage means the action did better than expected; negative advantage means it did worse. PPO learns a value baseline, while GRPO estimates one from sibling responses.
30
+
31
+ PPO means Proximal Policy Optimization. It is a policy-gradient method that trains a second network, called a critic or value function, to estimate expected return. The observed return minus that expectation becomes advantage. PPO also clips the probability-ratio incentive so one sampled action cannot drive an arbitrarily large update.
32
+
33
+ GRPO means Group Relative Policy Optimization. It removes the learned critic and samples several sibling responses for the same prompt. Each response is compared with the group's reward mean and standard deviation. Better-than-group responses receive positive advantage; worse responses receive negative advantage. GRPO keeps policy ratios and clipping.
34
+
35
+ Both methods need a baseline because raw reward lacks context. PPO's baseline comes from the critic, and Generalized Advantage Estimation combines value predictions across time. GRPO's baseline comes from sibling rewards. If every sibling receives the same reward, the group standard deviation is zero and there is no relative advantage to learn from.
36
+
37
+ An objective, often written L of theta, turns the training goal into one number. The gradient is the local slope of that objective with respect to every adjustable parameter. Backpropagation computes those slopes. The optimizer applies a learning rate alpha to make a small update from theta to theta prime. In policy-gradient training, the RL method determines the credit weights inside the loss; ordinary differentiation performs the parameter update.
38
+
39
+ PPO and GRPO use a probability ratio. The numerator is the current policy's probability for the stored action. The denominator is the behavior probability recorded during rollout. A ratio above one means that action became more likely. Clipping with epsilon caps the incentive once the ratio moves too far in one update. It limits this sampled term, but does not prove that the entire model stayed close.
40
+
41
+ Entropy measures how spread out one policy distribution is. KL divergence compares two distributions and is commonly used to measure movement from a reference model or mismatch between teacher and student. Its direction matters. Cross-entropy uses target probabilities q to weight the log of student probabilities p. Distillation usually minimizes this quantity so the student assigns probability where the teacher does.
42
+
43
+ RL means reinforcement learning. RFT means reinforcement fine-tuning: applying an RL-style objective to an already pretrained model. RLVR adds verifiable rewards, such as executable tests. PPO and GRPO are policy update rules. PPO is Proximal Policy Optimization. GRPO is Group Relative Policy Optimization. GAE is Generalized Advantage Estimation. Pass-at-k asks whether at least one of k sampled attempts succeeds. On-policy and off-policy describe who generated the data.
44
+
45
+ The teacher-guided methods use dense distributions instead of only scalar outcomes. OPSD is On-Policy Self-Distillation and conditions the teacher on a trusted solution. SDFT is Self-Distillation Fine-Tuning and supplies a demonstration. SDPO is Self-Distillation Policy Optimization and supplies failure feedback. In each case the student creates the prefix and the frozen teacher receives extra evidence. SRPO, Self-Rewarding Policy Optimization, routes successful and unsuccessful samples through different signals.
46
+
47
+ ## Provenance
48
+
49
+ This is the production narration for the OpenPond learning series, generated from the canonical course script. Equations, diagrams, and cited paper results remain in the accompanying video and research document.
@@ -0,0 +1,35 @@
1
+ # On-policy and off-policy data
2
+
3
+ Post-training from first principles · Lesson 3 of 10 · 1:06
4
+
5
+ ## Learning objective
6
+
7
+ Why learner rollouts and teacher or stored data support different updates.
8
+
9
+ ## Using this with an LLM
10
+
11
+ Use this script as lesson source material. Preserve its distinctions and caveats when summarizing, making flash cards, proposing experiments, or answering questions. The narration is the source of truth; ask for missing experimental details instead of inventing them.
12
+
13
+ ## Visual context
14
+
15
+ on- and off-policy sources, a concrete rollout record, stored-data schemas, and objective routing.
16
+
17
+ ## Narration transcript
18
+
19
+ On-policy data comes from the learner being updated. Policy version twelve generates an attempt that will update version twelve.
20
+
21
+ Off-policy data comes from a teacher, human, or older checkpoint. These labels describe source, not quality.
22
+
23
+ An RL rollout stores the problem, sampled actions, generating checkpoint, behavior probabilities, environment observations, and reward.
24
+
25
+ Text alone cannot connect evaluated behavior to an update. The missing record matters as much as the final response.
26
+
27
+ Other sources carry different signals: teacher targets, chosen and rejected responses, or old actions with original probabilities and rewards.
28
+
29
+ PPO and GRPO usually score learner attempts. Offline distillation uses teacher examples. On-policy distillation labels the student's current prefix.
30
+
31
+ Inside any RL rollout, the final patch is only one part of a sequence of actions and observations.
32
+
33
+ ## Provenance
34
+
35
+ This is the production narration for the OpenPond learning series, generated from the canonical course script. Equations, diagrams, and cited paper results remain in the accompanying video and research document.
@@ -0,0 +1,43 @@
1
+ # Rewards and credit assignment
2
+
3
+ Post-training from first principles · Lesson 4 of 10 · 3:00
4
+
5
+ ## Learning objective
6
+
7
+ Follow a code-repair trajectory from actions and observations to advantage.
8
+
9
+ ## Using this with an LLM
10
+
11
+ Use this script as lesson source material. Preserve its distinctions and caveats when summarizing, making flash cards, proposing experiments, or answering questions. The narration is the source of truth; ask for missing experimental details instead of inventing them.
12
+
13
+ ## Visual context
14
+
15
+ trajectories, rollout tuples, feedback versus reward, credit assignment, discounted return, advantage, entropy, pass@k, reference KL.
16
+
17
+ ## Narration transcript
18
+
19
+ An RL rollout is a sequence, not only a final response. Credit assignment asks which decisions inside that sequence should become more or less likely.
20
+
21
+ The issue and repository form state zero. The behavior policy samples “inspect files.” File contents are an environment observation, not another sampled action. The next state contains those contents. The policy then writes raise Cancelled Error, and test output closes the trajectory.
22
+
23
+ The tuple makes those roles explicit. Tau contains states, sampled actions, behavior log-probabilities, observations, and terminal reward. Gradients apply to action positions controlled by the policy. Observations condition later actions but are masked out of the policy-gradient loss.
24
+
25
+ Reward and feedback carry different information. A patch can receive reward zero while the environment still knows that eight tests passed, cancellation failed, and the expected exception was Cancelled Error. Reward is the scalar consumed by an objective. Feedback is the richer evidence the environment produced.
26
+
27
+ A terminal score also leaves the cause ambiguous. Was the wrong file inspected? Was the timeout branch incorrect? Was “None” the decisive token? Response-level credit applies one outcome broadly. Step- or token-level signals can be denser, but density alone does not prove causal accuracy.
28
+
29
+ For multi-step problems, return carries future rewards backward. G at time t is the discounted sum of later rewards. With gamma point nine and success two steps later, the earlier action receives return point eight one. PPO critics and Generalized Advantage Estimation, or GAE, build on this idea.
30
+
31
+ Advantage then asks: better than what was expected? If observed return is point eight and the baseline is point five, advantage is positive point three. A return of point two gives negative point three.
32
+
33
+ Positive advantage raises the sampled behavior’s probability; negative advantage lowers it. PPO learns a critic for the baseline. GRPO replaces that critic with statistics from sibling responses.
34
+
35
+ Learning also changes exploration. Pass-at-one can rise while entropy and pass-at-k fall if the model collapses onto one strategy. Reliability and breadth therefore need separate metrics.
36
+
37
+ Finally, a frozen reference policy anchors behavior outside the optimized task. Kullback–Leibler divergence, shortened to KL, measures distribution shift. Too little constraint permits drift and reward exploits; too much prevents useful learning.
38
+
39
+ A test-based reward is useful only when the test measures the intended outcome reliably.
40
+
41
+ ## Provenance
42
+
43
+ This is the production narration for the OpenPond learning series, generated from the canonical course script. Equations, diagrams, and cited paper results remain in the accompanying video and research document.
@@ -0,0 +1,43 @@
1
+ # Verifiable rewards
2
+
3
+ Post-training from first principles · Lesson 5 of 10 · 2:53
4
+
5
+ ## Learning objective
6
+
7
+ See how tests create scalable rewards—and how a model can exploit the checker.
8
+
9
+ ## Using this with an LLM
10
+
11
+ Use this script as lesson source material. Preserve its distinctions and caveats when summarizing, making flash cards, proposing experiments, or answering questions. The narration is the source of truth; ask for missing experimental details instead of inventing them.
12
+
13
+ ## Visual context
14
+
15
+ the cancellation test as verifier, RLVR definition, verifier equation, executable loop, secondary applications, verifier errors, reward hacking, and hidden evaluation.
16
+
17
+ ## Narration transcript
18
+
19
+ Every reward-based update depends on whether its evaluator measures the right thing. This is the verifier problem.
20
+
21
+ RLVR means Reinforcement Learning with Verifiable Rewards. For our repair, the verifier applies the patch, runs the cancellation tests, and returns a reproducible outcome. The same idea applies to checked math answers, tool states, and agent environments.
22
+
23
+ A preference such as “patch A is cleaner” depends on a judge. A verifiable reward applies an explicit rule. Verifiable does not mean infallible; it means the same patch should receive the same result under the same evaluator state.
24
+
25
+ Write the rule as r equals V of x, y, and e. X is the public task. Y is the sampled attempt. E is evaluator-only state such as hidden tests. For a binary verifier, reward is zero or one.
26
+
27
+ This equation separates the setting from the optimizer. RLVR supplies reward. PPO, GRPO, or another policy optimizer supplies the update. DeepSeek-R1 is an important example of reasoning training built from this combination.
28
+
29
+ The full loop is prompt, policy attempt, execution, verification, reward, and policy update. Automatic checks scale because a person does not need to score every trajectory. The verifier also becomes both the task specification and an attack surface.
30
+
31
+ RLVR applies beyond short math answers. Code can compile and run tests. Tool agents can be checked against a target database state. Search can verify retrieved evidence. Scientific tasks can use simulators and constraints. It is a poor fit when quality is inherently subjective, outcomes cannot be reproduced, or the checker is easy to game.
32
+
33
+ Verifier errors have two directions. A false negative rejects a valid solution. A false positive is more dangerous under optimization because a wrong solution receives positive gradient and can become an exploit.
34
+
35
+ Suppose visible tests cover only three examples. Hard-coding those outputs and implementing the general rule both receive reward one. The optimizer cannot distinguish them. Hidden tests, randomized cases, invariants, and shadow verifiers make the measured objective closer to the intended task.
36
+
37
+ That requires an information boundary. The policy sees the public task and allowed observations. The evaluator alone sees hidden checks and anti-cheat state. Training-only teacher evidence belongs in a third protected channel.
38
+
39
+ So “RLVR with GRPO” is precise: the first term names how outcomes are labeled, and the second names how those labels become an update.
40
+
41
+ ## Provenance
42
+
43
+ This is the production narration for the OpenPond learning series, generated from the canonical course script. Equations, diagrams, and cited paper results remain in the accompanying video and research document.
@@ -0,0 +1,41 @@
1
+ # PPO and GRPO
2
+
3
+ Post-training from first principles · Lesson 6 of 10 · 2:54
4
+
5
+ ## Learning objective
6
+
7
+ Compare PPO's learned critic with GRPO's sibling-response baseline.
8
+
9
+ ## Using this with an LLM
10
+
11
+ Use this script as lesson source material. Preserve its distinctions and caveats when summarizing, making flash cards, proposing experiments, or answering questions. The narration is the source of truth; ask for missing experimental details instead of inventing them.
12
+
13
+ ## Visual context
14
+
15
+ PPO's critic on the cancellation trajectory, GRPO sibling patch attempts, worked group advantages, clipping, mixed-group probability, zero-variance groups, and direct comparison.
16
+
17
+ ## Narration transcript
18
+
19
+ A trustworthy reward still has to become a learning signal. PPO and GRPO differ mainly in how they decide whether an outcome was better or worse than expected.
20
+
21
+ PPO means Proximal Policy Optimization. For the successful inspect, patch, and test trajectory, observed return is point eight. A learned critic expected point five. Their difference, positive point three, is the advantage. PPO combines that advantage with a clipped probability ratio and updates the policy.
22
+
23
+ GRPO means Group Relative Policy Optimization. It keeps rollouts, rewards, probability ratios, and clipping, but removes the critic. Instead, it samples several patches for the same cancellation bug and compares their verifier scores.
24
+
25
+ Our six sibling patches produce rewards one, zero, zero, one, one, zero. The mean is one half and the standard deviation is one half. Normalizing gives every passing patch advantage positive one and every failing patch negative one.
26
+
27
+ That comparison is useful but coarse. Raise Timeout Error and return None both failed, so binary GRPO treats them alike even if one was closer. The group says which siblings did better; it does not explain why.
28
+
29
+ The clipping graph controls how much one sampled action can influence a single update. If its probability moves from point one zero to point one three, the ratio is one point three. With a cap at one point two, additional positive credit stops growing. Clipping limits a local incentive; it does not guarantee that the whole policy stayed close.
30
+
31
+ A group teaches only when it contains contrast. If one patch succeeds with probability p, a group of size G is mixed with probability one minus p to the G minus one minus p, all to the G. That subtracts all-success and all-failure groups.
32
+
33
+ Near zero success, almost every group fails. Near one, almost every group passes. Either case produces little relative signal. Larger groups widen the useful middle but require more patch generation and test execution.
34
+
35
+ This is why prompts need to sit near the learning frontier: difficult enough to fail, possible enough to solve. If every cancellation patch receives the same reward, normalized group advantage is zero.
36
+
37
+ Both PPO and GRPO need trajectories, rewards, behavior log-probabilities, clipped ratios, and gradients. PPO learns expected value with a critic. GRPO estimates a baseline from sibling patches. The cancellation environment stays the same; only the baseline estimator changes.
38
+
39
+ ## Provenance
40
+
41
+ This is the production narration for the OpenPond learning series, generated from the canonical course script. Equations, diagrams, and cited paper results remain in the accompanying video and research document.
@@ -0,0 +1,37 @@
1
+ # Distillation
2
+
3
+ Post-training from first principles · Lesson 7 of 10 · 2:42
4
+
5
+ ## Learning objective
6
+
7
+ Transfer a teacher's token distribution instead of copying one final answer.
8
+
9
+ ## Using this with an LLM
10
+
11
+ Use this script as lesson source material. Preserve its distinctions and caveats when summarizing, making flash cards, proposing experiments, or answering questions. The narration is the source of truth; ask for missing experimental details instead of inventing them.
12
+
13
+ ## Visual context
14
+
15
+ cancellation-token hard and soft targets, token cross-entropy, student update, teacher temperature, KL direction, prefix provenance, and privileged teacher context.
16
+
17
+ ## Narration transcript
18
+
19
+ Outcome rewards compress an attempt to a scalar. Distillation supplies richer guidance by specifying a probability distribution over plausible next tokens.
20
+
21
+ A one-hot target says “Cancelled Error” and assigns every alternative zero. A teacher distribution can say that “Cancelled Error” is most likely, “Timeout Error” is also plausible, “return” is weak, and “retry” is unlikely. That structure contains more information than the one sampled token.
22
+
23
+ Teacher probability q weights the log of student probability p over the vocabulary. Minimizing that cross-entropy moves the student toward the teacher. In our example, the student initially favors “return,” while the teacher favors “Cancelled Error.” After an update, the two distributions move closer.
24
+
25
+ Teacher temperature controls how much of that structure is visible. Low temperature makes the target almost one-hot. Higher temperature reveals alternatives in the tail; too much can magnify noise. Teacher-target temperature is separate from the rollout temperature used to sample behavior.
26
+
27
+ KL direction changes the lesson too. Forward KL penalizes the student for missing teacher-supported alternatives. Reverse KL strongly penalizes student probability where the teacher assigns little mass, often favoring a narrower mode. “We used KL” is incomplete without direction and temperature.
28
+
29
+ Prefix provenance determines which states receive teacher targets. Offline distillation uses fixed teacher trajectories. On-policy distillation lets the student reach its own strange cancellation state, then asks the teacher what should come next there.
30
+
31
+ The teacher does not need larger weights. The same frozen model can see the student's failed prefix plus extra training-only evidence: a verified repair, an expert demonstration, or a test explanation. That evidence changes the teacher distribution.
32
+
33
+ Training transfers the useful part of that privileged view into student weights. Deployment removes the evidence. A verified solution, demonstration, or failure explanation produces a different teacher target.
34
+
35
+ ## Provenance
36
+
37
+ This is the production narration for the OpenPond learning series, generated from the canonical course script. Equations, diagrams, and cited paper results remain in the accompanying video and research document.
@@ -0,0 +1,37 @@
1
+ # OPSD, SDFT, and SDPO
2
+
3
+ Post-training from first principles · Lesson 8 of 10 · 2:43
4
+
5
+ ## Learning objective
6
+
7
+ Compare trusted solutions, demonstrations, and failure feedback at one prefix.
8
+
9
+ ## Using this with an LLM
10
+
11
+ Use this script as lesson source material. Preserve its distinctions and caveats when summarizing, making flash cards, proposing experiments, or answering questions. The narration is the source of truth; ask for missing experimental details instead of inventing them.
12
+
13
+ ## Visual context
14
+
15
+ one failed cancellation patch with three teacher evidence channels, OPSD/SDFT/SDPO worked examples, evidence comparison, and shared failure modes.
16
+
17
+ ## Narration transcript
18
+
19
+ OPSD, SDFT, and SDPO share one mechanism. A frozen teacher scores the student's prefix while receiving additional evidence. The evidence source defines the method.
20
+
21
+ The student distribution is p theta given the issue and its prefix. Teacher target q scores the same next token at the same prefix, but also conditions on evidence e. Training matches q into p; deployment removes e.
22
+
23
+ On-Policy Self-Distillation, or OPSD, gives the teacher a trusted solution. Here that is the verified repair: preserve cleanup and raise Cancelled Error. The teacher uses it to guide the student's own prefix without placing the privileged patch in the deployed student's prompt. The OPSD paper evaluated this mechanism on reasoning tasks; the animation maps the information flow onto our code case.
24
+
25
+ Self-Distillation Fine-Tuning, or SDFT, gives the teacher an expert demonstration. A related repair shows the sequence inspect the flag, run cleanup, then raise the exception. Unlike offline imitation, SDFT scores the current student's return-None prefix, so the demonstration guides the state the student actually reached.
26
+
27
+ Self-Distillation Policy Optimization, or SDPO, gives the teacher feedback about the current failure. The scalar reward is zero, but the test trace says test-cancel failed, expected Cancelled Error, and return skipped cancellation. The teacher converts that explanation into dense token probabilities along the failed prefix.
28
+
29
+ The choice follows the evidence. A trusted solution supports OPSD. A demonstration supports SDFT. An explanatory failure trace supports SDPO. If all you have is a binary outcome, GRPO may be the honest baseline; inventing a generic “feedback” field would hide these different trust boundaries.
30
+
31
+ None of the dense methods is automatically safe. A privileged patch can leak. A demonstration can be irrelevant. A teacher can misread feedback and densify an error. New capability can overwrite old behavior, entropy can collapse, and teacher forward passes can erase claimed efficiency savings.
32
+
33
+ So the durable comparison is not acronym against acronym. Hold the student prefix fixed, vary the evidence, record the cost, and measure both the cancellation repair and unrelated retained capability.
34
+
35
+ ## Provenance
36
+
37
+ This is the production narration for the OpenPond learning series, generated from the canonical course script. Equations, diagrams, and cited paper results remain in the accompanying video and research document.
@@ -0,0 +1,45 @@
1
+ # Credible experiments
2
+
3
+ Post-training from first principles · Lesson 9 of 10 · 3:32
4
+
5
+ ## Learning objective
6
+
7
+ Build versioned datasets, fair baselines, and claims that survive scrutiny.
8
+
9
+ ## Using this with an LLM
10
+
11
+ Use this script as lesson source material. Preserve its distinctions and caveats when summarizing, making flash cards, proposing experiments, or answering questions. The narration is the source of truth; ask for missing experimental details instead of inventing them.
12
+
13
+ ## Visual context
14
+
15
+ Taskset boundaries, Hugging Face import, baselines, evaluation, compute, experimental campaigns, claim table, synthesis.
16
+
17
+ ## Narration transcript
18
+
19
+ A research result requires many versioned tasks, protected information boundaries, matched baselines, and held-out evaluation.
20
+
21
+ Policy-visible input includes the issue, repository revision, allowed tools, and public tests. Training-only evidence may contain an expert repair, privileged patch, or public diagnostic. Evaluator-only hidden tests, anti-exploit checks, and held-out repository clusters must remain outside both.
22
+
23
+ The Taskset preserves those boundaries together with source revision, license, split, verifier version, and content hash. Related issues from the same repository should be clustered before splitting; otherwise memorization can look like generalization.
24
+
25
+ Hugging Face support is worthwhile when it behaves as a reproducible importer. Pin the repository revision. Inspect the dataset card and license. Choose configuration and split explicitly. Map columns through a reviewed transform, log rejected rows, preserve original identifiers, and materialize an immutable snapshot. A training run should never silently follow a changing main branch.
26
+
27
+ Every proposed method needs a simple credible baseline. Begin with the base model and the most direct token-imitation objective supported by the data. Add offline distillation or preference learning when those signals are relevant. Compare with GRPO or another RLVR baseline before claiming value from OPSD, SDFT, or SDPO.
28
+
29
+ Equal optimizer tokens and equal total compute answer different questions. A distillation method may use fewer rollouts while spending additional teacher forward passes. Report both controls.
30
+
31
+ Evaluation should span four dimensions. Capability measures task success. Diversity measures pass-at-k, entropy, and distinct strategies. Retention tests old domains and distribution drift. Integrity audits reward exploits, leakage, and disagreement with a shadow verifier. A single average can hide serious regressions.
32
+
33
+ Compute accounting includes student rollout tokens, teacher-scored tokens, backward tokens, verifier time, external calls, failed jobs, memory, storage, and wall clock.
34
+
35
+ The primary study compares GRPO, SDPO, and a routed success-and-failure method on code repair while separating public diagnostics from hidden tests. Verified math can replicate the GRPO and OPSD claims with exact graders. A tool protocol can replicate token imitation and SDFT while measuring retention. The extra domains test whether a conclusion transfers; they do not replace the main story halfway through it.
36
+
37
+ The paper should bind every claim to a dataset revision, metric, run set, seed, confidence interval, and ablation. That structure keeps a local result from turning into a universal claim.
38
+
39
+ The final map is simple. Demonstrations produce imitation targets. Preferences produce relative targets. Verifiable outcomes produce rewards. Privileged solutions and failure explanations produce context-conditioned teacher distributions.
40
+
41
+ Evidence determines the target. The target determines the update. The update changes future behavior. Evaluation measures what improved—and what changed elsewhere.
42
+
43
+ ## Provenance
44
+
45
+ This is the production narration for the OpenPond learning series, generated from the canonical course script. Equations, diagrams, and cited paper results remain in the accompanying video and research document.
@@ -0,0 +1,47 @@
1
+ # Technical appendix
2
+
3
+ Post-training from first principles · Lesson 10 of 10 · 3:12
4
+
5
+ ## Learning objective
6
+
7
+ Inspect implementation choices and paper-specific results after the core mechanisms are clear.
8
+
9
+ ## Using this with an LLM
10
+
11
+ Use this script as lesson source material. Preserve its distinctions and caveats when summarizing, making flash cards, proposing experiments, or answering questions. The narration is the source of truth; ask for missing experimental details instead of inventing them.
12
+
13
+ ## Visual context
14
+
15
+ Length-normalization equations and pass-at-k curves; full versus top-k teacher-logit storage; paper-reported OPSD, SDFT, and SDPO bars; and an SRPO success/failure routing diagram.
16
+
17
+ ## Narration transcript
18
+
19
+ ### GRPO details
20
+
21
+ The first appendix isolates two GRPO evaluation choices that matter after the main mechanism is understood.
22
+
23
+ Length normalization changes gradient weight. A fourteen-hundred-token correct answer and a one-hundred-eighty-token correct answer may receive the same reward but contribute very different numbers of token terms. Sequence averaging, token averaging, overlong penalties, and group normalization therefore change the effective objective.
24
+
25
+ Pass-at-one measures reliability from one sample. Pass-at-k asks whether any of k samples succeeds and exposes remaining search diversity. A model can improve pass-at-one while collapsing onto one strategy, so report both metrics with the sampling configuration.
26
+
27
+ ### Distillation systems
28
+
29
+ The systems problem is storage. A large vocabulary creates one teacher logit for every token at every trajectory position.
30
+
31
+ Top-k compression stores the most likely teacher logits and approximates the remaining tail. This reduces memory and bandwidth but changes the target distribution. Measure divergence from full logits on a validation sample, and count teacher-forward cost when comparing efficiency.
32
+
33
+ ### Paper details and SRPO
34
+
35
+ The reported graphs belong in an appendix because they come from different models, datasets, and experimental budgets.
36
+
37
+ The OPSD study reports its largest displayed aggregate gain at the smallest Qwen3 model. At one point seven billion parameters, the displayed base score is thirty-seven point one, GRPO is thirty-seven point seven, and OPSD is forty-three point four. The gaps are smaller at four and eight billion parameters. That is a hypothesis for replication, not a universal model-size law.
38
+
39
+ The SDFT study emphasizes knowledge acquisition and retention. Its displayed strict aggregate moves from eighty to eighty-nine, while its out-of-distribution aggregate moves from eighty to ninety-eight. The SDPO study reports forty-one point two for GRPO and forty-eight point eight for SDPO on its code benchmark. Their bars answer different questions and should not be read as a controlled head-to-head comparison.
40
+
41
+ SRPO means Sample-Routed Policy Optimization. It sends verified successes toward a GRPO-style outcome update and failures with explanations toward an SDPO-style corrective target.
42
+
43
+ The useful principle is signal routing. A successful rollout already shows what worked and can support sparse selection. A failed rollout with diagnostics may reveal how to recover and can support dense correction. The router preserves that asymmetry instead of forcing every sample through one uniform loss.
44
+
45
+ ## Provenance
46
+
47
+ This is the production narration for the OpenPond learning series, generated from the canonical course script. Equations, diagrams, and cited paper results remain in the accompanying video and research document.
@@ -7,8 +7,8 @@
7
7
  <link rel="icon" type="image/png" sizes="16x16" href="./favicon-16x16.png" />
8
8
  <link rel="apple-touch-icon" href="./openpond-icon.png" />
9
9
  <title>OpenPond App</title>
10
- <script type="module" crossorigin src="./assets/index-CKmte5ok.js"></script>
11
- <link rel="stylesheet" crossorigin href="./assets/index-8H-GDlM5.css">
10
+ <script type="module" crossorigin src="./assets/index-B5AN29EO.js"></script>
11
+ <link rel="stylesheet" crossorigin href="./assets/index-COGlKIJ4.css">
12
12
  </head>
13
13
  <body>
14
14
  <div id="root"></div>
@@ -0,0 +1,49 @@
1
+ WEBVTT
2
+
3
+ C01
4
+ 00:00:03.670 --> 00:00:09.855
5
+ Open Lab, choose New agent, and start the guided creation flow.
6
+
7
+ C02
8
+ 00:00:09.575 --> 00:00:17.817
9
+ Choose From prompt to describe the Agent directly, or From chats to include examples of the work it should handle.
10
+
11
+ C04
12
+ 00:00:17.537 --> 00:00:27.159
13
+ This synthetic example starts with three chats: Acme has renewal blockers, Northstar is expanding, and the team needs a weekly account review. Add them with the Agent purpose.
14
+
15
+ C06
16
+ 00:00:26.879 --> 00:00:36.819
17
+ Review the selected excerpts as examples. OpenPond turns their recurring goals, decisions, corrections, and answer patterns into the first Agent plan.
18
+
19
+ C07
20
+ 00:00:36.539 --> 00:00:46.658
21
+ The draft converts those examples into the Agent's purpose, chat behavior, runnable actions, output formats, and Evals for your review.
22
+
23
+ C08
24
+ 00:00:46.378 --> 00:00:52.652
25
+ The right side panel asks how the Agent should prioritize conflicting account risks.
26
+
27
+ C09
28
+ 00:00:52.372 --> 00:01:00.753
29
+ Before creation, review the proposed name, chat endpoint, runnable actions, Evals, and output formats.
30
+
31
+ C09A
32
+ 00:01:00.473 --> 00:01:09.071
33
+ If the plan misses a requirement, open Edit plan and describe the change in plain language before any Agent source is created.
34
+
35
+ C09B
36
+ 00:01:08.791 --> 00:01:16.049
37
+ Review the revised plan, then click Confirm plan to create the Agent and run its Evals.
38
+
39
+ C10
40
+ 00:01:15.769 --> 00:01:24.189
41
+ This is the creation result: OpenPond built the Agent and passed its behavior Evals against the Account Health examples.
42
+
43
+ C11
44
+ 00:01:23.909 --> 00:01:31.346
45
+ After the Evals pass, the Agent is saved to your Profile and is ready to use.
46
+
47
+ C12
48
+ 00:01:31.066 --> 00:01:37.658
49
+ The Agent page shows its purpose, chat entrypoint, and runnable account actions.
@@ -0,0 +1,45 @@
1
+ WEBVTT
2
+
3
+ I00
4
+ 00:00:03.670 --> 00:00:14.365
5
+ The current Agent finds the right facts, but buries the priority after the account summary. Improve will make the billing and P1 decision lead every high-risk result.
6
+
7
+ I01
8
+ 00:00:14.085 --> 00:00:23.707
9
+ Open Improve agent on the existing Agent, then choose whether to describe the correction directly or include a chat that demonstrates it.
10
+
11
+ I02
12
+ 00:00:23.427 --> 00:00:33.853
13
+ Choose From chats, ask every high-risk response to lead with the billing and P1 decision, and attach the correction conversation that demonstrates it.
14
+
15
+ I03
16
+ 00:00:33.573 --> 00:00:43.156
17
+ Review the new lead-with-priority behavior, the existing capabilities that must stay intact, and the Evals OpenPond will rerun.
18
+
19
+ I04
20
+ 00:00:42.876 --> 00:00:49.468
21
+ Review the planned change and confirm that it preserves citations and every existing action.
22
+
23
+ I05
24
+ 00:00:49.188 --> 00:00:58.900
25
+ The same locked Evals make the decision clear: the current Agent misses the new lead sentence, while the updated Agent passes every original and new Eval.
26
+
27
+ I06
28
+ 00:00:58.620 --> 00:01:07.666
29
+ Apply the update only after that comparison. The existing Agent updates in place, with the successful Evals recorded.
30
+
31
+ I07
32
+ 00:01:07.386 --> 00:01:16.879
33
+ The improved chat now places disputed billing and the open P1 case ahead of adoption decline while preserving source citations.
34
+
35
+ I08
36
+ 00:01:16.599 --> 00:01:25.069
37
+ The direct triage action repeats the billing-first result, proving the change works outside chat without breaking the existing workflow.
38
+
39
+ I09
40
+ 00:01:24.789 --> 00:01:32.940
41
+ Commit the final Agent source with a clear Git message so the exact Profile version remains reviewable and shareable.
42
+
43
+ I10
44
+ 00:01:32.660 --> 00:01:45.898
45
+ Open Sync to review the committed Profile changes. Confirm sync publishes this version so teammates can use the Agent in chat and hosted sandboxes can attach the same source.
@@ -0,0 +1,33 @@
1
+ WEBVTT
2
+
3
+ C13Q
4
+ 00:00:03.670 --> 00:00:12.050
5
+ Use opens a normal chat with the Agent attached. Type a natural-language question about Acme before sending it.
6
+
7
+ C13
8
+ 00:00:11.770 --> 00:00:21.532
9
+ Acme is the at-risk example: the Agent answers in chat with renewal timing, usage decline, the disputed invoice, and the open P1 case.
10
+
11
+ C14
12
+ 00:00:21.252 --> 00:00:29.850
13
+ A follow-up keeps Acme context and assigns the first actions to Revenue Operations and Support instead of restarting the conversation.
14
+
15
+ C15
16
+ 00:00:29.570 --> 00:00:38.208
17
+ The action picker offers chat plus each runnable account action, so you can choose conversation or a focused task.
18
+
19
+ C16
20
+ 00:00:37.928 --> 00:00:48.493
21
+ Northstar is the healthy expansion example from the source chats. Its direct summary identifies rising usage and the request for twenty-five more seats.
22
+
23
+ C17
24
+ 00:00:48.213 --> 00:00:56.772
25
+ Running renewal triage for Acme returns high risk, cites the billing and support blockers, and orders the next steps.
26
+
27
+ C18
28
+ 00:00:56.492 --> 00:01:05.309
29
+ The weekly review action combines the portfolio and returns visible Markdown, CSV, and JSON artifacts for downstream use.
30
+
31
+ C19
32
+ 00:01:05.029 --> 00:01:12.337
33
+ After restarting OpenPond, the same Agent, chat history, and actions remain available.