pulseml 0.2.0__tar.gz → 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pulseml
3
- Version: 0.2.0
3
+ Version: 0.2.2
4
4
  Summary: A live ML training debugger - GUI or CLI, any backend (NumPy, PyTorch, TensorFlow, CuPy, JAX).
5
5
  Author-email: Yash Patel <codeyash09@gmail.com>
6
6
  License: Proprietary
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "pulseml"
7
- version = "0.2.0"
7
+ version = "0.2.2"
8
8
  description = "A live ML training debugger - GUI or CLI, any backend (NumPy, PyTorch, TensorFlow, CuPy, JAX)."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.9"
@@ -3373,27 +3373,58 @@ def auto_track(train_fn=None, throttle_interval=1.0, code_text=None, project_roo
3373
3373
  # macOS/Linux: Resets raw terminal modes and clears the screen
3374
3374
  subprocess.run('stty sane', shell=True)
3375
3375
  subprocess.run('clear', shell=True)
3376
- # Replace this part in your RESTART block:
3377
- try:
3378
- restart_result = subprocess.run(
3379
- [sys.executable, script_path] + sys.argv[1:],
3380
- capture_output=True,
3381
- text=True
3376
+ # Retry the restart itself instead of falling back to
3377
+ # "keep running the old, already-in-memory process" on
3378
+ # a bad exit code. A nonzero exit almost always means
3379
+ # the replacement process crashed immediately (the
3380
+ # agent's fix didn't fully take, or introduced a new
3381
+ # bug) -- silently resuming the OLD code just means
3382
+ # the same already-broken run limps along unsupervised
3383
+ # with nobody watching. Retrying gives a transient
3384
+ # cause (a file/port briefly held by the exiting old
3385
+ # process, a flaky import) a real chance to clear.
3386
+ # Capped, not infinite, so a deterministically-broken
3387
+ # script can't spin forever burning compute unattended.
3388
+ MAX_RESTART_ATTEMPTS = 5
3389
+ RETRY_BACKOFF_SECONDS = 3
3390
+
3391
+ restart_argv = [sys.executable, script_path] + sys.argv[1:]
3392
+ attempt = 0
3393
+ restarted_ok = False
3394
+ while attempt < MAX_RESTART_ATTEMPTS:
3395
+ attempt += 1
3396
+ try:
3397
+ restart_result = subprocess.run(
3398
+ restart_argv,
3399
+ capture_output=True,
3400
+ text=True,
3401
+ )
3402
+ except Exception as exc:
3403
+ print(f"[PULSE] Restart attempt {attempt}/{MAX_RESTART_ATTEMPTS} failed to launch ({exc}).")
3404
+ if attempt < MAX_RESTART_ATTEMPTS:
3405
+ time.sleep(min(RETRY_BACKOFF_SECONDS * attempt, 30))
3406
+ continue
3407
+
3408
+ if restart_result.returncode == 0:
3409
+ restarted_ok = True
3410
+ break
3411
+
3412
+ print(
3413
+ f"\n[PULSE] Replacement training process exited with code "
3414
+ f"{restart_result.returncode} (attempt {attempt}/{MAX_RESTART_ATTEMPTS})."
3382
3415
  )
3383
- except Exception as exc:
3384
- print(f"[PULSE] Restart failed ({exc}); continuing the current training process.")
3385
- continue
3386
-
3387
- if restart_result.returncode == 0:
3416
+ if restart_result.stderr:
3417
+ print("[PULSE] STDERR from failed run:\n", restart_result.stderr)
3418
+ if restart_result.stdout:
3419
+ print("[PULSE] STDOUT from failed run:\n", restart_result.stdout)
3420
+ if attempt < MAX_RESTART_ATTEMPTS:
3421
+ print("[PULSE] Retrying restart...")
3422
+ time.sleep(min(RETRY_BACKOFF_SECONDS * attempt, 30))
3423
+
3424
+ if restarted_ok:
3388
3425
  sys.exit(0)
3389
-
3390
- # Print the captured errors so you can see why it exited with code 1!
3391
- print(f"\n[PULSE] Replacement training process exited with code {restart_result.returncode}.")
3392
- if restart_result.stderr:
3393
- print("[PULSE] STDERR from failed run:\n", restart_result.stderr)
3394
- if restart_result.stdout:
3395
- print("[PULSE] STDOUT from failed run:\n", restart_result.stdout)
3396
- print("[PULSE] Continuing the current process.")
3426
+
3427
+ print(f"[PULSE] ⚠ Giving up after {MAX_RESTART_ATTEMPTS} restart attempts -- continuing the current process.")
3397
3428
  # bad (NaN/inf, a loss spike) and asked training to hold here until
3398
3429
  # the user hits "Resume Training" -- or a fix gets applied and they
3399
3430
  # resume manually. Blocks this exact line from executing further,
@@ -3510,6 +3541,24 @@ def _install_cli_excepthook(cli):
3510
3541
  previous_hook = sys.excepthook
3511
3542
 
3512
3543
  def _hook(exc_type, exc_value, exc_tb):
3544
+ # Disarm the global line/call tracer immediately. Once the script
3545
+ # has crashed, nothing below should still be traced -- and if we
3546
+ # leave `sys.settrace` pointed at `cli_tracer`, it stays the
3547
+ # process-wide trace function into interpreter shutdown, where it
3548
+ # gets invoked on unrelated __del__ calls (e.g. litellm's
3549
+ # AsyncHTTPHandler) after module globals have already been cleared
3550
+ # to None. That produces a second, uncatchable "Exception ignored
3551
+ # in ..." failure from *inside* the tracer, separate from (and
3552
+ # after) whatever we report here.
3553
+ sys.settrace(None)
3554
+ try:
3555
+ frame = exc_tb.tb_frame if exc_tb else None
3556
+ while frame is not None:
3557
+ frame.f_trace = None
3558
+ frame = frame.f_back
3559
+ except Exception:
3560
+ pass
3561
+
3513
3562
  previous_hook(exc_type, exc_value, exc_tb)
3514
3563
  if issubclass(exc_type, KeyboardInterrupt):
3515
3564
  return
@@ -3603,14 +3652,29 @@ def _start_cli_tracker(
3603
3652
  cli.extra_files = dict(extra_files or {})
3604
3653
  cli.pending_startup_error = startup_error
3605
3654
  cli.print_banner()
3606
- _install_cli_excepthook(cli)
3655
+ # Agent/provider setup used to be deferred until cli_tracer saw its
3656
+ # first 'line' event with a trackable/resolved local in scope -- fine
3657
+ # for a training loop, but it meant any crash before that point (e.g.
3658
+ # failing during data loading, before the loop even starts) left
3659
+ # cli.agent_provider unset, so the crash hook below had no agent to
3660
+ # hand the traceback to and could only print "no AI agent is
3661
+ # configured". discover_variables() already falls back to the
3662
+ # statically-discovered `cli.discovered` names (populated from AST
3663
+ # parsing at auto_track()-call time, before any user code runs) when
3664
+ # there are no live locals yet, so setup doesn't actually need to wait
3665
+ # for a live frame -- run it eagerly, before the excepthook can matter.
3666
+ cli.interactive_setup()
3607
3667
 
3608
3668
  if startup_error:
3609
3669
  print("[Pulse] The function passed to auto_track() raised an exception during its dry run:")
3610
3670
  print(startup_error)
3611
- print("[Pulse] Set up an AI agent below and Pulse will offer to diagnose/fix it before continuing.\n")
3671
+ if cli.agent_provider:
3672
+ print("[Pulse] Pulse will offer to diagnose/fix it (via the agent set up above) before continuing.\n")
3673
+ else:
3674
+ print("[Pulse] No agent configured (see summary above) -- run /agent to set one up before continuing.\n")
3675
+
3676
+ _install_cli_excepthook(cli)
3612
3677
 
3613
- setup_done = {"value": False}
3614
3678
  last_logged = {"t": 0.0}
3615
3679
 
3616
3680
  # ------------------------------------------------------------------
@@ -3641,7 +3705,7 @@ def _start_cli_tracker(
3641
3705
  # next window. Between windows, the training loop runs at native,
3642
3706
  # untraced speed.
3643
3707
  _CAPTURE_SPAN = min(0.05, throttle_interval / 4 if throttle_interval > 0 else 0.05)
3644
- window = {"open": True, "closes_at": 0.0} # start open so setup can run immediately
3708
+ window = {"open": True, "closes_at": 0.0} # start open so the first snapshot can run immediately
3645
3709
 
3646
3710
  def _close_window():
3647
3711
  window["open"] = False
@@ -3663,6 +3727,11 @@ def _start_cli_tracker(
3663
3727
  if window["open"]:
3664
3728
  _close_window()
3665
3729
 
3730
+ # Setup now runs eagerly, before tracing even starts (see above), so
3731
+ # the ticker can start right away too -- no need to wait for the
3732
+ # tracer to see a live frame first.
3733
+ threading.Thread(target=_ticker, daemon=True).start()
3734
+
3666
3735
  def cli_tracer(frame, event, arg):
3667
3736
  filename = frame.f_code.co_filename
3668
3737
  if _is_library_frame(filename):
@@ -3709,21 +3778,6 @@ def _start_cli_tracker(
3709
3778
  ):
3710
3779
  cli.var_states[name] = "track"
3711
3780
 
3712
- if not setup_done["value"]:
3713
- has_resolved_shape = any(shape is not None for shape in cli.discovered.values())
3714
- has_trackable_local = any(
3715
- not n.startswith("__") and is_trackable(v) for n, v in local_vars.items()
3716
- )
3717
- if has_resolved_shape or has_trackable_local:
3718
- cli.watch_locals = local_vars
3719
- cli.interactive_setup()
3720
- setup_done["value"] = True
3721
- # Setup is interactive/blocking; once it's done, start the
3722
- # background ticker that opens/closes future windows. (Kept
3723
- # off until now so setup itself isn't racing a window close.)
3724
- threading.Thread(target=_ticker, daemon=True).start()
3725
- return cli_tracer
3726
-
3727
3781
  cli.watch_locals = local_vars
3728
3782
 
3729
3783
  now = time.time()
@@ -184,6 +184,68 @@ _YELLOW = "\033[93m"
184
184
  _PULSE_RE = re.compile(r"Pulse(?:\s(?:CLI|AI))?")
185
185
 
186
186
 
187
+ # ----------------------------------------------------------------------------
188
+ # Terms of Service -- the actual license Pulse ships under (mirrors the
189
+ # LICENSE file in the Pulse GitHub repo). Shown in full at sign-up (see
190
+ # _prompt_tos_acceptance) and its acceptance is what cloud.record_tos_
191
+ # acceptance timestamps server-side. PULSE_TOS_URL, if set, is shown
192
+ # alongside this as a link to the canonical hosted copy -- update that env
193
+ # var (and this text, if the license is ever revised) together so the two
194
+ # never drift apart.
195
+ # ----------------------------------------------------------------------------
196
+ PULSE_LICENSE_TEXT = """\
197
+ PULSE PROPRIETARY SOFTWARE LICENSE
198
+
199
+ Copyright (c) 2026 Yash Patel. All Rights Reserved.
200
+
201
+ This software and associated files (the "Software") are the proprietary
202
+ property of the copyright holder. The Software is licensed, not sold.
203
+
204
+ 1. GRANT OF LICENSE
205
+ Subject to the terms of this license, the copyright holder grants you a
206
+ limited, non-exclusive, non-transferable, revocable license to install
207
+ and run the Software for your own internal use.
208
+
209
+ 2. RESTRICTIONS
210
+ Except as expressly permitted above, you may NOT, and may not permit
211
+ others to:
212
+ (a) copy, reproduce, distribute, sublicense, rent, lease, or resell the
213
+ Software;
214
+ (b) modify, adapt, translate, or create derivative works based on the
215
+ Software;
216
+ (c) reverse engineer, decompile, disassemble, or otherwise attempt to
217
+ derive the source code of the Software, except to the extent this
218
+ restriction is prohibited by applicable law;
219
+ (d) remove, obscure, or alter any proprietary notices on the Software;
220
+ (e) use the Software to build a competing product or service.
221
+
222
+ 3. FUTURE PAID TERMS
223
+ The copyright holder reserves the right to change pricing, introduce
224
+ paid tiers or license keys, limit functionality, or discontinue free
225
+ distribution of future versions at any time. Continued use of any
226
+ future version may be subject to additional terms, including payment.
227
+
228
+ 4. NO WARRANTY
229
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS
230
+ OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
231
+ MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE, AND
232
+ NONINFRINGEMENT.
233
+
234
+ 5. LIMITATION OF LIABILITY
235
+ IN NO EVENT SHALL THE COPYRIGHT HOLDER BE LIABLE FOR ANY CLAIM,
236
+ DAMAGES, OR OTHER LIABILITY ARISING FROM, OUT OF, OR IN CONNECTION
237
+ WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
238
+
239
+ 6. TERMINATION
240
+ This license is effective until terminated. It will terminate
241
+ automatically without notice if you fail to comply with any of its
242
+ terms. Upon termination, you must cease all use of the Software and
243
+ destroy all copies in your possession.
244
+
245
+ For licensing inquiries, contact: codeyash09@gmail.com
246
+ """
247
+
248
+
187
249
  def _highlight_pulse(text: str, base_color: Optional[str] = None) -> str:
188
250
  """Color every occurrence of 'Pulse' (and 'Pulse CLI'/'Pulse AI') orange,
189
251
  resuming `base_color` afterward so nesting inside a red/blue line works."""
@@ -297,7 +359,16 @@ SYSTEM_PROMPT = (
297
359
  "of training, adjust the dial so future auto-intervention checks match reality instead of "
298
360
  "either missing real trouble or crying wolf on normal noise. Only send this when you have an "
299
361
  "actual reason from the data, not by default.\n"
300
- " Put CALC:/PROMOTE:/GPUTRACK:/GPUUNTRACK:/SENSITIVITY: lines anywhere in your Reasoning, not in the "
362
+ " NORMAL_START: <comma-separated var=value pairs, or 'none'>\n"
363
+ " Spike detection normally needs several real data points before it has a baseline to "
364
+ "compare against, which means a genuine explosion in the first few steps of training can go "
365
+ "undetected. If you can estimate a loss-like tracked variable's expected value at the very "
366
+ "start of training from the code alone (e.g. a randomly-initialized N-class classifier's "
367
+ "cross-entropy loss starts near ln(N); a policy's initial reward is often near a known "
368
+ "random-policy baseline), send it here so Pulse can catch a real explosion from step one "
369
+ "instead of waiting for history to accumulate. Only estimate what you can actually justify "
370
+ "from the code -- omit a variable (or send 'none') rather than guess.\n"
371
+ " Put CALC:/PROMOTE:/GPUTRACK:/GPUUNTRACK:/SENSITIVITY:/NORMAL_START: lines anywhere in your Reasoning, not in the "
301
372
  "Diagnosis or Fix.\n\n"
302
373
  "PERIODIC GPU CHECK-IN:\n"
303
374
  "Every 15 minutes after Pulse starts, if you have an AI provider configured, Pulse will send "
@@ -653,8 +724,20 @@ class PulseCLI:
653
724
  # one signal instead of moving the whole dial.
654
725
  self.explosion_multiplier: Optional[float] = None # loss spike: latest > baseline * N
655
726
  self.plateau_range_frac: Optional[float] = None # plateau: window range < frac * |latest|
656
- self.oscillation_flip_threshold: Optional[int] = None # oscillation: sign flips required (of 19)
657
- self.oscillation_delta_frac: Optional[float] = None # oscillation: avg |delta| > frac * |latest|
727
+ self.oscillation_flip_threshold: Optional[int] = None # oscillation: sign flips required (of 18)
728
+ self.oscillation_delta_frac: Optional[float] = None # oscillation: avg |delta| > frac * scale
729
+ self.stagnation_frac: Optional[float] = None # stagnation: long-window improvement < frac * |early mean|
730
+ # Agent-estimated (from reading the code, via a NORMAL_START:
731
+ # directive -- see _apply_directives) expected starting value for
732
+ # a loss-like tracked variable. Explosion detection normally needs
733
+ # several real finite data points before it has anything to
734
+ # compare `latest` against, which means a genuine blow-up in the
735
+ # first few steps of training can slip through undetected -- this
736
+ # gives it a code-derived anchor to compare against from step one,
737
+ # before real history exists. See _prime_at_start, which asks the
738
+ # agent for this automatically, once, before the first step.
739
+ self._normal_start_baselines: Dict[str, float] = {}
740
+ self._start_primed: bool = False # _prime_at_start runs at most once per process
658
741
  self._last_intervention_signature: Optional[str] = None
659
742
  # /code is on by default -- every manually-asked question includes
660
743
  # the training code (and any cross-file context) unless turned off.
@@ -1480,6 +1563,39 @@ class PulseCLI:
1480
1563
  # deliberately NOT marked dirty here -- it hasn't changed
1481
1564
  # (see above), so there's nothing to re-send.
1482
1565
  self._cloud_dirty_fields.update({"uptime_seconds", "downtime_seconds"})
1566
+ # Preload this session's already-synced history so THIS
1567
+ # process's flushes extend it instead of quietly replacing
1568
+ # it. self._agent_logs/_error_tracebacks/_telemetry/
1569
+ # _incidents all start empty in __init__ -- with nothing
1570
+ # more done, the first flush from here would PATCH the
1571
+ # whole array column (see _build_cloud_patch_body) down to
1572
+ # just whatever this process adds, erasing every entry
1573
+ # logged before the restart even though it's still sitting
1574
+ # in local memory of a process that's gone. Best-effort:
1575
+ # if this fetch fails (offline blip, etc.) we still carry
1576
+ # on -- worst case is the pre-restart history is briefly
1577
+ # at risk on the next flush rather than the run refusing
1578
+ # to continue.
1579
+ try:
1580
+ existing = cloud.fetch_debug_session(self.debug_session_id)
1581
+ except cloud.SupabaseError:
1582
+ existing = None
1583
+ if existing:
1584
+ self._agent_logs = cloud.decode_entries(existing.get("agent_logs"))
1585
+ self._error_tracebacks = cloud.decode_entries(existing.get("error_tracebacks"))
1586
+ self._telemetry = cloud.decode_entries(existing.get("telemetry"))
1587
+ self._incidents = cloud.decode_entries(existing.get("incidents"))
1588
+ cprint(
1589
+ f"[Pulse] Loaded {len(self._agent_logs)} agent log(s), {len(self._incidents)} "
1590
+ f"incident(s) from before the restart -- history preserved."
1591
+ )
1592
+ else:
1593
+ cprint(
1594
+ "[Pulse] ⚠ Could not load this session's history before the restart -- "
1595
+ "new entries will be appended locally, but the next sync may not include "
1596
+ "everything logged before the restart.",
1597
+ color=_YELLOW,
1598
+ )
1483
1599
  cprint(f"[Pulse] Resumed git commit {sha[:10]}… -- carried over, not re-detected (see /commit to update it manually).")
1484
1600
  cprint(f"[Pulse] Resumed debug session (id={self.debug_session_id[:8]}…) after restart -- uptime/downtime counters carried over.")
1485
1601
  else:
@@ -1608,6 +1724,17 @@ class PulseCLI:
1608
1724
  self.email = env_user
1609
1725
  token = cloud.attach_session_token(self.user_id)
1610
1726
  cloud.save_cached_credentials(self.user_id, self.email, self.team_id, session_token=token)
1727
+ except cloud.SupabaseEmailConfirmationRequired:
1728
+ # Account created, but this deployment requires email
1729
+ # confirmation before a session can be issued -- an
1730
+ # unattended job can't click that link, so there's
1731
+ # nothing more to do here this run. Not a "login
1732
+ # failed" -- just fall through to local (no-cloud) mode
1733
+ # for this run and let a human confirm and re-run.
1734
+ cprint(
1735
+ f"[Pulse] Confirm your account: check {env_user}'s email for a confirmation link, "
1736
+ "then log in. Continuing this run in local (no-cloud) mode.",
1737
+ )
1611
1738
  except cloud.SupabaseError as exc:
1612
1739
  cprint(f"[Pulse] ⚠ Non-interactive login failed ({exc}). Continuing in local (no-cloud) mode.", color=_RED)
1613
1740
  else:
@@ -1624,31 +1751,39 @@ class PulseCLI:
1624
1751
  _flush_stdin()
1625
1752
  cprint("\n--- Pulse Cloud Authentication ---")
1626
1753
  resp = input(
1627
- "[Pulse] Select an option: (s)ign up / (l)og in / (r)ecover account > "
1628
- ).strip().lower()
1754
+ "[Pulse] (s)ign up [default] / (l)og in / (r)ecover account > "
1755
+ ).strip().lower() or "s"
1629
1756
 
1630
1757
  if resp in ("s", "signup", "sign up"):
1631
1758
  _flush_stdin()
1632
- email = input("email (>=8 chars: letters, numbers, _, -) > ").strip()
1633
- if not email:
1634
- cprint("[Pulse] email cannot be blank.", color=_RED)
1635
- continue
1636
- if not cloud.is_valid_email(email):
1637
- cprint("[Pulse] emails must be >=8 characters: letters, numbers, underscores, and hyphens only.", color=_RED)
1638
- continue
1639
- password = getpass.getpass("Password (min. 8 characters) > ")
1640
- if len(password) < 8:
1641
- cprint("[Pulse] Password must be at least 8 characters.", color=_RED)
1642
- continue
1643
- confirm_password = getpass.getpass("Confirm password > ")
1644
- if password != confirm_password:
1645
- # A typo here with no confirmation step means signing up
1646
- # with a password the user doesn't actually know -- and
1647
- # with no email on file yet at this point, there'd be no
1648
- # way back in. Re-prompt from the top of sign-up rather
1649
- # than silently keeping whichever one was typed first.
1650
- cprint("[Pulse] Passwords didn't match. Let's try again.", color=_RED)
1651
- continue
1759
+ # Retry only the field that actually failed (email or
1760
+ # password) instead of bouncing back to the top-level
1761
+ # "sign up / log in / recover" menu on every typo -- that
1762
+ # used to mean a single password mismatch cost you a full
1763
+ # re-type of the email address too.
1764
+ while True:
1765
+ email = input("email (e.g. name@example.com, min. 8 characters) > ").strip()
1766
+ if not email:
1767
+ cprint("[Pulse] email cannot be blank.", color=_RED)
1768
+ continue
1769
+ if not cloud.is_valid_email(email):
1770
+ cprint("[Pulse] enter a valid email address (min. 8 characters), e.g. name@example.com.", color=_RED)
1771
+ continue
1772
+ break
1773
+ while True:
1774
+ password = getpass.getpass("Password (min. 8 characters) > ")
1775
+ if len(password) < 8:
1776
+ cprint("[Pulse] Password must be at least 8 characters.", color=_RED)
1777
+ continue
1778
+ confirm_password = getpass.getpass("Confirm password > ")
1779
+ if password != confirm_password:
1780
+ # A typo here with no confirmation step means signing up
1781
+ # with a password the user doesn't actually know -- and
1782
+ # with no email on file yet at this point, there'd be no
1783
+ # way back in. Re-prompt for password only, not email.
1784
+ cprint("[Pulse] Passwords didn't match. Let's try again.", color=_RED)
1785
+ continue
1786
+ break
1652
1787
  accepted = self._prompt_tos_acceptance()
1653
1788
  if not accepted:
1654
1789
  cprint("[Pulse] Sign up cancelled -- acceptance is required to create an account.")
@@ -1664,6 +1799,38 @@ class PulseCLI:
1664
1799
  cloud.save_cached_credentials(self.user_id, self.email, self.team_id, session_token=token)
1665
1800
  self._show_recovery_code(cloud.attach_recovery_code(self.user_id))
1666
1801
  return
1802
+ except cloud.SupabaseEmailConfirmationRequired:
1803
+ # The account WAS created -- this isn't a failure. The
1804
+ # OLD behavior here just printed a message and fell
1805
+ # back to the top-level menu, which meant: leave the
1806
+ # CLI, go confirm in email, come back, remember to
1807
+ # pick "log in" instead of "sign up", and retype the
1808
+ # email+password you just typed 10 seconds ago. That's
1809
+ # exactly the kind of drop-off point that loses a
1810
+ # just-converted cold-email lead. Instead, stay in
1811
+ # this flow and retry with the SAME credentials
1812
+ # already in memory -- the only thing left to do is
1813
+ # click the link and hit Enter.
1814
+ cprint("[Pulse] Almost there -- check your email for a confirmation link and click it.")
1815
+ while True:
1816
+ _flush_stdin()
1817
+ resp2 = input(
1818
+ "Press Enter once confirmed to continue (or type 'skip' to do this later) > "
1819
+ ).strip().lower()
1820
+ if resp2 == "skip":
1821
+ cprint("[Pulse] No problem -- run Pulse again and choose (l)og in once you've confirmed.")
1822
+ break
1823
+ try:
1824
+ user = cloud.log_in(email, password)
1825
+ except cloud.SupabaseError as exc2:
1826
+ cprint(f"[Pulse] Not confirmed yet ({exc2}). Check your email and try again.", color=_YELLOW)
1827
+ continue
1828
+ cprint(f"[Pulse] ✓ Confirmed! Signed in as {email}.")
1829
+ self.user_id = user["id"]
1830
+ self.email = email
1831
+ token = cloud.attach_session_token(self.user_id)
1832
+ cloud.save_cached_credentials(self.user_id, self.email, self.team_id, session_token=token)
1833
+ return
1667
1834
  except cloud.SupabaseError as exc:
1668
1835
  cprint(f"[Pulse] ⚠ Sign up failed: {exc}", color=_RED)
1669
1836
 
@@ -1735,19 +1902,43 @@ class PulseCLI:
1735
1902
  self.email = None
1736
1903
 
1737
1904
  def _prompt_tos_acceptance(self) -> bool:
1738
- """Legal note for whoever's deploying this: the URLs below are
1739
- placeholders -- point PULSE_TOS_URL/PULSE_PRIVACY_URL at your
1740
- actual Terms of Service and Privacy Policy (drafted by an actual
1741
- lawyer, not by Pulse) before relying on this for anything. This
1742
- function only builds the ACCEPTANCE MECHANISM (the prompt, and
1743
- recording a timestamp -- see cloud.record_tos_acceptance) that a
1744
- real ToS/Privacy Policy needs; it doesn't write the legal text
1745
- itself."""
1746
- tos_url = os.environ.get("PULSE_TOS_URL", "(set PULSE_TOS_URL to link your Terms of Service here)")
1905
+ """Show a short summary of the Pulse license up front (full text
1906
+ is PULSE_LICENSE_TEXT, mirrored from the LICENSE file in the
1907
+ Pulse GitHub repo, and is always one keystroke away via 'full')
1908
+ and require typing 'yes' to accept before an account is created.
1909
+
1910
+ The full ~50-line license used to be printed unconditionally
1911
+ before every sign-up -- legally fine, but it's exactly the kind
1912
+ of wall of text that makes someone who just clicked through from
1913
+ a cold email bail before ever seeing the product. A short,
1914
+ scannable summary gets the same acceptance (still gated on
1915
+ actually typing 'yes', still timestamped server-side by
1916
+ cloud.record_tos_acceptance right after this returns True) without
1917
+ that drop-off risk, while anyone who wants the full text still
1918
+ gets it by typing 'full'.
1919
+
1920
+ PULSE_PRIVACY_URL, if set, is shown alongside it for a separate
1921
+ privacy policy (this license covers software use, not data
1922
+ handling) -- point that env var at your actual Privacy Policy
1923
+ before relying on this for anything.
1924
+ """
1747
1925
  privacy_url = os.environ.get("PULSE_PRIVACY_URL", "(set PULSE_PRIVACY_URL to link your Privacy Policy here)")
1748
- print(f"\nTerms of Service: {tos_url}\nPrivacy Policy: {privacy_url}")
1926
+ tos_url = os.environ.get("PULSE_TOS_URL")
1927
+ print(
1928
+ "\nQuick terms before you create an account:\n"
1929
+ " • Pulse is proprietary software, licensed (not sold) for your own use.\n"
1930
+ " • No redistributing, reselling, or reverse-engineering it, and no using it to build a competing product.\n"
1931
+ " • Provided as-is, no warranty -- pricing/paid tiers may be introduced for future versions.\n"
1932
+ )
1933
+ if tos_url:
1934
+ print(f"Full license: {tos_url}")
1935
+ print(f"Privacy Policy: {privacy_url}\n")
1749
1936
  _flush_stdin()
1750
- resp = input("Type 'yes' to accept and create your account > ").strip().lower()
1937
+ resp = input("Type 'yes' to accept (or 'full' to read the complete license first) > ").strip().lower()
1938
+ if resp == "full":
1939
+ print(f"\n{PULSE_LICENSE_TEXT}")
1940
+ _flush_stdin()
1941
+ resp = input("Type 'yes' to accept and create your account > ").strip().lower()
1751
1942
  return resp == "yes"
1752
1943
 
1753
1944
  def _show_recovery_code(self, code: Optional[str]) -> None:
@@ -1878,6 +2069,28 @@ class PulseCLI:
1878
2069
  cprint("[Pulse] Non-interactive mode and no unambiguous workspace to pick -- continuing without a team.")
1879
2070
  return
1880
2071
 
2072
+ # A brand-new sign-up (or anyone who currently belongs to zero
2073
+ # workspaces) has nothing to actually pick between here -- the
2074
+ # old behavior forced everyone through this menu regardless, with
2075
+ # no Enter-key default, meaning a just-signed-up trial user's very
2076
+ # next required action was "type c, then optionally type a GitHub
2077
+ # repo URL" before they could do anything else. Skip straight to
2078
+ # a silently auto-created personal workspace instead; joining a
2079
+ # teammate's workspace with a code is one command away (/repo can
2080
+ # set a repo on it later, too), but creating your own shouldn't
2081
+ # need a prompt at all when there's nothing else it could be.
2082
+ if not existing:
2083
+ try:
2084
+ team = cloud.create_team(self.user_id, repo=cloud.git_remote_url(self._repo_cwd), cwd=self._repo_cwd)
2085
+ self.team_id = team["team_id"]
2086
+ self.team_join_code = team.get("join_code")
2087
+ self.team_admin_ids = list(team.get("admin_ids") or [])
2088
+ cprint(f"[Pulse] ✓ Workspace ready (join code: {self.team_join_code}) -- share this to invite teammates, or /repo to set a repo.")
2089
+ cloud.save_cached_credentials(self.user_id, self.email, self.team_id)
2090
+ except cloud.SupabaseError as exc:
2091
+ cprint(f"[Pulse] ⚠ Could not create a workspace ({exc}) -- continuing without a team. Try /repo or restart to pick one.", color=_RED)
2092
+ return
2093
+
1881
2094
  while True:
1882
2095
  _flush_stdin()
1883
2096
  cprint("\n--- Pulse Workspace ---")
@@ -2092,6 +2305,66 @@ class PulseCLI:
2092
2305
  self.team_id = None
2093
2306
  self.debug_session_id = None
2094
2307
 
2308
+ def _cmd_logout(self, arg: str) -> None:
2309
+ """/logout -- sign out of the account currently in use for this
2310
+ run, then immediately offers to sign back in (as the same account
2311
+ or a different one) without having to restart Pulse. Revokes the
2312
+ server-side session token (so the cached credentials.json on this
2313
+ machine can't silently log back in on its own) and clears the
2314
+ local cache. If a new account signs in, this also re-runs the
2315
+ workspace picker and opens a fresh cloud debug session under that
2316
+ identity -- the old session is flushed and left as-is on the
2317
+ dashboard rather than mixing two identities' data into one row.
2318
+ """
2319
+ if not self.user_id:
2320
+ cprint("[Pulse] Not signed in to Pulse Cloud -- nothing to log out of.")
2321
+ return
2322
+ if self.non_interactive:
2323
+ cprint("[Pulse] /logout requires an interactive session.")
2324
+ return
2325
+
2326
+ old_email = self.email
2327
+
2328
+ # Flush anything still pending for the CURRENT session/identity
2329
+ # before switching -- otherwise a dirty field written after
2330
+ # logout could still get attributed to the old debug session.
2331
+ self._flush_cloud_now()
2332
+
2333
+ cloud.revoke_session_token(self.user_id) # best-effort; never raises
2334
+ cloud.clear_cached_credentials()
2335
+ self.user_id = None
2336
+ self.email = None
2337
+ self.team_id = None
2338
+ self.team_admin_ids = []
2339
+ self.debug_session_id = None
2340
+ cprint(f"[Pulse] ✓ Logged out of {old_email}.")
2341
+
2342
+ try:
2343
+ self._auth_flow()
2344
+ except cloud.SupabaseError as exc:
2345
+ cprint(f"[Pulse] ⚠ Could not sign in to Pulse cloud, continuing locally: {exc}", color=_RED)
2346
+ return
2347
+
2348
+ if not self.user_id:
2349
+ cprint("[Pulse] Continuing this run in local (no-cloud) mode.")
2350
+ return
2351
+
2352
+ try:
2353
+ self._team_flow()
2354
+ except cloud.SupabaseError as exc:
2355
+ cprint(f"[Pulse] ⚠ Team setup failed, continuing without a team: {exc}", color=_RED)
2356
+
2357
+ try:
2358
+ sha = self._last_synced_commit_sha or cloud.current_git_commit_sha(self._repo_cwd) or "unknown"
2359
+ self.debug_session_id = cloud.create_debug_session(self.team_id, self.user_id, git_commit_sha=sha)
2360
+ self._last_synced_commit_sha = sha
2361
+ if self.debug_session_id:
2362
+ cprint(f"[Pulse] Debug session started (id={self.debug_session_id}, commit={sha[:10] if sha != 'unknown' else 'unknown'}).")
2363
+ atexit.register(self._flush_cloud_now)
2364
+ self._last_cloud_flush = time.monotonic()
2365
+ except cloud.SupabaseError as exc:
2366
+ cprint(f"[Pulse] ⚠ Could not start a cloud debug session, continuing locally: {exc}", color=_RED)
2367
+
2095
2368
  def _cmd_webhook(self, arg: str) -> None:
2096
2369
  """/webhook set <url> | /webhook test | /webhook off | /webhook
2097
2370
  -- a Slack-incoming-webhook-compatible URL that gets a message
@@ -2444,6 +2717,7 @@ class PulseCLI:
2444
2717
  ("/code", "toggle whether questions include your training code"),
2445
2718
  ("/password", "change your Pulse account password"),
2446
2719
  ("/recover", "reset a forgotten password with a recovery code"),
2720
+ ("/logout", "sign out, then sign back in as the same or a different account"),
2447
2721
  ]),
2448
2722
  ("Cloud & team", [
2449
2723
  ("/cloud", "show sign-in, workspace, and sync status at a glance"),
@@ -3069,19 +3343,59 @@ class PulseCLI:
3069
3343
  # to route around.
3070
3344
  argv = [python_exe, script_path] + sys.argv[1:]
3071
3345
 
3072
- try:
3073
- # Synchronous run keeps stdin attached and handles spaces in paths correctly on Windows
3074
- result = subprocess.run(argv)
3075
- except Exception as exc:
3076
- cprint(f"[Pulse] ⚠ Restart failed ({exc}). Continuing current run.", color=_RED)
3077
- self._log_incident("restart_failed", str(exc))
3078
- return
3346
+ # Retry the restart itself instead of ever falling back to "keep
3347
+ # running the old, already-in-memory process" on a bad exit code.
3348
+ # A nonzero exit here almost always means the replacement process
3349
+ # crashed immediately (e.g. the agent's fix didn't fully fix it,
3350
+ # or introduced a new bug) -- silently resuming the OLD code path
3351
+ # used to look like a safe fallback, but for an unsupervised run
3352
+ # it just means the SAME already-crashed-once code keeps limping
3353
+ # along, or the loop quietly stalls, with nobody watching to
3354
+ # notice. Retrying the launch instead gives a transient problem
3355
+ # (e.g. a port/file briefly locked by the process that's still
3356
+ # exiting, a flaky import) a real chance to clear, and gives the
3357
+ # agent's fix -- which is already saved to disk -- more chances to
3358
+ # actually take effect. Attempts are capped (not infinite) so a
3359
+ # deterministically-broken script can't spin forever burning
3360
+ # compute/cost unattended; MAX_RESTART_ATTEMPTS is the one knob to
3361
+ # raise if that cap is ever too low for a given job.
3362
+ MAX_RESTART_ATTEMPTS = 5
3363
+ RETRY_BACKOFF_SECONDS = 3 # multiplied by attempt number, capped below
3364
+
3365
+ attempt = 0
3366
+ while True:
3367
+ attempt += 1
3368
+ try:
3369
+ # Synchronous run keeps stdin attached and handles spaces in paths correctly on Windows
3370
+ result = subprocess.run(argv)
3371
+ except Exception as exc:
3372
+ cprint(f"[Pulse] ⚠ Restart attempt {attempt}/{MAX_RESTART_ATTEMPTS} failed to launch ({exc}).", color=_RED)
3373
+ if attempt >= MAX_RESTART_ATTEMPTS:
3374
+ cprint(
3375
+ f"[Pulse] ⚠ Giving up after {MAX_RESTART_ATTEMPTS} restart attempts -- "
3376
+ "continuing current run with the old in-memory code.",
3377
+ color=_RED,
3378
+ )
3379
+ self._log_incident("restart_failed", f"Could not launch replacement process after {attempt} attempts: {exc}")
3380
+ return
3381
+ time.sleep(min(RETRY_BACKOFF_SECONDS * attempt, 30))
3382
+ continue
3079
3383
 
3080
- if result.returncode == 0:
3081
- sys.exit(0)
3082
- message = f"Replacement training process exited with code {result.returncode}"
3083
- cprint(f"[Pulse] ⚠ {message}. Continuing current run.", color=_RED)
3084
- self._log_incident("restart_failed", message)
3384
+ if result.returncode == 0:
3385
+ sys.exit(0)
3386
+
3387
+ message = f"Replacement training process exited with code {result.returncode} (attempt {attempt}/{MAX_RESTART_ATTEMPTS})"
3388
+ if attempt >= MAX_RESTART_ATTEMPTS:
3389
+ cprint(
3390
+ f"[Pulse] ⚠ {message}. Giving up after {MAX_RESTART_ATTEMPTS} attempts -- "
3391
+ "continuing current run with the old in-memory code.",
3392
+ color=_RED,
3393
+ )
3394
+ self._log_incident("restart_failed", message)
3395
+ return
3396
+
3397
+ cprint(f"[Pulse] ⚠ {message}. Retrying restart...", color=_YELLOW)
3398
+ time.sleep(min(RETRY_BACKOFF_SECONDS * attempt, 30))
3085
3399
 
3086
3400
  def _print_variable_summary(self) -> None:
3087
3401
  variables = self.discover_variables()
@@ -3448,18 +3762,21 @@ class PulseCLI:
3448
3762
  _GPUTRACK_RE = re.compile(r"^\s*GPUTRACK:\s*(.+)$", re.MULTILINE)
3449
3763
  _GPUUNTRACK_RE = re.compile(r"^\s*GPUUNTRACK:\s*(.+)$", re.MULTILINE)
3450
3764
  _SENSITIVITY_RE = re.compile(r"^\s*SENSITIVITY:\s*(.+)$", re.MULTILINE)
3765
+ _NORMAL_START_RE = re.compile(r"^\s*NORMAL_START:\s*(.+)$", re.MULTILINE)
3451
3766
 
3452
3767
  @classmethod
3453
3768
  def _extract_directives(cls, text: str):
3454
- """Pull CALC:/PROMOTE:/GPUTRACK:/GPUUNTRACK:/SENSITIVITY: lines out
3455
- of an agent response, returning (cleaned_text, calc_exprs,
3456
- promote_names, gputrack_names, gpuuntrack_names,
3457
- sensitivity_args). Cleaned text has those lines stripped so they
3458
- don't clutter what's printed/stored. A bare 'none' value (as
3459
- instructed for the periodic GPU check-in reply format) is dropped
3460
- rather than treated as a variable name. sensitivity_args is a list
3461
- of raw SENSITIVITY: argument strings (usually 0 or 1) -- applied
3462
- via _cmd_sensitivity(..., quiet=True) same as a manual /sensitivity.
3769
+ """Pull CALC:/PROMOTE:/GPUTRACK:/GPUUNTRACK:/SENSITIVITY:/
3770
+ NORMAL_START: lines out of an agent response, returning
3771
+ (cleaned_text, calc_exprs, promote_names, gputrack_names,
3772
+ gpuuntrack_names, sensitivity_args, normal_start_args). Cleaned
3773
+ text has those lines stripped so they don't clutter what's
3774
+ printed/stored. A bare 'none' value (as instructed for the
3775
+ periodic GPU check-in reply format) is dropped rather than
3776
+ treated as a variable name. sensitivity_args/normal_start_args
3777
+ are lists of raw argument strings (usually 0 or 1) -- applied via
3778
+ _cmd_sensitivity(..., quiet=True) / the NORMAL_START parsing in
3779
+ _apply_directives, same as a manual /sensitivity.
3463
3780
  """
3464
3781
  calc_exprs = [m.strip() for m in cls._CALC_RE.findall(text) if m.strip()]
3465
3782
  promote_names = []
@@ -3476,14 +3793,16 @@ class PulseCLI:
3476
3793
  n.strip() for n in m.split(",") if n.strip() and n.strip().lower() != "none"
3477
3794
  )
3478
3795
  sensitivity_args = [m.strip() for m in cls._SENSITIVITY_RE.findall(text) if m.strip()]
3796
+ normal_start_args = [m.strip() for m in cls._NORMAL_START_RE.findall(text) if m.strip()]
3479
3797
 
3480
3798
  cleaned = cls._CALC_RE.sub("", text)
3481
3799
  cleaned = cls._PROMOTE_RE.sub("", cleaned)
3482
3800
  cleaned = cls._GPUTRACK_RE.sub("", cleaned)
3483
3801
  cleaned = cls._GPUUNTRACK_RE.sub("", cleaned)
3484
3802
  cleaned = cls._SENSITIVITY_RE.sub("", cleaned)
3803
+ cleaned = cls._NORMAL_START_RE.sub("", cleaned)
3485
3804
  cleaned = re.sub(r"\n{3,}", "\n\n", cleaned).strip()
3486
- return cleaned, calc_exprs, promote_names, gputrack_names, gpuuntrack_names, sensitivity_args
3805
+ return cleaned, calc_exprs, promote_names, gputrack_names, gpuuntrack_names, sensitivity_args, normal_start_args
3487
3806
 
3488
3807
  def _apply_directives(
3489
3808
  self,
@@ -3492,12 +3811,14 @@ class PulseCLI:
3492
3811
  gputrack_names: Optional[List[str]] = None,
3493
3812
  gpuuntrack_names: Optional[List[str]] = None,
3494
3813
  sensitivity_args: Optional[List[str]] = None,
3814
+ normal_start_args: Optional[List[str]] = None,
3495
3815
  ) -> str:
3496
3816
  """Deterministically compute any CALC: expressions and apply any
3497
- PROMOTE:/GPUTRACK:/GPUUNTRACK:/SENSITIVITY: requests, returning a
3498
- short human-readable summary to print and to feed back into the
3499
- agent's own history (so it sees the verified numbers/state on the
3500
- next turn instead of trusting its own arithmetic or memory).
3817
+ PROMOTE:/GPUTRACK:/GPUUNTRACK:/SENSITIVITY:/NORMAL_START: requests,
3818
+ returning a short human-readable summary to print and to feed
3819
+ back into the agent's own history (so it sees the verified
3820
+ numbers/state on the next turn instead of trusting its own
3821
+ arithmetic or memory).
3501
3822
  """
3502
3823
  notes = []
3503
3824
 
@@ -3550,6 +3871,33 @@ class PulseCLI:
3550
3871
  print(f" 🎚 Sensitivity adjusted (agent request): {result}")
3551
3872
  notes.append(result)
3552
3873
 
3874
+ if normal_start_args:
3875
+ applied = []
3876
+ for raw in normal_start_args:
3877
+ if raw.strip().lower() == "none":
3878
+ continue
3879
+ for pair in raw.split(","):
3880
+ pair = pair.strip()
3881
+ if not pair or "=" not in pair:
3882
+ continue
3883
+ name, _, val_str = pair.partition("=")
3884
+ name = name.strip()
3885
+ if not name:
3886
+ continue
3887
+ try:
3888
+ val = float(val_str.strip())
3889
+ except ValueError:
3890
+ continue
3891
+ self._normal_start_baselines[name] = val
3892
+ applied.append(f"{name}={val:.4g}")
3893
+ if applied:
3894
+ print(f" 🌱 Seeded starting baseline (agent estimate, from code): {', '.join(applied)}")
3895
+ notes.append(
3896
+ f"Seeded expected starting value(s) from the code: {', '.join(applied)}. "
3897
+ "These act as a spike-detection baseline until real data accumulates, so a "
3898
+ "genuine explosion in the first few steps can still be caught."
3899
+ )
3900
+
3553
3901
  return "\n\n".join(notes)
3554
3902
 
3555
3903
  _GPU_CHECKIN_PROMPT = (
@@ -3597,7 +3945,7 @@ class PulseCLI:
3597
3945
  # and try again at the next interval.
3598
3946
  cprint(f"[Pulse] ⚠ GPU check-in skipped (agent request failed: {exc})", color=_RED)
3599
3947
  return
3600
- _, _calc, _promote, gputrack_names, gpuuntrack_names, _sens = self._extract_directives(answer)
3948
+ _, _calc, _promote, gputrack_names, gpuuntrack_names, _sens, _norm = self._extract_directives(answer)
3601
3949
  summary = self._apply_directives([], [], gputrack_names, gpuuntrack_names)
3602
3950
  if summary:
3603
3951
  self.agent_history.append({"role": "user", "content": prompt})
@@ -3815,9 +4163,9 @@ class PulseCLI:
3815
4163
  raw_analysis = self._call_model(
3816
4164
  _PASS2_ANALYZE_TMPL.format(regions=regions), max_tokens=700
3817
4165
  )
3818
- analysis, calc_exprs, promote_names, gputrack_names, gpuuntrack_names, sensitivity_args = self._extract_directives(raw_analysis)
4166
+ analysis, calc_exprs, promote_names, gputrack_names, gpuuntrack_names, sensitivity_args, normal_start_args = self._extract_directives(raw_analysis)
3819
4167
  print(f"[2] Diagnosis & reasoning\n{analysis}\n")
3820
- directive_note = self._apply_directives(calc_exprs, promote_names, gputrack_names, gpuuntrack_names, sensitivity_args)
4168
+ directive_note = self._apply_directives(calc_exprs, promote_names, gputrack_names, gpuuntrack_names, sensitivity_args, normal_start_args)
3821
4169
  if directive_note:
3822
4170
  self.agent_history.append({"role": "user", "content": directive_note})
3823
4171
 
@@ -4455,7 +4803,9 @@ class PulseCLI:
4455
4803
  self.plateau_range_frac if self.plateau_range_frac is not None
4456
4804
  else 10 ** (-5 + 3 * s)
4457
4805
  ),
4458
- # Loose: needs 15/19 directional reversals. Tight: only 6.
4806
+ # Loose: needs 15/18 directional reversals. Tight: only 6.
4807
+ # (A 20-point window yields 19 deltas and only 18 consecutive
4808
+ # delta-pairs to check for a sign flip -- see _check_for_trouble.)
4459
4809
  "oscillation_flip_threshold": (
4460
4810
  self.oscillation_flip_threshold if self.oscillation_flip_threshold is not None
4461
4811
  else round(15 - 9 * s)
@@ -4466,6 +4816,26 @@ class PulseCLI:
4466
4816
  self.oscillation_delta_frac if self.oscillation_delta_frac is not None
4467
4817
  else 0.10 - 0.08 * s
4468
4818
  ),
4819
+ # Stagnation: has a loss-like variable meaningfully improved
4820
+ # over a LONG window, once normal per-step noise is averaged
4821
+ # out? This is deliberately separate from "plateau" above --
4822
+ # plateau looks at the *range* of the last 20 steps, which a
4823
+ # loss with completely normal noise (bouncing around by, say,
4824
+ # 0.02 every step while never actually trending down) will
4825
+ # never look "flat enough" to trigger, even after a thousand
4826
+ # steps of zero real progress. Stagnation instead compares
4827
+ # the mean of the first quarter of a long window against the
4828
+ # mean of the last quarter, so per-step noise washes out and
4829
+ # only genuine lack of improvement is left.
4830
+ # Loose (s=0): needs 500 steps of history, and even a 0.3%
4831
+ # improvement over that window counts as "still improving".
4832
+ # Tight (s=1): needs only 150 steps, and demands a full 8%
4833
+ # improvement before it stops flagging.
4834
+ "stagnation_window": round(500 - 350 * s),
4835
+ "stagnation_frac": (
4836
+ self.stagnation_frac if self.stagnation_frac is not None
4837
+ else 0.003 + 0.077 * s
4838
+ ),
4469
4839
  }
4470
4840
 
4471
4841
  def _cmd_sensitivity(self, arg: str, quiet: bool = False) -> Optional[str]:
@@ -4486,10 +4856,11 @@ class PulseCLI:
4486
4856
  f"Sensitivity is {self.sensitivity:.2f} (0=loosest, 1=tightest). Derived thresholds: "
4487
4857
  f"spike >{th['explosion_multiplier']:.1f}x baseline, "
4488
4858
  f"plateau range <{th['plateau_range_frac']:.1e} of latest, "
4489
- f"oscillation >={int(th['oscillation_flip_threshold'])} reversals/19 "
4490
- f"with avg swing >{th['oscillation_delta_frac']*100:.0f}% of latest. "
4859
+ f"oscillation >={int(th['oscillation_flip_threshold'])} reversals/18 "
4860
+ f"with avg swing >{th['oscillation_delta_frac']*100:.0f}% of latest, "
4861
+ f"stagnation <{th['stagnation_frac']*100:.1f}% improvement over {int(th['stagnation_window'])} steps. "
4491
4862
  "Usage: /sensitivity <0.0-1.0|loose|medium|tight> or "
4492
- "/sensitivity <spike|plateau|oscillation> <value|auto>"
4863
+ "/sensitivity <spike|plateau|oscillation|stagnation> <value|auto>"
4493
4864
  )
4494
4865
  if quiet:
4495
4866
  return msg
@@ -4498,12 +4869,13 @@ class PulseCLI:
4498
4869
 
4499
4870
  parts = arg.split(None, 1)
4500
4871
  sub = parts[0].lower()
4501
- if sub in ("spike", "plateau", "oscillation") and len(parts) == 2:
4872
+ if sub in ("spike", "plateau", "oscillation", "stagnation") and len(parts) == 2:
4502
4873
  val_str = parts[1].strip().lower()
4503
4874
  field = {
4504
4875
  "spike": "explosion_multiplier",
4505
4876
  "plateau": "plateau_range_frac",
4506
4877
  "oscillation": "oscillation_flip_threshold", # flips; delta_frac follows the dial
4878
+ "stagnation": "stagnation_frac", # window length always follows the dial
4507
4879
  }[sub]
4508
4880
  if val_str == "auto":
4509
4881
  setattr(self, field, None)
@@ -4523,6 +4895,7 @@ class PulseCLI:
4523
4895
  if sub == "reset":
4524
4896
  self.explosion_multiplier = self.plateau_range_frac = None
4525
4897
  self.oscillation_flip_threshold = self.oscillation_delta_frac = None
4898
+ self.stagnation_frac = None
4526
4899
  msg = "✓ Cleared per-signal overrides -- all thresholds now follow the overall dial."
4527
4900
  if quiet:
4528
4901
  return msg
@@ -4537,7 +4910,7 @@ class PulseCLI:
4537
4910
  except ValueError:
4538
4911
  msg = (
4539
4912
  f"Usage: /sensitivity <0.0-1.0|{'|'.join(self._SENSITIVITY_PRESETS)}> or "
4540
- "/sensitivity <spike|plateau|oscillation> <value|auto>"
4913
+ "/sensitivity <spike|plateau|oscillation|stagnation> <value|auto>"
4541
4914
  )
4542
4915
  if quiet:
4543
4916
  return msg
@@ -4565,30 +4938,100 @@ class PulseCLI:
4565
4938
 
4566
4939
  if _looks_like_loss(var_name) and latest is not None:
4567
4940
  finite_recent = [v for v in hist[-50:] if v is not None and math.isfinite(v)]
4568
-
4569
- # Explosion detection
4570
- if len(finite_recent) >= 5:
4571
- baseline = min(finite_recent[:-1])
4941
+
4942
+ # Explosion detection. Normally needs a few real finite
4943
+ # points to compute a "recent minimum" baseline -- which
4944
+ # meant a genuine explosion in the first few steps of
4945
+ # training could never be caught, since there just wasn't
4946
+ # enough history yet to compare against. If the agent has
4947
+ # seeded an expected starting value for this variable from
4948
+ # reading the code (a NORMAL_START: directive, sent
4949
+ # automatically once at the start of training -- see
4950
+ # _apply_directives / _prime_at_start), fold it into the
4951
+ # baseline pool so a step-1 blow-up has something real to
4952
+ # compare against too.
4953
+ seed = self._normal_start_baselines.get(var_name)
4954
+ baseline_pool = finite_recent[:-1] if len(finite_recent) >= 5 else []
4955
+ if seed is not None:
4956
+ baseline_pool = baseline_pool + [seed]
4957
+ if baseline_pool:
4958
+ baseline = min(baseline_pool)
4572
4959
  if baseline > 0 and latest > baseline * th["explosion_multiplier"]:
4960
+ basis = (
4961
+ "its expected starting value (estimated from the code)"
4962
+ if not finite_recent[:-1] else "its recent minimum"
4963
+ )
4573
4964
  reasons.append(
4574
4965
  f"'{var_name}' spiked to {latest:.4g}, "
4575
- f"{latest / baseline:.1f}x its recent minimum ({baseline:.4g})"
4966
+ f"{latest / baseline:.1f}x {basis} ({baseline:.4g})"
4576
4967
  )
4577
-
4968
+
4578
4969
  if len(finite_recent) >= 20:
4579
4970
  recent_window = finite_recent[-20:]
4580
-
4971
+
4581
4972
  deltas = [recent_window[i] - recent_window[i-1] for i in range(1, len(recent_window))]
4582
-
4973
+
4974
+ # Scale the plateau/oscillation thresholds off the
4975
+ # window's own typical magnitude (mean |value| across
4976
+ # all 20 points), not off `latest` alone. `latest` is
4977
+ # a single sample that can itself land at a momentary
4978
+ # peak, trough, or near-zero crossing of the very
4979
+ # curve being judged -- using it as the sole scale
4980
+ # reference made both checks unreliable: too
4981
+ # trigger-happy right as a curve crossed zero, too lax
4982
+ # whenever `latest` happened to be sitting at a local
4983
+ # extreme instead of a typical value.
4984
+ scale = sum(abs(v) for v in recent_window) / len(recent_window)
4985
+ if scale == 0:
4986
+ scale = abs(latest) # degenerate all-zero window; fall back rather than lose the check entirely
4987
+
4583
4988
  window_range = max(recent_window) - min(recent_window)
4584
- if window_range < (abs(latest) * th["plateau_range_frac"]) and window_range > 0:
4989
+ if window_range == 0:
4990
+ # Identical value for 20 straight steps (dead
4991
+ # gradient, lr=0, a frozen model) is the most
4992
+ # extreme plateau there is, not an edge case to
4993
+ # skip. The old `window_range > 0` guard here
4994
+ # excluded exactly this, so a totally frozen loss
4995
+ # -- arguably the easiest plateau to catch -- was
4996
+ # the one case that could never be flagged.
4997
+ reasons.append(f"'{var_name}' has completely frozen (identical value for the last 20 steps).")
4998
+ elif window_range < (scale * th["plateau_range_frac"]):
4585
4999
  reasons.append(f"'{var_name}' has plateaued (range across last 20 steps is {window_range:.2e}).")
4586
-
5000
+
4587
5001
  sign_flips = sum(1 for i in range(1, len(deltas)) if (deltas[i] * deltas[i-1]) < 0)
4588
5002
  avg_delta_mag = sum(abs(d) for d in deltas) / len(deltas)
4589
-
4590
- if sign_flips >= th["oscillation_flip_threshold"] and avg_delta_mag > (abs(latest) * th["oscillation_delta_frac"]):
4591
- reasons.append(f"'{var_name}' is heavily oscillating ({sign_flips} directional reversals in 20 steps).")
5003
+
5004
+ if sign_flips >= th["oscillation_flip_threshold"] and avg_delta_mag > (scale * th["oscillation_delta_frac"]):
5005
+ reasons.append(f"'{var_name}' is heavily oscillating ({sign_flips} directional reversals in the last 20 steps).")
5006
+
5007
+ # Long-horizon stagnation check -- deliberately separate
5008
+ # from the plateau check above, and computed from its own
5009
+ # independently-sized slice of `hist` (not finite_recent,
5010
+ # which stays capped at 50 so it doesn't change the
5011
+ # explosion baseline's "recent minimum" into a
5012
+ # much-older, possibly stale minimum). Plateau looks at
5013
+ # the *range* of just the last 20 steps, so a loss
5014
+ # bouncing around by completely normal per-step noise
5015
+ # (e.g. +/-0.02 every step, never trending down) will
5016
+ # never look "flat enough" to trip it, no matter how many
5017
+ # hundreds of steps go by with zero real progress --
5018
+ # which is exactly what a loss stuck oscillating in a
5019
+ # narrow band around the same value for 1000+ steps looks
5020
+ # like. This instead compares the mean of the first
5021
+ # quarter of a long window against the mean of the last
5022
+ # quarter, so per-step noise averages out and only
5023
+ # genuine lack of improvement is left standing.
5024
+ window = int(th["stagnation_window"])
5025
+ long_window_raw = [v for v in hist[-window:] if v is not None and math.isfinite(v)]
5026
+ if len(long_window_raw) >= window:
5027
+ quarter = max(1, window // 4)
5028
+ early_mean = sum(long_window_raw[:quarter]) / quarter
5029
+ late_mean = sum(long_window_raw[-quarter:]) / quarter
5030
+ if early_mean != 0 and abs(early_mean - late_mean) < abs(early_mean) * th["stagnation_frac"]:
5031
+ reasons.append(
5032
+ f"'{var_name}' hasn't meaningfully improved over the last {window} steps "
5033
+ f"(from {early_mean:.4g} to {late_mean:.4g}, despite normal step-to-step noise)."
5034
+ )
4592
5035
 
4593
5036
  for sub_name, entry in self._matrix_cache.items():
4594
5037
  stats = entry.get("stats", {})
@@ -4596,6 +5039,72 @@ class PulseCLI:
4596
5039
  reasons.append(f"'{sub_name}' has nan={stats.get('nan')} inf={stats.get('inf')}")
4597
5040
 
4598
5041
  return "; ".join(reasons) if reasons else None
5042
+ _START_PRIME_PROMPT = (
5043
+ "[Automatic start-of-run check -- sent once, automatically, before the first training step, "
5044
+ "so this is your only chance to set these from the code alone, before any real data exists] "
5045
+ "Look at the training code and the tracked variables above. Three things:\n"
5046
+ "1. Judge how noisy this run's loss/metric curves are likely to be, given the model type, "
5047
+ "batch size, learning rate, and loss function, and set an appropriate sensitivity for "
5048
+ "spike/plateau/oscillation detection.\n"
5049
+ "2. For each loss-like tracked variable, estimate its expected value at the very start of "
5050
+ "training from the code alone if you can justify one (e.g. a randomly-initialized N-class "
5051
+ "classifier's cross-entropy loss starts near ln(N); a policy's initial reward is often near "
5052
+ "a known random-policy baseline). This is the ONLY way Pulse can catch a real explosion in "
5053
+ "the first few steps of training -- normally spike detection needs several real data points "
5054
+ "before it has anything to compare against, so a blow-up before then would otherwise go "
5055
+ "completely undetected.\n"
5056
+ "3. Identify any critical variables -- e.g. loss, or core tensors that live exclusively in "
5057
+ "GPU memory -- that are worth the closer, GPU-synced look from step one, to catch memory "
5058
+ "spikes or numerical instability early rather than after they've already caused visible "
5059
+ "damage.\n\n"
5060
+ "Reply with ONLY these three lines, in exactly this format, and nothing else -- no diagnosis, "
5061
+ "no prose:\n"
5062
+ "SENSITIVITY: <0.0-1.0, a preset (loose/medium/tight), or 'spike|plateau|oscillation <value|auto>'>\n"
5063
+ "NORMAL_START: <comma-separated var=value pairs for loss-like tracked variables you can justify, or 'none'>\n"
5064
+ "GPUTRACK: <comma-separated variable names to track closely from the start, or 'none'>"
5065
+ )
5066
+
5067
+ def _prime_at_start(self) -> None:
5068
+ """Ask the agent, once, automatically, before the very first
5069
+ training step, to (a) set a sensitivity appropriate to this run's
5070
+ code, (b) estimate a starting-value baseline for loss-like tracked
5071
+ variables by reading the code alone (a NORMAL_START: directive --
5072
+ see _apply_directives), and (c) flag any variables worth GPU-level
5073
+ tracking from step one (a GPUTRACK: directive).
5074
+
5075
+ All three of these used to only ever happen reactively: sensitivity
5076
+ only got tuned once the agent had already seen live data (e.g. a
5077
+ GPU check-in or a manual /ask), GPU-tracking only ever got turned
5078
+ on after something had already looked suspicious enough to ask
5079
+ about, and there was no mechanism at all for seeding a starting
5080
+ baseline -- which meant an explosion in the first few steps of
5081
+ training, before 5 real data points existed, could never be
5082
+ caught (see _check_for_trouble). Doing all three once up front,
5083
+ from the code alone, closes those gaps before training even
5084
+ starts.
5085
+
5086
+ Best-effort and silent on failure -- this must never be the thing
5087
+ that makes a training run fail to start. Runs at most once per
5088
+ process (guarded by self._start_primed), and only if an agent
5089
+ provider/key is actually configured and the training code was
5090
+ made available via set_code_text.
5091
+ """
5092
+ if self._start_primed:
5093
+ return
5094
+ self._start_primed = True
5095
+ if not self.agent_provider or not self.agent_key or not self.code_text:
5096
+ return
5097
+ try:
5098
+ context = self._build_agent_context(include_code=True)
5099
+ answer = self._call_model(f"{context}\n\n{self._START_PRIME_PROMPT}", max_tokens=300)
5100
+ except AgentRequestFailed as exc:
5101
+ cprint(f"[Pulse] ⚠ Start-of-run sensitivity check skipped (agent request failed: {exc})", color=_YELLOW)
5102
+ return
5103
+ _, _calc, _promote, gputrack_names, _gpuu, sensitivity_args, normal_start_args = self._extract_directives(answer)
5104
+ summary = self._apply_directives([], [], gputrack_names, None, sensitivity_args, normal_start_args)
5105
+ if summary:
5106
+ cprint(f"[Pulse] Start-of-run check: {summary}", color=_YELLOW)
5107
+
4599
5108
  def update(self, step: Optional[int] = None, generate_pdfs: Optional[bool] = None) -> None:
4600
5109
  """Called at every training step/checkpoint.
4601
5110
 
@@ -4622,6 +5131,8 @@ class PulseCLI:
4622
5131
  loss_var = next((v for v in self.tracked_vars if _looks_like_loss(v)), None)
4623
5132
  new_loss_value: Optional[float] = None
4624
5133
 
5134
+ self._prime_at_start()
5135
+
4625
5136
  # Uptime: wall-clock time since the *previous* update() call handed
4626
5137
  # control back to the training loop, i.e. time actually spent in
4627
5138
  # the user's own training code (forward/backward/optimizer step).
@@ -5032,6 +5543,9 @@ class PulseCLI:
5032
5543
  if cmd.lower().startswith("/deleteaccount"):
5033
5544
  self._cmd_deleteaccount(cmd[14:].strip())
5034
5545
  continue
5546
+ if cmd.lower() == "/logout":
5547
+ self._cmd_logout("")
5548
+ continue
5035
5549
  if cmd.lower().startswith("/commit"):
5036
5550
  self._cmd_commit(cmd[7:].strip())
5037
5551
  continue
@@ -191,6 +191,17 @@ class SupabaseError(Exception):
191
191
  pass
192
192
 
193
193
 
194
+ class SupabaseEmailConfirmationRequired(SupabaseError):
195
+ """Raised by sign_up() when the account was actually created but this
196
+ project's Auth settings require the user to confirm their email
197
+ before a session can be issued. A subclass of SupabaseError (not a
198
+ separate flag) so any existing `except SupabaseError` call site keeps
199
+ working unchanged, while a call site that wants to show a "go check
200
+ your email" message instead of a generic "sign up failed" message can
201
+ catch this specifically."""
202
+ pass
203
+
204
+
194
205
  # ----------------------------------------------------------------------------
195
206
  # Auth session state -- the real Supabase Auth JWT for the signed-in user,
196
207
  # as distinct from the app-level "session_token"/session_token_hash pair
@@ -602,7 +613,7 @@ def sign_up(email: str, password: str, plan: Optional[str] = None) -> Dict[str,
602
613
  Returns the user profile from the profiles table."""
603
614
  if not is_valid_email(email):
604
615
  raise SupabaseError(
605
- "Emails must be greater than 8 characters: letters, numbers, underscores, and hyphens only."
616
+ "Enter a valid email address (min. 8 characters), e.g. name@example.com."
606
617
  )
607
618
  if len(password) < 8:
608
619
  raise SupabaseError("Password must be at least 8 characters.")
@@ -644,8 +655,8 @@ def sign_up(email: str, password: str, plan: Optional[str] = None) -> Dict[str,
644
655
  # user confirms their email and logs in for real -- surface that
645
656
  # plainly instead of the misleading "profile was not created"
646
657
  # below (which is what happens if you try to fetch it anyway).
647
- raise SupabaseError(
648
- "Account created -- check your email to confirm your address, then log in."
658
+ raise SupabaseEmailConfirmationRequired(
659
+ "Confirm your account: check your email for a confirmation link, then log in."
649
660
  )
650
661
 
651
662
  # Authenticate every subsequent request in this process (starting with
@@ -1225,6 +1236,41 @@ def fetch_recent_sessions(
1225
1236
  return _request("GET", "Debug_Sessions", params=params, timeout=_TIMEOUT_INTERACTIVE) or []
1226
1237
 
1227
1238
 
1239
+ def fetch_debug_session(session_id: str) -> Optional[Dict[str, Any]]:
1240
+ """Fetch a single Debug_Sessions row by id (fields still
1241
+ compressed -- caller decodes with decode_entries).
1242
+
1243
+ Used when resuming an EXISTING debug session -- currently only after
1244
+ an auto-fix restart, which carries the same session id into the new
1245
+ process via PULSE_AUTO_SESSION_ID -- so the resumed process can
1246
+ preload its local agent_logs/error_tracebacks/telemetry/incidents
1247
+ lists with whatever's already saved server-side. Without this, a
1248
+ freshly-started process's lists start empty and, since a PATCH
1249
+ replaces the whole array column (PostgREST has no array "append"
1250
+ verb -- see _build_cloud_patch_body), its very first flush after
1251
+ the restart would silently overwrite the pre-restart history with
1252
+ just the handful of entries logged since. Falls back to the legacy
1253
+ select list for deployments that don't have the incidents/uptime/
1254
+ downtime columns yet.
1255
+ """
1256
+ try:
1257
+ rows = _request(
1258
+ "GET", "Debug_Sessions",
1259
+ params={"id": f"eq.{session_id}", "select": _SESSION_SELECT},
1260
+ )
1261
+ except SupabaseError:
1262
+ try:
1263
+ rows = _request(
1264
+ "GET", "Debug_Sessions",
1265
+ params={"id": f"eq.{session_id}", "select": _SESSION_SELECT_LEGACY},
1266
+ )
1267
+ except SupabaseError:
1268
+ return None
1269
+ if not rows:
1270
+ return None
1271
+ return rows[0]
1272
+
1273
+
1228
1274
  def patch_debug_session(session_id: str, fields: Dict[str, Any], timeout: float = _TIMEOUT_BACKGROUND) -> None:
1229
1275
  _request(
1230
1276
  "PATCH", "Debug_Sessions",
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pulseml
3
- Version: 0.2.0
3
+ Version: 0.2.2
4
4
  Summary: A live ML training debugger - GUI or CLI, any backend (NumPy, PyTorch, TensorFlow, CuPy, JAX).
5
5
  Author-email: Yash Patel <codeyash09@gmail.com>
6
6
  License: Proprietary
File without changes
File without changes
File without changes
File without changes
File without changes