pulseml 0.2.0__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pulseml-0.2.0 → pulseml-0.2.2}/PKG-INFO +1 -1
- {pulseml-0.2.0 → pulseml-0.2.2}/pyproject.toml +1 -1
- {pulseml-0.2.0 → pulseml-0.2.2}/src/pulse/pulse.py +92 -38
- {pulseml-0.2.0 → pulseml-0.2.2}/src/pulse/pulse_cli.py +598 -84
- {pulseml-0.2.0 → pulseml-0.2.2}/src/pulse/pulse_supabase.py +49 -3
- {pulseml-0.2.0 → pulseml-0.2.2}/src/pulseml.egg-info/PKG-INFO +1 -1
- {pulseml-0.2.0 → pulseml-0.2.2}/LICENSE +0 -0
- {pulseml-0.2.0 → pulseml-0.2.2}/README.md +0 -0
- {pulseml-0.2.0 → pulseml-0.2.2}/setup.cfg +0 -0
- {pulseml-0.2.0 → pulseml-0.2.2}/src/pulse/__init__.py +0 -0
- {pulseml-0.2.0 → pulseml-0.2.2}/src/pulse/pulse_backend.py +0 -0
- {pulseml-0.2.0 → pulseml-0.2.2}/src/pulse/pulse_pdf.py +0 -0
- {pulseml-0.2.0 → pulseml-0.2.2}/src/pulseml.egg-info/SOURCES.txt +0 -0
- {pulseml-0.2.0 → pulseml-0.2.2}/src/pulseml.egg-info/dependency_links.txt +0 -0
- {pulseml-0.2.0 → pulseml-0.2.2}/src/pulseml.egg-info/requires.txt +0 -0
- {pulseml-0.2.0 → pulseml-0.2.2}/src/pulseml.egg-info/top_level.txt +0 -0
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "pulseml"
|
|
7
|
-
version = "0.2.
|
|
7
|
+
version = "0.2.2"
|
|
8
8
|
description = "A live ML training debugger - GUI or CLI, any backend (NumPy, PyTorch, TensorFlow, CuPy, JAX)."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.9"
|
|
@@ -3373,27 +3373,58 @@ def auto_track(train_fn=None, throttle_interval=1.0, code_text=None, project_roo
|
|
|
3373
3373
|
# macOS/Linux: Resets raw terminal modes and clears the screen
|
|
3374
3374
|
subprocess.run('stty sane', shell=True)
|
|
3375
3375
|
subprocess.run('clear', shell=True)
|
|
3376
|
-
#
|
|
3377
|
-
|
|
3378
|
-
|
|
3379
|
-
|
|
3380
|
-
|
|
3381
|
-
|
|
3376
|
+
# Retry the restart itself instead of falling back to
|
|
3377
|
+
# "keep running the old, already-in-memory process" on
|
|
3378
|
+
# a bad exit code. A nonzero exit almost always means
|
|
3379
|
+
# the replacement process crashed immediately (the
|
|
3380
|
+
# agent's fix didn't fully take, or introduced a new
|
|
3381
|
+
# bug) -- silently resuming the OLD code just means
|
|
3382
|
+
# the same already-broken run limps along unsupervised
|
|
3383
|
+
# with nobody watching. Retrying gives a transient
|
|
3384
|
+
# cause (a file/port briefly held by the exiting old
|
|
3385
|
+
# process, a flaky import) a real chance to clear.
|
|
3386
|
+
# Capped, not infinite, so a deterministically-broken
|
|
3387
|
+
# script can't spin forever burning compute unattended.
|
|
3388
|
+
MAX_RESTART_ATTEMPTS = 5
|
|
3389
|
+
RETRY_BACKOFF_SECONDS = 3
|
|
3390
|
+
|
|
3391
|
+
restart_argv = [sys.executable, script_path] + sys.argv[1:]
|
|
3392
|
+
attempt = 0
|
|
3393
|
+
restarted_ok = False
|
|
3394
|
+
while attempt < MAX_RESTART_ATTEMPTS:
|
|
3395
|
+
attempt += 1
|
|
3396
|
+
try:
|
|
3397
|
+
restart_result = subprocess.run(
|
|
3398
|
+
restart_argv,
|
|
3399
|
+
capture_output=True,
|
|
3400
|
+
text=True,
|
|
3401
|
+
)
|
|
3402
|
+
except Exception as exc:
|
|
3403
|
+
print(f"[PULSE] Restart attempt {attempt}/{MAX_RESTART_ATTEMPTS} failed to launch ({exc}).")
|
|
3404
|
+
if attempt < MAX_RESTART_ATTEMPTS:
|
|
3405
|
+
time.sleep(min(RETRY_BACKOFF_SECONDS * attempt, 30))
|
|
3406
|
+
continue
|
|
3407
|
+
|
|
3408
|
+
if restart_result.returncode == 0:
|
|
3409
|
+
restarted_ok = True
|
|
3410
|
+
break
|
|
3411
|
+
|
|
3412
|
+
print(
|
|
3413
|
+
f"\n[PULSE] Replacement training process exited with code "
|
|
3414
|
+
f"{restart_result.returncode} (attempt {attempt}/{MAX_RESTART_ATTEMPTS})."
|
|
3382
3415
|
)
|
|
3383
|
-
|
|
3384
|
-
|
|
3385
|
-
|
|
3386
|
-
|
|
3387
|
-
|
|
3416
|
+
if restart_result.stderr:
|
|
3417
|
+
print("[PULSE] STDERR from failed run:\n", restart_result.stderr)
|
|
3418
|
+
if restart_result.stdout:
|
|
3419
|
+
print("[PULSE] STDOUT from failed run:\n", restart_result.stdout)
|
|
3420
|
+
if attempt < MAX_RESTART_ATTEMPTS:
|
|
3421
|
+
print("[PULSE] Retrying restart...")
|
|
3422
|
+
time.sleep(min(RETRY_BACKOFF_SECONDS * attempt, 30))
|
|
3423
|
+
|
|
3424
|
+
if restarted_ok:
|
|
3388
3425
|
sys.exit(0)
|
|
3389
|
-
|
|
3390
|
-
|
|
3391
|
-
print(f"\n[PULSE] Replacement training process exited with code {restart_result.returncode}.")
|
|
3392
|
-
if restart_result.stderr:
|
|
3393
|
-
print("[PULSE] STDERR from failed run:\n", restart_result.stderr)
|
|
3394
|
-
if restart_result.stdout:
|
|
3395
|
-
print("[PULSE] STDOUT from failed run:\n", restart_result.stdout)
|
|
3396
|
-
print("[PULSE] Continuing the current process.")
|
|
3426
|
+
|
|
3427
|
+
print(f"[PULSE] ⚠ Giving up after {MAX_RESTART_ATTEMPTS} restart attempts -- continuing the current process.")
|
|
3397
3428
|
# bad (NaN/inf, a loss spike) and asked training to hold here until
|
|
3398
3429
|
# the user hits "Resume Training" -- or a fix gets applied and they
|
|
3399
3430
|
# resume manually. Blocks this exact line from executing further,
|
|
@@ -3510,6 +3541,24 @@ def _install_cli_excepthook(cli):
|
|
|
3510
3541
|
previous_hook = sys.excepthook
|
|
3511
3542
|
|
|
3512
3543
|
def _hook(exc_type, exc_value, exc_tb):
|
|
3544
|
+
# Disarm the global line/call tracer immediately. Once the script
|
|
3545
|
+
# has crashed, nothing below should still be traced -- and if we
|
|
3546
|
+
# leave `sys.settrace` pointed at `cli_tracer`, it stays the
|
|
3547
|
+
# process-wide trace function into interpreter shutdown, where it
|
|
3548
|
+
# gets invoked on unrelated __del__ calls (e.g. litellm's
|
|
3549
|
+
# AsyncHTTPHandler) after module globals have already been cleared
|
|
3550
|
+
# to None. That produces a second, uncatchable "Exception ignored
|
|
3551
|
+
# in ..." failure from *inside* the tracer, separate from (and
|
|
3552
|
+
# after) whatever we report here.
|
|
3553
|
+
sys.settrace(None)
|
|
3554
|
+
try:
|
|
3555
|
+
frame = exc_tb.tb_frame if exc_tb else None
|
|
3556
|
+
while frame is not None:
|
|
3557
|
+
frame.f_trace = None
|
|
3558
|
+
frame = frame.f_back
|
|
3559
|
+
except Exception:
|
|
3560
|
+
pass
|
|
3561
|
+
|
|
3513
3562
|
previous_hook(exc_type, exc_value, exc_tb)
|
|
3514
3563
|
if issubclass(exc_type, KeyboardInterrupt):
|
|
3515
3564
|
return
|
|
@@ -3603,14 +3652,29 @@ def _start_cli_tracker(
|
|
|
3603
3652
|
cli.extra_files = dict(extra_files or {})
|
|
3604
3653
|
cli.pending_startup_error = startup_error
|
|
3605
3654
|
cli.print_banner()
|
|
3606
|
-
|
|
3655
|
+
# Agent/provider setup used to be deferred until cli_tracer saw its
|
|
3656
|
+
# first 'line' event with a trackable/resolved local in scope -- fine
|
|
3657
|
+
# for a training loop, but it meant any crash before that point (e.g.
|
|
3658
|
+
# failing during data loading, before the loop even starts) left
|
|
3659
|
+
# cli.agent_provider unset, so the crash hook below had no agent to
|
|
3660
|
+
# hand the traceback to and could only print "no AI agent is
|
|
3661
|
+
# configured". discover_variables() already falls back to the
|
|
3662
|
+
# statically-discovered `cli.discovered` names (populated from AST
|
|
3663
|
+
# parsing at auto_track()-call time, before any user code runs) when
|
|
3664
|
+
# there are no live locals yet, so setup doesn't actually need to wait
|
|
3665
|
+
# for a live frame -- run it eagerly, before the excepthook can matter.
|
|
3666
|
+
cli.interactive_setup()
|
|
3607
3667
|
|
|
3608
3668
|
if startup_error:
|
|
3609
3669
|
print("[Pulse] The function passed to auto_track() raised an exception during its dry run:")
|
|
3610
3670
|
print(startup_error)
|
|
3611
|
-
|
|
3671
|
+
if cli.agent_provider:
|
|
3672
|
+
print("[Pulse] Pulse will offer to diagnose/fix it (via the agent set up above) before continuing.\n")
|
|
3673
|
+
else:
|
|
3674
|
+
print("[Pulse] No agent configured (see summary above) -- run /agent to set one up before continuing.\n")
|
|
3675
|
+
|
|
3676
|
+
_install_cli_excepthook(cli)
|
|
3612
3677
|
|
|
3613
|
-
setup_done = {"value": False}
|
|
3614
3678
|
last_logged = {"t": 0.0}
|
|
3615
3679
|
|
|
3616
3680
|
# ------------------------------------------------------------------
|
|
@@ -3641,7 +3705,7 @@ def _start_cli_tracker(
|
|
|
3641
3705
|
# next window. Between windows, the training loop runs at native,
|
|
3642
3706
|
# untraced speed.
|
|
3643
3707
|
_CAPTURE_SPAN = min(0.05, throttle_interval / 4 if throttle_interval > 0 else 0.05)
|
|
3644
|
-
window = {"open": True, "closes_at": 0.0} # start open so
|
|
3708
|
+
window = {"open": True, "closes_at": 0.0} # start open so the first snapshot can run immediately
|
|
3645
3709
|
|
|
3646
3710
|
def _close_window():
|
|
3647
3711
|
window["open"] = False
|
|
@@ -3663,6 +3727,11 @@ def _start_cli_tracker(
|
|
|
3663
3727
|
if window["open"]:
|
|
3664
3728
|
_close_window()
|
|
3665
3729
|
|
|
3730
|
+
# Setup now runs eagerly, before tracing even starts (see above), so
|
|
3731
|
+
# the ticker can start right away too -- no need to wait for the
|
|
3732
|
+
# tracer to see a live frame first.
|
|
3733
|
+
threading.Thread(target=_ticker, daemon=True).start()
|
|
3734
|
+
|
|
3666
3735
|
def cli_tracer(frame, event, arg):
|
|
3667
3736
|
filename = frame.f_code.co_filename
|
|
3668
3737
|
if _is_library_frame(filename):
|
|
@@ -3709,21 +3778,6 @@ def _start_cli_tracker(
|
|
|
3709
3778
|
):
|
|
3710
3779
|
cli.var_states[name] = "track"
|
|
3711
3780
|
|
|
3712
|
-
if not setup_done["value"]:
|
|
3713
|
-
has_resolved_shape = any(shape is not None for shape in cli.discovered.values())
|
|
3714
|
-
has_trackable_local = any(
|
|
3715
|
-
not n.startswith("__") and is_trackable(v) for n, v in local_vars.items()
|
|
3716
|
-
)
|
|
3717
|
-
if has_resolved_shape or has_trackable_local:
|
|
3718
|
-
cli.watch_locals = local_vars
|
|
3719
|
-
cli.interactive_setup()
|
|
3720
|
-
setup_done["value"] = True
|
|
3721
|
-
# Setup is interactive/blocking; once it's done, start the
|
|
3722
|
-
# background ticker that opens/closes future windows. (Kept
|
|
3723
|
-
# off until now so setup itself isn't racing a window close.)
|
|
3724
|
-
threading.Thread(target=_ticker, daemon=True).start()
|
|
3725
|
-
return cli_tracer
|
|
3726
|
-
|
|
3727
3781
|
cli.watch_locals = local_vars
|
|
3728
3782
|
|
|
3729
3783
|
now = time.time()
|
|
@@ -184,6 +184,68 @@ _YELLOW = "\033[93m"
|
|
|
184
184
|
_PULSE_RE = re.compile(r"Pulse(?:\s(?:CLI|AI))?")
|
|
185
185
|
|
|
186
186
|
|
|
187
|
+
# ----------------------------------------------------------------------------
|
|
188
|
+
# Terms of Service -- the actual license Pulse ships under (mirrors the
|
|
189
|
+
# LICENSE file in the Pulse GitHub repo). Shown in full at sign-up (see
|
|
190
|
+
# _prompt_tos_acceptance) and its acceptance is what cloud.record_tos_
|
|
191
|
+
# acceptance timestamps server-side. PULSE_TOS_URL, if set, is shown
|
|
192
|
+
# alongside this as a link to the canonical hosted copy -- update that env
|
|
193
|
+
# var (and this text, if the license is ever revised) together so the two
|
|
194
|
+
# never drift apart.
|
|
195
|
+
# ----------------------------------------------------------------------------
|
|
196
|
+
PULSE_LICENSE_TEXT = """\
|
|
197
|
+
PULSE PROPRIETARY SOFTWARE LICENSE
|
|
198
|
+
|
|
199
|
+
Copyright (c) 2026 Yash Patel. All Rights Reserved.
|
|
200
|
+
|
|
201
|
+
This software and associated files (the "Software") are the proprietary
|
|
202
|
+
property of the copyright holder. The Software is licensed, not sold.
|
|
203
|
+
|
|
204
|
+
1. GRANT OF LICENSE
|
|
205
|
+
Subject to the terms of this license, the copyright holder grants you a
|
|
206
|
+
limited, non-exclusive, non-transferable, revocable license to install
|
|
207
|
+
and run the Software for your own internal use.
|
|
208
|
+
|
|
209
|
+
2. RESTRICTIONS
|
|
210
|
+
Except as expressly permitted above, you may NOT, and may not permit
|
|
211
|
+
others to:
|
|
212
|
+
(a) copy, reproduce, distribute, sublicense, rent, lease, or resell the
|
|
213
|
+
Software;
|
|
214
|
+
(b) modify, adapt, translate, or create derivative works based on the
|
|
215
|
+
Software;
|
|
216
|
+
(c) reverse engineer, decompile, disassemble, or otherwise attempt to
|
|
217
|
+
derive the source code of the Software, except to the extent this
|
|
218
|
+
restriction is prohibited by applicable law;
|
|
219
|
+
(d) remove, obscure, or alter any proprietary notices on the Software;
|
|
220
|
+
(e) use the Software to build a competing product or service.
|
|
221
|
+
|
|
222
|
+
3. FUTURE PAID TERMS
|
|
223
|
+
The copyright holder reserves the right to change pricing, introduce
|
|
224
|
+
paid tiers or license keys, limit functionality, or discontinue free
|
|
225
|
+
distribution of future versions at any time. Continued use of any
|
|
226
|
+
future version may be subject to additional terms, including payment.
|
|
227
|
+
|
|
228
|
+
4. NO WARRANTY
|
|
229
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS
|
|
230
|
+
OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
|
231
|
+
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE, AND
|
|
232
|
+
NONINFRINGEMENT.
|
|
233
|
+
|
|
234
|
+
5. LIMITATION OF LIABILITY
|
|
235
|
+
IN NO EVENT SHALL THE COPYRIGHT HOLDER BE LIABLE FOR ANY CLAIM,
|
|
236
|
+
DAMAGES, OR OTHER LIABILITY ARISING FROM, OUT OF, OR IN CONNECTION
|
|
237
|
+
WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
|
238
|
+
|
|
239
|
+
6. TERMINATION
|
|
240
|
+
This license is effective until terminated. It will terminate
|
|
241
|
+
automatically without notice if you fail to comply with any of its
|
|
242
|
+
terms. Upon termination, you must cease all use of the Software and
|
|
243
|
+
destroy all copies in your possession.
|
|
244
|
+
|
|
245
|
+
For licensing inquiries, contact: codeyash09@gmail.com
|
|
246
|
+
"""
|
|
247
|
+
|
|
248
|
+
|
|
187
249
|
def _highlight_pulse(text: str, base_color: Optional[str] = None) -> str:
|
|
188
250
|
"""Color every occurrence of 'Pulse' (and 'Pulse CLI'/'Pulse AI') orange,
|
|
189
251
|
resuming `base_color` afterward so nesting inside a red/blue line works."""
|
|
@@ -297,7 +359,16 @@ SYSTEM_PROMPT = (
|
|
|
297
359
|
"of training, adjust the dial so future auto-intervention checks match reality instead of "
|
|
298
360
|
"either missing real trouble or crying wolf on normal noise. Only send this when you have an "
|
|
299
361
|
"actual reason from the data, not by default.\n"
|
|
300
|
-
"
|
|
362
|
+
" NORMAL_START: <comma-separated var=value pairs, or 'none'>\n"
|
|
363
|
+
" Spike detection normally needs several real data points before it has a baseline to "
|
|
364
|
+
"compare against, which means a genuine explosion in the first few steps of training can go "
|
|
365
|
+
"undetected. If you can estimate a loss-like tracked variable's expected value at the very "
|
|
366
|
+
"start of training from the code alone (e.g. a randomly-initialized N-class classifier's "
|
|
367
|
+
"cross-entropy loss starts near ln(N); a policy's initial reward is often near a known "
|
|
368
|
+
"random-policy baseline), send it here so Pulse can catch a real explosion from step one "
|
|
369
|
+
"instead of waiting for history to accumulate. Only estimate what you can actually justify "
|
|
370
|
+
"from the code -- omit a variable (or send 'none') rather than guess.\n"
|
|
371
|
+
" Put CALC:/PROMOTE:/GPUTRACK:/GPUUNTRACK:/SENSITIVITY:/NORMAL_START: lines anywhere in your Reasoning, not in the "
|
|
301
372
|
"Diagnosis or Fix.\n\n"
|
|
302
373
|
"PERIODIC GPU CHECK-IN:\n"
|
|
303
374
|
"Every 15 minutes after Pulse starts, if you have an AI provider configured, Pulse will send "
|
|
@@ -653,8 +724,20 @@ class PulseCLI:
|
|
|
653
724
|
# one signal instead of moving the whole dial.
|
|
654
725
|
self.explosion_multiplier: Optional[float] = None # loss spike: latest > baseline * N
|
|
655
726
|
self.plateau_range_frac: Optional[float] = None # plateau: window range < frac * |latest|
|
|
656
|
-
self.oscillation_flip_threshold: Optional[int] = None # oscillation: sign flips required (of
|
|
657
|
-
self.oscillation_delta_frac: Optional[float] = None # oscillation: avg |delta| > frac *
|
|
727
|
+
self.oscillation_flip_threshold: Optional[int] = None # oscillation: sign flips required (of 18)
|
|
728
|
+
self.oscillation_delta_frac: Optional[float] = None # oscillation: avg |delta| > frac * scale
|
|
729
|
+
self.stagnation_frac: Optional[float] = None # stagnation: long-window improvement < frac * |early mean|
|
|
730
|
+
# Agent-estimated (from reading the code, via a NORMAL_START:
|
|
731
|
+
# directive -- see _apply_directives) expected starting value for
|
|
732
|
+
# a loss-like tracked variable. Explosion detection normally needs
|
|
733
|
+
# several real finite data points before it has anything to
|
|
734
|
+
# compare `latest` against, which means a genuine blow-up in the
|
|
735
|
+
# first few steps of training can slip through undetected -- this
|
|
736
|
+
# gives it a code-derived anchor to compare against from step one,
|
|
737
|
+
# before real history exists. See _prime_at_start, which asks the
|
|
738
|
+
# agent for this automatically, once, before the first step.
|
|
739
|
+
self._normal_start_baselines: Dict[str, float] = {}
|
|
740
|
+
self._start_primed: bool = False # _prime_at_start runs at most once per process
|
|
658
741
|
self._last_intervention_signature: Optional[str] = None
|
|
659
742
|
# /code is on by default -- every manually-asked question includes
|
|
660
743
|
# the training code (and any cross-file context) unless turned off.
|
|
@@ -1480,6 +1563,39 @@ class PulseCLI:
|
|
|
1480
1563
|
# deliberately NOT marked dirty here -- it hasn't changed
|
|
1481
1564
|
# (see above), so there's nothing to re-send.
|
|
1482
1565
|
self._cloud_dirty_fields.update({"uptime_seconds", "downtime_seconds"})
|
|
1566
|
+
# Preload this session's already-synced history so THIS
|
|
1567
|
+
# process's flushes extend it instead of quietly replacing
|
|
1568
|
+
# it. self._agent_logs/_error_tracebacks/_telemetry/
|
|
1569
|
+
# _incidents all start empty in __init__ -- with nothing
|
|
1570
|
+
# more done, the first flush from here would PATCH the
|
|
1571
|
+
# whole array column (see _build_cloud_patch_body) down to
|
|
1572
|
+
# just whatever this process adds, erasing every entry
|
|
1573
|
+
# logged before the restart even though it's still sitting
|
|
1574
|
+
# in local memory of a process that's gone. Best-effort:
|
|
1575
|
+
# if this fetch fails (offline blip, etc.) we still carry
|
|
1576
|
+
# on -- worst case is the pre-restart history is briefly
|
|
1577
|
+
# at risk on the next flush rather than the run refusing
|
|
1578
|
+
# to continue.
|
|
1579
|
+
try:
|
|
1580
|
+
existing = cloud.fetch_debug_session(self.debug_session_id)
|
|
1581
|
+
except cloud.SupabaseError:
|
|
1582
|
+
existing = None
|
|
1583
|
+
if existing:
|
|
1584
|
+
self._agent_logs = cloud.decode_entries(existing.get("agent_logs"))
|
|
1585
|
+
self._error_tracebacks = cloud.decode_entries(existing.get("error_tracebacks"))
|
|
1586
|
+
self._telemetry = cloud.decode_entries(existing.get("telemetry"))
|
|
1587
|
+
self._incidents = cloud.decode_entries(existing.get("incidents"))
|
|
1588
|
+
cprint(
|
|
1589
|
+
f"[Pulse] Loaded {len(self._agent_logs)} agent log(s), {len(self._incidents)} "
|
|
1590
|
+
f"incident(s) from before the restart -- history preserved."
|
|
1591
|
+
)
|
|
1592
|
+
else:
|
|
1593
|
+
cprint(
|
|
1594
|
+
"[Pulse] ⚠ Could not load this session's history before the restart -- "
|
|
1595
|
+
"new entries will be appended locally, but the next sync may not include "
|
|
1596
|
+
"everything logged before the restart.",
|
|
1597
|
+
color=_YELLOW,
|
|
1598
|
+
)
|
|
1483
1599
|
cprint(f"[Pulse] Resumed git commit {sha[:10]}… -- carried over, not re-detected (see /commit to update it manually).")
|
|
1484
1600
|
cprint(f"[Pulse] Resumed debug session (id={self.debug_session_id[:8]}…) after restart -- uptime/downtime counters carried over.")
|
|
1485
1601
|
else:
|
|
@@ -1608,6 +1724,17 @@ class PulseCLI:
|
|
|
1608
1724
|
self.email = env_user
|
|
1609
1725
|
token = cloud.attach_session_token(self.user_id)
|
|
1610
1726
|
cloud.save_cached_credentials(self.user_id, self.email, self.team_id, session_token=token)
|
|
1727
|
+
except cloud.SupabaseEmailConfirmationRequired:
|
|
1728
|
+
# Account created, but this deployment requires email
|
|
1729
|
+
# confirmation before a session can be issued -- an
|
|
1730
|
+
# unattended job can't click that link, so there's
|
|
1731
|
+
# nothing more to do here this run. Not a "login
|
|
1732
|
+
# failed" -- just fall through to local (no-cloud) mode
|
|
1733
|
+
# for this run and let a human confirm and re-run.
|
|
1734
|
+
cprint(
|
|
1735
|
+
f"[Pulse] Confirm your account: check {env_user}'s email for a confirmation link, "
|
|
1736
|
+
"then log in. Continuing this run in local (no-cloud) mode.",
|
|
1737
|
+
)
|
|
1611
1738
|
except cloud.SupabaseError as exc:
|
|
1612
1739
|
cprint(f"[Pulse] ⚠ Non-interactive login failed ({exc}). Continuing in local (no-cloud) mode.", color=_RED)
|
|
1613
1740
|
else:
|
|
@@ -1624,31 +1751,39 @@ class PulseCLI:
|
|
|
1624
1751
|
_flush_stdin()
|
|
1625
1752
|
cprint("\n--- Pulse Cloud Authentication ---")
|
|
1626
1753
|
resp = input(
|
|
1627
|
-
"[Pulse]
|
|
1628
|
-
).strip().lower()
|
|
1754
|
+
"[Pulse] (s)ign up [default] / (l)og in / (r)ecover account > "
|
|
1755
|
+
).strip().lower() or "s"
|
|
1629
1756
|
|
|
1630
1757
|
if resp in ("s", "signup", "sign up"):
|
|
1631
1758
|
_flush_stdin()
|
|
1632
|
-
|
|
1633
|
-
|
|
1634
|
-
|
|
1635
|
-
|
|
1636
|
-
|
|
1637
|
-
|
|
1638
|
-
|
|
1639
|
-
|
|
1640
|
-
|
|
1641
|
-
|
|
1642
|
-
|
|
1643
|
-
|
|
1644
|
-
|
|
1645
|
-
|
|
1646
|
-
|
|
1647
|
-
|
|
1648
|
-
|
|
1649
|
-
|
|
1650
|
-
|
|
1651
|
-
|
|
1759
|
+
# Retry only the field that actually failed (email or
|
|
1760
|
+
# password) instead of bouncing back to the top-level
|
|
1761
|
+
# "sign up / log in / recover" menu on every typo -- that
|
|
1762
|
+
# used to mean a single password mismatch cost you a full
|
|
1763
|
+
# re-type of the email address too.
|
|
1764
|
+
while True:
|
|
1765
|
+
email = input("email (e.g. name@example.com, min. 8 characters) > ").strip()
|
|
1766
|
+
if not email:
|
|
1767
|
+
cprint("[Pulse] email cannot be blank.", color=_RED)
|
|
1768
|
+
continue
|
|
1769
|
+
if not cloud.is_valid_email(email):
|
|
1770
|
+
cprint("[Pulse] enter a valid email address (min. 8 characters), e.g. name@example.com.", color=_RED)
|
|
1771
|
+
continue
|
|
1772
|
+
break
|
|
1773
|
+
while True:
|
|
1774
|
+
password = getpass.getpass("Password (min. 8 characters) > ")
|
|
1775
|
+
if len(password) < 8:
|
|
1776
|
+
cprint("[Pulse] Password must be at least 8 characters.", color=_RED)
|
|
1777
|
+
continue
|
|
1778
|
+
confirm_password = getpass.getpass("Confirm password > ")
|
|
1779
|
+
if password != confirm_password:
|
|
1780
|
+
# A typo here with no confirmation step means signing up
|
|
1781
|
+
# with a password the user doesn't actually know -- and
|
|
1782
|
+
# with no email on file yet at this point, there'd be no
|
|
1783
|
+
# way back in. Re-prompt for password only, not email.
|
|
1784
|
+
cprint("[Pulse] Passwords didn't match. Let's try again.", color=_RED)
|
|
1785
|
+
continue
|
|
1786
|
+
break
|
|
1652
1787
|
accepted = self._prompt_tos_acceptance()
|
|
1653
1788
|
if not accepted:
|
|
1654
1789
|
cprint("[Pulse] Sign up cancelled -- acceptance is required to create an account.")
|
|
@@ -1664,6 +1799,38 @@ class PulseCLI:
|
|
|
1664
1799
|
cloud.save_cached_credentials(self.user_id, self.email, self.team_id, session_token=token)
|
|
1665
1800
|
self._show_recovery_code(cloud.attach_recovery_code(self.user_id))
|
|
1666
1801
|
return
|
|
1802
|
+
except cloud.SupabaseEmailConfirmationRequired:
|
|
1803
|
+
# The account WAS created -- this isn't a failure. The
|
|
1804
|
+
# OLD behavior here just printed a message and fell
|
|
1805
|
+
# back to the top-level menu, which meant: leave the
|
|
1806
|
+
# CLI, go confirm in email, come back, remember to
|
|
1807
|
+
# pick "log in" instead of "sign up", and retype the
|
|
1808
|
+
# email+password you just typed 10 seconds ago. That's
|
|
1809
|
+
# exactly the kind of drop-off point that loses a
|
|
1810
|
+
# just-converted cold-email lead. Instead, stay in
|
|
1811
|
+
# this flow and retry with the SAME credentials
|
|
1812
|
+
# already in memory -- the only thing left to do is
|
|
1813
|
+
# click the link and hit Enter.
|
|
1814
|
+
cprint("[Pulse] Almost there -- check your email for a confirmation link and click it.")
|
|
1815
|
+
while True:
|
|
1816
|
+
_flush_stdin()
|
|
1817
|
+
resp2 = input(
|
|
1818
|
+
"Press Enter once confirmed to continue (or type 'skip' to do this later) > "
|
|
1819
|
+
).strip().lower()
|
|
1820
|
+
if resp2 == "skip":
|
|
1821
|
+
cprint("[Pulse] No problem -- run Pulse again and choose (l)og in once you've confirmed.")
|
|
1822
|
+
break
|
|
1823
|
+
try:
|
|
1824
|
+
user = cloud.log_in(email, password)
|
|
1825
|
+
except cloud.SupabaseError as exc2:
|
|
1826
|
+
cprint(f"[Pulse] Not confirmed yet ({exc2}). Check your email and try again.", color=_YELLOW)
|
|
1827
|
+
continue
|
|
1828
|
+
cprint(f"[Pulse] ✓ Confirmed! Signed in as {email}.")
|
|
1829
|
+
self.user_id = user["id"]
|
|
1830
|
+
self.email = email
|
|
1831
|
+
token = cloud.attach_session_token(self.user_id)
|
|
1832
|
+
cloud.save_cached_credentials(self.user_id, self.email, self.team_id, session_token=token)
|
|
1833
|
+
return
|
|
1667
1834
|
except cloud.SupabaseError as exc:
|
|
1668
1835
|
cprint(f"[Pulse] ⚠ Sign up failed: {exc}", color=_RED)
|
|
1669
1836
|
|
|
@@ -1735,19 +1902,43 @@ class PulseCLI:
|
|
|
1735
1902
|
self.email = None
|
|
1736
1903
|
|
|
1737
1904
|
def _prompt_tos_acceptance(self) -> bool:
|
|
1738
|
-
"""
|
|
1739
|
-
|
|
1740
|
-
|
|
1741
|
-
|
|
1742
|
-
|
|
1743
|
-
|
|
1744
|
-
|
|
1745
|
-
|
|
1746
|
-
|
|
1905
|
+
"""Show a short summary of the Pulse license up front (full text
|
|
1906
|
+
is PULSE_LICENSE_TEXT, mirrored from the LICENSE file in the
|
|
1907
|
+
Pulse GitHub repo, and is always one keystroke away via 'full')
|
|
1908
|
+
and require typing 'yes' to accept before an account is created.
|
|
1909
|
+
|
|
1910
|
+
The full ~50-line license used to be printed unconditionally
|
|
1911
|
+
before every sign-up -- legally fine, but it's exactly the kind
|
|
1912
|
+
of wall of text that makes someone who just clicked through from
|
|
1913
|
+
a cold email bail before ever seeing the product. A short,
|
|
1914
|
+
scannable summary gets the same acceptance (still gated on
|
|
1915
|
+
actually typing 'yes', still timestamped server-side by
|
|
1916
|
+
cloud.record_tos_acceptance right after this returns True) without
|
|
1917
|
+
that drop-off risk, while anyone who wants the full text still
|
|
1918
|
+
gets it by typing 'full'.
|
|
1919
|
+
|
|
1920
|
+
PULSE_PRIVACY_URL, if set, is shown alongside it for a separate
|
|
1921
|
+
privacy policy (this license covers software use, not data
|
|
1922
|
+
handling) -- point that env var at your actual Privacy Policy
|
|
1923
|
+
before relying on this for anything.
|
|
1924
|
+
"""
|
|
1747
1925
|
privacy_url = os.environ.get("PULSE_PRIVACY_URL", "(set PULSE_PRIVACY_URL to link your Privacy Policy here)")
|
|
1748
|
-
|
|
1926
|
+
tos_url = os.environ.get("PULSE_TOS_URL")
|
|
1927
|
+
print(
|
|
1928
|
+
"\nQuick terms before you create an account:\n"
|
|
1929
|
+
" • Pulse is proprietary software, licensed (not sold) for your own use.\n"
|
|
1930
|
+
" • No redistributing, reselling, or reverse-engineering it, and no using it to build a competing product.\n"
|
|
1931
|
+
" • Provided as-is, no warranty -- pricing/paid tiers may be introduced for future versions.\n"
|
|
1932
|
+
)
|
|
1933
|
+
if tos_url:
|
|
1934
|
+
print(f"Full license: {tos_url}")
|
|
1935
|
+
print(f"Privacy Policy: {privacy_url}\n")
|
|
1749
1936
|
_flush_stdin()
|
|
1750
|
-
resp = input("Type 'yes' to accept
|
|
1937
|
+
resp = input("Type 'yes' to accept (or 'full' to read the complete license first) > ").strip().lower()
|
|
1938
|
+
if resp == "full":
|
|
1939
|
+
print(f"\n{PULSE_LICENSE_TEXT}")
|
|
1940
|
+
_flush_stdin()
|
|
1941
|
+
resp = input("Type 'yes' to accept and create your account > ").strip().lower()
|
|
1751
1942
|
return resp == "yes"
|
|
1752
1943
|
|
|
1753
1944
|
def _show_recovery_code(self, code: Optional[str]) -> None:
|
|
@@ -1878,6 +2069,28 @@ class PulseCLI:
|
|
|
1878
2069
|
cprint("[Pulse] Non-interactive mode and no unambiguous workspace to pick -- continuing without a team.")
|
|
1879
2070
|
return
|
|
1880
2071
|
|
|
2072
|
+
# A brand-new sign-up (or anyone who currently belongs to zero
|
|
2073
|
+
# workspaces) has nothing to actually pick between here -- the
|
|
2074
|
+
# old behavior forced everyone through this menu regardless, with
|
|
2075
|
+
# no Enter-key default, meaning a just-signed-up trial user's very
|
|
2076
|
+
# next required action was "type c, then optionally type a GitHub
|
|
2077
|
+
# repo URL" before they could do anything else. Skip straight to
|
|
2078
|
+
# a silently auto-created personal workspace instead; joining a
|
|
2079
|
+
# teammate's workspace with a code is one command away (/repo can
|
|
2080
|
+
# set a repo on it later, too), but creating your own shouldn't
|
|
2081
|
+
# need a prompt at all when there's nothing else it could be.
|
|
2082
|
+
if not existing:
|
|
2083
|
+
try:
|
|
2084
|
+
team = cloud.create_team(self.user_id, repo=cloud.git_remote_url(self._repo_cwd), cwd=self._repo_cwd)
|
|
2085
|
+
self.team_id = team["team_id"]
|
|
2086
|
+
self.team_join_code = team.get("join_code")
|
|
2087
|
+
self.team_admin_ids = list(team.get("admin_ids") or [])
|
|
2088
|
+
cprint(f"[Pulse] ✓ Workspace ready (join code: {self.team_join_code}) -- share this to invite teammates, or /repo to set a repo.")
|
|
2089
|
+
cloud.save_cached_credentials(self.user_id, self.email, self.team_id)
|
|
2090
|
+
except cloud.SupabaseError as exc:
|
|
2091
|
+
cprint(f"[Pulse] ⚠ Could not create a workspace ({exc}) -- continuing without a team. Try /repo or restart to pick one.", color=_RED)
|
|
2092
|
+
return
|
|
2093
|
+
|
|
1881
2094
|
while True:
|
|
1882
2095
|
_flush_stdin()
|
|
1883
2096
|
cprint("\n--- Pulse Workspace ---")
|
|
@@ -2092,6 +2305,66 @@ class PulseCLI:
|
|
|
2092
2305
|
self.team_id = None
|
|
2093
2306
|
self.debug_session_id = None
|
|
2094
2307
|
|
|
2308
|
+
def _cmd_logout(self, arg: str) -> None:
|
|
2309
|
+
"""/logout -- sign out of the account currently in use for this
|
|
2310
|
+
run, then immediately offers to sign back in (as the same account
|
|
2311
|
+
or a different one) without having to restart Pulse. Revokes the
|
|
2312
|
+
server-side session token (so the cached credentials.json on this
|
|
2313
|
+
machine can't silently log back in on its own) and clears the
|
|
2314
|
+
local cache. If a new account signs in, this also re-runs the
|
|
2315
|
+
workspace picker and opens a fresh cloud debug session under that
|
|
2316
|
+
identity -- the old session is flushed and left as-is on the
|
|
2317
|
+
dashboard rather than mixing two identities' data into one row.
|
|
2318
|
+
"""
|
|
2319
|
+
if not self.user_id:
|
|
2320
|
+
cprint("[Pulse] Not signed in to Pulse Cloud -- nothing to log out of.")
|
|
2321
|
+
return
|
|
2322
|
+
if self.non_interactive:
|
|
2323
|
+
cprint("[Pulse] /logout requires an interactive session.")
|
|
2324
|
+
return
|
|
2325
|
+
|
|
2326
|
+
old_email = self.email
|
|
2327
|
+
|
|
2328
|
+
# Flush anything still pending for the CURRENT session/identity
|
|
2329
|
+
# before switching -- otherwise a dirty field written after
|
|
2330
|
+
# logout could still get attributed to the old debug session.
|
|
2331
|
+
self._flush_cloud_now()
|
|
2332
|
+
|
|
2333
|
+
cloud.revoke_session_token(self.user_id) # best-effort; never raises
|
|
2334
|
+
cloud.clear_cached_credentials()
|
|
2335
|
+
self.user_id = None
|
|
2336
|
+
self.email = None
|
|
2337
|
+
self.team_id = None
|
|
2338
|
+
self.team_admin_ids = []
|
|
2339
|
+
self.debug_session_id = None
|
|
2340
|
+
cprint(f"[Pulse] ✓ Logged out of {old_email}.")
|
|
2341
|
+
|
|
2342
|
+
try:
|
|
2343
|
+
self._auth_flow()
|
|
2344
|
+
except cloud.SupabaseError as exc:
|
|
2345
|
+
cprint(f"[Pulse] ⚠ Could not sign in to Pulse cloud, continuing locally: {exc}", color=_RED)
|
|
2346
|
+
return
|
|
2347
|
+
|
|
2348
|
+
if not self.user_id:
|
|
2349
|
+
cprint("[Pulse] Continuing this run in local (no-cloud) mode.")
|
|
2350
|
+
return
|
|
2351
|
+
|
|
2352
|
+
try:
|
|
2353
|
+
self._team_flow()
|
|
2354
|
+
except cloud.SupabaseError as exc:
|
|
2355
|
+
cprint(f"[Pulse] ⚠ Team setup failed, continuing without a team: {exc}", color=_RED)
|
|
2356
|
+
|
|
2357
|
+
try:
|
|
2358
|
+
sha = self._last_synced_commit_sha or cloud.current_git_commit_sha(self._repo_cwd) or "unknown"
|
|
2359
|
+
self.debug_session_id = cloud.create_debug_session(self.team_id, self.user_id, git_commit_sha=sha)
|
|
2360
|
+
self._last_synced_commit_sha = sha
|
|
2361
|
+
if self.debug_session_id:
|
|
2362
|
+
cprint(f"[Pulse] Debug session started (id={self.debug_session_id}, commit={sha[:10] if sha != 'unknown' else 'unknown'}).")
|
|
2363
|
+
atexit.register(self._flush_cloud_now)
|
|
2364
|
+
self._last_cloud_flush = time.monotonic()
|
|
2365
|
+
except cloud.SupabaseError as exc:
|
|
2366
|
+
cprint(f"[Pulse] ⚠ Could not start a cloud debug session, continuing locally: {exc}", color=_RED)
|
|
2367
|
+
|
|
2095
2368
|
def _cmd_webhook(self, arg: str) -> None:
|
|
2096
2369
|
"""/webhook set <url> | /webhook test | /webhook off | /webhook
|
|
2097
2370
|
-- a Slack-incoming-webhook-compatible URL that gets a message
|
|
@@ -2444,6 +2717,7 @@ class PulseCLI:
|
|
|
2444
2717
|
("/code", "toggle whether questions include your training code"),
|
|
2445
2718
|
("/password", "change your Pulse account password"),
|
|
2446
2719
|
("/recover", "reset a forgotten password with a recovery code"),
|
|
2720
|
+
("/logout", "sign out, then sign back in as the same or a different account"),
|
|
2447
2721
|
]),
|
|
2448
2722
|
("Cloud & team", [
|
|
2449
2723
|
("/cloud", "show sign-in, workspace, and sync status at a glance"),
|
|
@@ -3069,19 +3343,59 @@ class PulseCLI:
|
|
|
3069
3343
|
# to route around.
|
|
3070
3344
|
argv = [python_exe, script_path] + sys.argv[1:]
|
|
3071
3345
|
|
|
3072
|
-
|
|
3073
|
-
|
|
3074
|
-
|
|
3075
|
-
|
|
3076
|
-
|
|
3077
|
-
|
|
3078
|
-
|
|
3346
|
+
# Retry the restart itself instead of ever falling back to "keep
|
|
3347
|
+
# running the old, already-in-memory process" on a bad exit code.
|
|
3348
|
+
# A nonzero exit here almost always means the replacement process
|
|
3349
|
+
# crashed immediately (e.g. the agent's fix didn't fully fix it,
|
|
3350
|
+
# or introduced a new bug) -- silently resuming the OLD code path
|
|
3351
|
+
# used to look like a safe fallback, but for an unsupervised run
|
|
3352
|
+
# it just means the SAME already-crashed-once code keeps limping
|
|
3353
|
+
# along, or the loop quietly stalls, with nobody watching to
|
|
3354
|
+
# notice. Retrying the launch instead gives a transient problem
|
|
3355
|
+
# (e.g. a port/file briefly locked by the process that's still
|
|
3356
|
+
# exiting, a flaky import) a real chance to clear, and gives the
|
|
3357
|
+
# agent's fix -- which is already saved to disk -- more chances to
|
|
3358
|
+
# actually take effect. Attempts are capped (not infinite) so a
|
|
3359
|
+
# deterministically-broken script can't spin forever burning
|
|
3360
|
+
# compute/cost unattended; MAX_RESTART_ATTEMPTS is the one knob to
|
|
3361
|
+
# raise if that cap is ever too low for a given job.
|
|
3362
|
+
MAX_RESTART_ATTEMPTS = 5
|
|
3363
|
+
RETRY_BACKOFF_SECONDS = 3 # multiplied by attempt number, capped below
|
|
3364
|
+
|
|
3365
|
+
attempt = 0
|
|
3366
|
+
while True:
|
|
3367
|
+
attempt += 1
|
|
3368
|
+
try:
|
|
3369
|
+
# Synchronous run keeps stdin attached and handles spaces in paths correctly on Windows
|
|
3370
|
+
result = subprocess.run(argv)
|
|
3371
|
+
except Exception as exc:
|
|
3372
|
+
cprint(f"[Pulse] ⚠ Restart attempt {attempt}/{MAX_RESTART_ATTEMPTS} failed to launch ({exc}).", color=_RED)
|
|
3373
|
+
if attempt >= MAX_RESTART_ATTEMPTS:
|
|
3374
|
+
cprint(
|
|
3375
|
+
f"[Pulse] ⚠ Giving up after {MAX_RESTART_ATTEMPTS} restart attempts -- "
|
|
3376
|
+
"continuing current run with the old in-memory code.",
|
|
3377
|
+
color=_RED,
|
|
3378
|
+
)
|
|
3379
|
+
self._log_incident("restart_failed", f"Could not launch replacement process after {attempt} attempts: {exc}")
|
|
3380
|
+
return
|
|
3381
|
+
time.sleep(min(RETRY_BACKOFF_SECONDS * attempt, 30))
|
|
3382
|
+
continue
|
|
3079
3383
|
|
|
3080
|
-
|
|
3081
|
-
|
|
3082
|
-
|
|
3083
|
-
|
|
3084
|
-
|
|
3384
|
+
if result.returncode == 0:
|
|
3385
|
+
sys.exit(0)
|
|
3386
|
+
|
|
3387
|
+
message = f"Replacement training process exited with code {result.returncode} (attempt {attempt}/{MAX_RESTART_ATTEMPTS})"
|
|
3388
|
+
if attempt >= MAX_RESTART_ATTEMPTS:
|
|
3389
|
+
cprint(
|
|
3390
|
+
f"[Pulse] ⚠ {message}. Giving up after {MAX_RESTART_ATTEMPTS} attempts -- "
|
|
3391
|
+
"continuing current run with the old in-memory code.",
|
|
3392
|
+
color=_RED,
|
|
3393
|
+
)
|
|
3394
|
+
self._log_incident("restart_failed", message)
|
|
3395
|
+
return
|
|
3396
|
+
|
|
3397
|
+
cprint(f"[Pulse] ⚠ {message}. Retrying restart...", color=_YELLOW)
|
|
3398
|
+
time.sleep(min(RETRY_BACKOFF_SECONDS * attempt, 30))
|
|
3085
3399
|
|
|
3086
3400
|
def _print_variable_summary(self) -> None:
|
|
3087
3401
|
variables = self.discover_variables()
|
|
@@ -3448,18 +3762,21 @@ class PulseCLI:
|
|
|
3448
3762
|
_GPUTRACK_RE = re.compile(r"^\s*GPUTRACK:\s*(.+)$", re.MULTILINE)
|
|
3449
3763
|
_GPUUNTRACK_RE = re.compile(r"^\s*GPUUNTRACK:\s*(.+)$", re.MULTILINE)
|
|
3450
3764
|
_SENSITIVITY_RE = re.compile(r"^\s*SENSITIVITY:\s*(.+)$", re.MULTILINE)
|
|
3765
|
+
_NORMAL_START_RE = re.compile(r"^\s*NORMAL_START:\s*(.+)$", re.MULTILINE)
|
|
3451
3766
|
|
|
3452
3767
|
@classmethod
|
|
3453
3768
|
def _extract_directives(cls, text: str):
|
|
3454
|
-
"""Pull CALC:/PROMOTE:/GPUTRACK:/GPUUNTRACK:/SENSITIVITY
|
|
3455
|
-
of an agent response, returning
|
|
3456
|
-
promote_names, gputrack_names,
|
|
3457
|
-
sensitivity_args). Cleaned
|
|
3458
|
-
don't clutter what's
|
|
3459
|
-
|
|
3460
|
-
|
|
3461
|
-
|
|
3462
|
-
|
|
3769
|
+
"""Pull CALC:/PROMOTE:/GPUTRACK:/GPUUNTRACK:/SENSITIVITY:/
|
|
3770
|
+
NORMAL_START: lines out of an agent response, returning
|
|
3771
|
+
(cleaned_text, calc_exprs, promote_names, gputrack_names,
|
|
3772
|
+
gpuuntrack_names, sensitivity_args, normal_start_args). Cleaned
|
|
3773
|
+
text has those lines stripped so they don't clutter what's
|
|
3774
|
+
printed/stored. A bare 'none' value (as instructed for the
|
|
3775
|
+
periodic GPU check-in reply format) is dropped rather than
|
|
3776
|
+
treated as a variable name. sensitivity_args/normal_start_args
|
|
3777
|
+
are lists of raw argument strings (usually 0 or 1) -- applied via
|
|
3778
|
+
_cmd_sensitivity(..., quiet=True) / the NORMAL_START parsing in
|
|
3779
|
+
_apply_directives, same as a manual /sensitivity.
|
|
3463
3780
|
"""
|
|
3464
3781
|
calc_exprs = [m.strip() for m in cls._CALC_RE.findall(text) if m.strip()]
|
|
3465
3782
|
promote_names = []
|
|
@@ -3476,14 +3793,16 @@ class PulseCLI:
|
|
|
3476
3793
|
n.strip() for n in m.split(",") if n.strip() and n.strip().lower() != "none"
|
|
3477
3794
|
)
|
|
3478
3795
|
sensitivity_args = [m.strip() for m in cls._SENSITIVITY_RE.findall(text) if m.strip()]
|
|
3796
|
+
normal_start_args = [m.strip() for m in cls._NORMAL_START_RE.findall(text) if m.strip()]
|
|
3479
3797
|
|
|
3480
3798
|
cleaned = cls._CALC_RE.sub("", text)
|
|
3481
3799
|
cleaned = cls._PROMOTE_RE.sub("", cleaned)
|
|
3482
3800
|
cleaned = cls._GPUTRACK_RE.sub("", cleaned)
|
|
3483
3801
|
cleaned = cls._GPUUNTRACK_RE.sub("", cleaned)
|
|
3484
3802
|
cleaned = cls._SENSITIVITY_RE.sub("", cleaned)
|
|
3803
|
+
cleaned = cls._NORMAL_START_RE.sub("", cleaned)
|
|
3485
3804
|
cleaned = re.sub(r"\n{3,}", "\n\n", cleaned).strip()
|
|
3486
|
-
return cleaned, calc_exprs, promote_names, gputrack_names, gpuuntrack_names, sensitivity_args
|
|
3805
|
+
return cleaned, calc_exprs, promote_names, gputrack_names, gpuuntrack_names, sensitivity_args, normal_start_args
|
|
3487
3806
|
|
|
3488
3807
|
def _apply_directives(
|
|
3489
3808
|
self,
|
|
@@ -3492,12 +3811,14 @@ class PulseCLI:
|
|
|
3492
3811
|
gputrack_names: Optional[List[str]] = None,
|
|
3493
3812
|
gpuuntrack_names: Optional[List[str]] = None,
|
|
3494
3813
|
sensitivity_args: Optional[List[str]] = None,
|
|
3814
|
+
normal_start_args: Optional[List[str]] = None,
|
|
3495
3815
|
) -> str:
|
|
3496
3816
|
"""Deterministically compute any CALC: expressions and apply any
|
|
3497
|
-
PROMOTE:/GPUTRACK:/GPUUNTRACK:/SENSITIVITY: requests,
|
|
3498
|
-
short human-readable summary to print and to feed
|
|
3499
|
-
agent's own history (so it sees the verified
|
|
3500
|
-
next turn instead of trusting its own
|
|
3817
|
+
PROMOTE:/GPUTRACK:/GPUUNTRACK:/SENSITIVITY:/NORMAL_START: requests,
|
|
3818
|
+
returning a short human-readable summary to print and to feed
|
|
3819
|
+
back into the agent's own history (so it sees the verified
|
|
3820
|
+
numbers/state on the next turn instead of trusting its own
|
|
3821
|
+
arithmetic or memory).
|
|
3501
3822
|
"""
|
|
3502
3823
|
notes = []
|
|
3503
3824
|
|
|
@@ -3550,6 +3871,33 @@ class PulseCLI:
|
|
|
3550
3871
|
print(f" 🎚 Sensitivity adjusted (agent request): {result}")
|
|
3551
3872
|
notes.append(result)
|
|
3552
3873
|
|
|
3874
|
+
if normal_start_args:
|
|
3875
|
+
applied = []
|
|
3876
|
+
for raw in normal_start_args:
|
|
3877
|
+
if raw.strip().lower() == "none":
|
|
3878
|
+
continue
|
|
3879
|
+
for pair in raw.split(","):
|
|
3880
|
+
pair = pair.strip()
|
|
3881
|
+
if not pair or "=" not in pair:
|
|
3882
|
+
continue
|
|
3883
|
+
name, _, val_str = pair.partition("=")
|
|
3884
|
+
name = name.strip()
|
|
3885
|
+
if not name:
|
|
3886
|
+
continue
|
|
3887
|
+
try:
|
|
3888
|
+
val = float(val_str.strip())
|
|
3889
|
+
except ValueError:
|
|
3890
|
+
continue
|
|
3891
|
+
self._normal_start_baselines[name] = val
|
|
3892
|
+
applied.append(f"{name}={val:.4g}")
|
|
3893
|
+
if applied:
|
|
3894
|
+
print(f" 🌱 Seeded starting baseline (agent estimate, from code): {', '.join(applied)}")
|
|
3895
|
+
notes.append(
|
|
3896
|
+
f"Seeded expected starting value(s) from the code: {', '.join(applied)}. "
|
|
3897
|
+
"These act as a spike-detection baseline until real data accumulates, so a "
|
|
3898
|
+
"genuine explosion in the first few steps can still be caught."
|
|
3899
|
+
)
|
|
3900
|
+
|
|
3553
3901
|
return "\n\n".join(notes)
|
|
3554
3902
|
|
|
3555
3903
|
_GPU_CHECKIN_PROMPT = (
|
|
@@ -3597,7 +3945,7 @@ class PulseCLI:
|
|
|
3597
3945
|
# and try again at the next interval.
|
|
3598
3946
|
cprint(f"[Pulse] ⚠ GPU check-in skipped (agent request failed: {exc})", color=_RED)
|
|
3599
3947
|
return
|
|
3600
|
-
_, _calc, _promote, gputrack_names, gpuuntrack_names, _sens = self._extract_directives(answer)
|
|
3948
|
+
_, _calc, _promote, gputrack_names, gpuuntrack_names, _sens, _norm = self._extract_directives(answer)
|
|
3601
3949
|
summary = self._apply_directives([], [], gputrack_names, gpuuntrack_names)
|
|
3602
3950
|
if summary:
|
|
3603
3951
|
self.agent_history.append({"role": "user", "content": prompt})
|
|
@@ -3815,9 +4163,9 @@ class PulseCLI:
|
|
|
3815
4163
|
raw_analysis = self._call_model(
|
|
3816
4164
|
_PASS2_ANALYZE_TMPL.format(regions=regions), max_tokens=700
|
|
3817
4165
|
)
|
|
3818
|
-
analysis, calc_exprs, promote_names, gputrack_names, gpuuntrack_names, sensitivity_args = self._extract_directives(raw_analysis)
|
|
4166
|
+
analysis, calc_exprs, promote_names, gputrack_names, gpuuntrack_names, sensitivity_args, normal_start_args = self._extract_directives(raw_analysis)
|
|
3819
4167
|
print(f"[2] Diagnosis & reasoning\n{analysis}\n")
|
|
3820
|
-
directive_note = self._apply_directives(calc_exprs, promote_names, gputrack_names, gpuuntrack_names, sensitivity_args)
|
|
4168
|
+
directive_note = self._apply_directives(calc_exprs, promote_names, gputrack_names, gpuuntrack_names, sensitivity_args, normal_start_args)
|
|
3821
4169
|
if directive_note:
|
|
3822
4170
|
self.agent_history.append({"role": "user", "content": directive_note})
|
|
3823
4171
|
|
|
@@ -4455,7 +4803,9 @@ class PulseCLI:
|
|
|
4455
4803
|
self.plateau_range_frac if self.plateau_range_frac is not None
|
|
4456
4804
|
else 10 ** (-5 + 3 * s)
|
|
4457
4805
|
),
|
|
4458
|
-
# Loose: needs 15/
|
|
4806
|
+
# Loose: needs 15/18 directional reversals. Tight: only 6.
|
|
4807
|
+
# (A 20-point window yields 19 deltas and only 18 consecutive
|
|
4808
|
+
# delta-pairs to check for a sign flip -- see _check_for_trouble.)
|
|
4459
4809
|
"oscillation_flip_threshold": (
|
|
4460
4810
|
self.oscillation_flip_threshold if self.oscillation_flip_threshold is not None
|
|
4461
4811
|
else round(15 - 9 * s)
|
|
@@ -4466,6 +4816,26 @@ class PulseCLI:
|
|
|
4466
4816
|
self.oscillation_delta_frac if self.oscillation_delta_frac is not None
|
|
4467
4817
|
else 0.10 - 0.08 * s
|
|
4468
4818
|
),
|
|
4819
|
+
# Stagnation: has a loss-like variable meaningfully improved
|
|
4820
|
+
# over a LONG window, once normal per-step noise is averaged
|
|
4821
|
+
# out? This is deliberately separate from "plateau" above --
|
|
4822
|
+
# plateau looks at the *range* of the last 20 steps, which a
|
|
4823
|
+
# loss with completely normal noise (bouncing around by, say,
|
|
4824
|
+
# 0.02 every step while never actually trending down) will
|
|
4825
|
+
# never look "flat enough" to trigger, even after a thousand
|
|
4826
|
+
# steps of zero real progress. Stagnation instead compares
|
|
4827
|
+
# the mean of the first quarter of a long window against the
|
|
4828
|
+
# mean of the last quarter, so per-step noise washes out and
|
|
4829
|
+
# only genuine lack of improvement is left.
|
|
4830
|
+
# Loose (s=0): needs 500 steps of history, and even a 0.3%
|
|
4831
|
+
# improvement over that window counts as "still improving".
|
|
4832
|
+
# Tight (s=1): needs only 150 steps, and demands a full 8%
|
|
4833
|
+
# improvement before it stops flagging.
|
|
4834
|
+
"stagnation_window": round(500 - 350 * s),
|
|
4835
|
+
"stagnation_frac": (
|
|
4836
|
+
self.stagnation_frac if self.stagnation_frac is not None
|
|
4837
|
+
else 0.003 + 0.077 * s
|
|
4838
|
+
),
|
|
4469
4839
|
}
|
|
4470
4840
|
|
|
4471
4841
|
def _cmd_sensitivity(self, arg: str, quiet: bool = False) -> Optional[str]:
|
|
@@ -4486,10 +4856,11 @@ class PulseCLI:
|
|
|
4486
4856
|
f"Sensitivity is {self.sensitivity:.2f} (0=loosest, 1=tightest). Derived thresholds: "
|
|
4487
4857
|
f"spike >{th['explosion_multiplier']:.1f}x baseline, "
|
|
4488
4858
|
f"plateau range <{th['plateau_range_frac']:.1e} of latest, "
|
|
4489
|
-
f"oscillation >={int(th['oscillation_flip_threshold'])} reversals/
|
|
4490
|
-
f"with avg swing >{th['oscillation_delta_frac']*100:.0f}% of latest
|
|
4859
|
+
f"oscillation >={int(th['oscillation_flip_threshold'])} reversals/18 "
|
|
4860
|
+
f"with avg swing >{th['oscillation_delta_frac']*100:.0f}% of latest, "
|
|
4861
|
+
f"stagnation <{th['stagnation_frac']*100:.1f}% improvement over {int(th['stagnation_window'])} steps. "
|
|
4491
4862
|
"Usage: /sensitivity <0.0-1.0|loose|medium|tight> or "
|
|
4492
|
-
"/sensitivity <spike|plateau|oscillation> <value|auto>"
|
|
4863
|
+
"/sensitivity <spike|plateau|oscillation|stagnation> <value|auto>"
|
|
4493
4864
|
)
|
|
4494
4865
|
if quiet:
|
|
4495
4866
|
return msg
|
|
@@ -4498,12 +4869,13 @@ class PulseCLI:
|
|
|
4498
4869
|
|
|
4499
4870
|
parts = arg.split(None, 1)
|
|
4500
4871
|
sub = parts[0].lower()
|
|
4501
|
-
if sub in ("spike", "plateau", "oscillation") and len(parts) == 2:
|
|
4872
|
+
if sub in ("spike", "plateau", "oscillation", "stagnation") and len(parts) == 2:
|
|
4502
4873
|
val_str = parts[1].strip().lower()
|
|
4503
4874
|
field = {
|
|
4504
4875
|
"spike": "explosion_multiplier",
|
|
4505
4876
|
"plateau": "plateau_range_frac",
|
|
4506
4877
|
"oscillation": "oscillation_flip_threshold", # flips; delta_frac follows the dial
|
|
4878
|
+
"stagnation": "stagnation_frac", # window length always follows the dial
|
|
4507
4879
|
}[sub]
|
|
4508
4880
|
if val_str == "auto":
|
|
4509
4881
|
setattr(self, field, None)
|
|
@@ -4523,6 +4895,7 @@ class PulseCLI:
|
|
|
4523
4895
|
if sub == "reset":
|
|
4524
4896
|
self.explosion_multiplier = self.plateau_range_frac = None
|
|
4525
4897
|
self.oscillation_flip_threshold = self.oscillation_delta_frac = None
|
|
4898
|
+
self.stagnation_frac = None
|
|
4526
4899
|
msg = "✓ Cleared per-signal overrides -- all thresholds now follow the overall dial."
|
|
4527
4900
|
if quiet:
|
|
4528
4901
|
return msg
|
|
@@ -4537,7 +4910,7 @@ class PulseCLI:
|
|
|
4537
4910
|
except ValueError:
|
|
4538
4911
|
msg = (
|
|
4539
4912
|
f"Usage: /sensitivity <0.0-1.0|{'|'.join(self._SENSITIVITY_PRESETS)}> or "
|
|
4540
|
-
"/sensitivity <spike|plateau|oscillation> <value|auto>"
|
|
4913
|
+
"/sensitivity <spike|plateau|oscillation|stagnation> <value|auto>"
|
|
4541
4914
|
)
|
|
4542
4915
|
if quiet:
|
|
4543
4916
|
return msg
|
|
@@ -4565,30 +4938,100 @@ class PulseCLI:
|
|
|
4565
4938
|
|
|
4566
4939
|
if _looks_like_loss(var_name) and latest is not None:
|
|
4567
4940
|
finite_recent = [v for v in hist[-50:] if v is not None and math.isfinite(v)]
|
|
4568
|
-
|
|
4569
|
-
# Explosion detection
|
|
4570
|
-
|
|
4571
|
-
|
|
4941
|
+
|
|
4942
|
+
# Explosion detection. Normally needs a few real finite
|
|
4943
|
+
# points to compute a "recent minimum" baseline -- which
|
|
4944
|
+
# meant a genuine explosion in the first few steps of
|
|
4945
|
+
# training could never be caught, since there just wasn't
|
|
4946
|
+
# enough history yet to compare against. If the agent has
|
|
4947
|
+
# seeded an expected starting value for this variable from
|
|
4948
|
+
# reading the code (a NORMAL_START: directive, sent
|
|
4949
|
+
# automatically once at the start of training -- see
|
|
4950
|
+
# _apply_directives / _prime_at_start), fold it into the
|
|
4951
|
+
# baseline pool so a step-1 blow-up has something real to
|
|
4952
|
+
# compare against too.
|
|
4953
|
+
seed = self._normal_start_baselines.get(var_name)
|
|
4954
|
+
baseline_pool = finite_recent[:-1] if len(finite_recent) >= 5 else []
|
|
4955
|
+
if seed is not None:
|
|
4956
|
+
baseline_pool = baseline_pool + [seed]
|
|
4957
|
+
if baseline_pool:
|
|
4958
|
+
baseline = min(baseline_pool)
|
|
4572
4959
|
if baseline > 0 and latest > baseline * th["explosion_multiplier"]:
|
|
4960
|
+
basis = (
|
|
4961
|
+
"its expected starting value (estimated from the code)"
|
|
4962
|
+
if not finite_recent[:-1] else "its recent minimum"
|
|
4963
|
+
)
|
|
4573
4964
|
reasons.append(
|
|
4574
4965
|
f"'{var_name}' spiked to {latest:.4g}, "
|
|
4575
|
-
f"{latest / baseline:.1f}x
|
|
4966
|
+
f"{latest / baseline:.1f}x {basis} ({baseline:.4g})"
|
|
4576
4967
|
)
|
|
4577
|
-
|
|
4968
|
+
|
|
4578
4969
|
if len(finite_recent) >= 20:
|
|
4579
4970
|
recent_window = finite_recent[-20:]
|
|
4580
|
-
|
|
4971
|
+
|
|
4581
4972
|
deltas = [recent_window[i] - recent_window[i-1] for i in range(1, len(recent_window))]
|
|
4582
|
-
|
|
4973
|
+
|
|
4974
|
+
# Scale the plateau/oscillation thresholds off the
|
|
4975
|
+
# window's own typical magnitude (mean |value| across
|
|
4976
|
+
# all 20 points), not off `latest` alone. `latest` is
|
|
4977
|
+
# a single sample that can itself land at a momentary
|
|
4978
|
+
# peak, trough, or near-zero crossing of the very
|
|
4979
|
+
# curve being judged -- using it as the sole scale
|
|
4980
|
+
# reference made both checks unreliable: too
|
|
4981
|
+
# trigger-happy right as a curve crossed zero, too lax
|
|
4982
|
+
# whenever `latest` happened to be sitting at a local
|
|
4983
|
+
# extreme instead of a typical value.
|
|
4984
|
+
scale = sum(abs(v) for v in recent_window) / len(recent_window)
|
|
4985
|
+
if scale == 0:
|
|
4986
|
+
scale = abs(latest) # degenerate all-zero window; fall back rather than lose the check entirely
|
|
4987
|
+
|
|
4583
4988
|
window_range = max(recent_window) - min(recent_window)
|
|
4584
|
-
if window_range
|
|
4989
|
+
if window_range == 0:
|
|
4990
|
+
# Identical value for 20 straight steps (dead
|
|
4991
|
+
# gradient, lr=0, a frozen model) is the most
|
|
4992
|
+
# extreme plateau there is, not an edge case to
|
|
4993
|
+
# skip. The old `window_range > 0` guard here
|
|
4994
|
+
# excluded exactly this, so a totally frozen loss
|
|
4995
|
+
# -- arguably the easiest plateau to catch -- was
|
|
4996
|
+
# the one case that could never be flagged.
|
|
4997
|
+
reasons.append(f"'{var_name}' has completely frozen (identical value for the last 20 steps).")
|
|
4998
|
+
elif window_range < (scale * th["plateau_range_frac"]):
|
|
4585
4999
|
reasons.append(f"'{var_name}' has plateaued (range across last 20 steps is {window_range:.2e}).")
|
|
4586
|
-
|
|
5000
|
+
|
|
4587
5001
|
sign_flips = sum(1 for i in range(1, len(deltas)) if (deltas[i] * deltas[i-1]) < 0)
|
|
4588
5002
|
avg_delta_mag = sum(abs(d) for d in deltas) / len(deltas)
|
|
4589
|
-
|
|
4590
|
-
if sign_flips >= th["oscillation_flip_threshold"] and avg_delta_mag > (
|
|
4591
|
-
reasons.append(f"'{var_name}' is heavily oscillating ({sign_flips} directional reversals in 20 steps).")
|
|
5003
|
+
|
|
5004
|
+
if sign_flips >= th["oscillation_flip_threshold"] and avg_delta_mag > (scale * th["oscillation_delta_frac"]):
|
|
5005
|
+
reasons.append(f"'{var_name}' is heavily oscillating ({sign_flips} directional reversals in the last 20 steps).")
|
|
5006
|
+
|
|
5007
|
+
# Long-horizon stagnation check -- deliberately separate
|
|
5008
|
+
# from the plateau check above, and computed from its own
|
|
5009
|
+
# independently-sized slice of `hist` (not finite_recent,
|
|
5010
|
+
# which stays capped at 50 so it doesn't change the
|
|
5011
|
+
# explosion baseline's "recent minimum" into a
|
|
5012
|
+
# much-older, possibly stale minimum). Plateau looks at
|
|
5013
|
+
# the *range* of just the last 20 steps, so a loss
|
|
5014
|
+
# bouncing around by completely normal per-step noise
|
|
5015
|
+
# (e.g. +/-0.02 every step, never trending down) will
|
|
5016
|
+
# never look "flat enough" to trip it, no matter how many
|
|
5017
|
+
# hundreds of steps go by with zero real progress --
|
|
5018
|
+
# which is exactly what a loss stuck oscillating in a
|
|
5019
|
+
# narrow band around the same value for 1000+ steps looks
|
|
5020
|
+
# like. This instead compares the mean of the first
|
|
5021
|
+
# quarter of a long window against the mean of the last
|
|
5022
|
+
# quarter, so per-step noise averages out and only
|
|
5023
|
+
# genuine lack of improvement is left standing.
|
|
5024
|
+
window = int(th["stagnation_window"])
|
|
5025
|
+
long_window_raw = [v for v in hist[-window:] if v is not None and math.isfinite(v)]
|
|
5026
|
+
if len(long_window_raw) >= window:
|
|
5027
|
+
quarter = max(1, window // 4)
|
|
5028
|
+
early_mean = sum(long_window_raw[:quarter]) / quarter
|
|
5029
|
+
late_mean = sum(long_window_raw[-quarter:]) / quarter
|
|
5030
|
+
if early_mean != 0 and abs(early_mean - late_mean) < abs(early_mean) * th["stagnation_frac"]:
|
|
5031
|
+
reasons.append(
|
|
5032
|
+
f"'{var_name}' hasn't meaningfully improved over the last {window} steps "
|
|
5033
|
+
f"(from {early_mean:.4g} to {late_mean:.4g}, despite normal step-to-step noise)."
|
|
5034
|
+
)
|
|
4592
5035
|
|
|
4593
5036
|
for sub_name, entry in self._matrix_cache.items():
|
|
4594
5037
|
stats = entry.get("stats", {})
|
|
@@ -4596,6 +5039,72 @@ class PulseCLI:
|
|
|
4596
5039
|
reasons.append(f"'{sub_name}' has nan={stats.get('nan')} inf={stats.get('inf')}")
|
|
4597
5040
|
|
|
4598
5041
|
return "; ".join(reasons) if reasons else None
|
|
5042
|
+
_START_PRIME_PROMPT = (
|
|
5043
|
+
"[Automatic start-of-run check -- sent once, automatically, before the first training step, "
|
|
5044
|
+
"so this is your only chance to set these from the code alone, before any real data exists] "
|
|
5045
|
+
"Look at the training code and the tracked variables above. Three things:\n"
|
|
5046
|
+
"1. Judge how noisy this run's loss/metric curves are likely to be, given the model type, "
|
|
5047
|
+
"batch size, learning rate, and loss function, and set an appropriate sensitivity for "
|
|
5048
|
+
"spike/plateau/oscillation detection.\n"
|
|
5049
|
+
"2. For each loss-like tracked variable, estimate its expected value at the very start of "
|
|
5050
|
+
"training from the code alone if you can justify one (e.g. a randomly-initialized N-class "
|
|
5051
|
+
"classifier's cross-entropy loss starts near ln(N); a policy's initial reward is often near "
|
|
5052
|
+
"a known random-policy baseline). This is the ONLY way Pulse can catch a real explosion in "
|
|
5053
|
+
"the first few steps of training -- normally spike detection needs several real data points "
|
|
5054
|
+
"before it has anything to compare against, so a blow-up before then would otherwise go "
|
|
5055
|
+
"completely undetected.\n"
|
|
5056
|
+
"3. Identify any critical variables -- e.g. loss, or core tensors that live exclusively in "
|
|
5057
|
+
"GPU memory -- that are worth the closer, GPU-synced look from step one, to catch memory "
|
|
5058
|
+
"spikes or numerical instability early rather than after they've already caused visible "
|
|
5059
|
+
"damage.\n\n"
|
|
5060
|
+
"Reply with ONLY these three lines, in exactly this format, and nothing else -- no diagnosis, "
|
|
5061
|
+
"no prose:\n"
|
|
5062
|
+
"SENSITIVITY: <0.0-1.0, a preset (loose/medium/tight), or 'spike|plateau|oscillation <value|auto>'>\n"
|
|
5063
|
+
"NORMAL_START: <comma-separated var=value pairs for loss-like tracked variables you can justify, or 'none'>\n"
|
|
5064
|
+
"GPUTRACK: <comma-separated variable names to track closely from the start, or 'none'>"
|
|
5065
|
+
)
|
|
5066
|
+
|
|
5067
|
+
def _prime_at_start(self) -> None:
|
|
5068
|
+
"""Ask the agent, once, automatically, before the very first
|
|
5069
|
+
training step, to (a) set a sensitivity appropriate to this run's
|
|
5070
|
+
code, (b) estimate a starting-value baseline for loss-like tracked
|
|
5071
|
+
variables by reading the code alone (a NORMAL_START: directive --
|
|
5072
|
+
see _apply_directives), and (c) flag any variables worth GPU-level
|
|
5073
|
+
tracking from step one (a GPUTRACK: directive).
|
|
5074
|
+
|
|
5075
|
+
All three of these used to only ever happen reactively: sensitivity
|
|
5076
|
+
only got tuned once the agent had already seen live data (e.g. a
|
|
5077
|
+
GPU check-in or a manual /ask), GPU-tracking only ever got turned
|
|
5078
|
+
on after something had already looked suspicious enough to ask
|
|
5079
|
+
about, and there was no mechanism at all for seeding a starting
|
|
5080
|
+
baseline -- which meant an explosion in the first few steps of
|
|
5081
|
+
training, before 5 real data points existed, could never be
|
|
5082
|
+
caught (see _check_for_trouble). Doing all three once up front,
|
|
5083
|
+
from the code alone, closes those gaps before training even
|
|
5084
|
+
starts.
|
|
5085
|
+
|
|
5086
|
+
Best-effort and silent on failure -- this must never be the thing
|
|
5087
|
+
that makes a training run fail to start. Runs at most once per
|
|
5088
|
+
process (guarded by self._start_primed), and only if an agent
|
|
5089
|
+
provider/key is actually configured and the training code was
|
|
5090
|
+
made available via set_code_text.
|
|
5091
|
+
"""
|
|
5092
|
+
if self._start_primed:
|
|
5093
|
+
return
|
|
5094
|
+
self._start_primed = True
|
|
5095
|
+
if not self.agent_provider or not self.agent_key or not self.code_text:
|
|
5096
|
+
return
|
|
5097
|
+
try:
|
|
5098
|
+
context = self._build_agent_context(include_code=True)
|
|
5099
|
+
answer = self._call_model(f"{context}\n\n{self._START_PRIME_PROMPT}", max_tokens=300)
|
|
5100
|
+
except AgentRequestFailed as exc:
|
|
5101
|
+
cprint(f"[Pulse] ⚠ Start-of-run sensitivity check skipped (agent request failed: {exc})", color=_YELLOW)
|
|
5102
|
+
return
|
|
5103
|
+
_, _calc, _promote, gputrack_names, _gpuu, sensitivity_args, normal_start_args = self._extract_directives(answer)
|
|
5104
|
+
summary = self._apply_directives([], [], gputrack_names, None, sensitivity_args, normal_start_args)
|
|
5105
|
+
if summary:
|
|
5106
|
+
cprint(f"[Pulse] Start-of-run check: {summary}", color=_YELLOW)
|
|
5107
|
+
|
|
4599
5108
|
def update(self, step: Optional[int] = None, generate_pdfs: Optional[bool] = None) -> None:
|
|
4600
5109
|
"""Called at every training step/checkpoint.
|
|
4601
5110
|
|
|
@@ -4622,6 +5131,8 @@ class PulseCLI:
|
|
|
4622
5131
|
loss_var = next((v for v in self.tracked_vars if _looks_like_loss(v)), None)
|
|
4623
5132
|
new_loss_value: Optional[float] = None
|
|
4624
5133
|
|
|
5134
|
+
self._prime_at_start()
|
|
5135
|
+
|
|
4625
5136
|
# Uptime: wall-clock time since the *previous* update() call handed
|
|
4626
5137
|
# control back to the training loop, i.e. time actually spent in
|
|
4627
5138
|
# the user's own training code (forward/backward/optimizer step).
|
|
@@ -5032,6 +5543,9 @@ class PulseCLI:
|
|
|
5032
5543
|
if cmd.lower().startswith("/deleteaccount"):
|
|
5033
5544
|
self._cmd_deleteaccount(cmd[14:].strip())
|
|
5034
5545
|
continue
|
|
5546
|
+
if cmd.lower() == "/logout":
|
|
5547
|
+
self._cmd_logout("")
|
|
5548
|
+
continue
|
|
5035
5549
|
if cmd.lower().startswith("/commit"):
|
|
5036
5550
|
self._cmd_commit(cmd[7:].strip())
|
|
5037
5551
|
continue
|
|
@@ -191,6 +191,17 @@ class SupabaseError(Exception):
|
|
|
191
191
|
pass
|
|
192
192
|
|
|
193
193
|
|
|
194
|
+
class SupabaseEmailConfirmationRequired(SupabaseError):
|
|
195
|
+
"""Raised by sign_up() when the account was actually created but this
|
|
196
|
+
project's Auth settings require the user to confirm their email
|
|
197
|
+
before a session can be issued. A subclass of SupabaseError (not a
|
|
198
|
+
separate flag) so any existing `except SupabaseError` call site keeps
|
|
199
|
+
working unchanged, while a call site that wants to show a "go check
|
|
200
|
+
your email" message instead of a generic "sign up failed" message can
|
|
201
|
+
catch this specifically."""
|
|
202
|
+
pass
|
|
203
|
+
|
|
204
|
+
|
|
194
205
|
# ----------------------------------------------------------------------------
|
|
195
206
|
# Auth session state -- the real Supabase Auth JWT for the signed-in user,
|
|
196
207
|
# as distinct from the app-level "session_token"/session_token_hash pair
|
|
@@ -602,7 +613,7 @@ def sign_up(email: str, password: str, plan: Optional[str] = None) -> Dict[str,
|
|
|
602
613
|
Returns the user profile from the profiles table."""
|
|
603
614
|
if not is_valid_email(email):
|
|
604
615
|
raise SupabaseError(
|
|
605
|
-
"
|
|
616
|
+
"Enter a valid email address (min. 8 characters), e.g. name@example.com."
|
|
606
617
|
)
|
|
607
618
|
if len(password) < 8:
|
|
608
619
|
raise SupabaseError("Password must be at least 8 characters.")
|
|
@@ -644,8 +655,8 @@ def sign_up(email: str, password: str, plan: Optional[str] = None) -> Dict[str,
|
|
|
644
655
|
# user confirms their email and logs in for real -- surface that
|
|
645
656
|
# plainly instead of the misleading "profile was not created"
|
|
646
657
|
# below (which is what happens if you try to fetch it anyway).
|
|
647
|
-
raise
|
|
648
|
-
"
|
|
658
|
+
raise SupabaseEmailConfirmationRequired(
|
|
659
|
+
"Confirm your account: check your email for a confirmation link, then log in."
|
|
649
660
|
)
|
|
650
661
|
|
|
651
662
|
# Authenticate every subsequent request in this process (starting with
|
|
@@ -1225,6 +1236,41 @@ def fetch_recent_sessions(
|
|
|
1225
1236
|
return _request("GET", "Debug_Sessions", params=params, timeout=_TIMEOUT_INTERACTIVE) or []
|
|
1226
1237
|
|
|
1227
1238
|
|
|
1239
|
+
def fetch_debug_session(session_id: str) -> Optional[Dict[str, Any]]:
|
|
1240
|
+
"""Fetch a single Debug_Sessions row by id (fields still
|
|
1241
|
+
compressed -- caller decodes with decode_entries).
|
|
1242
|
+
|
|
1243
|
+
Used when resuming an EXISTING debug session -- currently only after
|
|
1244
|
+
an auto-fix restart, which carries the same session id into the new
|
|
1245
|
+
process via PULSE_AUTO_SESSION_ID -- so the resumed process can
|
|
1246
|
+
preload its local agent_logs/error_tracebacks/telemetry/incidents
|
|
1247
|
+
lists with whatever's already saved server-side. Without this, a
|
|
1248
|
+
freshly-started process's lists start empty and, since a PATCH
|
|
1249
|
+
replaces the whole array column (PostgREST has no array "append"
|
|
1250
|
+
verb -- see _build_cloud_patch_body), its very first flush after
|
|
1251
|
+
the restart would silently overwrite the pre-restart history with
|
|
1252
|
+
just the handful of entries logged since. Falls back to the legacy
|
|
1253
|
+
select list for deployments that don't have the incidents/uptime/
|
|
1254
|
+
downtime columns yet.
|
|
1255
|
+
"""
|
|
1256
|
+
try:
|
|
1257
|
+
rows = _request(
|
|
1258
|
+
"GET", "Debug_Sessions",
|
|
1259
|
+
params={"id": f"eq.{session_id}", "select": _SESSION_SELECT},
|
|
1260
|
+
)
|
|
1261
|
+
except SupabaseError:
|
|
1262
|
+
try:
|
|
1263
|
+
rows = _request(
|
|
1264
|
+
"GET", "Debug_Sessions",
|
|
1265
|
+
params={"id": f"eq.{session_id}", "select": _SESSION_SELECT_LEGACY},
|
|
1266
|
+
)
|
|
1267
|
+
except SupabaseError:
|
|
1268
|
+
return None
|
|
1269
|
+
if not rows:
|
|
1270
|
+
return None
|
|
1271
|
+
return rows[0]
|
|
1272
|
+
|
|
1273
|
+
|
|
1228
1274
|
def patch_debug_session(session_id: str, fields: Dict[str, Any], timeout: float = _TIMEOUT_BACKGROUND) -> None:
|
|
1229
1275
|
_request(
|
|
1230
1276
|
"PATCH", "Debug_Sessions",
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|