crane-llm 0.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- crane_llm-0.0.0/LICENSE +28 -0
- crane_llm-0.0.0/MANIFEST.in +28 -0
- crane_llm-0.0.0/PKG-INFO +219 -0
- crane_llm-0.0.0/README.md +181 -0
- crane_llm-0.0.0/crane_llm/__init__.py +62 -0
- crane_llm-0.0.0/crane_llm/config.py +46 -0
- crane_llm-0.0.0/crane_llm/llms/__init__.py +0 -0
- crane_llm-0.0.0/crane_llm/llms/config_llms.py +319 -0
- crane_llm-0.0.0/crane_llm/llms/huggingface_model_loader.py +40 -0
- crane_llm-0.0.0/crane_llm/llms/llm_executor.py +482 -0
- crane_llm-0.0.0/crane_llm/llms/prompt_extractor.py +187 -0
- crane_llm-0.0.0/crane_llm/llms/result_check.py +641 -0
- crane_llm-0.0.0/crane_llm/llms/retry.py +116 -0
- crane_llm-0.0.0/crane_llm/nb_extension/__init__.py +27 -0
- crane_llm-0.0.0/crane_llm/nb_extension/api.py +215 -0
- crane_llm-0.0.0/crane_llm/nb_extension/assistant.py +45 -0
- crane_llm-0.0.0/crane_llm/nb_extension/cell_filter.py +34 -0
- crane_llm-0.0.0/crane_llm/nb_extension/extension.py +103 -0
- crane_llm-0.0.0/crane_llm/nb_extension/install.json +5 -0
- crane_llm-0.0.0/crane_llm/nb_extension/ipython_hooks.py +200 -0
- crane_llm-0.0.0/crane_llm/nb_extension/labextension/package.json +67 -0
- crane_llm-0.0.0/crane_llm/nb_extension/labextension/static/590.2ed9be6f9f7239ed.js +3 -0
- crane_llm-0.0.0/crane_llm/nb_extension/labextension/static/remoteEntry.bc1e8f0006d7929c.js +7 -0
- crane_llm-0.0.0/crane_llm/nb_extension/labextension/static/style.js +4 -0
- crane_llm-0.0.0/crane_llm/nb_extension/labextension/static/third-party-licenses.json +40 -0
- crane_llm-0.0.0/crane_llm/nb_extension/llm_client.py +217 -0
- crane_llm-0.0.0/crane_llm/nb_extension/magic.py +57 -0
- crane_llm-0.0.0/crane_llm/nb_extension/package.json +63 -0
- crane_llm-0.0.0/crane_llm/nb_extension/prompt_builder.py +46 -0
- crane_llm-0.0.0/crane_llm/nb_extension/runinfo.py +50 -0
- crane_llm-0.0.0/crane_llm/nb_extension/session_state.py +98 -0
- crane_llm-0.0.0/crane_llm/nb_extension/settings.py +321 -0
- crane_llm-0.0.0/crane_llm/nb_extension/smoke_test.py +403 -0
- crane_llm-0.0.0/crane_llm/nb_extension/src/index.ts +1139 -0
- crane_llm-0.0.0/crane_llm/nb_extension/tsconfig.json +19 -0
- crane_llm-0.0.0/crane_llm/nb_extension/ui.py +58 -0
- crane_llm-0.0.0/crane_llm/runinfo_parser/__init__.py +0 -0
- crane_llm-0.0.0/crane_llm/runinfo_parser/cell_executor.py +37 -0
- crane_llm-0.0.0/crane_llm/runinfo_parser/dependency_visitor.py +65 -0
- crane_llm-0.0.0/crane_llm/runinfo_parser/notebook_runtime_extractor.py +143 -0
- crane_llm-0.0.0/crane_llm/runinfo_parser/preprocess_notebook.py +127 -0
- crane_llm-0.0.0/crane_llm/runinfo_parser/runinfo_tracker.py +45 -0
- crane_llm-0.0.0/crane_llm/runinfo_parser/runtime_summary.py +202 -0
- crane_llm-0.0.0/crane_llm/runinfo_parser/summarize_config.json +47 -0
- crane_llm-0.0.0/crane_llm/runinfo_parser/summary_rules.py +1309 -0
- crane_llm-0.0.0/crane_llm/runinfo_parser/utils.py +45 -0
- crane_llm-0.0.0/crane_llm.egg-info/PKG-INFO +219 -0
- crane_llm-0.0.0/crane_llm.egg-info/SOURCES.txt +52 -0
- crane_llm-0.0.0/crane_llm.egg-info/dependency_links.txt +1 -0
- crane_llm-0.0.0/crane_llm.egg-info/requires.txt +18 -0
- crane_llm-0.0.0/crane_llm.egg-info/top_level.txt +1 -0
- crane_llm-0.0.0/pyproject.toml +79 -0
- crane_llm-0.0.0/setup.cfg +4 -0
- crane_llm-0.0.0/setup.py +80 -0
crane_llm-0.0.0/LICENSE
ADDED
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
BSD 3-Clause License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2025, Yiran
|
|
4
|
+
|
|
5
|
+
Redistribution and use in source and binary forms, with or without
|
|
6
|
+
modification, are permitted provided that the following conditions are met:
|
|
7
|
+
|
|
8
|
+
1. Redistributions of source code must retain the above copyright notice, this
|
|
9
|
+
list of conditions and the following disclaimer.
|
|
10
|
+
|
|
11
|
+
2. Redistributions in binary form must reproduce the above copyright notice,
|
|
12
|
+
this list of conditions and the following disclaimer in the documentation
|
|
13
|
+
and/or other materials provided with the distribution.
|
|
14
|
+
|
|
15
|
+
3. Neither the name of the copyright holder nor the names of its
|
|
16
|
+
contributors may be used to endorse or promote products derived from
|
|
17
|
+
this software without specific prior written permission.
|
|
18
|
+
|
|
19
|
+
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
|
20
|
+
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
|
21
|
+
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
|
22
|
+
DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
|
23
|
+
FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
|
24
|
+
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
|
25
|
+
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
|
26
|
+
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
|
27
|
+
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
|
28
|
+
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
include LICENSE
|
|
2
|
+
include README.md
|
|
3
|
+
include pyproject.toml
|
|
4
|
+
include crane_llm/runinfo_parser/summarize_config.json
|
|
5
|
+
|
|
6
|
+
# Frontend sources and the built bundle.
|
|
7
|
+
include crane_llm/nb_extension/package.json
|
|
8
|
+
include crane_llm/nb_extension/tsconfig.json
|
|
9
|
+
include crane_llm/nb_extension/install.json
|
|
10
|
+
recursive-include crane_llm/nb_extension/src *.ts
|
|
11
|
+
recursive-include crane_llm/nb_extension/labextension *
|
|
12
|
+
|
|
13
|
+
# Keep the build off the large directories. Without these, setuptools walks
|
|
14
|
+
# every one of them while assembling the file list, which dominates the
|
|
15
|
+
# install time. target_nbs alone is ~17 GB.
|
|
16
|
+
prune crane_llm/nb_extension/node_modules
|
|
17
|
+
prune crane_llm/nb_extension/.yarn
|
|
18
|
+
prune crane_llm/nb_extension/lib
|
|
19
|
+
prune llms/llms_inputs
|
|
20
|
+
prune llms/llms_outputs
|
|
21
|
+
prune target_nbs
|
|
22
|
+
prune target_nbs_thestackdedup
|
|
23
|
+
prune target_nbs_notebookerrorsdataset
|
|
24
|
+
prune test_nb
|
|
25
|
+
prune results
|
|
26
|
+
prune tmp
|
|
27
|
+
prune build
|
|
28
|
+
prune dist
|
crane_llm-0.0.0/PKG-INFO
ADDED
|
@@ -0,0 +1,219 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: crane-llm
|
|
3
|
+
Version: 0.0.0
|
|
4
|
+
Summary: CRANE-LLM: runtime-augmented crash prediction and diagnosis for ML notebooks
|
|
5
|
+
Author: PELAB, Linköping University
|
|
6
|
+
License: BSD-3-Clause
|
|
7
|
+
Project-URL: Homepage, https://github.com/yarinamomo/crane_llm
|
|
8
|
+
Project-URL: Repository, https://github.com/yarinamomo/crane_llm
|
|
9
|
+
Project-URL: Issues, https://github.com/yarinamomo/crane_llm/issues
|
|
10
|
+
Keywords: jupyter,jupyterlab,jupyterlab-extension,notebook,llm,crash-prediction
|
|
11
|
+
Classifier: Framework :: Jupyter
|
|
12
|
+
Classifier: Framework :: Jupyter :: JupyterLab
|
|
13
|
+
Classifier: Framework :: Jupyter :: JupyterLab :: 4
|
|
14
|
+
Classifier: Framework :: Jupyter :: JupyterLab :: Extensions
|
|
15
|
+
Classifier: Framework :: Jupyter :: JupyterLab :: Extensions :: Prebuilt
|
|
16
|
+
Classifier: License :: OSI Approved :: BSD License
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Intended Audience :: Science/Research
|
|
19
|
+
Requires-Python: >=3.9
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
License-File: LICENSE
|
|
22
|
+
Requires-Dist: ipython>=8.5
|
|
23
|
+
Requires-Dist: jupyterlab<5,>=4.0.0
|
|
24
|
+
Requires-Dist: nbformat>=5.7
|
|
25
|
+
Requires-Dist: openai>=1.40
|
|
26
|
+
Requires-Dist: python-dotenv>=1.0
|
|
27
|
+
Requires-Dist: rich>=13.0
|
|
28
|
+
Provides-Extra: widgets
|
|
29
|
+
Requires-Dist: ipywidgets>=8.0; extra == "widgets"
|
|
30
|
+
Provides-Extra: build
|
|
31
|
+
Requires-Dist: jupyter-builder<2,>=1.0.0; extra == "build"
|
|
32
|
+
Provides-Extra: experiments
|
|
33
|
+
Requires-Dist: google-genai; extra == "experiments"
|
|
34
|
+
Requires-Dist: ollama; extra == "experiments"
|
|
35
|
+
Requires-Dist: torch; extra == "experiments"
|
|
36
|
+
Requires-Dist: transformers; extra == "experiments"
|
|
37
|
+
Dynamic: license-file
|
|
38
|
+
|
|
39
|
+
# CRANE-LLM: Runtime-Augmented LLMs for Crash Prediction and Diagnosis in ML Notebooks
|
|
40
|
+
|
|
41
|
+
This is the official repository for our paper "CRANE-LLM: Runtime-Augmented LLMs for Crash Prediction and Diagnosis in ML Notebooks". In this paper, we propose CRANE-LLM, a novel approach that prompts LLMs with static code and runtime information extracted from the notebook kernel state to enhance their prediction and explanation of ML notebook crashes.
|
|
42
|
+
|
|
43
|
+
## Using the CRANE-LLM notebook extension
|
|
44
|
+
|
|
45
|
+
CRANE-LLM ships as a JupyterLab 4 / Notebook 7 extension. You do not need to
|
|
46
|
+
clone this repository or build anything to use it: the published wheel already
|
|
47
|
+
contains the compiled frontend.
|
|
48
|
+
|
|
49
|
+
### Installing
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
pip install crane-llm
|
|
53
|
+
jupyter lab
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
Or, to install a specific release directly from GitHub without PyPI. Take the
|
|
57
|
+
version from the [releases page](https://github.com/yarinamomo/crane_llm/releases)
|
|
58
|
+
and substitute it in both places:
|
|
59
|
+
|
|
60
|
+
```bash
|
|
61
|
+
pip install https://github.com/yarinamomo/crane_llm/releases/download/v0.0.0/crane_llm-0.0.0-py3-none-any.whl
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
Check that it registered:
|
|
65
|
+
|
|
66
|
+
```bash
|
|
67
|
+
jupyter labextension list # expect: crane-llm-jlab <version> enabled ok
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
Then open a notebook, run a few cells, select the cell you want to check, and
|
|
71
|
+
click **CRANE-LLM** in the toolbar. The full walkthrough is in
|
|
72
|
+
[Using the extension](./crane_llm/nb_extension/README.md#5-using-the-extension).
|
|
73
|
+
|
|
74
|
+
Two things the install deliberately leaves out. Add `pip install notebook` if
|
|
75
|
+
you want the Notebook 7 interface rather than JupyterLab, and note that the ML
|
|
76
|
+
stack is not pulled in: runtime summarisation covers pandas, numpy, torch,
|
|
77
|
+
sklearn and TensorFlow objects when those packages are present in your kernel,
|
|
78
|
+
and quietly skips them when they are not.
|
|
79
|
+
|
|
80
|
+
To work on the extension rather than use it, see
|
|
81
|
+
[the extension README](./crane_llm/nb_extension/README.md), which covers the
|
|
82
|
+
source build. To publish a new version of it, see [RELEASING.md](./RELEASING.md).
|
|
83
|
+
|
|
84
|
+
### Setting up a model
|
|
85
|
+
|
|
86
|
+
You need an API key for whichever model you want to use. The quickest way,
|
|
87
|
+
which is remembered across sessions, is one line in any notebook cell:
|
|
88
|
+
|
|
89
|
+
```python
|
|
90
|
+
import crane_llm
|
|
91
|
+
crane_llm.set_api_key("sk-...")
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
That writes `~/.crane_llm/config.json`. The extension looks for each setting in
|
|
95
|
+
this order, and takes the first one it finds:
|
|
96
|
+
|
|
97
|
+
1. an explicit argument, e.g. `%%crane_llm gpt-5-mini`
|
|
98
|
+
2. the `CRANE_LLM_API_KEY`, `CRANE_LLM_MODEL` and `CRANE_LLM_BASE_URL` environment variables
|
|
99
|
+
3. the provider's own variables, `OPENAI_API_KEY` and `OPENAI_BASE_URL`
|
|
100
|
+
4. `~/.crane_llm/config.json`
|
|
101
|
+
5. for the model only, the default in [`config_llms.py`](./crane_llm/llms/config_llms.py)
|
|
102
|
+
|
|
103
|
+
**A `.env` file is not a step of its own.** Before the lookup runs, any `.env`
|
|
104
|
+
in the directory you started Jupyter from, or in a directory above it, is read
|
|
105
|
+
and used to fill in whichever of those variables the environment does not
|
|
106
|
+
already define. Its values are then found at step 2 or step 3, under whatever
|
|
107
|
+
variable name they were written with. Two consequences:
|
|
108
|
+
|
|
109
|
+
- a variable already present in the environment beats the same name in `.env`,
|
|
110
|
+
because the file never overwrites something already set. This includes a
|
|
111
|
+
persistent variable, such as a Windows user environment variable, which a
|
|
112
|
+
Jupyter kernel inherits without your having exported anything. Keeping the
|
|
113
|
+
same key in both places is fine; just remember that editing only the `.env`
|
|
114
|
+
copy will appear to do nothing
|
|
115
|
+
- `OPENAI_API_KEY=...` in `.env` beats a key in `~/.crane_llm/config.json`,
|
|
116
|
+
because it is read at step 3 and the file is step 4
|
|
117
|
+
|
|
118
|
+
#### Using a model other than OpenAI
|
|
119
|
+
|
|
120
|
+
Any OpenAI-compatible endpoint works by adding a `base_url` and a model name.
|
|
121
|
+
This covers most providers, including Claude and Gemini through a gateway:
|
|
122
|
+
|
|
123
|
+
| Provider | `base_url` | Example model |
|
|
124
|
+
|---|---|---|
|
|
125
|
+
| OpenAI | *(none needed)* | `gpt-5` |
|
|
126
|
+
| OpenRouter (Claude, Gemini, Llama, …) | `https://openrouter.ai/api/v1` | `anthropic/claude-sonnet-4.5` |
|
|
127
|
+
| Google Gemini | `https://generativelanguage.googleapis.com/v1beta/openai/` | `gemini-2.5-flash` |
|
|
128
|
+
| Groq | `https://api.groq.com/openai/v1` | `qwen/qwen3-32b` |
|
|
129
|
+
| Ollama, local, no key needed | `http://localhost:11434/v1` | `qwen2.5-coder:32b` |
|
|
130
|
+
|
|
131
|
+
For example, to use Claude through OpenRouter:
|
|
132
|
+
|
|
133
|
+
```python
|
|
134
|
+
import crane_llm
|
|
135
|
+
crane_llm.set_api_key(
|
|
136
|
+
"sk-or-...",
|
|
137
|
+
base_url="https://openrouter.ai/api/v1",
|
|
138
|
+
model="anthropic/claude-sonnet-4.5",
|
|
139
|
+
)
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
Or a local model with no account at all:
|
|
143
|
+
|
|
144
|
+
```python
|
|
145
|
+
crane_llm.set_api_key(base_url="http://localhost:11434/v1", model="qwen2.5-coder:32b")
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
With no `base_url`, CRANE-LLM calls OpenAI's Responses API, which is what the
|
|
149
|
+
experiments in the paper used. As soon as a `base_url` is set it switches to
|
|
150
|
+
Chat Completions, because almost no compatible server implements `/responses`.
|
|
151
|
+
Azure OpenAI is the exception: it needs a `base_url` *and* the Responses API,
|
|
152
|
+
so set `CRANE_LLM_API_STYLE=responses` there.
|
|
153
|
+
|
|
154
|
+
## Paper Artefacts: Repository structure and reproducibility details
|
|
155
|
+
### Dataset:
|
|
156
|
+
We use [**Junobench**]((https://huggingface.co/datasets/PELAB-LiU/JunoBench)) dataset in our experiments.
|
|
157
|
+
|
|
158
|
+
### LLMs:
|
|
159
|
+
LLMs include Gemini (Gemini-2.5-Flash), Qwen (Qwen-2.5-Coder-32B-Instruct), GPT-5.
|
|
160
|
+
|
|
161
|
+
### Repository structure:
|
|
162
|
+
|
|
163
|
+
All importable Python code lives under the single package
|
|
164
|
+
[`crane_llm/`](./crane_llm), which is also what the installable wheel contains.
|
|
165
|
+
The scripts at the repository root drive the experiments and are run from a
|
|
166
|
+
checkout rather than installed. Generated experiment inputs and outputs stay in
|
|
167
|
+
[`llms/`](./llms) at the root, beside the code that produces them.
|
|
168
|
+
|
|
169
|
+
- [`crane_llm/`](./crane_llm): the installable package
|
|
170
|
+
- [`nb_extension/`](./crane_llm/nb_extension): the JupyterLab extension, documented in [its own README](./crane_llm/nb_extension/README.md)
|
|
171
|
+
- [`runinfo_parser/`](./crane_llm/runinfo_parser): scripts and configuration for *runtime information extraction*
|
|
172
|
+
- [`config_llms.py`](./crane_llm/llms/config_llms.py): configuration for experiments, including prompts, LLM configs, and related input and output paths
|
|
173
|
+
- [`prompt_extractor.py`](./crane_llm/llms/prompt_extractor.py): script for constructing prompts (i.e., `llms_inputs/`)
|
|
174
|
+
- [`llm_executor.py`](./crane_llm/llms/llm_executor.py): script for querying LLMs to generate outputs in `llms_outputs/`
|
|
175
|
+
- [`crane-llm.py`](./crane-llm.py): script to run CRANE-LLM given a target notebook
|
|
176
|
+
- [`main_LLM.py`](./main_LLM.py): script to run the experiment pipeline, including batch run all notebooks in the dataset for a specific task and experimental setting
|
|
177
|
+
- [`main.py`](./main.py): script to run result compilation and analysis
|
|
178
|
+
- [`llms`](./llms): LLM experiment inputs and outputs
|
|
179
|
+
- [`llms_inputs/`](./llms/llms_inputs): generated inputs (executed code cells only, executed code cells with runtime information) to the LLMs
|
|
180
|
+
- [`llms_outputs/`](./llms/llms_outputs): generated outputs by the LLMs
|
|
181
|
+
- [`results_raw/`](./llms/llms_outputs/results_raw/): LLM outputs organized into: prefix\_[*LLM*]\_[*experimental setup*]\_[*runtime category ablation setup or API grounding*]
|
|
182
|
+
- [`ground_truth_crash_prediction.xlsx`](./llms/llms_outputs/ground_truth_crash_prediction.xlsx): ground truth labels used for evaluating LLM outputs as well as downstream analysis, provided by JunoBench
|
|
183
|
+
- [`results`](./results): evaluated result outcomes and compiled statistics
|
|
184
|
+
- [`results_parsed_detection_and_diagnosis.xlsx`](./results/results_parsed_detection_and_diagnosis.xlsx): CRANE-LLM performance on the joint crash prediction and diagnosis task
|
|
185
|
+
- Sheet "Final_evaluation": Detailed outcomes per LLM per experimental setup. Settings include:
|
|
186
|
+
- code: -RT
|
|
187
|
+
- runinfo: +RT (CRANE-LLM)
|
|
188
|
+
- Sheet "Results_summary": Compiled results and statistics on crash prediction and diagnosis performance of CRANE-LLM
|
|
189
|
+
- [`results_parsed_detection_only.xlsx`](./results/results_parsed_detection_only.xlsx): CRANE-LLM performance on the crash prediction-only task
|
|
190
|
+
- Sheet "Final_evaluation": Detailed outcomes per LLM per experimental setup including *runtime information category ablation study*, and *API documentation grounding study*. All settings include:
|
|
191
|
+
- code: -RT
|
|
192
|
+
- runinfo: +RT (CRANE-LLM)
|
|
193
|
+
- runinfo_r_v: +RT-S (CRANE-LLM - S), ablated structural runtime information
|
|
194
|
+
- runinfo_s_r: +RT-V (CRANE-LLM - V), ablated value semantics runtime information
|
|
195
|
+
- runinfo_s_v: +RT-R (CRANE-LLM - R), ablated type-level (representation and type semantics) runtime information
|
|
196
|
+
- runinfo_full_doc: +RT+doc (CRANE-LLM + doc), full runtime information with additional API documentation information
|
|
197
|
+
- Sheet "Results_summary": Compiled results and statistics on crash prediction-only performance of CRANE-LLM, including runtime information category ablation study and API documentation grounding study results (and token analysis results)
|
|
198
|
+
- [`runtime_doc_token_analysis.txt`](./results/runtime_doc_token_analysis.txt): statistics of tokens of additional API documentation information, results gained by running script [`token_analysis.py`](./utils/token_analysis.py)
|
|
199
|
+
- [`pairwise_significance_detection_and_diagnosis.json`](./results/pairwise_significance_detection_and_diagnosis.json) and [`pairwise_significance_detection_only.json`](./results/pairwise_significance_detection_only.json): statistical test results of the joint crash prediction and diagnosis task and the crash prediction-only task, the statistics tests are ran by script [statistical_test.py](./utils/statistical_test.py)
|
|
200
|
+
- [`cohens_kappa_human_validation.txt`](./results/cohens_kappa_human_validation.txt): statistics of human evaluation on crash diagnosis outputs
|
|
201
|
+
- [`runtime_recording/`](./results/runtime_recording/): statistics of runtime for prior cell executions and querying CRANE-LLM (when using GPT-5).
|
|
202
|
+
|
|
203
|
+
### Environment
|
|
204
|
+
|
|
205
|
+
To ensure full reproducibility, we provide a docker image (digest: sha256:ecb5753d1cdfc9f0d5dfeb59818cde5be5be2f79541c5facf99761393919e171):
|
|
206
|
+
```bash
|
|
207
|
+
docker pull yarinamomo/crane_env:latest
|
|
208
|
+
```
|
|
209
|
+
Then run the docker container:
|
|
210
|
+
```bash
|
|
211
|
+
docker run -v [volumn_mount_windows_path]:/cranellm_env -w /cranellm_env -p 8888:8888 -it yarinamomo/crane_env:latest /bin/bash
|
|
212
|
+
```
|
|
213
|
+
Then you can attach this environment to **VS Code** "*Dev Containers: Attach to Running Container...*"
|
|
214
|
+
|
|
215
|
+
For the commercial LLMs used in the experiments (Gemini and GPT-5), please ensure that the API keys are properly set up before running the scripts (for example, set as global environment variable or config in `.env`). Open-source LLMs (Qwen) can be run directly; however, note that execution may take longer depending on the computational resources available.
|
|
216
|
+
|
|
217
|
+
## License
|
|
218
|
+
|
|
219
|
+
This project is licensed under the terms of the BSD 3-Clause License.
|
|
@@ -0,0 +1,181 @@
|
|
|
1
|
+
# CRANE-LLM: Runtime-Augmented LLMs for Crash Prediction and Diagnosis in ML Notebooks
|
|
2
|
+
|
|
3
|
+
This is the official repository for our paper "CRANE-LLM: Runtime-Augmented LLMs for Crash Prediction and Diagnosis in ML Notebooks". In this paper, we propose CRANE-LLM, a novel approach that prompts LLMs with static code and runtime information extracted from the notebook kernel state to enhance their prediction and explanation of ML notebook crashes.
|
|
4
|
+
|
|
5
|
+
## Using the CRANE-LLM notebook extension
|
|
6
|
+
|
|
7
|
+
CRANE-LLM ships as a JupyterLab 4 / Notebook 7 extension. You do not need to
|
|
8
|
+
clone this repository or build anything to use it: the published wheel already
|
|
9
|
+
contains the compiled frontend.
|
|
10
|
+
|
|
11
|
+
### Installing
|
|
12
|
+
|
|
13
|
+
```bash
|
|
14
|
+
pip install crane-llm
|
|
15
|
+
jupyter lab
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
Or, to install a specific release directly from GitHub without PyPI. Take the
|
|
19
|
+
version from the [releases page](https://github.com/yarinamomo/crane_llm/releases)
|
|
20
|
+
and substitute it in both places:
|
|
21
|
+
|
|
22
|
+
```bash
|
|
23
|
+
pip install https://github.com/yarinamomo/crane_llm/releases/download/v0.0.0/crane_llm-0.0.0-py3-none-any.whl
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
Check that it registered:
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
jupyter labextension list # expect: crane-llm-jlab <version> enabled ok
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
Then open a notebook, run a few cells, select the cell you want to check, and
|
|
33
|
+
click **CRANE-LLM** in the toolbar. The full walkthrough is in
|
|
34
|
+
[Using the extension](./crane_llm/nb_extension/README.md#5-using-the-extension).
|
|
35
|
+
|
|
36
|
+
Two things the install deliberately leaves out. Add `pip install notebook` if
|
|
37
|
+
you want the Notebook 7 interface rather than JupyterLab, and note that the ML
|
|
38
|
+
stack is not pulled in: runtime summarisation covers pandas, numpy, torch,
|
|
39
|
+
sklearn and TensorFlow objects when those packages are present in your kernel,
|
|
40
|
+
and quietly skips them when they are not.
|
|
41
|
+
|
|
42
|
+
To work on the extension rather than use it, see
|
|
43
|
+
[the extension README](./crane_llm/nb_extension/README.md), which covers the
|
|
44
|
+
source build. To publish a new version of it, see [RELEASING.md](./RELEASING.md).
|
|
45
|
+
|
|
46
|
+
### Setting up a model
|
|
47
|
+
|
|
48
|
+
You need an API key for whichever model you want to use. The quickest way,
|
|
49
|
+
which is remembered across sessions, is one line in any notebook cell:
|
|
50
|
+
|
|
51
|
+
```python
|
|
52
|
+
import crane_llm
|
|
53
|
+
crane_llm.set_api_key("sk-...")
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
That writes `~/.crane_llm/config.json`. The extension looks for each setting in
|
|
57
|
+
this order, and takes the first one it finds:
|
|
58
|
+
|
|
59
|
+
1. an explicit argument, e.g. `%%crane_llm gpt-5-mini`
|
|
60
|
+
2. the `CRANE_LLM_API_KEY`, `CRANE_LLM_MODEL` and `CRANE_LLM_BASE_URL` environment variables
|
|
61
|
+
3. the provider's own variables, `OPENAI_API_KEY` and `OPENAI_BASE_URL`
|
|
62
|
+
4. `~/.crane_llm/config.json`
|
|
63
|
+
5. for the model only, the default in [`config_llms.py`](./crane_llm/llms/config_llms.py)
|
|
64
|
+
|
|
65
|
+
**A `.env` file is not a step of its own.** Before the lookup runs, any `.env`
|
|
66
|
+
in the directory you started Jupyter from, or in a directory above it, is read
|
|
67
|
+
and used to fill in whichever of those variables the environment does not
|
|
68
|
+
already define. Its values are then found at step 2 or step 3, under whatever
|
|
69
|
+
variable name they were written with. Two consequences:
|
|
70
|
+
|
|
71
|
+
- a variable already present in the environment beats the same name in `.env`,
|
|
72
|
+
because the file never overwrites something already set. This includes a
|
|
73
|
+
persistent variable, such as a Windows user environment variable, which a
|
|
74
|
+
Jupyter kernel inherits without your having exported anything. Keeping the
|
|
75
|
+
same key in both places is fine; just remember that editing only the `.env`
|
|
76
|
+
copy will appear to do nothing
|
|
77
|
+
- `OPENAI_API_KEY=...` in `.env` beats a key in `~/.crane_llm/config.json`,
|
|
78
|
+
because it is read at step 3 and the file is step 4
|
|
79
|
+
|
|
80
|
+
#### Using a model other than OpenAI
|
|
81
|
+
|
|
82
|
+
Any OpenAI-compatible endpoint works by adding a `base_url` and a model name.
|
|
83
|
+
This covers most providers, including Claude and Gemini through a gateway:
|
|
84
|
+
|
|
85
|
+
| Provider | `base_url` | Example model |
|
|
86
|
+
|---|---|---|
|
|
87
|
+
| OpenAI | *(none needed)* | `gpt-5` |
|
|
88
|
+
| OpenRouter (Claude, Gemini, Llama, …) | `https://openrouter.ai/api/v1` | `anthropic/claude-sonnet-4.5` |
|
|
89
|
+
| Google Gemini | `https://generativelanguage.googleapis.com/v1beta/openai/` | `gemini-2.5-flash` |
|
|
90
|
+
| Groq | `https://api.groq.com/openai/v1` | `qwen/qwen3-32b` |
|
|
91
|
+
| Ollama, local, no key needed | `http://localhost:11434/v1` | `qwen2.5-coder:32b` |
|
|
92
|
+
|
|
93
|
+
For example, to use Claude through OpenRouter:
|
|
94
|
+
|
|
95
|
+
```python
|
|
96
|
+
import crane_llm
|
|
97
|
+
crane_llm.set_api_key(
|
|
98
|
+
"sk-or-...",
|
|
99
|
+
base_url="https://openrouter.ai/api/v1",
|
|
100
|
+
model="anthropic/claude-sonnet-4.5",
|
|
101
|
+
)
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
Or a local model with no account at all:
|
|
105
|
+
|
|
106
|
+
```python
|
|
107
|
+
crane_llm.set_api_key(base_url="http://localhost:11434/v1", model="qwen2.5-coder:32b")
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
With no `base_url`, CRANE-LLM calls OpenAI's Responses API, which is what the
|
|
111
|
+
experiments in the paper used. As soon as a `base_url` is set it switches to
|
|
112
|
+
Chat Completions, because almost no compatible server implements `/responses`.
|
|
113
|
+
Azure OpenAI is the exception: it needs a `base_url` *and* the Responses API,
|
|
114
|
+
so set `CRANE_LLM_API_STYLE=responses` there.
|
|
115
|
+
|
|
116
|
+
## Paper Artefacts: Repository structure and reproducibility details
|
|
117
|
+
### Dataset:
|
|
118
|
+
We use [**Junobench**]((https://huggingface.co/datasets/PELAB-LiU/JunoBench)) dataset in our experiments.
|
|
119
|
+
|
|
120
|
+
### LLMs:
|
|
121
|
+
LLMs include Gemini (Gemini-2.5-Flash), Qwen (Qwen-2.5-Coder-32B-Instruct), GPT-5.
|
|
122
|
+
|
|
123
|
+
### Repository structure:
|
|
124
|
+
|
|
125
|
+
All importable Python code lives under the single package
|
|
126
|
+
[`crane_llm/`](./crane_llm), which is also what the installable wheel contains.
|
|
127
|
+
The scripts at the repository root drive the experiments and are run from a
|
|
128
|
+
checkout rather than installed. Generated experiment inputs and outputs stay in
|
|
129
|
+
[`llms/`](./llms) at the root, beside the code that produces them.
|
|
130
|
+
|
|
131
|
+
- [`crane_llm/`](./crane_llm): the installable package
|
|
132
|
+
- [`nb_extension/`](./crane_llm/nb_extension): the JupyterLab extension, documented in [its own README](./crane_llm/nb_extension/README.md)
|
|
133
|
+
- [`runinfo_parser/`](./crane_llm/runinfo_parser): scripts and configuration for *runtime information extraction*
|
|
134
|
+
- [`config_llms.py`](./crane_llm/llms/config_llms.py): configuration for experiments, including prompts, LLM configs, and related input and output paths
|
|
135
|
+
- [`prompt_extractor.py`](./crane_llm/llms/prompt_extractor.py): script for constructing prompts (i.e., `llms_inputs/`)
|
|
136
|
+
- [`llm_executor.py`](./crane_llm/llms/llm_executor.py): script for querying LLMs to generate outputs in `llms_outputs/`
|
|
137
|
+
- [`crane-llm.py`](./crane-llm.py): script to run CRANE-LLM given a target notebook
|
|
138
|
+
- [`main_LLM.py`](./main_LLM.py): script to run the experiment pipeline, including batch run all notebooks in the dataset for a specific task and experimental setting
|
|
139
|
+
- [`main.py`](./main.py): script to run result compilation and analysis
|
|
140
|
+
- [`llms`](./llms): LLM experiment inputs and outputs
|
|
141
|
+
- [`llms_inputs/`](./llms/llms_inputs): generated inputs (executed code cells only, executed code cells with runtime information) to the LLMs
|
|
142
|
+
- [`llms_outputs/`](./llms/llms_outputs): generated outputs by the LLMs
|
|
143
|
+
- [`results_raw/`](./llms/llms_outputs/results_raw/): LLM outputs organized into: prefix\_[*LLM*]\_[*experimental setup*]\_[*runtime category ablation setup or API grounding*]
|
|
144
|
+
- [`ground_truth_crash_prediction.xlsx`](./llms/llms_outputs/ground_truth_crash_prediction.xlsx): ground truth labels used for evaluating LLM outputs as well as downstream analysis, provided by JunoBench
|
|
145
|
+
- [`results`](./results): evaluated result outcomes and compiled statistics
|
|
146
|
+
- [`results_parsed_detection_and_diagnosis.xlsx`](./results/results_parsed_detection_and_diagnosis.xlsx): CRANE-LLM performance on the joint crash prediction and diagnosis task
|
|
147
|
+
- Sheet "Final_evaluation": Detailed outcomes per LLM per experimental setup. Settings include:
|
|
148
|
+
- code: -RT
|
|
149
|
+
- runinfo: +RT (CRANE-LLM)
|
|
150
|
+
- Sheet "Results_summary": Compiled results and statistics on crash prediction and diagnosis performance of CRANE-LLM
|
|
151
|
+
- [`results_parsed_detection_only.xlsx`](./results/results_parsed_detection_only.xlsx): CRANE-LLM performance on the crash prediction-only task
|
|
152
|
+
- Sheet "Final_evaluation": Detailed outcomes per LLM per experimental setup including *runtime information category ablation study*, and *API documentation grounding study*. All settings include:
|
|
153
|
+
- code: -RT
|
|
154
|
+
- runinfo: +RT (CRANE-LLM)
|
|
155
|
+
- runinfo_r_v: +RT-S (CRANE-LLM - S), ablated structural runtime information
|
|
156
|
+
- runinfo_s_r: +RT-V (CRANE-LLM - V), ablated value semantics runtime information
|
|
157
|
+
- runinfo_s_v: +RT-R (CRANE-LLM - R), ablated type-level (representation and type semantics) runtime information
|
|
158
|
+
- runinfo_full_doc: +RT+doc (CRANE-LLM + doc), full runtime information with additional API documentation information
|
|
159
|
+
- Sheet "Results_summary": Compiled results and statistics on crash prediction-only performance of CRANE-LLM, including runtime information category ablation study and API documentation grounding study results (and token analysis results)
|
|
160
|
+
- [`runtime_doc_token_analysis.txt`](./results/runtime_doc_token_analysis.txt): statistics of tokens of additional API documentation information, results gained by running script [`token_analysis.py`](./utils/token_analysis.py)
|
|
161
|
+
- [`pairwise_significance_detection_and_diagnosis.json`](./results/pairwise_significance_detection_and_diagnosis.json) and [`pairwise_significance_detection_only.json`](./results/pairwise_significance_detection_only.json): statistical test results of the joint crash prediction and diagnosis task and the crash prediction-only task, the statistics tests are ran by script [statistical_test.py](./utils/statistical_test.py)
|
|
162
|
+
- [`cohens_kappa_human_validation.txt`](./results/cohens_kappa_human_validation.txt): statistics of human evaluation on crash diagnosis outputs
|
|
163
|
+
- [`runtime_recording/`](./results/runtime_recording/): statistics of runtime for prior cell executions and querying CRANE-LLM (when using GPT-5).
|
|
164
|
+
|
|
165
|
+
### Environment
|
|
166
|
+
|
|
167
|
+
To ensure full reproducibility, we provide a docker image (digest: sha256:ecb5753d1cdfc9f0d5dfeb59818cde5be5be2f79541c5facf99761393919e171):
|
|
168
|
+
```bash
|
|
169
|
+
docker pull yarinamomo/crane_env:latest
|
|
170
|
+
```
|
|
171
|
+
Then run the docker container:
|
|
172
|
+
```bash
|
|
173
|
+
docker run -v [volumn_mount_windows_path]:/cranellm_env -w /cranellm_env -p 8888:8888 -it yarinamomo/crane_env:latest /bin/bash
|
|
174
|
+
```
|
|
175
|
+
Then you can attach this environment to **VS Code** "*Dev Containers: Attach to Running Container...*"
|
|
176
|
+
|
|
177
|
+
For the commercial LLMs used in the experiments (Gemini and GPT-5), please ensure that the API keys are properly set up before running the scripts (for example, set as global environment variable or config in `.env`). Open-source LLMs (Qwen) can be run directly; however, note that execution may take longer depending on the computational resources available.
|
|
178
|
+
|
|
179
|
+
## License
|
|
180
|
+
|
|
181
|
+
This project is licensed under the terms of the BSD 3-Clause License.
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
"""CRANE-LLM: runtime-augmented crash prediction and diagnosis for ML notebooks.
|
|
2
|
+
|
|
3
|
+
Everything the installed distribution provides lives under this one top-level
|
|
4
|
+
name. That is deliberate. The kernel puts the notebook's own directory ahead of
|
|
5
|
+
``site-packages`` on ``sys.path``, so a package installed as ``config`` or
|
|
6
|
+
``utils`` would be shadowed by any notebook that happens to keep a file of that
|
|
7
|
+
name beside it, and the extension would then fail inside the user's session for
|
|
8
|
+
reasons that look nothing like the cause.
|
|
9
|
+
|
|
10
|
+
Subpackages:
|
|
11
|
+
|
|
12
|
+
- ``crane_llm.nb_extension`` -- the JupyterLab extension (frontend and the
|
|
13
|
+
kernel-side backend it talks to)
|
|
14
|
+
- ``crane_llm.runinfo_parser`` -- runtime information extraction
|
|
15
|
+
- ``crane_llm.llms`` -- prompts, LLM clients and the batch experiment pipeline
|
|
16
|
+
- ``crane_llm.config`` -- summarisation constants shared by the above
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
__version__ = "0.0.0"
|
|
20
|
+
|
|
21
|
+
__all__ = ["__version__", "load_ipython_extension", "set_api_key"]
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _jupyter_labextension_paths():
|
|
25
|
+
"""Tell ``jupyter labextension develop .`` where the built bundle is.
|
|
26
|
+
|
|
27
|
+
Declared on the top-level package because that command imports the
|
|
28
|
+
distribution's root module to find this hook. ``src`` is relative to this
|
|
29
|
+
file; ``dest`` is the name the bundle is served under.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
return [
|
|
33
|
+
{
|
|
34
|
+
"src": "nb_extension/labextension",
|
|
35
|
+
"dest": "crane-llm-jlab",
|
|
36
|
+
}
|
|
37
|
+
]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def load_ipython_extension(ipython):
|
|
41
|
+
"""Support ``%load_ext crane_llm``.
|
|
42
|
+
|
|
43
|
+
Delegates to the extension package. Imported lazily so that merely
|
|
44
|
+
importing ``crane_llm`` does not pull in ipywidgets or an LLM client.
|
|
45
|
+
"""
|
|
46
|
+
|
|
47
|
+
from .nb_extension import load_ipython_extension as _load
|
|
48
|
+
|
|
49
|
+
_load(ipython)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def set_api_key(*args, **kwargs):
|
|
53
|
+
"""Store an API key in the user configuration file.
|
|
54
|
+
|
|
55
|
+
Re-exported here because this is the one piece of setup a user of the
|
|
56
|
+
installed package has to do, and ``crane_llm.set_api_key`` is the obvious
|
|
57
|
+
place to look for it.
|
|
58
|
+
"""
|
|
59
|
+
|
|
60
|
+
from .nb_extension.settings import set_api_key as _set_api_key
|
|
61
|
+
|
|
62
|
+
return _set_api_key(*args, **kwargs)
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
from typing import NamedTuple
|
|
2
|
+
import os
|
|
3
|
+
|
|
4
|
+
here = os.path.dirname(__file__)
|
|
5
|
+
sum_rule_config_path = os.path.join(here, "runinfo_parser", "summarize_config.json")
|
|
6
|
+
|
|
7
|
+
max_len_value = 5
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class var_type(NamedTuple):
|
|
11
|
+
error = 'error',
|
|
12
|
+
unknown = 'unknown',
|
|
13
|
+
var = 'variable',
|
|
14
|
+
con = 'constant',
|
|
15
|
+
call = 'call',
|
|
16
|
+
tuple = 'tuple',
|
|
17
|
+
list = 'list',
|
|
18
|
+
set = 'set',
|
|
19
|
+
dict = 'dict',
|
|
20
|
+
nparray = 'np.array',
|
|
21
|
+
tensor = 'tf.tensor',
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class var_val(NamedTuple):
|
|
25
|
+
error = 'error',
|
|
26
|
+
unknown = 'unknown',
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class var_shape(NamedTuple):
|
|
30
|
+
error = 'error',
|
|
31
|
+
unknown = 'unknown',
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class rule_res(NamedTuple):
|
|
35
|
+
unsupported = -1,
|
|
36
|
+
success = 0,
|
|
37
|
+
log = 1,
|
|
38
|
+
warning = 2,
|
|
39
|
+
error = 3,
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class notify_level(NamedTuple):
|
|
43
|
+
ignore = -1,
|
|
44
|
+
log = 0,
|
|
45
|
+
warning = 1,
|
|
46
|
+
error = 2
|
|
File without changes
|