django-explain-errors 0.6.0__tar.gz → 0.7.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {django_explain_errors-0.6.0/django_explain_errors.egg-info → django_explain_errors-0.7.0}/PKG-INFO +118 -39
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/README.md +117 -38
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0/django_explain_errors.egg-info}/PKG-INFO +118 -39
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/django_explain_errors.egg-info/SOURCES.txt +5 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/explain_errors/middleware.py +11 -6
- django_explain_errors-0.7.0/explain_errors/tracebacks.py +105 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/setup.py +2 -2
- django_explain_errors-0.7.0/tests/test_eval_fixtures.py +110 -0
- django_explain_errors-0.7.0/tests/test_eval_judge.py +393 -0
- django_explain_errors-0.7.0/tests/test_eval_run.py +389 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/tests/test_middleware.py +46 -0
- django_explain_errors-0.7.0/tests/test_tracebacks.py +129 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/LICENSE +0 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/MANIFEST.in +0 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/django_explain_errors.egg-info/dependency_links.txt +0 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/django_explain_errors.egg-info/requires.txt +0 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/django_explain_errors.egg-info/top_level.txt +0 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/explain_errors/__init__.py +0 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/explain_errors/client.py +0 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/explain_errors/management/__init__.py +0 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/explain_errors/management/commands/__init__.py +0 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/explain_errors/management/commands/build_error_index.py +0 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/explain_errors/rag/__init__.py +0 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/explain_errors/rag/indexer.py +0 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/explain_errors/rag/retriever.py +0 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/explain_errors/rag/store.py +0 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/explain_errors/sanitize.py +0 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/explain_errors/throttle.py +0 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/setup.cfg +0 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/tests/test_build_error_index.py +0 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/tests/test_client.py +0 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/tests/test_language.py +0 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/tests/test_rag.py +0 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/tests/test_sanitize.py +0 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/tests/test_signals.py +0 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/tests/test_throttle.py +0 -0
- {django_explain_errors-0.6.0 → django_explain_errors-0.7.0}/tests/test_truncation.py +0 -0
{django_explain_errors-0.6.0/django_explain_errors.egg-info → django_explain_errors-0.7.0}/PKG-INFO
RENAMED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: django-explain-errors
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.7.0
|
|
4
4
|
Summary: Django middleware that explains unhandled exceptions in DEBUG using an LLM, optionally grounded in your own project source via a local vector index. Works with OpenAI, Claude, or any OpenAI-compatible endpoint including local models.
|
|
5
5
|
Home-page: https://github.com/topunix/django-explain-errors
|
|
6
6
|
Author: topunix
|
|
@@ -54,7 +54,8 @@ a local model if you prefer not to send code off your machine.
|
|
|
54
54
|
|
|
55
55
|
It can optionally ground explanations in your own project source using a
|
|
56
56
|
local vector index (RAG), so explanations reference the actual code that
|
|
57
|
-
failed instead of staying generic
|
|
57
|
+
failed instead of staying generic (measured — see "Does RAG actually
|
|
58
|
+
help?" below).
|
|
58
59
|
|
|
59
60
|
The middleware supports both synchronous (WSGI) and asynchronous (ASGI)
|
|
60
61
|
views. It auto-detects the view chain at startup and routes requests through
|
|
@@ -183,7 +184,7 @@ developer is actually investigating.
|
|
|
183
184
|
| `OPENAI_MODEL` | No | Model used for explanations. Defaults to `gpt-4o-mini`. |
|
|
184
185
|
| `OPENAI_MAX_TOKENS` | No | Ceiling on tokens generated for the explanation, not a target — the system prompt itself asks for a concise answer. Defaults to `1000`; scales up automatically when `EXPLAIN_ERRORS_LANGUAGE` is set (see below), unless you set this explicitly, which always overrides the scaling. |
|
|
185
186
|
| `OPENAI_TIMEOUT` | No | Request timeout in seconds for the OpenAI client. Defaults to `10`. |
|
|
186
|
-
| `OPENAI_MAX_TRACEBACK_CHARS` | No |
|
|
187
|
+
| `OPENAI_MAX_TRACEBACK_CHARS` | No | Total character budget for the traceback sent to the model. Application frames (your own code, as opposed to Django, the standard library, or installed packages) are always kept; library frames fill whatever budget remains, nearest the raise point first, with an `... N library frames omitted ...` line where frames are dropped. If the application frames alone exceed the budget, falls back to keeping the last N characters of the raw traceback. Defaults to `3000`. |
|
|
187
188
|
| `EXPLAIN_ERRORS_PRESERVE_DEBUG_PAGE` | No | When `True` (the default), the middleware prints the explanation to stdout and returns `None`, so exception handling continues normally and Django renders its standard debug page. Set to `False` to instead return a JSON 500 response, which ends exception handling early (see Compatibility above). |
|
|
188
189
|
| `OPENAI_BASE_URL` (env or settings) | No | Base URL for any OpenAI-compatible API (for example Ollama at `http://localhost:11434/v1`). When set, a missing API key is replaced with a placeholder since local servers do not require one. |
|
|
189
190
|
| `EXPLAIN_ERRORS_MAX_CALLS` | No | Together with `EXPLAIN_ERRORS_WINDOW_SECONDS`, caps API spend to at most this many explanations within a rolling window; once the cap is hit, further errors in that window are not sent for explanation until an earlier call ages out. Defaults to `5`. |
|
|
@@ -256,6 +257,13 @@ read them in another language instead:
|
|
|
256
257
|
EXPLAIN_ERRORS_LANGUAGE = "Spanish" # or the code form, "es"
|
|
257
258
|
```
|
|
258
259
|
|
|
260
|
+
There is no fixed list of supported languages. `EXPLAIN_ERRORS_LANGUAGE` accepts
|
|
261
|
+
any language the configured model can write, because the setting adds one clause
|
|
262
|
+
to the system prompt (see `LANGUAGE_CLAUSE_TEMPLATE` in
|
|
263
|
+
`explain_errors/middleware.py`), not a translation catalog with its own
|
|
264
|
+
maintained language list. How well it works varies by model; see the known
|
|
265
|
+
limitation below.
|
|
266
|
+
|
|
259
267
|
This is independent of Django's own `LANGUAGE_CODE`, which controls the language your
|
|
260
268
|
site serves to its users, not the language you read explanations in. Regardless of
|
|
261
269
|
`EXPLAIN_ERRORS_LANGUAGE`, exception type names, Django and Python identifiers, code,
|
|
@@ -270,13 +278,18 @@ generally solid against OpenAI and Anthropic's APIs, but a small local model tha
|
|
|
270
278
|
writes fluent English explanations may produce broken or mixed-language output once
|
|
271
279
|
asked to switch languages.
|
|
272
280
|
|
|
273
|
-
When a language is configured
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
281
|
+
When a language is configured and `OPENAI_MAX_TOKENS` isn't set explicitly, the
|
|
282
|
+
token ceiling defaults to 3,000 instead of 1,000, a flat 3x multiplier applied the
|
|
283
|
+
same way regardless of how well or poorly a given language is known to tokenize, so
|
|
284
|
+
explanations in languages that use more tokens per word than English aren't cut off
|
|
285
|
+
mid-sentence. This is a ceiling, not a target: billing follows tokens actually
|
|
286
|
+
generated, so the extra headroom costs nothing if unused. Setting `OPENAI_MAX_TOKENS`
|
|
287
|
+
explicitly always overrides this scaling, at any value, including one lower than the
|
|
288
|
+
unscaled 1,000 default. The multiplier itself is deliberately generous rather than
|
|
289
|
+
precise: the underlying tokens-per-word figures are estimates, not direct
|
|
290
|
+
measurements, and the default model (`gpt-4o-mini`) uses the `o200k_base` tokenizer,
|
|
291
|
+
which handles non-Latin scripts considerably better than the `cl100k_base`-era ratios
|
|
292
|
+
these estimates lean on.
|
|
280
293
|
|
|
281
294
|
## Codebase-aware explanations (RAG)
|
|
282
295
|
|
|
@@ -361,40 +374,106 @@ else.)
|
|
|
361
374
|
|
|
362
375
|
### Before / after
|
|
363
376
|
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
**With RAG**, grounded in the actual function:
|
|
371
|
-
|
|
372
|
-
> In `myapp/utils.py`, `foo()` calls `int(value)` on line 12 without a
|
|
373
|
-
> `try`/`except`, so any non-numeric `value` raises `ValueError` straight
|
|
374
|
-
> through to the caller. Since `foo()` is called from `myapp/views.py` with
|
|
375
|
-
> unvalidated form input, add validation there or wrap the `int()` call in
|
|
376
|
-
> `foo()` with a clear error message.
|
|
377
|
-
|
|
378
|
-
## Example
|
|
379
|
-
|
|
380
|
-
Here is an example of how to use the middleware in a Django project:
|
|
377
|
+
A real result from the eval harness (`missing_fk`, one of the fixtures in
|
|
378
|
+
`evals/fixtures.py`): a view creates a new `Post` without setting the
|
|
379
|
+
required `author` foreign key. The traceback the model actually received
|
|
380
|
+
was already truncated to `OPENAI_MAX_TRACEBACK_CHARS`, so it contains no
|
|
381
|
+
application code at all, only Django/SQLite internals:
|
|
381
382
|
|
|
382
|
-
```
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
383
|
+
```
|
|
384
|
+
...(truncated)...
|
|
385
|
+
File ".../django/db/models/base.py", line 904, in save
|
|
386
|
+
self.save_base(
|
|
387
|
+
...
|
|
388
|
+
File ".../django/db/backends/utils.py", line 105, in _execute
|
|
389
|
+
return self.cursor.execute(sql, params)
|
|
390
|
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
|
391
|
+
File ".../django/db/backends/sqlite3/base.py", line 359, in execute
|
|
392
|
+
return super().execute(query, params)
|
|
393
|
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
|
394
|
+
django.db.utils.IntegrityError: NOT NULL constraint failed: blog_post.author_id
|
|
395
|
+
```
|
|
386
396
|
|
|
387
|
-
|
|
388
|
-
...
|
|
389
|
-
'explain_errors.middleware.ExplainErrorsMiddleware',
|
|
390
|
-
]
|
|
397
|
+
**Without RAG**, traceback only:
|
|
391
398
|
|
|
392
|
-
|
|
399
|
+
> The error "django.db.utils.IntegrityError: NOT NULL constraint failed: blog_post.author_id" occurs when you attempt to save a model instance but the `author_id` field does not receive a value, even though it requires one (is defined as `NOT NULL`).
|
|
400
|
+
>
|
|
401
|
+
> This typically indicates that the `author` field in your `Post` model (or similar) is mandatory, but you did not provide an `author` when creating or updating the post.
|
|
402
|
+
>
|
|
403
|
+
> To fix this:
|
|
404
|
+
>
|
|
405
|
+
> 1. Ensure that you are assigning a valid user or author instance to the `author` field before saving the `Post` model. For example:
|
|
406
|
+
> ```python
|
|
407
|
+
> post.author = some_user_instance
|
|
408
|
+
> ```
|
|
409
|
+
>
|
|
410
|
+
> 2. If you intended for `author_id` to be optional, adjust the model definition to allow null values by modifying the field:
|
|
411
|
+
> ```python
|
|
412
|
+
> author = models.ForeignKey(User, on_delete=models.CASCADE, null=True)
|
|
413
|
+
> ```
|
|
414
|
+
>
|
|
415
|
+
> 3. If the migration is necessary (if you did change the model), run:
|
|
416
|
+
> ```bash
|
|
417
|
+
> python manage.py makemigrations
|
|
418
|
+
> python manage.py migrate
|
|
419
|
+
> ```
|
|
420
|
+
>
|
|
421
|
+
> Make sure that every time you save a `Post`, it has a valid author assigned.
|
|
393
422
|
|
|
394
|
-
|
|
395
|
-
```
|
|
423
|
+
**With RAG**, grounded in the actual function:
|
|
396
424
|
|
|
397
|
-
|
|
425
|
+
> The error `django.db.utils.IntegrityError: NOT NULL constraint failed: blog_post.author_id` occurs because when you're creating a new `Post` instance in `clone_latest_post`, you're not providing an `author` for the new draft. The `author` field in the `Post` model is a ForeignKey that cannot be null, and thus must be populated.
|
|
426
|
+
>
|
|
427
|
+
> To fix this, modify the `clone_latest_post` function to include an `author` when creating the draft. You could retrieve the author from the latest post and use it like so:
|
|
428
|
+
>
|
|
429
|
+
> ```python
|
|
430
|
+
> def clone_latest_post(request):
|
|
431
|
+
> """Duplicate the most recent post as a new draft, ready for editing."""
|
|
432
|
+
> latest = Post.objects.order_by("-id").first()
|
|
433
|
+
> draft = Post.objects.create(
|
|
434
|
+
> title=f"Copy of {latest.title}",
|
|
435
|
+
> slug=f"copy-of-{latest.slug}",
|
|
436
|
+
> body=latest.body,
|
|
437
|
+
> author=latest.author, # Add this line to specify the author
|
|
438
|
+
> )
|
|
439
|
+
> return HttpResponse(f"Created draft #{draft.id}")
|
|
440
|
+
> ```
|
|
441
|
+
>
|
|
442
|
+
> This ensures the `draft` has a valid `author`, satisfying the NOT NULL constraint.
|
|
443
|
+
|
|
444
|
+
### Does RAG actually help?
|
|
445
|
+
|
|
446
|
+
To check whether RAG-grounded explanations are actually better, not just
|
|
447
|
+
longer, the package ships an eval harness (`evals/`): fifteen deliberately
|
|
448
|
+
broken Django views, each explained twice (once from the traceback alone,
|
|
449
|
+
once with RAG enabled) and judged by a separate model, blind to which
|
|
450
|
+
explanation is which, against the error's known cause and correct fix
|
|
451
|
+
location. Which side the judge sees as "A" is randomized per comparison so
|
|
452
|
+
position can't bias the result.
|
|
453
|
+
|
|
454
|
+
Across three runs (45 judged comparisons, 2 judge failures, 43 scored),
|
|
455
|
+
RAG-on won 35, RAG-off 5, and 3 tied. The gap isn't spread evenly across
|
|
456
|
+
everything the judge checks. It's concentrated in whether the explanation
|
|
457
|
+
names the right file and function, and whether it invents details along
|
|
458
|
+
the way: on `points_to_fix_location`, RAG-on answered yes in 26 of the
|
|
459
|
+
group-A comparisons against RAG-off's 13; on `no_fabrication`, 26 against
|
|
460
|
+
17. Without source access, `gpt-4o-mini` tends to invent a
|
|
461
|
+
plausible-sounding function name or parameter rather than say it doesn't
|
|
462
|
+
know; given the actual code via RAG, it mostly does not.
|
|
463
|
+
|
|
464
|
+
Three limitations are worth knowing before trusting this uncritically: RAG
|
|
465
|
+
can anchor on the wrong retrieved chunk, as it did in one fixture
|
|
466
|
+
(`missing_post_key`) where the fix got redirected to a retrieved template
|
|
467
|
+
instead of the view; the judge is shown the failing function's own
|
|
468
|
+
source, which is the same source RAG-on's retriever draws from, so part
|
|
469
|
+
of RAG-on's `no_fabrication` advantage may be judge and generator
|
|
470
|
+
overlapping on material RAG-off never sees rather than RAG-on being more
|
|
471
|
+
careful; and claim statuses are spot-checked, not exhaustively audited --
|
|
472
|
+
a script that flagged 14 of 363 claims on one run, all correct on manual
|
|
473
|
+
inspection, is a sample that turned up no false positive, not a proof
|
|
474
|
+
that none exists. Full per-fixture results, the judge prompt, and how to
|
|
475
|
+
reproduce this (about $1.37 for a `--runs 3` pass, most of it judge cost)
|
|
476
|
+
are in [`evals/README.md`](evals/README.md).
|
|
398
477
|
|
|
399
478
|
## License
|
|
400
479
|
|
|
@@ -10,7 +10,8 @@ a local model if you prefer not to send code off your machine.
|
|
|
10
10
|
|
|
11
11
|
It can optionally ground explanations in your own project source using a
|
|
12
12
|
local vector index (RAG), so explanations reference the actual code that
|
|
13
|
-
failed instead of staying generic
|
|
13
|
+
failed instead of staying generic (measured — see "Does RAG actually
|
|
14
|
+
help?" below).
|
|
14
15
|
|
|
15
16
|
The middleware supports both synchronous (WSGI) and asynchronous (ASGI)
|
|
16
17
|
views. It auto-detects the view chain at startup and routes requests through
|
|
@@ -139,7 +140,7 @@ developer is actually investigating.
|
|
|
139
140
|
| `OPENAI_MODEL` | No | Model used for explanations. Defaults to `gpt-4o-mini`. |
|
|
140
141
|
| `OPENAI_MAX_TOKENS` | No | Ceiling on tokens generated for the explanation, not a target — the system prompt itself asks for a concise answer. Defaults to `1000`; scales up automatically when `EXPLAIN_ERRORS_LANGUAGE` is set (see below), unless you set this explicitly, which always overrides the scaling. |
|
|
141
142
|
| `OPENAI_TIMEOUT` | No | Request timeout in seconds for the OpenAI client. Defaults to `10`. |
|
|
142
|
-
| `OPENAI_MAX_TRACEBACK_CHARS` | No |
|
|
143
|
+
| `OPENAI_MAX_TRACEBACK_CHARS` | No | Total character budget for the traceback sent to the model. Application frames (your own code, as opposed to Django, the standard library, or installed packages) are always kept; library frames fill whatever budget remains, nearest the raise point first, with an `... N library frames omitted ...` line where frames are dropped. If the application frames alone exceed the budget, falls back to keeping the last N characters of the raw traceback. Defaults to `3000`. |
|
|
143
144
|
| `EXPLAIN_ERRORS_PRESERVE_DEBUG_PAGE` | No | When `True` (the default), the middleware prints the explanation to stdout and returns `None`, so exception handling continues normally and Django renders its standard debug page. Set to `False` to instead return a JSON 500 response, which ends exception handling early (see Compatibility above). |
|
|
144
145
|
| `OPENAI_BASE_URL` (env or settings) | No | Base URL for any OpenAI-compatible API (for example Ollama at `http://localhost:11434/v1`). When set, a missing API key is replaced with a placeholder since local servers do not require one. |
|
|
145
146
|
| `EXPLAIN_ERRORS_MAX_CALLS` | No | Together with `EXPLAIN_ERRORS_WINDOW_SECONDS`, caps API spend to at most this many explanations within a rolling window; once the cap is hit, further errors in that window are not sent for explanation until an earlier call ages out. Defaults to `5`. |
|
|
@@ -212,6 +213,13 @@ read them in another language instead:
|
|
|
212
213
|
EXPLAIN_ERRORS_LANGUAGE = "Spanish" # or the code form, "es"
|
|
213
214
|
```
|
|
214
215
|
|
|
216
|
+
There is no fixed list of supported languages. `EXPLAIN_ERRORS_LANGUAGE` accepts
|
|
217
|
+
any language the configured model can write, because the setting adds one clause
|
|
218
|
+
to the system prompt (see `LANGUAGE_CLAUSE_TEMPLATE` in
|
|
219
|
+
`explain_errors/middleware.py`), not a translation catalog with its own
|
|
220
|
+
maintained language list. How well it works varies by model; see the known
|
|
221
|
+
limitation below.
|
|
222
|
+
|
|
215
223
|
This is independent of Django's own `LANGUAGE_CODE`, which controls the language your
|
|
216
224
|
site serves to its users, not the language you read explanations in. Regardless of
|
|
217
225
|
`EXPLAIN_ERRORS_LANGUAGE`, exception type names, Django and Python identifiers, code,
|
|
@@ -226,13 +234,18 @@ generally solid against OpenAI and Anthropic's APIs, but a small local model tha
|
|
|
226
234
|
writes fluent English explanations may produce broken or mixed-language output once
|
|
227
235
|
asked to switch languages.
|
|
228
236
|
|
|
229
|
-
When a language is configured
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
237
|
+
When a language is configured and `OPENAI_MAX_TOKENS` isn't set explicitly, the
|
|
238
|
+
token ceiling defaults to 3,000 instead of 1,000, a flat 3x multiplier applied the
|
|
239
|
+
same way regardless of how well or poorly a given language is known to tokenize, so
|
|
240
|
+
explanations in languages that use more tokens per word than English aren't cut off
|
|
241
|
+
mid-sentence. This is a ceiling, not a target: billing follows tokens actually
|
|
242
|
+
generated, so the extra headroom costs nothing if unused. Setting `OPENAI_MAX_TOKENS`
|
|
243
|
+
explicitly always overrides this scaling, at any value, including one lower than the
|
|
244
|
+
unscaled 1,000 default. The multiplier itself is deliberately generous rather than
|
|
245
|
+
precise: the underlying tokens-per-word figures are estimates, not direct
|
|
246
|
+
measurements, and the default model (`gpt-4o-mini`) uses the `o200k_base` tokenizer,
|
|
247
|
+
which handles non-Latin scripts considerably better than the `cl100k_base`-era ratios
|
|
248
|
+
these estimates lean on.
|
|
236
249
|
|
|
237
250
|
## Codebase-aware explanations (RAG)
|
|
238
251
|
|
|
@@ -317,40 +330,106 @@ else.)
|
|
|
317
330
|
|
|
318
331
|
### Before / after
|
|
319
332
|
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
**With RAG**, grounded in the actual function:
|
|
327
|
-
|
|
328
|
-
> In `myapp/utils.py`, `foo()` calls `int(value)` on line 12 without a
|
|
329
|
-
> `try`/`except`, so any non-numeric `value` raises `ValueError` straight
|
|
330
|
-
> through to the caller. Since `foo()` is called from `myapp/views.py` with
|
|
331
|
-
> unvalidated form input, add validation there or wrap the `int()` call in
|
|
332
|
-
> `foo()` with a clear error message.
|
|
333
|
-
|
|
334
|
-
## Example
|
|
335
|
-
|
|
336
|
-
Here is an example of how to use the middleware in a Django project:
|
|
333
|
+
A real result from the eval harness (`missing_fk`, one of the fixtures in
|
|
334
|
+
`evals/fixtures.py`): a view creates a new `Post` without setting the
|
|
335
|
+
required `author` foreign key. The traceback the model actually received
|
|
336
|
+
was already truncated to `OPENAI_MAX_TRACEBACK_CHARS`, so it contains no
|
|
337
|
+
application code at all, only Django/SQLite internals:
|
|
337
338
|
|
|
338
|
-
```
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
339
|
+
```
|
|
340
|
+
...(truncated)...
|
|
341
|
+
File ".../django/db/models/base.py", line 904, in save
|
|
342
|
+
self.save_base(
|
|
343
|
+
...
|
|
344
|
+
File ".../django/db/backends/utils.py", line 105, in _execute
|
|
345
|
+
return self.cursor.execute(sql, params)
|
|
346
|
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
|
347
|
+
File ".../django/db/backends/sqlite3/base.py", line 359, in execute
|
|
348
|
+
return super().execute(query, params)
|
|
349
|
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
|
350
|
+
django.db.utils.IntegrityError: NOT NULL constraint failed: blog_post.author_id
|
|
351
|
+
```
|
|
342
352
|
|
|
343
|
-
|
|
344
|
-
...
|
|
345
|
-
'explain_errors.middleware.ExplainErrorsMiddleware',
|
|
346
|
-
]
|
|
353
|
+
**Without RAG**, traceback only:
|
|
347
354
|
|
|
348
|
-
|
|
355
|
+
> The error "django.db.utils.IntegrityError: NOT NULL constraint failed: blog_post.author_id" occurs when you attempt to save a model instance but the `author_id` field does not receive a value, even though it requires one (is defined as `NOT NULL`).
|
|
356
|
+
>
|
|
357
|
+
> This typically indicates that the `author` field in your `Post` model (or similar) is mandatory, but you did not provide an `author` when creating or updating the post.
|
|
358
|
+
>
|
|
359
|
+
> To fix this:
|
|
360
|
+
>
|
|
361
|
+
> 1. Ensure that you are assigning a valid user or author instance to the `author` field before saving the `Post` model. For example:
|
|
362
|
+
> ```python
|
|
363
|
+
> post.author = some_user_instance
|
|
364
|
+
> ```
|
|
365
|
+
>
|
|
366
|
+
> 2. If you intended for `author_id` to be optional, adjust the model definition to allow null values by modifying the field:
|
|
367
|
+
> ```python
|
|
368
|
+
> author = models.ForeignKey(User, on_delete=models.CASCADE, null=True)
|
|
369
|
+
> ```
|
|
370
|
+
>
|
|
371
|
+
> 3. If the migration is necessary (if you did change the model), run:
|
|
372
|
+
> ```bash
|
|
373
|
+
> python manage.py makemigrations
|
|
374
|
+
> python manage.py migrate
|
|
375
|
+
> ```
|
|
376
|
+
>
|
|
377
|
+
> Make sure that every time you save a `Post`, it has a valid author assigned.
|
|
349
378
|
|
|
350
|
-
|
|
351
|
-
```
|
|
379
|
+
**With RAG**, grounded in the actual function:
|
|
352
380
|
|
|
353
|
-
|
|
381
|
+
> The error `django.db.utils.IntegrityError: NOT NULL constraint failed: blog_post.author_id` occurs because when you're creating a new `Post` instance in `clone_latest_post`, you're not providing an `author` for the new draft. The `author` field in the `Post` model is a ForeignKey that cannot be null, and thus must be populated.
|
|
382
|
+
>
|
|
383
|
+
> To fix this, modify the `clone_latest_post` function to include an `author` when creating the draft. You could retrieve the author from the latest post and use it like so:
|
|
384
|
+
>
|
|
385
|
+
> ```python
|
|
386
|
+
> def clone_latest_post(request):
|
|
387
|
+
> """Duplicate the most recent post as a new draft, ready for editing."""
|
|
388
|
+
> latest = Post.objects.order_by("-id").first()
|
|
389
|
+
> draft = Post.objects.create(
|
|
390
|
+
> title=f"Copy of {latest.title}",
|
|
391
|
+
> slug=f"copy-of-{latest.slug}",
|
|
392
|
+
> body=latest.body,
|
|
393
|
+
> author=latest.author, # Add this line to specify the author
|
|
394
|
+
> )
|
|
395
|
+
> return HttpResponse(f"Created draft #{draft.id}")
|
|
396
|
+
> ```
|
|
397
|
+
>
|
|
398
|
+
> This ensures the `draft` has a valid `author`, satisfying the NOT NULL constraint.
|
|
399
|
+
|
|
400
|
+
### Does RAG actually help?
|
|
401
|
+
|
|
402
|
+
To check whether RAG-grounded explanations are actually better, not just
|
|
403
|
+
longer, the package ships an eval harness (`evals/`): fifteen deliberately
|
|
404
|
+
broken Django views, each explained twice (once from the traceback alone,
|
|
405
|
+
once with RAG enabled) and judged by a separate model, blind to which
|
|
406
|
+
explanation is which, against the error's known cause and correct fix
|
|
407
|
+
location. Which side the judge sees as "A" is randomized per comparison so
|
|
408
|
+
position can't bias the result.
|
|
409
|
+
|
|
410
|
+
Across three runs (45 judged comparisons, 2 judge failures, 43 scored),
|
|
411
|
+
RAG-on won 35, RAG-off 5, and 3 tied. The gap isn't spread evenly across
|
|
412
|
+
everything the judge checks. It's concentrated in whether the explanation
|
|
413
|
+
names the right file and function, and whether it invents details along
|
|
414
|
+
the way: on `points_to_fix_location`, RAG-on answered yes in 26 of the
|
|
415
|
+
group-A comparisons against RAG-off's 13; on `no_fabrication`, 26 against
|
|
416
|
+
17. Without source access, `gpt-4o-mini` tends to invent a
|
|
417
|
+
plausible-sounding function name or parameter rather than say it doesn't
|
|
418
|
+
know; given the actual code via RAG, it mostly does not.
|
|
419
|
+
|
|
420
|
+
Three limitations are worth knowing before trusting this uncritically: RAG
|
|
421
|
+
can anchor on the wrong retrieved chunk, as it did in one fixture
|
|
422
|
+
(`missing_post_key`) where the fix got redirected to a retrieved template
|
|
423
|
+
instead of the view; the judge is shown the failing function's own
|
|
424
|
+
source, which is the same source RAG-on's retriever draws from, so part
|
|
425
|
+
of RAG-on's `no_fabrication` advantage may be judge and generator
|
|
426
|
+
overlapping on material RAG-off never sees rather than RAG-on being more
|
|
427
|
+
careful; and claim statuses are spot-checked, not exhaustively audited --
|
|
428
|
+
a script that flagged 14 of 363 claims on one run, all correct on manual
|
|
429
|
+
inspection, is a sample that turned up no false positive, not a proof
|
|
430
|
+
that none exists. Full per-fixture results, the judge prompt, and how to
|
|
431
|
+
reproduce this (about $1.37 for a `--runs 3` pass, most of it judge cost)
|
|
432
|
+
are in [`evals/README.md`](evals/README.md).
|
|
354
433
|
|
|
355
434
|
## License
|
|
356
435
|
|
{django_explain_errors-0.6.0 → django_explain_errors-0.7.0/django_explain_errors.egg-info}/PKG-INFO
RENAMED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: django-explain-errors
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.7.0
|
|
4
4
|
Summary: Django middleware that explains unhandled exceptions in DEBUG using an LLM, optionally grounded in your own project source via a local vector index. Works with OpenAI, Claude, or any OpenAI-compatible endpoint including local models.
|
|
5
5
|
Home-page: https://github.com/topunix/django-explain-errors
|
|
6
6
|
Author: topunix
|
|
@@ -54,7 +54,8 @@ a local model if you prefer not to send code off your machine.
|
|
|
54
54
|
|
|
55
55
|
It can optionally ground explanations in your own project source using a
|
|
56
56
|
local vector index (RAG), so explanations reference the actual code that
|
|
57
|
-
failed instead of staying generic
|
|
57
|
+
failed instead of staying generic (measured — see "Does RAG actually
|
|
58
|
+
help?" below).
|
|
58
59
|
|
|
59
60
|
The middleware supports both synchronous (WSGI) and asynchronous (ASGI)
|
|
60
61
|
views. It auto-detects the view chain at startup and routes requests through
|
|
@@ -183,7 +184,7 @@ developer is actually investigating.
|
|
|
183
184
|
| `OPENAI_MODEL` | No | Model used for explanations. Defaults to `gpt-4o-mini`. |
|
|
184
185
|
| `OPENAI_MAX_TOKENS` | No | Ceiling on tokens generated for the explanation, not a target — the system prompt itself asks for a concise answer. Defaults to `1000`; scales up automatically when `EXPLAIN_ERRORS_LANGUAGE` is set (see below), unless you set this explicitly, which always overrides the scaling. |
|
|
185
186
|
| `OPENAI_TIMEOUT` | No | Request timeout in seconds for the OpenAI client. Defaults to `10`. |
|
|
186
|
-
| `OPENAI_MAX_TRACEBACK_CHARS` | No |
|
|
187
|
+
| `OPENAI_MAX_TRACEBACK_CHARS` | No | Total character budget for the traceback sent to the model. Application frames (your own code, as opposed to Django, the standard library, or installed packages) are always kept; library frames fill whatever budget remains, nearest the raise point first, with an `... N library frames omitted ...` line where frames are dropped. If the application frames alone exceed the budget, falls back to keeping the last N characters of the raw traceback. Defaults to `3000`. |
|
|
187
188
|
| `EXPLAIN_ERRORS_PRESERVE_DEBUG_PAGE` | No | When `True` (the default), the middleware prints the explanation to stdout and returns `None`, so exception handling continues normally and Django renders its standard debug page. Set to `False` to instead return a JSON 500 response, which ends exception handling early (see Compatibility above). |
|
|
188
189
|
| `OPENAI_BASE_URL` (env or settings) | No | Base URL for any OpenAI-compatible API (for example Ollama at `http://localhost:11434/v1`). When set, a missing API key is replaced with a placeholder since local servers do not require one. |
|
|
189
190
|
| `EXPLAIN_ERRORS_MAX_CALLS` | No | Together with `EXPLAIN_ERRORS_WINDOW_SECONDS`, caps API spend to at most this many explanations within a rolling window; once the cap is hit, further errors in that window are not sent for explanation until an earlier call ages out. Defaults to `5`. |
|
|
@@ -256,6 +257,13 @@ read them in another language instead:
|
|
|
256
257
|
EXPLAIN_ERRORS_LANGUAGE = "Spanish" # or the code form, "es"
|
|
257
258
|
```
|
|
258
259
|
|
|
260
|
+
There is no fixed list of supported languages. `EXPLAIN_ERRORS_LANGUAGE` accepts
|
|
261
|
+
any language the configured model can write, because the setting adds one clause
|
|
262
|
+
to the system prompt (see `LANGUAGE_CLAUSE_TEMPLATE` in
|
|
263
|
+
`explain_errors/middleware.py`), not a translation catalog with its own
|
|
264
|
+
maintained language list. How well it works varies by model; see the known
|
|
265
|
+
limitation below.
|
|
266
|
+
|
|
259
267
|
This is independent of Django's own `LANGUAGE_CODE`, which controls the language your
|
|
260
268
|
site serves to its users, not the language you read explanations in. Regardless of
|
|
261
269
|
`EXPLAIN_ERRORS_LANGUAGE`, exception type names, Django and Python identifiers, code,
|
|
@@ -270,13 +278,18 @@ generally solid against OpenAI and Anthropic's APIs, but a small local model tha
|
|
|
270
278
|
writes fluent English explanations may produce broken or mixed-language output once
|
|
271
279
|
asked to switch languages.
|
|
272
280
|
|
|
273
|
-
When a language is configured
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
281
|
+
When a language is configured and `OPENAI_MAX_TOKENS` isn't set explicitly, the
|
|
282
|
+
token ceiling defaults to 3,000 instead of 1,000, a flat 3x multiplier applied the
|
|
283
|
+
same way regardless of how well or poorly a given language is known to tokenize, so
|
|
284
|
+
explanations in languages that use more tokens per word than English aren't cut off
|
|
285
|
+
mid-sentence. This is a ceiling, not a target: billing follows tokens actually
|
|
286
|
+
generated, so the extra headroom costs nothing if unused. Setting `OPENAI_MAX_TOKENS`
|
|
287
|
+
explicitly always overrides this scaling, at any value, including one lower than the
|
|
288
|
+
unscaled 1,000 default. The multiplier itself is deliberately generous rather than
|
|
289
|
+
precise: the underlying tokens-per-word figures are estimates, not direct
|
|
290
|
+
measurements, and the default model (`gpt-4o-mini`) uses the `o200k_base` tokenizer,
|
|
291
|
+
which handles non-Latin scripts considerably better than the `cl100k_base`-era ratios
|
|
292
|
+
these estimates lean on.
|
|
280
293
|
|
|
281
294
|
## Codebase-aware explanations (RAG)
|
|
282
295
|
|
|
@@ -361,40 +374,106 @@ else.)
|
|
|
361
374
|
|
|
362
375
|
### Before / after
|
|
363
376
|
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
**With RAG**, grounded in the actual function:
|
|
371
|
-
|
|
372
|
-
> In `myapp/utils.py`, `foo()` calls `int(value)` on line 12 without a
|
|
373
|
-
> `try`/`except`, so any non-numeric `value` raises `ValueError` straight
|
|
374
|
-
> through to the caller. Since `foo()` is called from `myapp/views.py` with
|
|
375
|
-
> unvalidated form input, add validation there or wrap the `int()` call in
|
|
376
|
-
> `foo()` with a clear error message.
|
|
377
|
-
|
|
378
|
-
## Example
|
|
379
|
-
|
|
380
|
-
Here is an example of how to use the middleware in a Django project:
|
|
377
|
+
A real result from the eval harness (`missing_fk`, one of the fixtures in
|
|
378
|
+
`evals/fixtures.py`): a view creates a new `Post` without setting the
|
|
379
|
+
required `author` foreign key. The traceback the model actually received
|
|
380
|
+
was already truncated to `OPENAI_MAX_TRACEBACK_CHARS`, so it contains no
|
|
381
|
+
application code at all, only Django/SQLite internals:
|
|
381
382
|
|
|
382
|
-
```
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
383
|
+
```
|
|
384
|
+
...(truncated)...
|
|
385
|
+
File ".../django/db/models/base.py", line 904, in save
|
|
386
|
+
self.save_base(
|
|
387
|
+
...
|
|
388
|
+
File ".../django/db/backends/utils.py", line 105, in _execute
|
|
389
|
+
return self.cursor.execute(sql, params)
|
|
390
|
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
|
391
|
+
File ".../django/db/backends/sqlite3/base.py", line 359, in execute
|
|
392
|
+
return super().execute(query, params)
|
|
393
|
+
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
|
|
394
|
+
django.db.utils.IntegrityError: NOT NULL constraint failed: blog_post.author_id
|
|
395
|
+
```
|
|
386
396
|
|
|
387
|
-
|
|
388
|
-
...
|
|
389
|
-
'explain_errors.middleware.ExplainErrorsMiddleware',
|
|
390
|
-
]
|
|
397
|
+
**Without RAG**, traceback only:
|
|
391
398
|
|
|
392
|
-
|
|
399
|
+
> The error "django.db.utils.IntegrityError: NOT NULL constraint failed: blog_post.author_id" occurs when you attempt to save a model instance but the `author_id` field does not receive a value, even though it requires one (is defined as `NOT NULL`).
|
|
400
|
+
>
|
|
401
|
+
> This typically indicates that the `author` field in your `Post` model (or similar) is mandatory, but you did not provide an `author` when creating or updating the post.
|
|
402
|
+
>
|
|
403
|
+
> To fix this:
|
|
404
|
+
>
|
|
405
|
+
> 1. Ensure that you are assigning a valid user or author instance to the `author` field before saving the `Post` model. For example:
|
|
406
|
+
> ```python
|
|
407
|
+
> post.author = some_user_instance
|
|
408
|
+
> ```
|
|
409
|
+
>
|
|
410
|
+
> 2. If you intended for `author_id` to be optional, adjust the model definition to allow null values by modifying the field:
|
|
411
|
+
> ```python
|
|
412
|
+
> author = models.ForeignKey(User, on_delete=models.CASCADE, null=True)
|
|
413
|
+
> ```
|
|
414
|
+
>
|
|
415
|
+
> 3. If the migration is necessary (if you did change the model), run:
|
|
416
|
+
> ```bash
|
|
417
|
+
> python manage.py makemigrations
|
|
418
|
+
> python manage.py migrate
|
|
419
|
+
> ```
|
|
420
|
+
>
|
|
421
|
+
> Make sure that every time you save a `Post`, it has a valid author assigned.
|
|
393
422
|
|
|
394
|
-
|
|
395
|
-
```
|
|
423
|
+
**With RAG**, grounded in the actual function:
|
|
396
424
|
|
|
397
|
-
|
|
425
|
+
> The error `django.db.utils.IntegrityError: NOT NULL constraint failed: blog_post.author_id` occurs because when you're creating a new `Post` instance in `clone_latest_post`, you're not providing an `author` for the new draft. The `author` field in the `Post` model is a ForeignKey that cannot be null, and thus must be populated.
|
|
426
|
+
>
|
|
427
|
+
> To fix this, modify the `clone_latest_post` function to include an `author` when creating the draft. You could retrieve the author from the latest post and use it like so:
|
|
428
|
+
>
|
|
429
|
+
> ```python
|
|
430
|
+
> def clone_latest_post(request):
|
|
431
|
+
> """Duplicate the most recent post as a new draft, ready for editing."""
|
|
432
|
+
> latest = Post.objects.order_by("-id").first()
|
|
433
|
+
> draft = Post.objects.create(
|
|
434
|
+
> title=f"Copy of {latest.title}",
|
|
435
|
+
> slug=f"copy-of-{latest.slug}",
|
|
436
|
+
> body=latest.body,
|
|
437
|
+
> author=latest.author, # Add this line to specify the author
|
|
438
|
+
> )
|
|
439
|
+
> return HttpResponse(f"Created draft #{draft.id}")
|
|
440
|
+
> ```
|
|
441
|
+
>
|
|
442
|
+
> This ensures the `draft` has a valid `author`, satisfying the NOT NULL constraint.
|
|
443
|
+
|
|
444
|
+
### Does RAG actually help?
|
|
445
|
+
|
|
446
|
+
To check whether RAG-grounded explanations are actually better, not just
|
|
447
|
+
longer, the package ships an eval harness (`evals/`): fifteen deliberately
|
|
448
|
+
broken Django views, each explained twice (once from the traceback alone,
|
|
449
|
+
once with RAG enabled) and judged by a separate model, blind to which
|
|
450
|
+
explanation is which, against the error's known cause and correct fix
|
|
451
|
+
location. Which side the judge sees as "A" is randomized per comparison so
|
|
452
|
+
position can't bias the result.
|
|
453
|
+
|
|
454
|
+
Across three runs (45 judged comparisons, 2 judge failures, 43 scored),
|
|
455
|
+
RAG-on won 35, RAG-off 5, and 3 tied. The gap isn't spread evenly across
|
|
456
|
+
everything the judge checks. It's concentrated in whether the explanation
|
|
457
|
+
names the right file and function, and whether it invents details along
|
|
458
|
+
the way: on `points_to_fix_location`, RAG-on answered yes in 26 of the
|
|
459
|
+
group-A comparisons against RAG-off's 13; on `no_fabrication`, 26 against
|
|
460
|
+
17. Without source access, `gpt-4o-mini` tends to invent a
|
|
461
|
+
plausible-sounding function name or parameter rather than say it doesn't
|
|
462
|
+
know; given the actual code via RAG, it mostly does not.
|
|
463
|
+
|
|
464
|
+
Three limitations are worth knowing before trusting this uncritically: RAG
|
|
465
|
+
can anchor on the wrong retrieved chunk, as it did in one fixture
|
|
466
|
+
(`missing_post_key`) where the fix got redirected to a retrieved template
|
|
467
|
+
instead of the view; the judge is shown the failing function's own
|
|
468
|
+
source, which is the same source RAG-on's retriever draws from, so part
|
|
469
|
+
of RAG-on's `no_fabrication` advantage may be judge and generator
|
|
470
|
+
overlapping on material RAG-off never sees rather than RAG-on being more
|
|
471
|
+
careful; and claim statuses are spot-checked, not exhaustively audited --
|
|
472
|
+
a script that flagged 14 of 363 claims on one run, all correct on manual
|
|
473
|
+
inspection, is a sample that turned up no false positive, not a proof
|
|
474
|
+
that none exists. Full per-fixture results, the judge prompt, and how to
|
|
475
|
+
reproduce this (about $1.37 for a `--runs 3` pass, most of it judge cost)
|
|
476
|
+
are in [`evals/README.md`](evals/README.md).
|
|
398
477
|
|
|
399
478
|
## License
|
|
400
479
|
|
|
@@ -12,6 +12,7 @@ explain_errors/client.py
|
|
|
12
12
|
explain_errors/middleware.py
|
|
13
13
|
explain_errors/sanitize.py
|
|
14
14
|
explain_errors/throttle.py
|
|
15
|
+
explain_errors/tracebacks.py
|
|
15
16
|
explain_errors/management/__init__.py
|
|
16
17
|
explain_errors/management/commands/__init__.py
|
|
17
18
|
explain_errors/management/commands/build_error_index.py
|
|
@@ -21,10 +22,14 @@ explain_errors/rag/retriever.py
|
|
|
21
22
|
explain_errors/rag/store.py
|
|
22
23
|
tests/test_build_error_index.py
|
|
23
24
|
tests/test_client.py
|
|
25
|
+
tests/test_eval_fixtures.py
|
|
26
|
+
tests/test_eval_judge.py
|
|
27
|
+
tests/test_eval_run.py
|
|
24
28
|
tests/test_language.py
|
|
25
29
|
tests/test_middleware.py
|
|
26
30
|
tests/test_rag.py
|
|
27
31
|
tests/test_sanitize.py
|
|
28
32
|
tests/test_signals.py
|
|
29
33
|
tests/test_throttle.py
|
|
34
|
+
tests/test_tracebacks.py
|
|
30
35
|
tests/test_truncation.py
|