pipelinemd 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pipelinemd/__init__.py ADDED
@@ -0,0 +1,43 @@
1
+ """pipelinemd - GitLab CI/CD failure doctor.
2
+
3
+ Two halves, deliberately separable:
4
+
5
+ * a **deterministic distiller** that reduces a job trace to the lines that
6
+ explain it and matches them against a catalog of known failure signatures -
7
+ no network, no model, no API key;
8
+ * an optional **LLM diagnosis** that reads only the distilled evidence and
9
+ names the root cause.
10
+
11
+ The first half is the product. The second is the upgrade.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ __version__ = "0.1.0"
17
+
18
+ from .models import (
19
+ Category,
20
+ Citation,
21
+ Confidence,
22
+ Diagnosis,
23
+ DistilledLog,
24
+ Fix,
25
+ JobRef,
26
+ Report,
27
+ Rule,
28
+ RuleHit,
29
+ )
30
+
31
+ __all__ = [
32
+ "Category",
33
+ "Citation",
34
+ "Confidence",
35
+ "Diagnosis",
36
+ "DistilledLog",
37
+ "Fix",
38
+ "JobRef",
39
+ "Report",
40
+ "Rule",
41
+ "RuleHit",
42
+ "__version__",
43
+ ]
pipelinemd/__main__.py ADDED
@@ -0,0 +1,10 @@
1
+ """Allow `python -m pipelinemd`."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import sys
6
+
7
+ from .cli import main
8
+
9
+ if __name__ == "__main__":
10
+ sys.exit(main())
@@ -0,0 +1,94 @@
1
+ """How far to trust one analysis, and when to hand it to a person instead.
2
+
3
+ The confidence is the level an analysis has earned from every signal behind
4
+ it, never more than its weakest part:
5
+
6
+ * **The classification** - the top rule's confidence, or retry history's. With
7
+ nothing to classify by, it is low: v1 makes no claim about what it cannot
8
+ place, and a diagnosis no rule corroborates is the model's word alone.
9
+ * **The model's own rating** of its diagnosis, when there is one.
10
+ * **Two signs of trouble**, each costing one level: a diagnosis that cited
11
+ lines which are not in the evidence, and a model that places the failure in a
12
+ different class from the rules.
13
+
14
+ It stays an ordinal - high, medium, low - like every confidence in this tool.
15
+ A decimal would claim a precision nothing here was calibrated to; `make eval`
16
+ reports how often each level is right instead, which is the evidence the level
17
+ can actually be held to.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ from dataclasses import dataclass
23
+
24
+ from .models import Confidence, FailureClass, Report
25
+ from .taxonomy import classify, model_disagreement
26
+
27
+ _ORDER: tuple[Confidence, ...] = (Confidence.LOW, Confidence.MEDIUM, Confidence.HIGH)
28
+
29
+ #: An analysis below this is presented as needing human review, which makes
30
+ #: low the only level under it. For the classification signal, the corpus
31
+ #: draws the line here: `make eval` reports class accuracy per level, and low
32
+ #: is the one that is wrong more often than right. The three diagnosis-side
33
+ #: adjustments cannot occur in the eval, which scores no diagnosis, so sending
34
+ #: an analysis they demoted to review is a judgement, not a measurement - see
35
+ #: docs/evaluation.md. The JSON report carries the level as well as the flag,
36
+ #: so a consumer that wants a stricter bar can apply one.
37
+ REVIEW_BELOW = Confidence.MEDIUM
38
+
39
+
40
+ def rank(confidence: Confidence) -> int:
41
+ """Low 0, medium 1, high 2: for comparing levels, never for display."""
42
+ return _ORDER.index(confidence)
43
+
44
+
45
+ def _lower(confidence: Confidence) -> Confidence:
46
+ return _ORDER[max(0, rank(confidence) - 1)]
47
+
48
+
49
+ @dataclass(frozen=True, slots=True)
50
+ class Assessment:
51
+ confidence: Confidence
52
+ #: Every reason the confidence is below high, in the order they applied.
53
+ #: Empty exactly when the confidence is high.
54
+ reasons: tuple[str, ...] = ()
55
+
56
+ @property
57
+ def needs_review(self) -> bool:
58
+ return rank(self.confidence) < rank(REVIEW_BELOW)
59
+
60
+
61
+ def assess(report: Report) -> Assessment:
62
+ """Derived on demand from the report, like the classification and the cost."""
63
+ classification = classify(report)
64
+ level = classification.confidence
65
+ reasons: list[str] = []
66
+
67
+ rule = report.top_hit.rule if classification.source == "rules" and report.top_hit else None
68
+ if rule is not None and rule.confidence is not Confidence.HIGH:
69
+ reasons.append(f"{rule.id} is a {rule.confidence}-confidence rule")
70
+ # Below the rule's own level - a transient rule that failed every attempt -
71
+ # or with no rule behind it at all, the basis is what explains the level.
72
+ if level is not Confidence.HIGH and (rule is None or rank(level) < rank(rule.confidence)):
73
+ reasons.append(classification.basis)
74
+
75
+ if (diagnosis := report.diagnosis) is not None:
76
+ if diagnosis.confidence is not Confidence.HIGH:
77
+ reasons.append(f"the model rated its diagnosis {diagnosis.confidence}")
78
+ if rank(diagnosis.confidence) < rank(level):
79
+ level = diagnosis.confidence
80
+ if diagnosis.unresolved_citations:
81
+ invented = ", ".join(f"L{n}" for n in diagnosis.unresolved_citations)
82
+ reasons.append(f"the diagnosis cited {invented}, which is not in the evidence")
83
+ level = _lower(level)
84
+ # Only a disagreement with a class the rules actually chose. Against
85
+ # `unclassified` the model is not contradicting anything, and the low
86
+ # classification has already said what there is to say.
87
+ other = model_disagreement(report)
88
+ if other is not None and classification.failure_class is not FailureClass.UNCLASSIFIED:
89
+ reasons.append(
90
+ f"the model reads it as {other}, the rules as {classification.failure_class}"
91
+ )
92
+ level = _lower(level)
93
+
94
+ return Assessment(confidence=level, reasons=tuple(reasons))