@plainconceptsplatform/workflows 1.0.8 → 1.0.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -48,7 +48,7 @@ let thrown =
|
|
|
48
48
|
AppExceptions
|
|
49
49
|
| where TimeGenerated > window
|
|
50
50
|
| extend Problem = ProblemId, Kind = ExceptionType, Stack = Details,
|
|
51
|
-
Msg =
|
|
51
|
+
Msg = OuterMessage, Category = "";
|
|
52
52
|
let logged =
|
|
53
53
|
AppTraces
|
|
54
54
|
| where TimeGenerated > window
|
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
# Managed by @plainconceptsplatform/workflows. Source: loops/templates/agentics/agentics-app-errors.yml. Update with `workflows update --force`; consumer edits may be overwritten.
|
|
2
2
|
name: "Agentics: App Errors"
|
|
3
3
|
|
|
4
|
-
#
|
|
5
|
-
# the distinct problems as issues here, labelled so refine sizes them and implement
|
|
4
|
+
# Every eight hours, ask Application Insights what the deployed application actually threw,
|
|
5
|
+
# and file the distinct problems as issues here, labelled so refine sizes them and implement
|
|
6
|
+
# fixes them.
|
|
6
7
|
#
|
|
7
8
|
# THIS REPORT STAYS IN THIS REPOSITORY. That is the design constraint, and it is the opposite
|
|
8
9
|
# of the one in agentics-error-report.yml. That workflow exists to carry a *workflow* failure
|
|
@@ -28,14 +29,14 @@ name: "Agentics: App Errors"
|
|
|
28
29
|
# --role "Monitoring Reader" \
|
|
29
30
|
# --scope "$(az monitor log-analytics workspace show -g "$AZURE_RG" -n "$WORKSPACE" --query id -o tsv)"
|
|
30
31
|
|
|
31
|
-
run-name: "App errors: ${{ github.event.inputs.app-env || 'pre' }}, last ${{ github.event.inputs.lookback-hours || '
|
|
32
|
+
run-name: "App errors: ${{ github.event.inputs.app-env || 'pre' }}, last ${{ github.event.inputs.lookback-hours || '8' }}h"
|
|
32
33
|
|
|
33
34
|
on:
|
|
34
35
|
schedule:
|
|
35
|
-
#
|
|
36
|
-
# runs on a GitHub-hosted runner and never touches the agent fleet, so there is no
|
|
37
|
-
# stagger against other repositories.
|
|
38
|
-
- cron: "
|
|
36
|
+
# Every eight hours: a failure should surface the same working day, not tomorrow morning.
|
|
37
|
+
# This runs on a GitHub-hosted runner and never touches the agent fleet, so there is no
|
|
38
|
+
# slot to stagger against other repositories.
|
|
39
|
+
- cron: "17 1,9,17 * * *"
|
|
39
40
|
workflow_dispatch:
|
|
40
41
|
inputs:
|
|
41
42
|
app-env:
|
|
@@ -47,7 +48,7 @@ on:
|
|
|
47
48
|
description: "How far back to look"
|
|
48
49
|
required: false
|
|
49
50
|
type: string
|
|
50
|
-
default: '
|
|
51
|
+
default: '8'
|
|
51
52
|
dry-run:
|
|
52
53
|
description: "Work out what would be filed and file nothing"
|
|
53
54
|
required: false
|
|
@@ -64,7 +65,7 @@ env:
|
|
|
64
65
|
APP_ENV: "pre"
|
|
65
66
|
# Matches the cron. A window wider than the schedule counts the same problem twice; a
|
|
66
67
|
# narrower one loses whatever happened in the gap.
|
|
67
|
-
LOOKBACK_HOURS: "
|
|
68
|
+
LOOKBACK_HOURS: "8"
|
|
68
69
|
# The floor. A problem below this is real but not yet news, and a belt buried under
|
|
69
70
|
# single-occurrence reports is a belt somebody switches off.
|
|
70
71
|
MIN_OCCURRENCES: "5"
|
|
@@ -138,12 +138,19 @@ jobs:
|
|
|
138
138
|
# needs-maintainer while this run waited in a queue, the issue now carries review and
|
|
139
139
|
# the verdict must stand: the reserve step clears review blindly, so eligibility has to
|
|
140
140
|
# catch it first (Pliny-Bot #372).
|
|
141
|
+
#
|
|
142
|
+
# BUT: when mode is rerefine, the trigger was a human comment answering a question the
|
|
143
|
+
# bot asked and then parked for review. That comment IS the human review the label was
|
|
144
|
+
# waiting for, so refusing to run would silence exactly the feedback the bot asked for
|
|
145
|
+
# (Odyssey #226: the bot asked a question, the user answered, and the review label
|
|
146
|
+
# made the router skip the answer). In rerefine mode the label is cleared and the
|
|
147
|
+
# refine proceeds; in first mode the triage verdict still stands.
|
|
141
148
|
eligibility:
|
|
142
149
|
needs: [still_open]
|
|
143
150
|
if: needs.still_open.outputs.open == 'true'
|
|
144
151
|
runs-on: agents-arc
|
|
145
152
|
permissions:
|
|
146
|
-
issues:
|
|
153
|
+
issues: write
|
|
147
154
|
outputs:
|
|
148
155
|
eligible: ${{ steps.check.outputs.eligible }}
|
|
149
156
|
steps:
|
|
@@ -152,11 +159,21 @@ jobs:
|
|
|
152
159
|
env:
|
|
153
160
|
GH_TOKEN: ${{ github.token }}
|
|
154
161
|
ISSUE_NUMBER: ${{ inputs.issue-number }}
|
|
162
|
+
MODE: ${{ inputs.mode }}
|
|
155
163
|
run: |
|
|
156
164
|
set -euo pipefail
|
|
157
165
|
labels=$(gh issue view "$ISSUE_NUMBER" --repo "$GITHUB_REPOSITORY" --json labels \
|
|
158
166
|
--jq '[.labels[].name]')
|
|
159
167
|
if jq -e 'index("review")' >/dev/null <<<"$labels"; then
|
|
168
|
+
if [ "$MODE" = "rerefine" ]; then
|
|
169
|
+
# A rerefine is triggered by a human comment on an issue the bot parked for
|
|
170
|
+
# review. The comment is the review: clear the label and proceed.
|
|
171
|
+
gh issue edit "$ISSUE_NUMBER" --remove-label "review" 2>/dev/null ||
|
|
172
|
+
echo "::notice::review label already gone on #$ISSUE_NUMBER"
|
|
173
|
+
echo "eligible=true" >> "$GITHUB_OUTPUT"
|
|
174
|
+
echo "::notice::Issue #$ISSUE_NUMBER had review label; cleared it for rerefine (user answered the bot)."
|
|
175
|
+
exit 0
|
|
176
|
+
fi
|
|
160
177
|
echo "eligible=false" >> "$GITHUB_OUTPUT"
|
|
161
178
|
echo "::notice::Issue #$ISSUE_NUMBER has the review label. Automated refinement skipped."
|
|
162
179
|
exit 0
|