@walwal-harness/cli 6.1.4 → 6.1.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/CHANGELOG.md +40 -0
  2. package/README.md +43 -74
  3. package/assets/launchd/com.walwal.harness-wake.plist.template +32 -0
  4. package/assets/templates/CONVENTIONS.md +12 -0
  5. package/assets/templates/HARNESS.md +8 -8
  6. package/assets/templates/company-pipeline-manifest.json +71 -0
  7. package/assets/templates/config.json +29 -27
  8. package/assets/templates/memory.md +16 -2
  9. package/assets/templates/progress.json.template +6 -5
  10. package/bin/init.js +180 -90
  11. package/conventions/shared.md +34 -0
  12. package/gotchas/conductor.md +16 -0
  13. package/gotchas/dispatcher.md +16 -0
  14. package/gotchas/meeting-manager.md +8 -0
  15. package/gotchas/service-ops.md +8 -0
  16. package/package.json +4 -4
  17. package/scripts/conductor-tick.sh +157 -56
  18. package/scripts/harness-archive.sh +7 -5
  19. package/scripts/harness-dashboard-up.sh +11 -3
  20. package/scripts/harness-hourly-review.sh +582 -0
  21. package/scripts/harness-meeting-doc.sh +55 -16
  22. package/scripts/harness-next.sh +1 -1
  23. package/scripts/harness-parity-audit.sh +101 -0
  24. package/scripts/harness-queue-manager.sh +13 -6
  25. package/scripts/harness-service-ops-monitor.sh +273 -0
  26. package/scripts/harness-session-start.sh +60 -31
  27. package/scripts/harness-statusline.sh +8 -12
  28. package/scripts/harness-stop.sh +85 -0
  29. package/scripts/harness-user-prompt-submit.sh +11 -29
  30. package/scripts/harness-wake-install.sh +180 -0
  31. package/scripts/harness-wake.sh +350 -0
  32. package/scripts/harness-worker-dispatch.sh +188 -0
  33. package/scripts/lib/harness-agent-resolver.sh +127 -0
  34. package/scripts/lib/harness-progress-migrate.sh +6 -1
  35. package/scripts/lib/harness-render-progress.sh +3 -3
  36. package/skills/brainstorming/SKILL.md +2 -2
  37. package/skills/conductor/SKILL.md +67 -55
  38. package/skills/cqo/SKILL.md +1 -0
  39. package/skills/cto/SKILL.md +1 -0
  40. package/skills/dispatcher/SKILL.md +33 -14
  41. package/skills/evaluator-code-quality/SKILL.md +4 -4
  42. package/skills/evaluator-functional/SKILL.md +6 -6
  43. package/skills/evaluator-visual/SKILL.md +4 -4
  44. package/skills/generator-backend/SKILL.md +4 -4
  45. package/skills/generator-frontend/SKILL.md +5 -5
  46. package/skills/meeting-manager/SKILL.md +82 -12
  47. package/skills/planner/SKILL.md +3 -3
  48. package/skills/service-ops/SKILL.md +25 -0
  49. package/commands/harness-solo.md +0 -103
  50. package/commands/harness-stop.md +0 -53
  51. package/commands/harness-team.md +0 -530
  52. package/scripts/harness-dashboard.sh +0 -509
  53. package/scripts/harness-goal-init.sh +0 -72
  54. package/scripts/harness-goal-show.sh +0 -37
  55. package/scripts/harness-gotcha-memory.sh +0 -348
  56. package/scripts/harness-monitor.sh +0 -398
  57. package/scripts/harness-prompt-history.sh +0 -164
  58. package/scripts/harness-tmux.sh +0 -372
@@ -36,7 +36,7 @@ default_decision_json() {
36
36
  implementation_drift) owner="cto"; action_type="implement" ;;
37
37
  planning_drift) owner="planner"; action_type="replan" ;;
38
38
  ops_drift) owner="service-ops"; action_type="monitor" ;;
39
- goal_drift) owner="dispatcher"; action_type="escalate-owner" ;;
39
+ goal_drift) owner="planner"; action_type="goal-realignment" ;;
40
40
  esac
41
41
  if [ "$requested_reason" = "goal-intake" ]; then
42
42
  owner="planner"
@@ -158,33 +158,43 @@ EOF
158
158
 
159
159
  cat > "$meeting_dir/prep-dispatcher.md" <<'EOF'
160
160
  # Prep — Dispatcher/CEO
161
- - Goal 자체가 흔들렸는가?
162
- - Owner escalation 이 필요한가?
163
- - 사업 우선순위 또는 KPI 정의를 바꿔야 하는가?
161
+ - Role: 회사 내부 CEO. Owner와의 유일한 외부 창구이며, Owner 입력은 최초 GOAL 이후 interrupt/additional request로만 취급한다.
162
+ - Must answer: Goal 자체가 흔들렸는가, 아니면 내부 실행/품질/운영 문제인가?
163
+ - Must decide: Owner escalation 이 정말 필요한가, 아니면 내부 임원진이 처리할 수 있는가?
164
+ - Must not: Owner를 기다린다는 결론으로 회의를 끝내지 마라.
165
+ - Evidence to cite: active GOAL, KPI, previous decision, escalation trigger.
164
166
  EOF
165
167
  cat > "$meeting_dir/prep-planner.md" <<'EOF'
166
168
  # Prep — Planner/COO
167
- - 기획/가설/웹리서치/레퍼런스 보강이 필요한가?
168
- - planning_drift 또는 goal_drift 근거는 무엇인가?
169
- - 수정해야 할 plan/feature/api 는 무엇인가?
169
+ - Role: COO. GOAL을 실행 가능한 work package, 가설, queue로 바꾸고 조직이 옆길로 새지 않게 한다.
170
+ - Must answer: planning_drift 또는 goal_drift 인가, 아니면 구현/품질/운영 문제인가?
171
+ - Must decide: 다음 operating cycle에서 어떤 work package를 queue/backlog/track으로 만들 것인가?
172
+ - Must not: "다음 sprint에서", "Owner가 정하면" 같은 대기형 결론을 쓰지 마라.
173
+ - Evidence to cite: feature-list, sprint-contract, queue, hypothesis/report artifacts.
170
174
  EOF
171
175
  cat > "$meeting_dir/prep-cto.md" <<'EOF'
172
176
  # Prep — CTO
173
- - 구현/아키텍처/기술선택이 원인인가?
174
- - implementation_drift 근거는 무엇인가?
175
- - 어떤 generator/evaluator를 다시 태워야 하는가?
177
+ - Role: CTO. 구현, 아키텍처, 기술선택, runtime recovery의 책임자다.
178
+ - Must answer: implementation_drift 인가, 런타임 장애인가, 설계 변경이 필요한가?
179
+ - Must decide: 어떤 generator/devops/hotfix worker를 태우고 어떤 deliverable을 요구할 것인가?
180
+ - Must not: 운영 장애를 단순 Service-Ops 알림으로만 남기지 마라. 복구 owner를 지정하라.
181
+ - Evidence to cite: code path, build/test output, runtime health, cto-review.
176
182
  EOF
177
183
  cat > "$meeting_dir/prep-cqo.md" <<'EOF'
178
184
  # Prep — CQO
179
- - 품질/회귀/검증 부족이 원인인가?
180
- - 어떤 evidence가 이 결론을 지지하는가?
181
- - evidence 없는 추정은 금지한다.
185
+ - Role: CQO. 품질, 회귀, 검증 기준의 최종 감시자다.
186
+ - Must answer: evidence가 충분한가, PASS/FAIL 판정이 방어 가능한가?
187
+ - Must decide: 어떤 evaluator/check가 필요하고, 어떤 주장은 검증 전까지 보류해야 하는가?
188
+ - Must not: evidence 없는 낙관론을 회의 결론으로 통과시키지 마라.
189
+ - Evidence to cite: evaluator reports, regression results, cqo-audit, reproducible checks.
182
190
  EOF
183
191
  cat > "$meeting_dir/prep-service-ops.md" <<'EOF'
184
192
  # Prep — Service-Ops
185
- - 어떤 KPI/로그/incident가 Goal에서 벗어났는가?
186
- - ops_drift 여부를 먼저 판단하라.
187
- - 운영 측 evidence를 문서 경로와 함께 적어라.
193
+ - Role: Service-Ops. 운영 신호, KPI, incident, monitor cadence의 책임자다.
194
+ - Must answer: 어떤 KPI/로그/health/incident가 Goal에서 벗어났는가?
195
+ - Must decide: ops_drift 인가, incident-war-room 이 필요한가, monitor cadence를 바꿔야 하는가?
196
+ - Must not: 서버 down을 혼자 경고만 하고 끝내지 마라. 회의에 올려 CTO/CQO action으로 연결하라.
197
+ - Evidence to cite: ops-report path, health table, incident timeline, last_check.
188
198
  EOF
189
199
 
190
200
  decision_json="$(default_decision_json "$drift")"
@@ -217,6 +227,35 @@ $(jq '.prior_tracks' <<<"$resolved_fork_context")
217
227
  - goal_adherence: $goal_adherence
218
228
  $fork_section
219
229
 
230
+ ## Required Role Positions
231
+
232
+ Meeting-Manager must fill this section before reporting the meeting as complete.
233
+
234
+ ### Dispatcher/CEO
235
+ - Position:
236
+ - Evidence:
237
+ - Escalation needed: yes/no
238
+
239
+ ### Planner/COO
240
+ - Position:
241
+ - Evidence:
242
+ - Proposed work package:
243
+
244
+ ### CTO
245
+ - Position:
246
+ - Evidence:
247
+ - Engineering action:
248
+
249
+ ### CQO
250
+ - Position:
251
+ - Evidence:
252
+ - Quality gate:
253
+
254
+ ### Service-Ops
255
+ - Position:
256
+ - Evidence:
257
+ - Operational action:
258
+
220
259
  ## Decision JSON
221
260
  \`\`\`json
222
261
  $(jq -n \
@@ -196,7 +196,7 @@ run_pre_eval_gate() {
196
196
  run_conductor_tick() {
197
197
  local candidate="$1"
198
198
  local owner
199
- owner=$(jq -r '.mode_selection.owner // empty' "$CONFIG" 2>/dev/null || true)
199
+ owner=$(jq -r '.company_mode.owner // "conductor"' "$CONFIG" 2>/dev/null || true)
200
200
  if [ "$owner" != "conductor" ]; then return 0; fi
201
201
  if [ ! -x "$SCRIPT_DIR/conductor-tick.sh" ]; then return 0; fi
202
202
 
@@ -0,0 +1,101 @@
1
+ #!/bin/bash
2
+ # harness-parity-audit.sh — compare installed project company-loop surface to walwal-harness
3
+
4
+ set -euo pipefail
5
+
6
+ SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
7
+ HARNESS_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
8
+ MANIFEST="$HARNESS_ROOT/assets/templates/company-pipeline-manifest.json"
9
+
10
+ usage() {
11
+ echo "usage: bash scripts/harness-parity-audit.sh <project-root> [<project-root> ...]" >&2
12
+ }
13
+
14
+ [ "$#" -gt 0 ] || { usage; exit 2; }
15
+ [ -f "$MANIFEST" ] || { echo "MISSING manifest: $MANIFEST" >&2; exit 2; }
16
+ command -v jq >/dev/null 2>&1 || { echo "MISSING jq" >&2; exit 2; }
17
+
18
+ sha() {
19
+ LC_ALL=en_US.UTF-8 LANG=en_US.UTF-8 shasum -a 256 "$1" | awk '{print $1}'
20
+ }
21
+
22
+ project_path_for() {
23
+ local rel="$1"
24
+ case "$rel" in
25
+ conventions/*) echo ".harness/$rel" ;;
26
+ gotchas/*) echo ".harness/$rel" ;;
27
+ skills/*) echo ".harness/$rel" ;;
28
+ *) echo "$rel" ;;
29
+ esac
30
+ }
31
+
32
+ config_core_filter() {
33
+ jq -S '
34
+ del(.project, .name, .description, .paths, .production, .telegram, .services, .owners, .runtime, .tech_stack, .integrations)
35
+ ' "$1"
36
+ }
37
+
38
+ status=0
39
+ for project in "$@"; do
40
+ project="$(cd "$project" && pwd)"
41
+ echo "== $project =="
42
+
43
+ if [ ! -d "$project/.harness" ]; then
44
+ echo "DRIFT missing .harness"
45
+ status=1
46
+ continue
47
+ fi
48
+
49
+ while IFS= read -r rel; do
50
+ src="$HARNESS_ROOT/$rel"
51
+ dst="$project/$(project_path_for "$rel")"
52
+ if [ ! -f "$src" ]; then
53
+ echo "UPSTREAM_REQUIRED missing-source $rel"
54
+ status=1
55
+ elif [ ! -f "$dst" ]; then
56
+ echo "DRIFT missing $rel -> $(project_path_for "$rel")"
57
+ status=1
58
+ elif [ "$(sha "$src")" != "$(sha "$dst")" ]; then
59
+ echo "DRIFT hash $rel -> $(project_path_for "$rel")"
60
+ status=1
61
+ else
62
+ echo "PASS $rel"
63
+ fi
64
+ done < <(jq -r '.identical_files[]' "$MANIFEST")
65
+
66
+ while IFS=$'\t' read -r src_rel dst_rel mode; do
67
+ src="$HARNESS_ROOT/$src_rel"
68
+ dst="$project/$dst_rel"
69
+ if [ ! -f "$dst" ]; then
70
+ echo "DRIFT missing $dst_rel"
71
+ status=1
72
+ elif [ "$mode" = "identical" ] && [ "$(sha "$src")" != "$(sha "$dst")" ]; then
73
+ echo "DRIFT hash $src_rel -> $dst_rel"
74
+ status=1
75
+ else
76
+ echo "PASS $src_rel -> $dst_rel"
77
+ fi
78
+ done < <(jq -r '.template_mappings[] | [.source, .target, .mode] | @tsv' "$MANIFEST")
79
+
80
+ if [ -f "$project/.harness/config.json" ]; then
81
+ tmp_src="$(mktemp)"
82
+ tmp_dst="$(mktemp)"
83
+ config_core_filter "$HARNESS_ROOT/assets/templates/config.json" > "$tmp_src"
84
+ config_core_filter "$project/.harness/config.json" > "$tmp_dst"
85
+ if jq -e -s '.[0] == .[1]' "$tmp_src" "$tmp_dst" >/dev/null; then
86
+ echo "PASS config-core"
87
+ else
88
+ echo "DRIFT config-core"
89
+ diff -u "$tmp_src" "$tmp_dst" | sed 's/^/ /'
90
+ status=1
91
+ fi
92
+ rm -f "$tmp_src" "$tmp_dst"
93
+ else
94
+ echo "DRIFT missing .harness/config.json"
95
+ status=1
96
+ fi
97
+
98
+ echo
99
+ done
100
+
101
+ exit "$status"
@@ -5,7 +5,8 @@
5
5
  # feature-queue.json을 생성/관리한다.
6
6
  #
7
7
  # Commands:
8
- # init feature-list.json → feature-queue.json 초기 생성
8
+ # init [sprint|all] [project-root]
9
+ # feature-list.json → feature-queue.json 초기 생성
9
10
  # dequeue <team> ready 큐에서 feature를 꺼내 team에 배정
10
11
  # pass <fid> feature를 passed로 이동, blocked→ready 전이
11
12
  # fail <fid> feature를 failed로 이동
@@ -71,8 +72,10 @@ fi
71
72
 
72
73
  # ══════════════════════════════════════════
73
74
  # init — Build queue from feature-list.json
74
- # Usage: init [sprint_number]
75
- # sprint_number: optional, filter features by sprint (default: all)
75
+ # Usage: init [sprint_number|all] [project-root]
76
+ # sprint_number: optional, filter features by sprint (default: all).
77
+ # If passing a project root, use `init all /path/to/project`; otherwise the
78
+ # first argument is interpreted as the sprint filter.
76
79
  # ══════════════════════════════════════════
77
80
  cmd_init() {
78
81
  local sprint_filter="${1:-all}"
@@ -85,11 +88,15 @@ cmd_init() {
85
88
  # Build dependency graph and topological sort
86
89
  # Output: feature-queue.json with ready (no deps) and blocked (has deps)
87
90
  jq --argjson concurrency "$CONCURRENCY" --arg sprint "$sprint_filter" '
88
- # Build passed set (features already passed by evaluator)
91
+ # Build passed set (features already passed by evaluator or imported audit).
92
+ # Legacy projects use string passes; newer NEXUS artifacts may use objects.
89
93
  def passed_set:
90
94
  [.features[] | select(
91
95
  ((.passes // []) | length > 0) and
92
- ((.passes // []) | any(. == "evaluator-functional"))
96
+ ((.passes // []) | any(
97
+ (. == "evaluator-functional") or
98
+ ((type == "object") and ((.status // "") == "PASS"))
99
+ ))
93
100
  ) | .id] ;
94
101
 
95
102
  # Filter features by sprint if specified
@@ -134,7 +141,7 @@ cmd_init() {
134
141
  }) | from_entries
135
142
  )
136
143
  }
137
- ' "$FEATURES" > "$QUEUE"
144
+ ' "$FEATURES" > "${QUEUE}.tmp.$$" && mv "${QUEUE}.tmp.$$" "$QUEUE"
138
145
 
139
146
  local ready_count blocked_count passed_count
140
147
  ready_count=$(jq '.queue.ready | length' "$QUEUE")
@@ -0,0 +1,273 @@
1
+ #!/bin/bash
2
+ # harness-service-ops-monitor.sh — deterministic Service-Ops production check
3
+ #
4
+ # Reads .harness/config.json runtime.production.services[] and writes:
5
+ # - progress.json.service_ops.health[]
6
+ # - progress.json.service_ops.monitor.last_check
7
+ # - .harness/actions/ops-report-hourly-<timestamp>.md
8
+ #
9
+ # This script is intentionally deterministic and does not require an LLM. Claude
10
+ # can add analysis later, but the Owner must always have a disk-backed record.
11
+
12
+ set -uo pipefail
13
+
14
+ SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
15
+ source "$SCRIPT_DIR/lib/harness-render-progress.sh"
16
+
17
+ PROJECT_ROOT="$(resolve_harness_root "${1:-.}")" || exit 0
18
+ PROGRESS="$PROJECT_ROOT/.harness/progress.json"
19
+ CONFIG="$PROJECT_ROOT/.harness/config.json"
20
+ ACTIONS_DIR="$PROJECT_ROOT/.harness/actions"
21
+ OPS_DIR="$PROJECT_ROOT/.harness/ops"
22
+
23
+ [ -f "$PROGRESS" ] || exit 0
24
+ [ -f "$CONFIG" ] || exit 0
25
+ command -v jq >/dev/null 2>&1 || exit 0
26
+
27
+ mkdir -p "$ACTIONS_DIR" "$OPS_DIR"
28
+
29
+ live=$(jq -r '.runtime.production.live // false' "$CONFIG" 2>/dev/null || echo false)
30
+ service_count=$(jq -r '(.runtime.production.services // []) | length' "$CONFIG" 2>/dev/null || echo 0)
31
+ ts="$(date -u +%Y-%m-%dT%H:%M:%SZ)"
32
+ stamp="$(date -u +%Y%m%dT%H%M%SZ)"
33
+ report_rel=".harness/actions/ops-report-hourly-${stamp}.md"
34
+ report_path="$PROJECT_ROOT/$report_rel"
35
+
36
+ json_escape() {
37
+ jq -Rn --arg v "$1" '$v'
38
+ }
39
+
40
+ tcp_check() {
41
+ local host="$1"
42
+ local port="$2"
43
+ if command -v nc >/dev/null 2>&1; then
44
+ nc -z -w 2 "$host" "$port" >/dev/null 2>&1
45
+ return $?
46
+ fi
47
+ bash -c ":</dev/tcp/$host/$port" >/dev/null 2>&1
48
+ }
49
+
50
+ http_status() {
51
+ local url="$1"
52
+ if command -v curl >/dev/null 2>&1; then
53
+ curl -s -o /dev/null -m 3 -w "%{http_code}" "$url" 2>/dev/null || echo "000"
54
+ else
55
+ echo "000"
56
+ fi
57
+ }
58
+
59
+ log_summary_json() {
60
+ local log_path="$1"
61
+ if [ -z "$log_path" ] || [ "$log_path" = "null" ]; then
62
+ jq -n '{configured:false, exists:false, recent_errors:0, last_line:null}'
63
+ return
64
+ fi
65
+ if [ ! -f "$log_path" ]; then
66
+ jq -n --arg path "$log_path" '{configured:true, exists:false, path:$path, recent_errors:0, last_line:null}'
67
+ return
68
+ fi
69
+ local recent errors last_line
70
+ recent="$(tail -200 "$log_path" 2>/dev/null || true)"
71
+ errors="$(printf '%s\n' "$recent" | grep -Eic 'error|exception|fatal|panic|traceback|fail' || true)"
72
+ last_line="$(tail -1 "$log_path" 2>/dev/null || true)"
73
+ jq -n --arg path "$log_path" --arg last "$last_line" --argjson errors "${errors:-0}" \
74
+ '{configured:true, exists:true, path:$path, recent_errors:$errors, last_line:$last}'
75
+ }
76
+
77
+ if [ "$live" != "true" ] || [ "${service_count:-0}" -eq 0 ]; then
78
+ jq --arg ts "$ts" --arg report "$report_rel" '
79
+ .service_ops.monitor.last_check = $ts |
80
+ .service_ops.monitor.stream_active = false |
81
+ .service_ops.monitor.last_report = $report |
82
+ .service_ops.health = [] |
83
+ .service_ops.incident.open = [] |
84
+ .service_ops.incident.signature = "" |
85
+ .service_ops.incident.repeat_count = 0 |
86
+ .service_ops.incident.recovery_required = false |
87
+ .service_ops.incident.partial_recovery = false |
88
+ .service_ops.incident.close_candidate = false |
89
+ .service_ops.incident.closed = false |
90
+ .service_ops.incident.last_seen_at = $ts
91
+ ' "$PROGRESS" > "${PROGRESS}.tmp" && mv "${PROGRESS}.tmp" "$PROGRESS"
92
+
93
+ {
94
+ echo "# Service-Ops Hourly Report"
95
+ echo ""
96
+ echo "- ts: $ts"
97
+ echo "- production.live: $live"
98
+ echo "- services: $service_count"
99
+ echo "- verdict: NO_PRODUCTION_SERVICES"
100
+ echo ""
101
+ echo "No production service endpoints are configured. Owner+CEO+CTO must define runtime.production.services[] before Service-Ops can monitor live servers."
102
+ } > "$report_path"
103
+ echo "$report_rel"
104
+ exit 0
105
+ fi
106
+
107
+ results_file="$(mktemp)"
108
+ : > "$results_file"
109
+
110
+ i=0
111
+ while [ "$i" -lt "$service_count" ]; do
112
+ svc="$(jq -c ".runtime.production.services[$i]" "$CONFIG")"
113
+ name="$(jq -r '.name // ("service-" + (input_line_number|tostring))' <<<"$svc")"
114
+ host="$(jq -r '.host // "127.0.0.1"' <<<"$svc")"
115
+ port="$(jq -r '.port // empty' <<<"$svc")"
116
+ health_path="$(jq -r '.health_path // empty' <<<"$svc")"
117
+ expected_status="$(jq -r '.expected_status // 200' <<<"$svc")"
118
+ log_path="$(jq -r '.log_path // empty' <<<"$svc")"
119
+
120
+ port_state="unknown"
121
+ if [ -n "$port" ] && tcp_check "$host" "$port"; then
122
+ port_state="listening"
123
+ else
124
+ port_state="down"
125
+ fi
126
+
127
+ health_status="null"
128
+ health_ok="null"
129
+ health_url=""
130
+ if [ -n "$health_path" ] && [ "$health_path" != "null" ] && [ -n "$port" ]; then
131
+ health_url="http://${host}:${port}${health_path}"
132
+ health_status="$(http_status "$health_url")"
133
+ if [ "$health_status" = "$expected_status" ]; then health_ok="true"; else health_ok="false"; fi
134
+ fi
135
+
136
+ log_json="$(log_summary_json "$log_path")"
137
+
138
+ status="ok"
139
+ if [ "$port_state" != "listening" ]; then
140
+ status="down"
141
+ elif [ "$health_ok" = "false" ]; then
142
+ status="degraded"
143
+ elif [ "$(jq -r '.configured and (.exists|not)' <<<"$log_json")" = "true" ]; then
144
+ status="log_missing"
145
+ elif [ "$(jq -r '.recent_errors // 0' <<<"$log_json")" -gt 0 ]; then
146
+ status="warn"
147
+ fi
148
+
149
+ jq -n \
150
+ --arg ts "$ts" \
151
+ --arg name "$name" \
152
+ --arg host "$host" \
153
+ --arg port "$port" \
154
+ --arg port_state "$port_state" \
155
+ --arg health_path "$health_path" \
156
+ --arg health_url "$health_url" \
157
+ --arg expected_status "$expected_status" \
158
+ --arg health_status "$health_status" \
159
+ --arg health_ok "$health_ok" \
160
+ --arg status "$status" \
161
+ --argjson log "$log_json" \
162
+ '{
163
+ ts:$ts,
164
+ name:$name,
165
+ host:$host,
166
+ port:($port|tonumber?),
167
+ port_state:$port_state,
168
+ health_path:(if $health_path == "" or $health_path == "null" then null else $health_path end),
169
+ health_url:(if $health_url == "" then null else $health_url end),
170
+ expected_status:($expected_status|tonumber?),
171
+ health_status:(if $health_status == "null" then null else ($health_status|tonumber?) end),
172
+ health_ok:(if $health_ok == "true" then true elif $health_ok == "false" then false else null end),
173
+ log:$log,
174
+ status:$status
175
+ }' >> "$results_file"
176
+ i=$((i + 1))
177
+ done
178
+
179
+ results_json="$(jq -s '.' "$results_file")"
180
+ rm -f "$results_file"
181
+
182
+ down_count="$(jq '[.[] | select(.status == "down" or .status == "degraded")] | length' <<<"$results_json")"
183
+ warn_count="$(jq '[.[] | select(.status == "warn" or .status == "log_missing")] | length' <<<"$results_json")"
184
+ ok_count="$(jq '[.[] | select(.status == "ok")] | length' <<<"$results_json")"
185
+ verdict="OK"
186
+ if [ "$down_count" -gt 0 ]; then verdict="INCIDENT"; elif [ "$warn_count" -gt 0 ]; then verdict="WARN"; fi
187
+
188
+ incidents_json="$(jq '
189
+ [ .[] | select(.status == "down" or .status == "degraded") |
190
+ {
191
+ id: ("OPS-" + (.name | ascii_upcase | gsub("[^A-Z0-9]+";"-"))),
192
+ dept: "Operations",
193
+ severity: (if .status == "down" then "critical" else "high" end),
194
+ message: (.name + " " + .status + " (" + .host + ":" + (.port|tostring) + ")"),
195
+ ts: .ts
196
+ }
197
+ ]' <<<"$results_json")"
198
+ incident_signature="$(jq -r '[.[].id] | sort | join(",")' <<<"$incidents_json")"
199
+ prev_signature="$(jq -r '(.service_ops.incident.signature // "") as $s | if $s != "" then $s else ([.service_ops.incident.open[]?.id] | sort | join(",")) end' "$PROGRESS" 2>/dev/null || echo "")"
200
+ prev_repeat="$(jq -r '.service_ops.incident.repeat_count // 0' "$PROGRESS" 2>/dev/null || echo 0)"
201
+ repeat_count=0
202
+ if [ -n "$incident_signature" ]; then
203
+ if [ -n "$prev_signature" ] && [ "${prev_repeat:-0}" -lt 1 ]; then
204
+ prev_repeat=1
205
+ fi
206
+ if [ "$incident_signature" = "$prev_signature" ]; then
207
+ repeat_count=$((prev_repeat + 1))
208
+ else
209
+ repeat_count=1
210
+ fi
211
+ fi
212
+ recovery_required=false
213
+ if [ "$repeat_count" -ge 2 ]; then
214
+ recovery_required=true
215
+ fi
216
+ partial_recovery=false
217
+ close_candidate=false
218
+ if [ "$ok_count" -gt 0 ] && [ "$down_count" -gt 0 ]; then
219
+ partial_recovery=true
220
+ elif [ "$down_count" -eq 0 ] && [ "$warn_count" -eq 0 ]; then
221
+ close_candidate=true
222
+ fi
223
+
224
+ jq --arg ts "$ts" --arg report "$report_rel" --arg signature "$incident_signature" --argjson repeat "$repeat_count" --argjson recovery "$recovery_required" --argjson partial "$partial_recovery" --argjson close_candidate "$close_candidate" --argjson health "$results_json" --argjson incidents "$incidents_json" --argjson warn "$warn_count" --argjson alert "$down_count" '
225
+ .service_ops.monitor.stream_active = false |
226
+ .service_ops.monitor.last_check = $ts |
227
+ .service_ops.monitor.last_report = $report |
228
+ .service_ops.monitor.warns_this_sprint = ((.service_ops.monitor.warns_this_sprint // 0) + $warn) |
229
+ .service_ops.monitor.alerts_this_sprint = ((.service_ops.monitor.alerts_this_sprint // 0) + $alert) |
230
+ .service_ops.health = $health |
231
+ .service_ops.incident.open = $incidents |
232
+ .service_ops.incident.signature = $signature |
233
+ .service_ops.incident.repeat_count = $repeat |
234
+ .service_ops.incident.recovery_required = $recovery |
235
+ .service_ops.incident.partial_recovery = $partial |
236
+ .service_ops.incident.close_candidate = $close_candidate |
237
+ .service_ops.incident.closed = (if $close_candidate then true else false end) |
238
+ .service_ops.incident.last_seen_at = $ts
239
+ ' "$PROGRESS" > "${PROGRESS}.tmp" && mv "${PROGRESS}.tmp" "$PROGRESS"
240
+
241
+ {
242
+ echo "# Service-Ops Hourly Report"
243
+ echo ""
244
+ echo "- ts: $ts"
245
+ echo "- production.live: true"
246
+ echo "- services: $service_count"
247
+ echo "- verdict: $verdict"
248
+ echo "- ok: $ok_count"
249
+ echo "- warnings: $warn_count"
250
+ echo "- alerts: $down_count"
251
+ echo "- incident_signature: ${incident_signature:-none}"
252
+ echo "- repeat_count: $repeat_count"
253
+ echo "- recovery_required: $recovery_required"
254
+ echo "- partial_recovery: $partial_recovery"
255
+ echo "- close_candidate: $close_candidate"
256
+ echo ""
257
+ echo "| Service | Port | Health | Logs | Status |"
258
+ echo "|---|---:|---|---|---|"
259
+ jq -r '.[] |
260
+ "| \(.name) | \(.host):\(.port) \(.port_state) | " +
261
+ (if .health_path == null then "n/a" else ((.health_status // "000")|tostring) + " expected " + ((.expected_status // 200)|tostring) end) +
262
+ " | " +
263
+ (if (.log.configured|not) then "not configured" elif (.log.exists|not) then "missing" else ((.log.recent_errors|tostring) + " recent errors") end) +
264
+ " | \(.status) |"' <<<"$results_json"
265
+ echo ""
266
+ echo "## Raw"
267
+ echo ""
268
+ echo '```json'
269
+ jq '.' <<<"$results_json"
270
+ echo '```'
271
+ } > "$report_path"
272
+
273
+ echo "$report_rel"