@walwal-harness/cli 5.6.5 → 5.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,8 +1,12 @@
1
1
  {
2
- "version": 2,
2
+ "version": 3,
3
3
  "mode": "solo",
4
4
  "project_name": "",
5
5
  "pipeline": null,
6
+ "dispatch": {
7
+ "counter": 0,
8
+ "id": null
9
+ },
6
10
  "sprint": {
7
11
  "number": 0,
8
12
  "status": "init",
package/bin/init.js CHANGED
@@ -346,6 +346,14 @@ function scaffoldHarness() {
346
346
  fs.writeFileSync(progressPath, JSON.stringify(progress, null, 2) + '\n');
347
347
  log('progress.json migrated to v2 (mode + team_state added)');
348
348
  }
349
+ if (progress.version < 3) {
350
+ progress.version = 3;
351
+ if (!progress.dispatch) {
352
+ progress.dispatch = { counter: 0, id: null };
353
+ }
354
+ fs.writeFileSync(progressPath, JSON.stringify(progress, null, 2) + '\n');
355
+ log('progress.json migrated to v3 (dispatch counter added)');
356
+ }
349
357
  } catch (e) {
350
358
  log('WARNING: Could not migrate progress.json');
351
359
  }
@@ -59,6 +59,39 @@ bash scripts/harness-tmux.sh --team --force-tmux
59
59
 
60
60
  **`--force-tmux` 필수**: iTerm2 감지 경로는 백그라운드에 iTerm2가 떠 있기만 해도 활성화되어 AppleScript 실패 시 팀 레이아웃이 조용히 사라짐. Team Mode는 항상 tmux로 강제하여 재현 가능한 레이아웃을 보장.
61
61
 
62
+ ### Step 2.5: Worker Pre-flight Bundle 빌드 (v5.6.6+)
63
+
64
+ Worker 는 plain Agent 로 실행되어 SKILL.md Startup 체크리스트를 자동 주입받지 못한다. Lead 가 Worker spawn 직전에 **역할별 바인딩 문서를 프롬프트에 직접 주입**한다. 이렇게 하면 "Worker 가 읽어야 함" → "이미 읽은 상태로 시작" 으로 전환되어 스킵이 구조적으로 불가능해진다.
65
+
66
+ ```bash
67
+ # Generator-{be|fe} / Evaluator-{functional|visual|code-quality} 별 번들 빌드
68
+ build_preflight_bundle() {
69
+ local role="$1" # generator-frontend | generator-backend | evaluator-functional | ...
70
+ local hroot="$2" # HARNESS_ROOT (worktree 가 아닌 원본 루트)
71
+ {
72
+ echo "===== ROOT CONVENTIONS.md ====="
73
+ [ -f "$hroot/CONVENTIONS.md" ] && cat "$hroot/CONVENTIONS.md" || echo "(none)"
74
+ echo
75
+ echo "===== AGENTS.md ====="
76
+ [ -f "$hroot/AGENTS.md" ] && cat "$hroot/AGENTS.md" || echo "(none)"
77
+ echo
78
+ echo "===== .harness/conventions/shared.md ====="
79
+ [ -f "$hroot/.harness/conventions/shared.md" ] && cat "$hroot/.harness/conventions/shared.md" || echo "(empty)"
80
+ echo
81
+ echo "===== .harness/conventions/$role.md ====="
82
+ [ -f "$hroot/.harness/conventions/$role.md" ] && cat "$hroot/.harness/conventions/$role.md" || echo "(empty)"
83
+ echo
84
+ echo "===== .harness/gotchas/$role.md ====="
85
+ [ -f "$hroot/.harness/gotchas/$role.md" ] && cat "$hroot/.harness/gotchas/$role.md" || echo "(empty)"
86
+ echo
87
+ echo "===== .harness/memory.md ====="
88
+ [ -f "$hroot/.harness/memory.md" ] && cat "$hroot/.harness/memory.md" || echo "(empty)"
89
+ }
90
+ }
91
+ ```
92
+
93
+ Worker/내부 Evaluator Agent 프롬프트 상단에 이 번들 출력을 `## Binding Rules (pre-loaded)` 섹션으로 삽입한다. Worker 는 이를 **추가 조회 없이 이미 적용되는 규범**으로 취급한다.
94
+
62
95
  ### Step 3: 초기 Worker 생성 (Auto-Dispatch)
63
96
 
64
97
  **v5.6.4+**: 개별 dequeue 대신 **`auto-dispatch`** 한 번으로 모든 idle team 에 ready feature 를 원자적으로 배정합니다. 의존성 없는 작업은 병렬로 즉시 시작됩니다.
@@ -97,10 +130,11 @@ ORCHESTRATION LOOP:
97
130
 
98
131
  Background Agent 완료 알림을 받으면:
99
132
 
100
- 1. Worker 결과 분석:
101
- - PASS인 경우 → Step 4a (Merge + Unblock)
102
- - FAIL (재시도 가능)인 경우 → Step 4b (Retry)
103
- - ESCALATED인 경우 → 사용자에게 알림, 해당 팀 유휴
133
+ 1. Worker 결과 분석 (반환 메시지 첫 줄 태그로 분기):
134
+ - `PASS` → Step 4a (Merge + Unblock)
135
+ - `FAIL` (재시도 가능) → Step 4b (Retry)
136
+ - `RATE_LIMIT` → Step 4c (Rate-Limit Hold, 10m probe)
137
+ - `ESCALATED` → 사용자에게 알림, 해당 팀 유휴
104
138
 
105
139
  2. **Auto-Dispatch (필수)** — worker 완료 직후 idle 이 된 팀뿐 아니라
106
140
  모든 idle team 에 ready feature 를 즉시 재배정:
@@ -145,6 +179,47 @@ Worker가 PASS로 반환되면:
145
179
  echo "$(date +'%Y-%m-%d %H:%M') | lead | pass | {FEATURE_ID} merged + unblocked deps" >> .harness/progress.log
146
180
  ```
147
181
 
182
+ #### Step 4c: Rate-Limit Hold (토큰 한도 대응 · v5.6.7+)
183
+
184
+ Worker 반환 첫 줄이 `RATE_LIMIT` 으로 시작하면 Lead 는 **에러 아닌 hold 모드**로 전환한다. 나머지 Worker 들은 자연 완료까지 계속 실행되고, 그 결과도 RATE_LIMIT 이면 합쳐서 hold 상태에 누적된다.
185
+
186
+ ```bash
187
+ # 1) Checkpoint 기록 (current in_progress + ready 스냅샷 저장)
188
+ bash "$HARNESS_ROOT/scripts/harness-queue-manager.sh" hold rate_limit 600 .
189
+
190
+ # 2) 로그 + tmux pane 타이틀 변경
191
+ echo "$(date +'%Y-%m-%d %H:%M') | lead | hold | rate-limit detected, pausing 10m" >> .harness/progress.log
192
+ tmux rename-window "⏸ HOLD (resume ~$(date -v+10M +%H:%M 2>/dev/null || date -d '+10 min' +%H:%M))" 2>/dev/null || true
193
+
194
+ # 3) 실패한 feature 는 requeue (WIP worktree 는 유지 — merge 없이 재사용)
195
+ bash "$HARNESS_ROOT/scripts/harness-queue-manager.sh" requeue {FEATURE_ID} .
196
+ ```
197
+
198
+ 4) **ScheduleWakeup 으로 10분 뒤 재진입 스케줄**:
199
+ ```
200
+ ScheduleWakeup({
201
+ delaySeconds: 600,
202
+ prompt: "/harness-team resume",
203
+ reason: "rate-limit hold — 10m probe"
204
+ })
205
+ ```
206
+
207
+ 5) Lead LOOP return (중단 아님 — wake-up 이 재진입 트리거).
208
+
209
+ **Wake-up 재진입 시 Lead 동작** (`/harness-team resume` 처리):
210
+
211
+ ```bash
212
+ # Probe: claude CLI 가 실제로 응답하는지 최소 호출로 확인
213
+ bash "$HARNESS_ROOT/scripts/harness-queue-manager.sh" resume-probe .
214
+ # 종료 코드: 0=clear, 1=still held, 2=escalated(>12h)
215
+ ```
216
+
217
+ - **0 (clear)** → 체크포인트 삭제됨. 즉시 `auto-dispatch` 실행 → Step 4 LOOP 복귀.
218
+ - **1 (still held)** → 다시 `ScheduleWakeup(600, "/harness-team resume", "rate-limit still held — cycle N")` 스케줄. hold_count 증가.
219
+ - **2 (escalated)** → 72 사이클(12시간) 초과. 사용자 개입 알림 후 LOOP 종료. 체크포인트 파일 (`.harness/actions/team-checkpoint.json`) 에 전체 상태가 남아있으므로 사용자가 수동 복구 가능.
220
+
221
+ **핵심 원칙**: 토큰 리밋은 "실패"가 아닌 "일시 정지". 진행 중이던 worktree/queue 상태는 그대로 보존되고, 10분 간격 probe 로 해제 즉시 이어서 진행한다. Session 을 닫아도 이어지길 원한다면 `ScheduleWakeup` 대신 `schedule` 스킬(CronCreate) 로 cron-backed 재시도 설정 가능.
222
+
148
223
  #### Step 4b: Retry (FAIL 처리)
149
224
 
150
225
  Worker가 FAIL (재시도 가능)로 반환되면:
@@ -171,6 +246,15 @@ Worker가 FAIL (재시도 가능)로 반환되면:
171
246
  당신은 Harness Team-{N} 워커입니다. **단일 Feature**에 대해 Gen→Eval 사이클을 수행합니다.
172
247
  완료 후 결과를 반환합니다. 다음 Feature는 Lead가 할당합니다.
173
248
 
249
+ ## Binding Rules (pre-loaded — 스킵 금지, 이미 적용됨)
250
+
251
+ Lead 가 Step 2.5 에서 build_preflight_bundle 로 생성한 번들이 아래에 주입됩니다.
252
+ 당신은 이 규칙을 이미 읽은 상태로 시작합니다. 추가 조회 불필요:
253
+
254
+ {PREFLIGHT_BUNDLE}
255
+
256
+ **작업 시작 전 필수 출력**: 위 번들에서 이번 Feature 작업에 **적용되는 규칙**을 3~8 줄로 요약한 뒤 진행하라. 비어있으면 "(empty)" 로 명시. 이 요약 없이 Phase 1 로 진입하면 Self-FAIL 처리하고 재시작한다. 내부 Evaluator Agent 를 생성할 때도 같은 번들을 `{PREFLIGHT_BUNDLE}` 자리에 그대로 전달하라 (Evaluator 도 plain Agent 이므로 자동주입 없음).
257
+
174
258
  ## 할당된 Feature
175
259
  - Feature ID: {FEATURE_ID}
176
260
  - 프로젝트 루트: 현재 디렉토리 (worktree 복사본)
@@ -301,6 +385,18 @@ logev fail "{FEATURE_ID} FAIL #{ATTEMPT} — {사유}"
301
385
  logev fail "{FEATURE_ID} FINAL FAIL after 5 attempts"
302
386
  ```
303
387
  Lead에게 반환: `ESCALATED | {FEATURE_ID} | attempts=5 | last_feedback={마지막_피드백}`
388
+
389
+ ### RATE_LIMIT (토큰 한도 감지 시 — v5.6.7+)
390
+
391
+ Gen 또는 Eval Phase 중 429 / "rate_limit" / "quota" / "overloaded_error" / "token limit" / "usage limit" 메시지를 만나면:
392
+
393
+ ```bash
394
+ logev hold "{FEATURE_ID} rate-limit hit — returning RATE_LIMIT to Lead"
395
+ ```
396
+ Lead에게 반환 **첫 줄에 반드시 `RATE_LIMIT` 태그 포함**:
397
+ `RATE_LIMIT | {FEATURE_ID} | phase={gen|eval} | attempt={N} | err={원문요약}`
398
+
399
+ 작업을 **포기하지 말고** 현재까지의 변경분을 worktree 에 그대로 commit (WIP). Lead 가 hold 해제 후 같은 worktree 로 resume.
304
400
  ```
305
401
 
306
402
  ---
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@walwal-harness/cli",
3
- "version": "5.6.5",
3
+ "version": "5.7.0",
4
4
  "description": "Production harness for AI agent engineering — Solo/Team mode, Planner, Generator(BE/FE), Evaluator chain (Code-Quality → Functional → Visual), optional Brainstormer. Supports React, Next.js, and Flutter FE stacks.",
5
5
  "bin": {
6
6
  "walwal-harness": "bin/init.js"
@@ -0,0 +1,109 @@
1
+ #!/bin/bash
2
+ # harness-archive.sh — 스프린트 종료 시 자동 아카이빙
3
+ #
4
+ # 동작:
5
+ # 1. .harness/actions/ 의 스프린트 산출물을 .harness/archive/D-NNN/S-NNN/ 로 이동
6
+ # 2. progress.json 의 sprint 상태를 초기화 (신규 dispatch 준비)
7
+ # 3. dispatch.id 가 없으면 counter++ 로 새 dispatch 시작
8
+ #
9
+ # 유지되는 파일 (이동하지 않음):
10
+ # - .harness/gotchas/**, .harness/conventions/**, .harness/ref/**
11
+ # - .harness/config.json, .harness/memory.md
12
+ # - .harness/progress.json (초기화만)
13
+ #
14
+ # 호출 주체: harness-next.sh (next_agent="archive" 도달 시 자동)
15
+
16
+ set -e
17
+
18
+ SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
19
+ source "$SCRIPT_DIR/lib/harness-render-progress.sh" 2>/dev/null || true
20
+
21
+ PROJECT_ROOT="${1:-.}"
22
+ PROJECT_ROOT="$(cd "$PROJECT_ROOT" && pwd)"
23
+
24
+ PROGRESS="$PROJECT_ROOT/.harness/progress.json"
25
+ ACTIONS_DIR="$PROJECT_ROOT/.harness/actions"
26
+ ARCHIVE_ROOT="$PROJECT_ROOT/.harness/archive"
27
+
28
+ if [ ! -f "$PROGRESS" ]; then
29
+ echo "[archive] ERROR: progress.json not found" >&2
30
+ exit 1
31
+ fi
32
+
33
+ command -v jq >/dev/null 2>&1 || { echo "[archive] ERROR: jq required" >&2; exit 1; }
34
+
35
+ # ── Read current dispatch/sprint numbers ──
36
+ dispatch_counter=$(jq -r '.dispatch.counter // 0' "$PROGRESS")
37
+ dispatch_id=$(jq -r '.dispatch.id // empty' "$PROGRESS")
38
+ sprint_num=$(jq -r '.sprint.number // 0' "$PROGRESS")
39
+
40
+ # Ensure dispatch id exists (fallback for legacy progress.json without dispatch)
41
+ if [ -z "$dispatch_id" ] || [ "$dispatch_id" = "null" ]; then
42
+ if [ "$dispatch_counter" -lt 1 ]; then
43
+ dispatch_counter=1
44
+ fi
45
+ dispatch_id=$(printf 'D-%03d' "$dispatch_counter")
46
+ fi
47
+
48
+ sprint_id=$(printf 'S-%03d' "$sprint_num")
49
+ target_dir="$ARCHIVE_ROOT/$dispatch_id/$sprint_id"
50
+
51
+ echo ""
52
+ echo " ── Archive ────────────────────────────"
53
+ echo " Dispatch : $dispatch_id"
54
+ echo " Sprint : $sprint_id"
55
+ echo " Target : ${target_dir#$PROJECT_ROOT/}"
56
+
57
+ # ── Move actions/ contents into archive ──
58
+ if [ -d "$ACTIONS_DIR" ] && [ "$(ls -A "$ACTIONS_DIR" 2>/dev/null)" ]; then
59
+ mkdir -p "$target_dir"
60
+ moved=0
61
+ for f in "$ACTIONS_DIR"/*; do
62
+ [ -e "$f" ] || continue
63
+ name="$(basename "$f")"
64
+ if [ -e "$target_dir/$name" ]; then
65
+ # Collision — suffix with timestamp to avoid overwrite
66
+ ts=$(date +%Y%m%d-%H%M%S)
67
+ mv "$f" "$target_dir/${name%.}.${ts}"
68
+ else
69
+ mv "$f" "$target_dir/"
70
+ fi
71
+ moved=$((moved + 1))
72
+ done
73
+ echo " Moved : $moved file(s)"
74
+ else
75
+ echo " Moved : 0 file(s) (actions/ empty)"
76
+ fi
77
+
78
+ # ── Reset progress.json for next dispatch ──
79
+ # - sprint → init
80
+ # - agents → cleared
81
+ # - artifacts → pending
82
+ # - dispatch.id cleared (next dispatcher run will allocate new D-NNN)
83
+ # - failure → cleared
84
+ jq --arg now "$(date -u +%Y-%m-%dT%H:%M:%SZ)" '
85
+ .pipeline = null
86
+ | .dispatch.id = null
87
+ | .sprint = { number: 0, status: "init", retry_count: 0 }
88
+ | .current_agent = null
89
+ | .agent_status = "pending"
90
+ | .completed_agents = []
91
+ | .next_agent = "dispatcher"
92
+ | .failure = { agent: null, location: null, message: null, retry_target: null }
93
+ | (.artifacts // {}) as $a
94
+ | .artifacts = ($a | with_entries(
95
+ if (.value | type) == "object"
96
+ then .value = { status: "pending", updated_by: null, updated_at: null }
97
+ else .
98
+ end))
99
+ | .updated_at = $now
100
+ ' "$PROGRESS" > "${PROGRESS}.tmp" && mv "${PROGRESS}.tmp" "$PROGRESS"
101
+
102
+ # ── Clear handoff so next session starts fresh ──
103
+ HANDOFF="$PROJECT_ROOT/.harness/handoff.json"
104
+ echo '{}' > "$HANDOFF"
105
+
106
+ echo " Status : archived, progress reset"
107
+ echo ""
108
+ echo " ✓ 다음 요청은 새로운 dispatch 로 시작합니다."
109
+ echo ""
@@ -11,6 +11,7 @@ set -uo pipefail
11
11
 
12
12
  SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
13
13
  source "$SCRIPT_DIR/lib/harness-render-progress.sh"
14
+ source "$SCRIPT_DIR/lib/harness-keywait.sh"
14
15
 
15
16
  PROJECT_ROOT="${1:-}"
16
17
  if [ -z "$PROJECT_ROOT" ]; then
@@ -420,5 +421,6 @@ while true; do
420
421
  # from any previous frame (fixes wrapped shell-prompt bleed-through).
421
422
  printf '%s\n' "$buf" | awk '{printf "%s\033[K\n", $0}'
422
423
  tput ed 2>/dev/null
423
- sleep 3
424
+ printf "${DIM} [r] refresh [q] quit${RESET}\033[K\n"
425
+ wait_or_refresh 3 || true
424
426
  done
@@ -7,6 +7,7 @@ set -uo pipefail
7
7
 
8
8
  SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
9
9
  source "$SCRIPT_DIR/lib/harness-render-progress.sh"
10
+ source "$SCRIPT_DIR/lib/harness-keywait.sh"
10
11
 
11
12
  PROJECT_ROOT="${1:-}"
12
13
  if [ -z "$PROJECT_ROOT" ]; then
@@ -181,7 +182,11 @@ while true; do
181
182
  sig=$(compute_signature)
182
183
  if [ "$sig" != "$LAST_SIG" ]; then
183
184
  render
185
+ printf "${DIM} [r] refresh [q] quit${RESET}\033[K\n"
184
186
  LAST_SIG="$sig"
185
187
  fi
186
- sleep "$REFRESH_SEC"
188
+ if ! wait_or_refresh "$REFRESH_SEC"; then
189
+ # 'r' 키 → 캐시 무효화해서 다음 루프에서 강제 렌더
190
+ LAST_SIG=""
191
+ fi
187
192
  done
@@ -9,6 +9,7 @@
9
9
  set -uo pipefail
10
10
 
11
11
  SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
12
+ source "$SCRIPT_DIR/lib/harness-keywait.sh"
12
13
 
13
14
  # ── Args ──
14
15
  # Usage: harness-monitor.sh [project-root] [--team N]
@@ -382,6 +383,7 @@ while true; do
382
383
  check_transitions
383
384
  fi
384
385
 
385
- sleep 3
386
+ printf "${DIM} [r] refresh [q] quit${RESET}\033[K\n"
387
+ wait_or_refresh 3 || true
386
388
  done
387
389
 
@@ -499,19 +499,29 @@ if [ "$next_agent" != "null" ] && [ "$next_agent" != "archive" ] && [ "$agent_st
499
499
 
500
500
  elif [ "$next_agent" = "archive" ]; then
501
501
  audit_log "system" "archive" "start" "sprint-${sprint_num}" "sprint cycle complete"
502
- jq -n \
503
- --arg from "${current_agent:-evaluator}" \
504
- --argjson sprint "$sprint_num" \
505
- --arg timestamp "$(date -u +%Y-%m-%dT%H:%M:%SZ)" \
506
- '{
507
- from: $from,
508
- to: "archive",
509
- prompt: "Sprint 문서를 아카이브하세요.\n.harness/actions/의 스프린트 문서를 .harness/archive/sprint-NNN/으로 이동합니다.\n.harness/handoff.json을 읽고 sprint 번호를 확인하세요.",
510
- sprint: $sprint,
511
- model: "opus",
512
- thinking_mode: null,
513
- timestamp: $timestamp
514
- }' > "$HANDOFF"
502
+
503
+ # ── Auto-archive: run archive script synchronously ──
504
+ # harness-archive.sh moves .harness/actions/* to .harness/archive/D-NNN/S-NNN/
505
+ # and resets progress.json. On next user prompt the flow starts fresh as a new dispatch.
506
+ if bash "$SCRIPT_DIR/harness-archive.sh" "$PROJECT_ROOT"; then
507
+ audit_log "system" "archive" "complete" "sprint-${sprint_num}" "auto-archived"
508
+ else
509
+ echo " ⚠ archive script failed — falling back to manual handoff" >&2
510
+ audit_log "system" "archive" "fail" "sprint-${sprint_num}" "script failed"
511
+ jq -n \
512
+ --arg from "${current_agent:-evaluator}" \
513
+ --argjson sprint "$sprint_num" \
514
+ --arg timestamp "$(date -u +%Y-%m-%dT%H:%M:%SZ)" \
515
+ '{
516
+ from: $from,
517
+ to: "archive",
518
+ prompt: "Sprint 문서를 아카이브하세요. harness-archive.sh 가 실패했습니다 — 수동으로 .harness/actions/* 를 .harness/archive/D-NNN/S-NNN/ 로 이동하세요.",
519
+ sprint: $sprint,
520
+ model: "opus",
521
+ thinking_mode: null,
522
+ timestamp: $timestamp
523
+ }' > "$HANDOFF"
524
+ fi
515
525
 
516
526
  else
517
527
  echo '{}' > "$HANDOFF"
@@ -7,6 +7,7 @@ set -uo pipefail
7
7
 
8
8
  SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
9
9
  source "$SCRIPT_DIR/lib/harness-render-progress.sh"
10
+ source "$SCRIPT_DIR/lib/harness-keywait.sh"
10
11
 
11
12
  PROJECT_ROOT="${1:-}"
12
13
  if [ -z "$PROJECT_ROOT" ]; then
@@ -158,5 +159,6 @@ while true; do
158
159
  tput cup 0 0 2>/dev/null
159
160
  echo "$buf"
160
161
  tput ed 2>/dev/null
161
- sleep 3
162
+ printf "${DIM} [r] refresh [q] quit${RESET}\033[K\n"
163
+ wait_or_refresh 3 || true
162
164
  done
@@ -487,6 +487,68 @@ cmd_idle_slots() {
487
487
  }' "$QUEUE"
488
488
  }
489
489
 
490
+ # ── Rate-Limit Hold checkpoint (v5.6.7+) ──
491
+ CHECKPOINT="$PROJECT_ROOT/.harness/actions/team-checkpoint.json"
492
+
493
+ cmd_hold() {
494
+ # Args: <reason> [retry_seconds=600]
495
+ local reason="${1:-rate_limit}"
496
+ local retry_secs="${2:-600}"
497
+ local now_iso retry_iso
498
+ now_iso="$(date -u +%Y-%m-%dT%H:%M:%SZ)"
499
+ retry_iso="$(date -u -v+"${retry_secs}"S +%Y-%m-%dT%H:%M:%SZ 2>/dev/null \
500
+ || date -u -d "+${retry_secs} seconds" +%Y-%m-%dT%H:%M:%SZ)"
501
+ acquire_queue_lock
502
+ trap release_queue_lock EXIT
503
+ local in_progress pending
504
+ in_progress="$(jq '[.teams[] | select(.current_feature != null) |
505
+ {team: .team_id, feature: .current_feature, phase: (.current_phase // "gen")}]' "$QUEUE")"
506
+ pending="$(jq '[.features[] | select(.status == "ready") | .id]' "$QUEUE")"
507
+ local tmp="${CHECKPOINT}.tmp"
508
+ jq -n --arg ts "$now_iso" --arg retry "$retry_iso" --arg reason "$reason" \
509
+ --argjson inprog "$in_progress" --argjson ready "$pending" \
510
+ '{timestamp:$ts, retry_at:$retry, last_error:$reason,
511
+ in_progress:$inprog, ready_features:$ready,
512
+ hold_count: 1, escalate_after: 72}' > "$tmp" && mv "$tmp" "$CHECKPOINT"
513
+ release_queue_lock
514
+ trap - EXIT
515
+ echo "[hold] checkpoint written — retry_at=$retry_iso reason=$reason"
516
+ }
517
+
518
+ cmd_resume_probe() {
519
+ # Lightweight probe — called on ScheduleWakeup resume.
520
+ # Exits 0 if ready to resume, 1 if still held, 2 if escalated.
521
+ [ ! -f "$CHECKPOINT" ] && { echo "[resume] no checkpoint — nothing to resume"; exit 0; }
522
+ local hold_count escalate_after
523
+ hold_count="$(jq -r '.hold_count // 1' "$CHECKPOINT")"
524
+ escalate_after="$(jq -r '.escalate_after // 72' "$CHECKPOINT")"
525
+ if [ "$hold_count" -ge "$escalate_after" ]; then
526
+ echo "[resume] ESCALATED — held $hold_count cycles (max=$escalate_after). Manual intervention required."
527
+ exit 2
528
+ fi
529
+ # Probe: minimal claude call — if rate-limited, exits non-zero quickly.
530
+ if command -v claude >/dev/null 2>&1; then
531
+ if echo "ping" | timeout 30 claude -p "reply only: pong" >/dev/null 2>&1; then
532
+ echo "[resume] probe OK — clearing hold"
533
+ rm -f "$CHECKPOINT"
534
+ exit 0
535
+ fi
536
+ fi
537
+ # Still held — increment and report
538
+ local tmp="${CHECKPOINT}.tmp"
539
+ jq '.hold_count = (.hold_count + 1) | .last_probe_at = now | todate' "$CHECKPOINT" > "$tmp" && mv "$tmp" "$CHECKPOINT"
540
+ echo "[resume] still held (cycle $((hold_count+1))/$escalate_after) — schedule another wake-up"
541
+ exit 1
542
+ }
543
+
544
+ cmd_hold_status() {
545
+ if [ ! -f "$CHECKPOINT" ]; then
546
+ echo '{"held":false}'
547
+ return
548
+ fi
549
+ jq '. + {held:true}' "$CHECKPOINT"
550
+ }
551
+
490
552
  # ── Dispatch ──
491
553
  case "$CMD" in
492
554
  init) cmd_init "$@" ;;
@@ -500,8 +562,11 @@ case "$CMD" in
500
562
  recover) cmd_recover ;;
501
563
  next-sprint) cmd_next_sprint ;;
502
564
  status) cmd_status ;;
565
+ hold) cmd_hold "$@" ;;
566
+ resume-probe) cmd_resume_probe ;;
567
+ hold-status) cmd_hold_status ;;
503
568
  *)
504
- echo "Usage: harness-queue-manager.sh <init|dequeue|auto-dispatch|idle-slots|pass|fail|requeue|recover|next-sprint|update_phase|status> [args]"
569
+ echo "Usage: harness-queue-manager.sh <init|dequeue|auto-dispatch|idle-slots|pass|fail|requeue|recover|next-sprint|update_phase|status|hold|resume-probe|hold-status> [args]"
505
570
  exit 1
506
571
  ;;
507
572
  esac
@@ -0,0 +1,57 @@
1
+ #!/bin/bash
2
+ # harness-keywait.sh — 패널 refresh/quit 키 대기 헬퍼
3
+ #
4
+ # 사용법:
5
+ # source scripts/lib/harness-keywait.sh
6
+ # while true; do
7
+ # render_panel
8
+ # echo " [r] refresh [q] quit"
9
+ # if wait_or_refresh 3; then
10
+ # # 타임아웃 → 일반 주기 리프레시
11
+ # :
12
+ # else
13
+ # # 'r' 키 즉시 리프레시 (반환값 != 0)
14
+ # :
15
+ # fi
16
+ # done
17
+ #
18
+ # 키 처리:
19
+ # 'r' / 'R' / Enter → 즉시 복귀 (return 1 — 강제 리프레시 신호)
20
+ # 'q' / 'Q' → exit 0 (패널 종료)
21
+ # 그 외 / 타임아웃 → return 0 (정상 주기 리프레시)
22
+ #
23
+ # TTY 가 아니면 plain sleep 으로 동작 (Claude Code Bash 도구 등에서도 안전).
24
+
25
+ wait_or_refresh() {
26
+ local secs="${1:-3}"
27
+
28
+ # Non-TTY fallback
29
+ if ! [ -t 0 ]; then
30
+ sleep "$secs"
31
+ return 0
32
+ fi
33
+
34
+ local key=""
35
+ # -t: timeout, -n 1: 1글자, -s: 에코 안함
36
+ # read 는 Enter 에서도 타임아웃 전에 즉시 복귀 (빈 키)
37
+ if IFS= read -rsn 1 -t "$secs" key 2>/dev/null; then
38
+ case "$key" in
39
+ q|Q)
40
+ # 커서 복원 후 종료
41
+ tput cnorm 2>/dev/null || true
42
+ clear
43
+ exit 0
44
+ ;;
45
+ r|R|"")
46
+ # 강제 리프레시 시그널
47
+ return 1
48
+ ;;
49
+ *)
50
+ # 기타 키는 무시하고 일반 복귀
51
+ return 0
52
+ ;;
53
+ esac
54
+ fi
55
+ # 타임아웃
56
+ return 0
57
+ }
@@ -42,6 +42,18 @@ jq '.agent_status = "completed" | .completed_agents += ["planner"]' .harness/p
42
42
  - Gotcha 교정 후 재작업 → `failure.retry_target` (해당 에이전트)
43
43
  - `pipeline` → 선택된 파이프라인 (FULLSTACK/FE-ONLY/BE-ONLY)
44
44
  - `sprint.number` → `1`, `sprint.status` → `"in_progress"` (신규 파이프라인인 경우에만)
45
+ - **신규 파이프라인인 경우** `dispatch.id` 가 `null` 이면 counter 를 올리고 새 ID 를 발급 (v5.7+):
46
+ ```bash
47
+ # dispatch.id 가 이미 있으면 기존 dispatch 유지, 없으면 새로 발급
48
+ cur=$(jq -r '.dispatch.id // ""' .harness/progress.json)
49
+ if [ -z "$cur" ]; then
50
+ next=$(jq -r '((.dispatch.counter // 0) + 1)' .harness/progress.json)
51
+ new_id=$(printf 'D-%03d' "$next")
52
+ bash scripts/harness-progress-set.sh . \
53
+ ".dispatch.counter = $next | .dispatch.id = \"$new_id\""
54
+ fi
55
+ ```
56
+ 아카이빙 후 `dispatch.id` 는 `null` 로 리셋되므로, 다음 dispatcher 실행 시 새 D-NNN 이 할당된다.
45
57
  2. `.harness/progress.log`에 요약 한 줄 추가
46
58
  3. **STOP. 다음 에이전트를 직접 호출하지 않는다.**
47
59
  4. 출력: `"✓ Dispatcher 완료. bash scripts/harness-next.sh 실행하여 다음 단계 확인."`