@walwal-harness/cli 5.0.9 → 5.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/commands/harness-team.md
CHANGED
|
@@ -210,6 +210,22 @@ Agent({
|
|
|
210
210
|
prompt: "당신은 독립 Evaluator입니다. Generator가 작성한 코드를 AC 기준으로 냉정하게 평가합니다.
|
|
211
211
|
Generator의 의도나 추론 과정은 알 수 없습니다. 오직 코드와 결과만 봅니다.
|
|
212
212
|
|
|
213
|
+
## 실시간 로깅 (필수)
|
|
214
|
+
|
|
215
|
+
평가 진행 상황을 실시간으로 기록합니다. **각 AC 검증마다 반드시 logev를 호출**하세요.
|
|
216
|
+
|
|
217
|
+
```bash
|
|
218
|
+
HARNESS_ROOT=$(git worktree list | head -1 | awk '{print $1}')
|
|
219
|
+
LOG=\"$HARNESS_ROOT/.harness/progress.log\"
|
|
220
|
+
logev() { echo \"$(date +'%Y-%m-%d %H:%M') | team-{N} | $1 | $2\" >> \"$LOG\"; }
|
|
221
|
+
```
|
|
222
|
+
|
|
223
|
+
**로깅 시점:**
|
|
224
|
+
1. 평가 시작 즉시: `logev eval-start \"{FEATURE_ID} evaluating — {AC수} ACs\"`
|
|
225
|
+
2. 각 AC 검증 후: `logev eval-check \"{FEATURE_ID} AC-{N}: [PASS/FAIL] {근거 요약}\"`
|
|
226
|
+
3. tsc/eslint 검증 후: `logev eval-check \"{FEATURE_ID} gate: tsc {OK/FAIL}, eslint {OK/FAIL}\"`
|
|
227
|
+
4. 최종 판정: `logev eval-done \"{FEATURE_ID} VERDICT={PASS/FAIL} SCORE={X.XX}/3.00\"`
|
|
228
|
+
|
|
213
229
|
## 평가 대상
|
|
214
230
|
- Feature ID: {FEATURE_ID}
|
|
215
231
|
- AC: jq '.features[] | select(.id == \"{FEATURE_ID}\").acceptance_criteria' .harness/actions/feature-list.json
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@walwal-harness/cli",
|
|
3
|
-
"version": "5.0
|
|
3
|
+
"version": "5.1.0",
|
|
4
4
|
"description": "Production harness for AI agent engineering — Solo/Team mode, Planner, Generator(BE/FE), Evaluator(Func/Visual), optional Brainstormer. Supports React, Next.js, and Flutter FE stacks.",
|
|
5
5
|
"bin": {
|
|
6
6
|
"walwal-harness": "bin/init.js"
|
|
@@ -40,7 +40,7 @@ strip_ansi() {
|
|
|
40
40
|
}
|
|
41
41
|
|
|
42
42
|
get_term_width() {
|
|
43
|
-
tput cols 2>/dev/null || echo 80
|
|
43
|
+
echo "${_COLS:-$(tput cols 2>/dev/null || echo 80)}"
|
|
44
44
|
}
|
|
45
45
|
|
|
46
46
|
get_mode() {
|
|
@@ -150,14 +150,28 @@ render_solo_prompt_history() {
|
|
|
150
150
|
local short_ts icon color
|
|
151
151
|
short_ts=$(echo "$ts" | sed 's/^[0-9]*-//')
|
|
152
152
|
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
153
|
+
# action 기반으로 먼저 분류 (team 모드에서 agent=team-N이므로)
|
|
154
|
+
case "$action" in
|
|
155
|
+
eval-start|eval-check|eval-done|eval)
|
|
156
|
+
icon="✦" ; color="$MAGENTA" ;;
|
|
157
|
+
gen-start|gen-read|gen-write|gen-test|gen-done|gen)
|
|
158
|
+
icon="▶" ; color="$GREEN" ;;
|
|
159
|
+
result|pass)
|
|
160
|
+
icon="✓" ; color="$GREEN" ;;
|
|
161
|
+
fail)
|
|
162
|
+
icon="✗" ; color="$RED" ;;
|
|
163
|
+
*)
|
|
164
|
+
# fallback: agent 기반
|
|
165
|
+
case "$agent" in
|
|
166
|
+
dispatcher*) icon="▸" ; color="$MAGENTA" ;;
|
|
167
|
+
planner*) icon="□" ; color="$YELLOW" ;;
|
|
168
|
+
generator*) icon="▶" ; color="$GREEN" ;;
|
|
169
|
+
eval*) icon="✦" ; color="$MAGENTA" ;;
|
|
170
|
+
team-*) icon="◆" ; color="$CYAN" ;;
|
|
171
|
+
user*) icon="★" ; color="$BOLD" ;;
|
|
172
|
+
*) icon="·" ; color="$DIM" ;;
|
|
173
|
+
esac
|
|
174
|
+
;;
|
|
161
175
|
esac
|
|
162
176
|
|
|
163
177
|
if [ ${#detail} -gt 40 ]; then detail="${detail:0:38}.."; fi
|
|
@@ -331,9 +345,10 @@ render_archive_prompts() {
|
|
|
331
345
|
total_prompts=$(echo "$all_prompts" | wc -l | tr -d ' ')
|
|
332
346
|
|
|
333
347
|
# 마지막 프롬프트를 제외한 모든 프롬프트 = archived (처리 완료)
|
|
348
|
+
# newest first (tac), 전체 출력 (제한 없음)
|
|
334
349
|
local archived
|
|
335
350
|
if [ "$total_prompts" -le 1 ]; then return; fi
|
|
336
|
-
archived=$(echo "$all_prompts" | head -n $((total_prompts - 1)) |
|
|
351
|
+
archived=$(echo "$all_prompts" | head -n $((total_prompts - 1)) | tac)
|
|
337
352
|
|
|
338
353
|
echo ""
|
|
339
354
|
echo -e "${BOLD}Archive Prompt${RESET} ${DIM}(처리 완료)${RESET}"
|
|
@@ -350,10 +365,6 @@ render_archive_prompts() {
|
|
|
350
365
|
|
|
351
366
|
echo -e " ${GREEN}✓${RESET} ${DIM}${short_ts}${RESET} ${detail}"
|
|
352
367
|
done
|
|
353
|
-
|
|
354
|
-
if [ "$total_prompts" -gt 9 ]; then
|
|
355
|
-
echo -e " ${DIM}… +$((total_prompts - 9)) more${RESET}"
|
|
356
|
-
fi
|
|
357
368
|
}
|
|
358
369
|
|
|
359
370
|
# ══════════════════════════════════════════
|
|
@@ -396,6 +407,9 @@ trap 'tput cnorm 2>/dev/null; exit 0' EXIT INT TERM
|
|
|
396
407
|
clear
|
|
397
408
|
|
|
398
409
|
while true; do
|
|
410
|
+
_ROWS=$(tput lines 2>/dev/null || echo 30)
|
|
411
|
+
_COLS=$(tput cols 2>/dev/null || echo 80)
|
|
412
|
+
export _ROWS _COLS
|
|
399
413
|
buf=$(render_dashboard 2>&1)
|
|
400
414
|
tput cup 0 0 2>/dev/null
|
|
401
415
|
echo "$buf"
|
|
@@ -159,8 +159,14 @@ render_team_section() {
|
|
|
159
159
|
case "$action" in
|
|
160
160
|
gen|gen-start|gen-read|gen-write|gen-test|gen-done)
|
|
161
161
|
a_icon="▶"; a_color="$GREEN"; a_label="Gen" ;;
|
|
162
|
-
eval
|
|
163
|
-
a_icon="✦"; a_color="$
|
|
162
|
+
eval-start)
|
|
163
|
+
a_icon="✦"; a_color="$MAGENTA"; a_label="Eval" ;;
|
|
164
|
+
eval-check)
|
|
165
|
+
a_icon="·"; a_color="$MAGENTA"; a_label="Eval" ;;
|
|
166
|
+
eval-done)
|
|
167
|
+
a_icon="✦"; a_color="${BOLD}${MAGENTA}"; a_label="Eval" ;;
|
|
168
|
+
eval)
|
|
169
|
+
a_icon="✦"; a_color="$MAGENTA"; a_label="Eval" ;;
|
|
164
170
|
result|pass)
|
|
165
171
|
a_icon="✓"; a_color="$GREEN"; a_label="Result" ;;
|
|
166
172
|
fail)
|
|
@@ -96,34 +96,30 @@ render_processing_status() {
|
|
|
96
96
|
echo ""
|
|
97
97
|
}
|
|
98
98
|
|
|
99
|
-
# 사용자 프롬프트만 필터하여 표시
|
|
99
|
+
# 사용자 프롬프트만 필터하여 표시 (newest first, 전체 출력)
|
|
100
100
|
render_user_prompts() {
|
|
101
101
|
if [ ! -f "$PROGRESS_LOG" ]; then
|
|
102
102
|
echo -e " ${DIM}(프롬프트 기록 없음)${RESET}"
|
|
103
103
|
return
|
|
104
104
|
fi
|
|
105
105
|
|
|
106
|
-
|
|
107
|
-
term_height=$(tput lines 2>/dev/null || echo 30)
|
|
108
|
-
local max_lines=$((term_height - 10))
|
|
109
|
-
if [ "$max_lines" -lt 5 ]; then max_lines=5; fi
|
|
110
|
-
|
|
111
|
-
# user-prompt 행만 필터
|
|
106
|
+
# user-prompt 행만 필터 → tac으로 newest first
|
|
112
107
|
local prompts
|
|
113
|
-
prompts=$(grep '| user-prompt |' "$PROGRESS_LOG" 2>/dev/null |
|
|
108
|
+
prompts=$(grep '| user-prompt |' "$PROGRESS_LOG" 2>/dev/null | tac)
|
|
114
109
|
|
|
115
110
|
if [ -z "$prompts" ]; then
|
|
116
111
|
echo -e " ${DIM}(사용자 프롬프트 없음)${RESET}"
|
|
117
112
|
return
|
|
118
113
|
fi
|
|
119
114
|
|
|
120
|
-
local line_num=0
|
|
121
115
|
local total_prompts
|
|
122
|
-
total_prompts=$(
|
|
116
|
+
total_prompts=$(echo "$prompts" | wc -l | tr -d ' ')
|
|
123
117
|
|
|
124
|
-
|
|
125
|
-
|
|
118
|
+
# 터미널 폭: main loop에서 측정한 _COLS 사용 (subshell 내 tput 불가 대비)
|
|
119
|
+
local max_width=$(( ${_COLS:-80} - 12 ))
|
|
120
|
+
if [ "$max_width" -lt 20 ]; then max_width=20; fi
|
|
126
121
|
|
|
122
|
+
echo "$prompts" | while IFS= read -r line; do
|
|
127
123
|
local ts detail
|
|
128
124
|
ts=$(echo "$line" | awk -F'|' '{gsub(/^ +| +$/,"",$1); print $1}')
|
|
129
125
|
detail=$(echo "$line" | awk -F'|' '{gsub(/^ +| +$/,"",$4); print $4}')
|
|
@@ -131,20 +127,10 @@ render_user_prompts() {
|
|
|
131
127
|
local short_ts
|
|
132
128
|
short_ts=$(echo "$ts" | grep -oE '[0-9]{2}:[0-9]{2}' | tail -1 || echo "$ts")
|
|
133
129
|
|
|
134
|
-
# 프롬프트 내용 truncate
|
|
135
|
-
local max_width
|
|
136
|
-
max_width=$(( $(tput cols 2>/dev/null || echo 40) - 12 ))
|
|
137
|
-
if [ "$max_width" -lt 20 ]; then max_width=20; fi
|
|
138
130
|
if [ ${#detail} -gt "$max_width" ]; then
|
|
139
131
|
detail="${detail:0:$((max_width - 2))}.."
|
|
140
132
|
fi
|
|
141
133
|
|
|
142
|
-
# 처리 상태 찾기: 이 프롬프트 이후의 첫 번째 에이전트 액션
|
|
143
|
-
local processed_by=""
|
|
144
|
-
local prompt_ts_epoch
|
|
145
|
-
prompt_ts_epoch=$(echo "$ts" | sed 's/[^0-9:]//g')
|
|
146
|
-
|
|
147
|
-
# 간단히: 마지막 프롬프트인지 확인 → 현재 처리 중 표시
|
|
148
134
|
echo -e " ${DIM}${short_ts}${RESET} ${WHITE}${detail}${RESET}"
|
|
149
135
|
done
|
|
150
136
|
|
|
@@ -164,6 +150,10 @@ trap 'tput cnorm 2>/dev/null; exit 0' EXIT INT TERM
|
|
|
164
150
|
clear
|
|
165
151
|
|
|
166
152
|
while true; do
|
|
153
|
+
# subshell 내 tput 불가 → 미리 터미널 크기 측정
|
|
154
|
+
_ROWS=$(tput lines 2>/dev/null || echo 30)
|
|
155
|
+
_COLS=$(tput cols 2>/dev/null || echo 80)
|
|
156
|
+
export _ROWS _COLS
|
|
167
157
|
buf=$(render_all 2>&1)
|
|
168
158
|
tput cup 0 0 2>/dev/null
|
|
169
159
|
echo "$buf"
|
package/skills/planner/SKILL.md
CHANGED
|
@@ -75,10 +75,58 @@ disable-model-invocation: true
|
|
|
75
75
|
## Constraints
|
|
76
76
|
|
|
77
77
|
- 기술 구현 세부사항은 Generator에 위임
|
|
78
|
-
- Sprint당 기능 3-5개 권장
|
|
79
78
|
- 각 기능에 `layer`, `service`, `depends_on` 명시
|
|
80
79
|
- API 계약의 스키마는 Pydantic/class-validator로 직접 변환 가능한 수준
|
|
81
80
|
|
|
81
|
+
## Team 병렬 스케줄링 규칙 (필수)
|
|
82
|
+
|
|
83
|
+
Team Mode는 **최대 3팀이 동시 작업**한다. Planner는 feature-list.json 설계 시 다음 규칙을 반드시 준수한다.
|
|
84
|
+
|
|
85
|
+
### 핵심 원칙: Sprint 시작 시 ready ≥ 3
|
|
86
|
+
|
|
87
|
+
Sprint 시작 시점에 `depends_on`이 모두 충족된(또는 비어 있는) feature가 **최소 3개** 있어야 3팀이 즉시 가동된다. ready가 1~2개이면 나머지 팀은 유휴 상태가 된다.
|
|
88
|
+
|
|
89
|
+
### 의존성 그래프 형태
|
|
90
|
+
|
|
91
|
+
**금지 — 직렬 체인:**
|
|
92
|
+
```
|
|
93
|
+
F-001 → F-002 → F-003 → F-004 (ready=1, 1팀만 작업)
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
**권장 — 넓은 DAG (fan-out):**
|
|
97
|
+
```
|
|
98
|
+
┌→ F-002 (no deps)
|
|
99
|
+
Foundation → F-003 (no deps) (ready=3, 3팀 동시)
|
|
100
|
+
└→ F-004 (no deps)
|
|
101
|
+
↓
|
|
102
|
+
F-005 (depends_on: [F-002, F-003])
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
### Sprint 설계 체크리스트
|
|
106
|
+
|
|
107
|
+
1. **Sprint당 feature 수**: `팀수 × 2` 이상 (3팀 = 최소 6개)
|
|
108
|
+
- 3개는 즉시 시작, 나머지는 앞선 feature 완료 시 투입
|
|
109
|
+
- 팀이 PASS 후 대기하지 않고 바로 다음 feature를 가져감
|
|
110
|
+
2. **동시 ready 보장**: Sprint 내 feature 중 `depends_on: []`인 것이 ≥ 3개
|
|
111
|
+
3. **의존성 깊이(critical path) 최소화**: 같은 Sprint 내 체인 깊이 ≤ 2단계
|
|
112
|
+
4. **layer 분산**: 같은 Sprint에 backend-only, frontend-only, fullstack을 혼합하여 서로 독립적으로 작업 가능
|
|
113
|
+
5. **Feature 분할**: 하나의 큰 feature 대신 독립 AC 그룹으로 분할
|
|
114
|
+
- 예: "대시보드 전체" (1개) → "통계 카드" + "차트 영역" + "최근 활동" (3개, 동시 작업 가능)
|
|
115
|
+
|
|
116
|
+
### 검증: ready count 시뮬레이션
|
|
117
|
+
|
|
118
|
+
feature-list.json 완성 후, Sprint별 ready count를 머릿속으로 시뮬레이션한다:
|
|
119
|
+
|
|
120
|
+
```
|
|
121
|
+
Sprint N 시작 → ready 목록 계산
|
|
122
|
+
ready ≥ 3 → OK (3팀 동시 가동)
|
|
123
|
+
ready = 2 → WARNING (1팀 유휴)
|
|
124
|
+
ready = 1 → FAIL → feature 분할 또는 의존성 제거 필요
|
|
125
|
+
ready = 0 → CRITICAL → 이전 Sprint 의존성 재설계 필요
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
ready가 3 미만인 Sprint가 있으면 **feature를 더 작게 분할하거나 의존성을 제거**하여 수정한다.
|
|
129
|
+
|
|
82
130
|
## After Completion
|
|
83
131
|
|
|
84
132
|
1. 사용자에게 plan.md + api-contract.json 리뷰 요청
|