@shawnstack/quickforge 1.7.11 → 1.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (122) hide show
  1. package/README.md +1 -1
  2. package/dist/assets/AgentProfilesPage-lKGIE5Dj.js +1 -0
  3. package/dist/assets/ChatPanelHost-CWUYIkXG.js +11 -0
  4. package/dist/assets/CloudAccountSettingsPage-BhXH0ups.js +1 -0
  5. package/dist/assets/{PluginsPage-mLvwvTmL.js → PluginsPage-DHWF13mN.js} +1 -1
  6. package/dist/assets/ScheduledTasksPage-DLAg-i8c.js +2 -0
  7. package/dist/assets/{SettingsWorkspacePage-0OINfEqK.js → SettingsWorkspacePage-BhUzL5oi.js} +313 -313
  8. package/dist/assets/ShareLinksSettingsPage-DogGa3lH.js +1 -0
  9. package/dist/assets/SharedConversationPage-7QpeaG_P.js +1 -0
  10. package/dist/assets/TerminalDock-CWSKcuiU.js +2 -0
  11. package/dist/assets/WorkspaceInspector-CEfyYqWy.js +36 -0
  12. package/dist/assets/{abnfDiagram-VRR7QNED-CMhVoYUh.js → abnfDiagram-VRR7QNED-DGklo3PO.js} +1 -1
  13. package/dist/assets/{architectureDiagram-ZJ3FMSHR-C52qL6Hg.js → architectureDiagram-ZJ3FMSHR-CioWLVTp.js} +1 -1
  14. package/dist/assets/{blockDiagram-677ZJIJ3-CM3XbLol.js → blockDiagram-677ZJIJ3-BJWBq5e9.js} +1 -1
  15. package/dist/assets/{c4Diagram-LMCZKHZV-D7Bh1ui1.js → c4Diagram-LMCZKHZV-BacRvqFK.js} +1 -1
  16. package/dist/assets/channel-CeXY-wZo.js +1 -0
  17. package/dist/assets/{chunk-32BRIVSS-xgrtbT8e.js → chunk-32BRIVSS-B53BNE2C.js} +1 -1
  18. package/dist/assets/{chunk-52WLFC77-CDbaHw80.js → chunk-52WLFC77-CYq4hT4f.js} +1 -1
  19. package/dist/assets/{chunk-C7G6YPKG-ZMLUvfLy.js → chunk-C7G6YPKG-KFaGpP_Z.js} +1 -1
  20. package/dist/assets/{chunk-EX3LRPZG-BcQhaqt7.js → chunk-EX3LRPZG-DqVd_ihG.js} +1 -1
  21. package/dist/assets/{chunk-FWX5IMBZ-DrnrdrSu.js → chunk-FWX5IMBZ-BxRjZkOP.js} +2 -2
  22. package/dist/assets/{chunk-HOUHSVGY-tdc-Z9eW.js → chunk-HOUHSVGY-C9QO8omv.js} +1 -1
  23. package/dist/assets/{chunk-ICXQ74PX-Cd7igVGt.js → chunk-ICXQ74PX-ClbGPgLV.js} +1 -1
  24. package/dist/assets/{chunk-MOJQB5TN-CLTjV5ov.js → chunk-MOJQB5TN-Ba4wlF1J.js} +1 -1
  25. package/dist/assets/{chunk-OGEWGWER-TgvkXkFX.js → chunk-OGEWGWER-DczhQmcx.js} +1 -1
  26. package/dist/assets/{chunk-PUDLZKDR-C5GMVYup.js → chunk-PUDLZKDR-CPWpcKUF.js} +1 -1
  27. package/dist/assets/{chunk-Q4XR5HBZ-fo5-HsqY.js → chunk-Q4XR5HBZ-CiF6T--X.js} +1 -1
  28. package/dist/assets/{chunk-V7JOEXUC-CEDazV6w.js → chunk-V7JOEXUC-Dxv0zwmw.js} +1 -1
  29. package/dist/assets/{chunk-VAUOI2AC-CWYjg8zE.js → chunk-VAUOI2AC-D3J5bw1P.js} +1 -1
  30. package/dist/assets/{chunk-VR4S4FIN-C_-eEj-9.js → chunk-VR4S4FIN-BXQSwwII.js} +1 -1
  31. package/dist/assets/{chunk-WYO6CB5R-NF-MsJJR.js → chunk-WYO6CB5R-C0azl-fk.js} +2 -2
  32. package/dist/assets/{chunk-ZGVPDNZ5-CRrAYi4g.js → chunk-ZGVPDNZ5-D_8F4Akx.js} +1 -1
  33. package/dist/assets/classDiagram-OUVF2IWQ-B8M6sf49.js +1 -0
  34. package/dist/assets/classDiagram-v2-EOCWNBFH-BY7fzBDH.js +1 -0
  35. package/dist/assets/{cynefinDiagram-TSTJHNR4-D4sZ7LYI.js → cynefinDiagram-TSTJHNR4-BFlsWz0B.js} +1 -1
  36. package/dist/assets/{dagre-VKFMJZFB-BlsnSh-Q.js → dagre-VKFMJZFB-DXhpjlEs.js} +1 -1
  37. package/dist/assets/{diagram-FQU43EPY-DW1yQiRO.js → diagram-FQU43EPY-DRVVOD9k.js} +1 -1
  38. package/dist/assets/{diagram-G47NLZAW-BTUAsFjb.js → diagram-G47NLZAW-BV9Gw1tL.js} +1 -1
  39. package/dist/assets/{diagram-NH7WQ7WH-BcaomqSd.js → diagram-NH7WQ7WH-CrUkXfRY.js} +1 -1
  40. package/dist/assets/{diagram-OA4YK3LP-DDLX7Xwr.js → diagram-OA4YK3LP-CuILjn7v.js} +1 -1
  41. package/dist/assets/{diagram-WEI45ONY-CIave0An.js → diagram-WEI45ONY-nlNN3tWB.js} +1 -1
  42. package/dist/assets/{ebnfDiagram-CCIWWBDH-Cr1Y6lPu.js → ebnfDiagram-CCIWWBDH-BuDgZt9n.js} +1 -1
  43. package/dist/assets/{erDiagram-Q63AITRT-D3EgiVVH.js → erDiagram-Q63AITRT-MEwxjmtP.js} +1 -1
  44. package/dist/assets/flowDiagram-23GEKE2U-BiUP-Scq.js +1 -0
  45. package/dist/assets/{ganttDiagram-NO4QXBWP-B0oyDIqK.js → ganttDiagram-NO4QXBWP-DNw5qghE.js} +1 -1
  46. package/dist/assets/{gitGraphDiagram-IHSO6WYX-uQtr6IqK.js → gitGraphDiagram-IHSO6WYX-CaTcakAt.js} +1 -1
  47. package/dist/assets/index-CX36XjVA.css +3 -0
  48. package/dist/assets/index-CgVTwN15.js +66 -0
  49. package/dist/assets/{infoDiagram-FWYZ7A6U-B_GzlLIM.js → infoDiagram-FWYZ7A6U-DwfyP7dU.js} +1 -1
  50. package/dist/assets/{ishikawaDiagram-FXEZZL3T-CKQIsSdU.js → ishikawaDiagram-FXEZZL3T-CmrzbOCV.js} +1 -1
  51. package/dist/assets/{journeyDiagram-5HDEW3XC-DrrvzZL1.js → journeyDiagram-5HDEW3XC-CYqe3sgr.js} +1 -1
  52. package/dist/assets/{kanban-definition-HUTT4EX6-CJQRTdXz.js → kanban-definition-HUTT4EX6-BMDO_mv0.js} +1 -1
  53. package/dist/assets/{line-B4DxX7tC.js → line-SbxJR8ZT.js} +1 -1
  54. package/dist/assets/lit-vendor-CinchGmI.js +2 -0
  55. package/dist/assets/local-tools-CJVNO1aW.js +388 -0
  56. package/dist/assets/{mcp-servers-dialog-B2lALtTE.js → mcp-servers-dialog-xf3UR49J.js} +2 -2
  57. package/dist/assets/{mermaid.core-DWpShXCs.js → mermaid.core-CwXTzYjs.js} +3 -3
  58. package/dist/assets/{mindmap-definition-LN4V7U3C-CEVwZD78.js → mindmap-definition-LN4V7U3C-CLyJ78mO.js} +1 -1
  59. package/dist/assets/{pegDiagram-2B236MQR-CBLkBdnQ.js → pegDiagram-2B236MQR-4qFznrSw.js} +1 -1
  60. package/dist/assets/{pi-web-ui-DV5jKjSg.js → pi-web-ui-BNuvqKDT.js} +289 -289
  61. package/dist/assets/{pieDiagram-ENE6RG2P-8tCnwtWS.js → pieDiagram-ENE6RG2P-141sGxVe.js} +1 -1
  62. package/dist/assets/{quadrantDiagram-ABIIQ3AL-CWd28cHF.js → quadrantDiagram-ABIIQ3AL-SUI44pAX.js} +1 -1
  63. package/dist/assets/{railroadDiagram-RFXS5EU6-CI4ouOBs.js → railroadDiagram-RFXS5EU6-1oiZuYBB.js} +1 -1
  64. package/dist/assets/{requirementDiagram-TGXJPOKE-DtFz7sw4.js → requirementDiagram-TGXJPOKE-CfVishLi.js} +1 -1
  65. package/dist/assets/{sankeyDiagram-HTMAVEWB--VeOwwk9.js → sankeyDiagram-HTMAVEWB-DVu7J_Z9.js} +1 -1
  66. package/dist/assets/{sequenceDiagram-DBY2YBRQ-B8AHk5Oe.js → sequenceDiagram-DBY2YBRQ-D-Ip2yiS.js} +1 -1
  67. package/dist/assets/{skills-dialog-DYp35JM3.js → skills-dialog-Bbqtuy2h.js} +3 -3
  68. package/dist/assets/{stateDiagram-2N3HPSRC-etXnJ8X8.js → stateDiagram-2N3HPSRC-Cg0hWwrp.js} +1 -1
  69. package/dist/assets/stateDiagram-v2-6OUMAXLB-DhmzNQd9.js +1 -0
  70. package/dist/assets/{swimlanes-5IMT3BWC-pabIKAmn.js → swimlanes-5IMT3BWC-Co5D-cGU.js} +1 -1
  71. package/dist/assets/swimlanesDiagram-G3AALYLV-BEvomQNa.js +8 -0
  72. package/dist/assets/{timeline-definition-FHXFAJF6-C55MKm3L.js → timeline-definition-FHXFAJF6-CBk4c_Ua.js} +1 -1
  73. package/dist/assets/{vennDiagram-L72KCM5P-CgNf1Q9W.js → vennDiagram-L72KCM5P-BII7wdRx.js} +1 -1
  74. package/dist/assets/{wardleyDiagram-EHGQE667-Drt5ytec.js → wardleyDiagram-EHGQE667-CkyJJIXP.js} +1 -1
  75. package/dist/assets/{xychartDiagram-FW5EYKEG-hhoEo_Ar.js → xychartDiagram-FW5EYKEG-B1A66Ks0.js} +1 -1
  76. package/dist/index.html +6 -6
  77. package/dist/licenses/material-icon-theme.txt +8 -8
  78. package/package.json +1 -1
  79. package/server/agent-manager.mjs +305 -49
  80. package/server/approval-store.mjs +1 -0
  81. package/server/ask-store.mjs +78 -0
  82. package/server/context-references.mjs +139 -0
  83. package/server/custom-commands.mjs +43 -0
  84. package/server/index.mjs +1 -1
  85. package/server/project-config.mjs +17 -5
  86. package/server/routes/agent-profiles.mjs +16 -1
  87. package/server/routes/agent.mjs +10 -1
  88. package/server/routes/shared-conversation.mjs +13 -1
  89. package/server/routes/skills.mjs +22 -0
  90. package/server/routes/workspace.mjs +166 -2
  91. package/server/selected-capabilities.mjs +76 -0
  92. package/server/session-state-import.mjs +20 -3
  93. package/server/skills.mjs +1 -1
  94. package/server/system-prompt.mjs +1 -3
  95. package/server/tools/definitions.mjs +18 -10
  96. package/server/tools/index.mjs +14 -2
  97. package/server/utils/workspace.mjs +13 -4
  98. package/skills/skill-creator/SKILL.md +485 -485
  99. package/skills/skill-creator/agents/analyzer.md +274 -274
  100. package/skills/skill-creator/agents/comparator.md +202 -202
  101. package/skills/skill-creator/agents/grader.md +223 -223
  102. package/skills/skill-creator/assets/eval_review.html +146 -146
  103. package/skills/skill-creator/eval-viewer/viewer.html +1325 -1325
  104. package/skills/skill-creator/references/schemas.md +430 -430
  105. package/dist/assets/AgentProfilesPage-Cvb1KTz5.js +0 -1
  106. package/dist/assets/ChatPanelHost-C7zV-xE_.js +0 -48
  107. package/dist/assets/CloudAccountSettingsPage-mAs_nbvw.js +0 -1
  108. package/dist/assets/ScheduledTasksPage-CuUW_9lq.js +0 -2
  109. package/dist/assets/ShareLinksSettingsPage-DqeyJYoD.js +0 -1
  110. package/dist/assets/SharedConversationPage-C7CDEoX9.js +0 -1
  111. package/dist/assets/TerminalDock-CZXcuMpq.js +0 -2
  112. package/dist/assets/WorkspaceInspector-CNSBOEvP.js +0 -36
  113. package/dist/assets/channel-Duu1CuH9.js +0 -1
  114. package/dist/assets/classDiagram-OUVF2IWQ-Cc63fLWP.js +0 -1
  115. package/dist/assets/classDiagram-v2-EOCWNBFH-CoXU0qT6.js +0 -1
  116. package/dist/assets/flowDiagram-23GEKE2U-BS7KXcqn.js +0 -1
  117. package/dist/assets/index-BgMvQW1b.js +0 -66
  118. package/dist/assets/index-CB1g2D0f.css +0 -3
  119. package/dist/assets/lit-vendor-DvaQX_UQ.js +0 -2
  120. package/dist/assets/local-tools-Dopjg9xq.js +0 -270
  121. package/dist/assets/stateDiagram-v2-6OUMAXLB-Dms3pICj.js +0 -1
  122. package/dist/assets/swimlanesDiagram-G3AALYLV-B14aWjv6.js +0 -8
@@ -1,202 +1,202 @@
1
- # Blind Comparator Agent
2
-
3
- Compare two outputs WITHOUT knowing which skill produced them.
4
-
5
- ## Role
6
-
7
- The Blind Comparator judges which output better accomplishes the eval task. You receive two outputs labeled A and B, but you do NOT know which skill produced which. This prevents bias toward a particular skill or approach.
8
-
9
- Your judgment is based purely on output quality and task completion.
10
-
11
- ## Inputs
12
-
13
- You receive these parameters in your prompt:
14
-
15
- - **output_a_path**: Path to the first output file or directory
16
- - **output_b_path**: Path to the second output file or directory
17
- - **eval_prompt**: The original task/prompt that was executed
18
- - **expectations**: List of expectations to check (optional - may be empty)
19
-
20
- ## Process
21
-
22
- ### Step 1: Read Both Outputs
23
-
24
- 1. Examine output A (file or directory)
25
- 2. Examine output B (file or directory)
26
- 3. Note the type, structure, and content of each
27
- 4. If outputs are directories, examine all relevant files inside
28
-
29
- ### Step 2: Understand the Task
30
-
31
- 1. Read the eval_prompt carefully
32
- 2. Identify what the task requires:
33
- - What should be produced?
34
- - What qualities matter (accuracy, completeness, format)?
35
- - What would distinguish a good output from a poor one?
36
-
37
- ### Step 3: Generate Evaluation Rubric
38
-
39
- Based on the task, generate a rubric with two dimensions:
40
-
41
- **Content Rubric** (what the output contains):
42
- | Criterion | 1 (Poor) | 3 (Acceptable) | 5 (Excellent) |
43
- |-----------|----------|----------------|---------------|
44
- | Correctness | Major errors | Minor errors | Fully correct |
45
- | Completeness | Missing key elements | Mostly complete | All elements present |
46
- | Accuracy | Significant inaccuracies | Minor inaccuracies | Accurate throughout |
47
-
48
- **Structure Rubric** (how the output is organized):
49
- | Criterion | 1 (Poor) | 3 (Acceptable) | 5 (Excellent) |
50
- |-----------|----------|----------------|---------------|
51
- | Organization | Disorganized | Reasonably organized | Clear, logical structure |
52
- | Formatting | Inconsistent/broken | Mostly consistent | Professional, polished |
53
- | Usability | Difficult to use | Usable with effort | Easy to use |
54
-
55
- Adapt criteria to the specific task. For example:
56
- - PDF form → "Field alignment", "Text readability", "Data placement"
57
- - Document → "Section structure", "Heading hierarchy", "Paragraph flow"
58
- - Data output → "Schema correctness", "Data types", "Completeness"
59
-
60
- ### Step 4: Evaluate Each Output Against the Rubric
61
-
62
- For each output (A and B):
63
-
64
- 1. **Score each criterion** on the rubric (1-5 scale)
65
- 2. **Calculate dimension totals**: Content score, Structure score
66
- 3. **Calculate overall score**: Average of dimension scores, scaled to 1-10
67
-
68
- ### Step 5: Check Assertions (if provided)
69
-
70
- If expectations are provided:
71
-
72
- 1. Check each expectation against output A
73
- 2. Check each expectation against output B
74
- 3. Count pass rates for each output
75
- 4. Use expectation scores as secondary evidence (not the primary decision factor)
76
-
77
- ### Step 6: Determine the Winner
78
-
79
- Compare A and B based on (in priority order):
80
-
81
- 1. **Primary**: Overall rubric score (content + structure)
82
- 2. **Secondary**: Assertion pass rates (if applicable)
83
- 3. **Tiebreaker**: If truly equal, declare a TIE
84
-
85
- Be decisive - ties should be rare. One output is usually better, even if marginally.
86
-
87
- ### Step 7: Write Comparison Results
88
-
89
- Save results to a JSON file at the path specified (or `comparison.json` if not specified).
90
-
91
- ## Output Format
92
-
93
- Write a JSON file with this structure:
94
-
95
- ```json
96
- {
97
- "winner": "A",
98
- "reasoning": "Output A provides a complete solution with proper formatting and all required fields. Output B is missing the date field and has formatting inconsistencies.",
99
- "rubric": {
100
- "A": {
101
- "content": {
102
- "correctness": 5,
103
- "completeness": 5,
104
- "accuracy": 4
105
- },
106
- "structure": {
107
- "organization": 4,
108
- "formatting": 5,
109
- "usability": 4
110
- },
111
- "content_score": 4.7,
112
- "structure_score": 4.3,
113
- "overall_score": 9.0
114
- },
115
- "B": {
116
- "content": {
117
- "correctness": 3,
118
- "completeness": 2,
119
- "accuracy": 3
120
- },
121
- "structure": {
122
- "organization": 3,
123
- "formatting": 2,
124
- "usability": 3
125
- },
126
- "content_score": 2.7,
127
- "structure_score": 2.7,
128
- "overall_score": 5.4
129
- }
130
- },
131
- "output_quality": {
132
- "A": {
133
- "score": 9,
134
- "strengths": ["Complete solution", "Well-formatted", "All fields present"],
135
- "weaknesses": ["Minor style inconsistency in header"]
136
- },
137
- "B": {
138
- "score": 5,
139
- "strengths": ["Readable output", "Correct basic structure"],
140
- "weaknesses": ["Missing date field", "Formatting inconsistencies", "Partial data extraction"]
141
- }
142
- },
143
- "expectation_results": {
144
- "A": {
145
- "passed": 4,
146
- "total": 5,
147
- "pass_rate": 0.80,
148
- "details": [
149
- {"text": "Output includes name", "passed": true},
150
- {"text": "Output includes date", "passed": true},
151
- {"text": "Format is PDF", "passed": true},
152
- {"text": "Contains signature", "passed": false},
153
- {"text": "Readable text", "passed": true}
154
- ]
155
- },
156
- "B": {
157
- "passed": 3,
158
- "total": 5,
159
- "pass_rate": 0.60,
160
- "details": [
161
- {"text": "Output includes name", "passed": true},
162
- {"text": "Output includes date", "passed": false},
163
- {"text": "Format is PDF", "passed": true},
164
- {"text": "Contains signature", "passed": false},
165
- {"text": "Readable text", "passed": true}
166
- ]
167
- }
168
- }
169
- }
170
- ```
171
-
172
- If no expectations were provided, omit the `expectation_results` field entirely.
173
-
174
- ## Field Descriptions
175
-
176
- - **winner**: "A", "B", or "TIE"
177
- - **reasoning**: Clear explanation of why the winner was chosen (or why it's a tie)
178
- - **rubric**: Structured rubric evaluation for each output
179
- - **content**: Scores for content criteria (correctness, completeness, accuracy)
180
- - **structure**: Scores for structure criteria (organization, formatting, usability)
181
- - **content_score**: Average of content criteria (1-5)
182
- - **structure_score**: Average of structure criteria (1-5)
183
- - **overall_score**: Combined score scaled to 1-10
184
- - **output_quality**: Summary quality assessment
185
- - **score**: 1-10 rating (should match rubric overall_score)
186
- - **strengths**: List of positive aspects
187
- - **weaknesses**: List of issues or shortcomings
188
- - **expectation_results**: (Only if expectations provided)
189
- - **passed**: Number of expectations that passed
190
- - **total**: Total number of expectations
191
- - **pass_rate**: Fraction passed (0.0 to 1.0)
192
- - **details**: Individual expectation results
193
-
194
- ## Guidelines
195
-
196
- - **Stay blind**: DO NOT try to infer which skill produced which output. Judge purely on output quality.
197
- - **Be specific**: Cite specific examples when explaining strengths and weaknesses.
198
- - **Be decisive**: Choose a winner unless outputs are genuinely equivalent.
199
- - **Output quality first**: Assertion scores are secondary to overall task completion.
200
- - **Be objective**: Don't favor outputs based on style preferences; focus on correctness and completeness.
201
- - **Explain your reasoning**: The reasoning field should make it clear why you chose the winner.
202
- - **Handle edge cases**: If both outputs fail, pick the one that fails less badly. If both are excellent, pick the one that's marginally better.
1
+ # Blind Comparator Agent
2
+
3
+ Compare two outputs WITHOUT knowing which skill produced them.
4
+
5
+ ## Role
6
+
7
+ The Blind Comparator judges which output better accomplishes the eval task. You receive two outputs labeled A and B, but you do NOT know which skill produced which. This prevents bias toward a particular skill or approach.
8
+
9
+ Your judgment is based purely on output quality and task completion.
10
+
11
+ ## Inputs
12
+
13
+ You receive these parameters in your prompt:
14
+
15
+ - **output_a_path**: Path to the first output file or directory
16
+ - **output_b_path**: Path to the second output file or directory
17
+ - **eval_prompt**: The original task/prompt that was executed
18
+ - **expectations**: List of expectations to check (optional - may be empty)
19
+
20
+ ## Process
21
+
22
+ ### Step 1: Read Both Outputs
23
+
24
+ 1. Examine output A (file or directory)
25
+ 2. Examine output B (file or directory)
26
+ 3. Note the type, structure, and content of each
27
+ 4. If outputs are directories, examine all relevant files inside
28
+
29
+ ### Step 2: Understand the Task
30
+
31
+ 1. Read the eval_prompt carefully
32
+ 2. Identify what the task requires:
33
+ - What should be produced?
34
+ - What qualities matter (accuracy, completeness, format)?
35
+ - What would distinguish a good output from a poor one?
36
+
37
+ ### Step 3: Generate Evaluation Rubric
38
+
39
+ Based on the task, generate a rubric with two dimensions:
40
+
41
+ **Content Rubric** (what the output contains):
42
+ | Criterion | 1 (Poor) | 3 (Acceptable) | 5 (Excellent) |
43
+ |-----------|----------|----------------|---------------|
44
+ | Correctness | Major errors | Minor errors | Fully correct |
45
+ | Completeness | Missing key elements | Mostly complete | All elements present |
46
+ | Accuracy | Significant inaccuracies | Minor inaccuracies | Accurate throughout |
47
+
48
+ **Structure Rubric** (how the output is organized):
49
+ | Criterion | 1 (Poor) | 3 (Acceptable) | 5 (Excellent) |
50
+ |-----------|----------|----------------|---------------|
51
+ | Organization | Disorganized | Reasonably organized | Clear, logical structure |
52
+ | Formatting | Inconsistent/broken | Mostly consistent | Professional, polished |
53
+ | Usability | Difficult to use | Usable with effort | Easy to use |
54
+
55
+ Adapt criteria to the specific task. For example:
56
+ - PDF form → "Field alignment", "Text readability", "Data placement"
57
+ - Document → "Section structure", "Heading hierarchy", "Paragraph flow"
58
+ - Data output → "Schema correctness", "Data types", "Completeness"
59
+
60
+ ### Step 4: Evaluate Each Output Against the Rubric
61
+
62
+ For each output (A and B):
63
+
64
+ 1. **Score each criterion** on the rubric (1-5 scale)
65
+ 2. **Calculate dimension totals**: Content score, Structure score
66
+ 3. **Calculate overall score**: Average of dimension scores, scaled to 1-10
67
+
68
+ ### Step 5: Check Assertions (if provided)
69
+
70
+ If expectations are provided:
71
+
72
+ 1. Check each expectation against output A
73
+ 2. Check each expectation against output B
74
+ 3. Count pass rates for each output
75
+ 4. Use expectation scores as secondary evidence (not the primary decision factor)
76
+
77
+ ### Step 6: Determine the Winner
78
+
79
+ Compare A and B based on (in priority order):
80
+
81
+ 1. **Primary**: Overall rubric score (content + structure)
82
+ 2. **Secondary**: Assertion pass rates (if applicable)
83
+ 3. **Tiebreaker**: If truly equal, declare a TIE
84
+
85
+ Be decisive - ties should be rare. One output is usually better, even if marginally.
86
+
87
+ ### Step 7: Write Comparison Results
88
+
89
+ Save results to a JSON file at the path specified (or `comparison.json` if not specified).
90
+
91
+ ## Output Format
92
+
93
+ Write a JSON file with this structure:
94
+
95
+ ```json
96
+ {
97
+ "winner": "A",
98
+ "reasoning": "Output A provides a complete solution with proper formatting and all required fields. Output B is missing the date field and has formatting inconsistencies.",
99
+ "rubric": {
100
+ "A": {
101
+ "content": {
102
+ "correctness": 5,
103
+ "completeness": 5,
104
+ "accuracy": 4
105
+ },
106
+ "structure": {
107
+ "organization": 4,
108
+ "formatting": 5,
109
+ "usability": 4
110
+ },
111
+ "content_score": 4.7,
112
+ "structure_score": 4.3,
113
+ "overall_score": 9.0
114
+ },
115
+ "B": {
116
+ "content": {
117
+ "correctness": 3,
118
+ "completeness": 2,
119
+ "accuracy": 3
120
+ },
121
+ "structure": {
122
+ "organization": 3,
123
+ "formatting": 2,
124
+ "usability": 3
125
+ },
126
+ "content_score": 2.7,
127
+ "structure_score": 2.7,
128
+ "overall_score": 5.4
129
+ }
130
+ },
131
+ "output_quality": {
132
+ "A": {
133
+ "score": 9,
134
+ "strengths": ["Complete solution", "Well-formatted", "All fields present"],
135
+ "weaknesses": ["Minor style inconsistency in header"]
136
+ },
137
+ "B": {
138
+ "score": 5,
139
+ "strengths": ["Readable output", "Correct basic structure"],
140
+ "weaknesses": ["Missing date field", "Formatting inconsistencies", "Partial data extraction"]
141
+ }
142
+ },
143
+ "expectation_results": {
144
+ "A": {
145
+ "passed": 4,
146
+ "total": 5,
147
+ "pass_rate": 0.80,
148
+ "details": [
149
+ {"text": "Output includes name", "passed": true},
150
+ {"text": "Output includes date", "passed": true},
151
+ {"text": "Format is PDF", "passed": true},
152
+ {"text": "Contains signature", "passed": false},
153
+ {"text": "Readable text", "passed": true}
154
+ ]
155
+ },
156
+ "B": {
157
+ "passed": 3,
158
+ "total": 5,
159
+ "pass_rate": 0.60,
160
+ "details": [
161
+ {"text": "Output includes name", "passed": true},
162
+ {"text": "Output includes date", "passed": false},
163
+ {"text": "Format is PDF", "passed": true},
164
+ {"text": "Contains signature", "passed": false},
165
+ {"text": "Readable text", "passed": true}
166
+ ]
167
+ }
168
+ }
169
+ }
170
+ ```
171
+
172
+ If no expectations were provided, omit the `expectation_results` field entirely.
173
+
174
+ ## Field Descriptions
175
+
176
+ - **winner**: "A", "B", or "TIE"
177
+ - **reasoning**: Clear explanation of why the winner was chosen (or why it's a tie)
178
+ - **rubric**: Structured rubric evaluation for each output
179
+ - **content**: Scores for content criteria (correctness, completeness, accuracy)
180
+ - **structure**: Scores for structure criteria (organization, formatting, usability)
181
+ - **content_score**: Average of content criteria (1-5)
182
+ - **structure_score**: Average of structure criteria (1-5)
183
+ - **overall_score**: Combined score scaled to 1-10
184
+ - **output_quality**: Summary quality assessment
185
+ - **score**: 1-10 rating (should match rubric overall_score)
186
+ - **strengths**: List of positive aspects
187
+ - **weaknesses**: List of issues or shortcomings
188
+ - **expectation_results**: (Only if expectations provided)
189
+ - **passed**: Number of expectations that passed
190
+ - **total**: Total number of expectations
191
+ - **pass_rate**: Fraction passed (0.0 to 1.0)
192
+ - **details**: Individual expectation results
193
+
194
+ ## Guidelines
195
+
196
+ - **Stay blind**: DO NOT try to infer which skill produced which output. Judge purely on output quality.
197
+ - **Be specific**: Cite specific examples when explaining strengths and weaknesses.
198
+ - **Be decisive**: Choose a winner unless outputs are genuinely equivalent.
199
+ - **Output quality first**: Assertion scores are secondary to overall task completion.
200
+ - **Be objective**: Don't favor outputs based on style preferences; focus on correctness and completeness.
201
+ - **Explain your reasoning**: The reasoning field should make it clear why you chose the winner.
202
+ - **Handle edge cases**: If both outputs fail, pick the one that fails less badly. If both are excellent, pick the one that's marginally better.