@lark-apaas/coding-miaoda-sandbox-skills 0.1.0-dev.28c4f05 → 0.1.0-dev.4e64c13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/miaoda/animation-skill/SKILL.md +348 -0
- package/miaoda/authz-cli/SKILL.md +1 -0
- package/miaoda/charts-skill/SKILL.md +264 -0
- package/miaoda/creative-to-fullstack/SKILL.md +157 -0
- package/miaoda/creative-to-fullstack/references/artifact-signals.md +46 -0
- package/miaoda/creative-to-fullstack/references/ui-to-function.md +134 -0
- package/miaoda/data-analysis/SKILL.md +151 -0
- package/miaoda/data-analysis/references/json-output-specification.md +277 -0
- package/miaoda/data-analysis/references/post-analysis-guide.md +77 -0
- package/miaoda/data-analysis/references/python-analysis-reference.md +272 -0
- package/miaoda/data-analysis/references/tmp-file-management-guide.md +100 -0
- package/miaoda/debug-investigation/SKILL.md +21 -18
- package/miaoda/extract-json-schema/SKILL.md +147 -0
- package/miaoda/lark-apps/SKILL.md +37 -0
- package/miaoda/lark-apps/references/openapi-key.md +80 -0
- package/miaoda/lark-apps-authz/SKILL.md +292 -0
- package/miaoda/lark-apps-authz/references/permission-points.md +39 -0
- package/miaoda/lark-apps-authz/references/role.md +122 -0
- package/miaoda/lark-apps-db/SKILL.md +226 -0
- package/miaoda/lark-apps-db/references/full-reference.md +302 -0
- package/miaoda/lark-apps-file/SKILL.md +216 -0
- package/miaoda/lark-apps-ops/SKILL.md +62 -0
- package/miaoda/lark-apps-ops/references/lark-apps-access-scope-get.md +30 -0
- package/miaoda/lark-apps-ops/references/lark-apps-access-scope-set.md +40 -0
- package/miaoda/lark-apps-ops/references/lark-apps-cache.md +62 -0
- package/miaoda/lark-apps-ops/references/lark-apps-env.md +46 -0
- package/miaoda/lark-apps-ops/references/lark-apps-local-dev.md +25 -0
- package/miaoda/lark-apps-ops/references/lark-apps-member.md +93 -0
- package/miaoda/lark-apps-ops/references/lark-apps-observability.md +46 -0
- package/miaoda/lark-apps-ops/references/lark-apps-plugin-install.md +36 -0
- package/miaoda/lark-apps-ops/references/lark-apps-plugin-list.md +23 -0
- package/miaoda/lark-apps-ops/references/lark-apps-plugin-uninstall.md +25 -0
- package/miaoda/lark-apps-ops/references/lark-apps-release-create.md +30 -0
- package/miaoda/lark-apps-ops/references/lark-apps-release-get.md +28 -0
- package/miaoda/lark-apps-ops/references/lark-apps-release-list.md +31 -0
- package/miaoda/lark-apps-ops/references/lark-apps-update.md +30 -0
- package/miaoda/lark-apps-ops/references/openapi-key.md +80 -0
- package/miaoda/miaoda-file/SKILL.md +1 -0
- package/miaoda/miaoda-sql/SKILL.md +6 -2
- package/miaoda/performance-review/SKILL.md +144 -0
- package/miaoda/performance-review/references/business-analyzer.md +139 -0
- package/miaoda/performance-review/references/examples.md +107 -0
- package/miaoda/reviewer-usage/SKILL.md +111 -0
- package/miaoda/testing-guide/SKILL.md +218 -0
- package/miaoda-design/lark-apps-comment/SKILL.md +110 -0
- package/miaoda-design/lark-apps-ops/SKILL.md +45 -0
- package/miaoda-design/lark-apps-ops/references/lark-apps-release-create.md +51 -0
- package/miaoda-design/lark-apps-ops/references/lark-apps-release-get.md +28 -0
- package/miaoda-design/lark-apps-ops/references/lark-apps-release-list.md +31 -0
- package/miaoda-design/lark-apps-ops/references/lark-apps-update.md +33 -0
- package/{shared → miaoda-modern}/lark-apps/SKILL.md +5 -5
- package/miaoda-modern/lark-apps/references/openapi-key.md +80 -0
- package/miaoda-modern/lark-apps-ops/SKILL.md +62 -0
- package/miaoda-modern/lark-apps-ops/references/lark-apps-access-scope-get.md +30 -0
- package/miaoda-modern/lark-apps-ops/references/lark-apps-access-scope-set.md +40 -0
- package/miaoda-modern/lark-apps-ops/references/lark-apps-cache.md +62 -0
- package/miaoda-modern/lark-apps-ops/references/lark-apps-env.md +46 -0
- package/miaoda-modern/lark-apps-ops/references/lark-apps-local-dev.md +25 -0
- package/miaoda-modern/lark-apps-ops/references/lark-apps-member.md +93 -0
- package/miaoda-modern/lark-apps-ops/references/lark-apps-observability.md +46 -0
- package/miaoda-modern/lark-apps-ops/references/lark-apps-plugin-install.md +36 -0
- package/miaoda-modern/lark-apps-ops/references/lark-apps-plugin-list.md +23 -0
- package/miaoda-modern/lark-apps-ops/references/lark-apps-plugin-uninstall.md +25 -0
- package/miaoda-modern/lark-apps-ops/references/lark-apps-release-create.md +30 -0
- package/miaoda-modern/lark-apps-ops/references/lark-apps-release-get.md +28 -0
- package/miaoda-modern/lark-apps-ops/references/lark-apps-release-list.md +31 -0
- package/miaoda-modern/lark-apps-ops/references/lark-apps-update.md +30 -0
- package/{shared/lark-apps → miaoda-modern/lark-apps-ops}/references/openapi-key.md +3 -3
- package/miaoda-modern/memory/SKILL.md +86 -0
- package/package.json +1 -1
- package/shared/lark-cli/SKILL.md +221 -0
- package/shared/lark-cli/lark-base/README.md +56 -0
- package/shared/lark-cli/lark-base/references/lark-base-commands.md +108 -0
- package/shared/lark-cli/lark-calendar/README.md +158 -0
- package/shared/lark-cli/lark-calendar/references/lark-calendar-meeting.md +30 -0
- package/shared/lark-cli/lark-calendar/references/lark-calendar-room-find.md +108 -0
- package/shared/lark-cli/lark-calendar/references/lark-calendar-suggestion.md +120 -0
- package/shared/lark-cli/lark-contact/README.md +35 -0
- package/shared/lark-cli/lark-contact/references/lark-contact-get-user.md +13 -0
- package/shared/lark-cli/lark-contact/references/lark-contact-search-user.md +121 -0
- package/shared/lark-cli/lark-doc/README.md +67 -0
- package/shared/lark-cli/lark-doc/references/lark-doc-fetch.md +138 -0
- package/shared/lark-cli/lark-doc/references/lark-doc-history.md +61 -0
- package/shared/lark-cli/lark-drive/README.md +129 -0
- package/shared/lark-cli/lark-drive/references/lark-drive-files-list.md +183 -0
- package/shared/lark-cli/lark-im/README.md +84 -0
- package/shared/lark-cli/lark-im/references/lark-im-chat-list.md +140 -0
- package/shared/lark-cli/lark-im/references/lark-im-chat-members-list.md +84 -0
- package/shared/lark-cli/lark-im/references/lark-im-chat-search.md +135 -0
- package/shared/lark-cli/lark-im/references/lark-im-reactions.md +232 -0
- package/shared/lark-cli/lark-minutes/README.md +51 -0
- package/shared/lark-cli/lark-minutes/references/lark-minutes-download.md +130 -0
- package/shared/lark-cli/lark-sheets/README.md +173 -0
- package/shared/lark-cli/lark-sheets/references/lark-sheets-changeset.md +105 -0
- package/shared/lark-cli/lark-sheets/references/lark-sheets-chart.md +45 -0
- package/shared/lark-cli/lark-sheets/references/lark-sheets-conditional-format.md +42 -0
- package/shared/lark-cli/lark-sheets/references/lark-sheets-filter-view.md +49 -0
- package/shared/lark-cli/lark-sheets/references/lark-sheets-filter.md +42 -0
- package/shared/lark-cli/lark-sheets/references/lark-sheets-float-image.md +43 -0
- package/shared/lark-cli/lark-sheets/references/lark-sheets-formula-verify.md +64 -0
- package/shared/lark-cli/lark-sheets/references/lark-sheets-history.md +70 -0
- package/shared/lark-cli/lark-sheets/references/lark-sheets-pivot-table.md +44 -0
- package/shared/lark-cli/lark-sheets/references/lark-sheets-read-data.md +216 -0
- package/shared/lark-cli/lark-sheets/references/lark-sheets-search-replace.md +67 -0
- package/shared/lark-cli/lark-sheets/references/lark-sheets-sheet-structure.md +52 -0
- package/shared/lark-cli/lark-sheets/references/lark-sheets-sparkline.md +47 -0
- package/shared/lark-cli/lark-sheets/references/lark-sheets-workbook.md +69 -0
- package/shared/lark-cli/lark-sheets/scripts/sheets_df.py +32 -0
- package/shared/lark-cli/lark-slides/README.md +86 -0
- package/shared/lark-cli/lark-slides/references/lark-slides-history.md +105 -0
- package/shared/lark-cli/lark-slides/references/lark-slides-xml-presentation-slide-get.md +108 -0
- package/shared/lark-cli/lark-slides/references/lark-slides-xml-presentations-get.md +77 -0
- package/shared/lark-cli/lark-task/README.md +93 -0
- package/shared/lark-cli/lark-task/references/lark-task-get-my-tasks.md +57 -0
- package/shared/lark-cli/lark-task/references/lark-task-get-related-tasks.md +49 -0
- package/shared/lark-cli/lark-task/references/lark-task-search.md +36 -0
- package/shared/lark-cli/lark-task/references/lark-task-tasklist-search.md +35 -0
- package/shared/lark-cli/lark-vc/README.md +40 -0
- package/shared/lark-cli/lark-vc/references/lark-vc-recording.md +31 -0
- package/shared/lark-cli/lark-whiteboard/README.md +35 -0
- package/shared/lark-cli/lark-whiteboard/references/lark-whiteboard-export.md +59 -0
- package/shared/lark-cli/lark-wiki/README.md +50 -0
- package/shared/lark-cli/lark-wiki/references/lark-wiki-node-get.md +59 -0
- package/shared/lark-cli/lark-wiki/references/lark-wiki-node-list.md +95 -0
- package/shared/lark-cli/lark-wiki/references/lark-wiki-space-list.md +68 -0
- package/miaoda-design/attachment/SKILL.md +0 -58
- /package/{shared → miaoda}/memory/SKILL.md +0 -0
- /package/{shared → miaoda-modern}/animation-skill/SKILL.md +0 -0
- /package/{shared → miaoda-modern}/charts-skill/SKILL.md +0 -0
- /package/{shared → miaoda-modern}/data-analysis/SKILL.md +0 -0
- /package/{shared → miaoda-modern}/data-analysis/references/json-output-specification.md +0 -0
- /package/{shared → miaoda-modern}/data-analysis/references/post-analysis-guide.md +0 -0
- /package/{shared → miaoda-modern}/data-analysis/references/python-analysis-reference.md +0 -0
- /package/{shared → miaoda-modern}/data-analysis/references/tmp-file-management-guide.md +0 -0
- /package/{shared → miaoda-modern}/extract-json-schema/SKILL.md +0 -0
- /package/{shared → miaoda-modern}/performance-review/SKILL.md +0 -0
- /package/{shared → miaoda-modern}/performance-review/references/business-analyzer.md +0 -0
- /package/{shared → miaoda-modern}/performance-review/references/examples.md +0 -0
- /package/{shared → miaoda-modern}/reviewer-usage/SKILL.md +0 -0
- /package/{shared → miaoda-modern}/testing-guide/SKILL.md +0 -0
|
@@ -0,0 +1,277 @@
|
|
|
1
|
+
# JSON Output Specification
|
|
2
|
+
|
|
3
|
+
## Complete Schema
|
|
4
|
+
|
|
5
|
+
```json
|
|
6
|
+
{
|
|
7
|
+
"analysis_id": "string (UUID v4)",
|
|
8
|
+
"timestamp": "string (ISO 8601 format)",
|
|
9
|
+
"metadata": {
|
|
10
|
+
"data_source": "string (file path or data description)",
|
|
11
|
+
"row_count": "integer",
|
|
12
|
+
"column_count": "integer",
|
|
13
|
+
"analysis_type": "exploratory | confirmatory | diagnostic | descriptive",
|
|
14
|
+
"business_question": "string (user's core question, optional)"
|
|
15
|
+
},
|
|
16
|
+
"data_quality": {
|
|
17
|
+
"completeness": "number (0-1, non-missing ratio)",
|
|
18
|
+
"missing_values": {
|
|
19
|
+
"column_name": "integer (missing count)"
|
|
20
|
+
},
|
|
21
|
+
"duplicates": "integer (duplicate row count)",
|
|
22
|
+
"outliers": {
|
|
23
|
+
"column_name": "integer (outlier count)"
|
|
24
|
+
},
|
|
25
|
+
"issues": ["string (issue description)"]
|
|
26
|
+
},
|
|
27
|
+
"statistics": {
|
|
28
|
+
"numeric_summary": {
|
|
29
|
+
"column_name": {
|
|
30
|
+
"count": "integer",
|
|
31
|
+
"mean": "number",
|
|
32
|
+
"std": "number",
|
|
33
|
+
"min": "number",
|
|
34
|
+
"max": "number",
|
|
35
|
+
"quartiles": ["number (Q1)", "number (Q2/median)", "number (Q3)"],
|
|
36
|
+
"skew": "number (optional)",
|
|
37
|
+
"kurtosis": "number (optional)"
|
|
38
|
+
}
|
|
39
|
+
},
|
|
40
|
+
"categorical_summary": {
|
|
41
|
+
"column_name": {
|
|
42
|
+
"unique_count": "integer",
|
|
43
|
+
"top_values": [
|
|
44
|
+
{"value": "string", "count": "integer", "percentage": "number"}
|
|
45
|
+
],
|
|
46
|
+
"mode": "string"
|
|
47
|
+
}
|
|
48
|
+
},
|
|
49
|
+
"correlations": [
|
|
50
|
+
{
|
|
51
|
+
"column1": "string",
|
|
52
|
+
"column2": "string",
|
|
53
|
+
"coefficient": "number",
|
|
54
|
+
"method": "pearson | spearman"
|
|
55
|
+
}
|
|
56
|
+
]
|
|
57
|
+
},
|
|
58
|
+
"insights": [
|
|
59
|
+
{
|
|
60
|
+
"id": "string (insight_1, insight_2, ...)",
|
|
61
|
+
"type": "correlation | trend | anomaly | distribution | comparison",
|
|
62
|
+
"title": "string (short title, within 10 characters)",
|
|
63
|
+
"description": "string (detailed description)",
|
|
64
|
+
"evidence": {
|
|
65
|
+
"metric": "string (metric name)",
|
|
66
|
+
"value": "number | string",
|
|
67
|
+
"p_value": "number (optional, for statistical tests)"
|
|
68
|
+
},
|
|
69
|
+
"significance": "high | medium | low",
|
|
70
|
+
"affected_columns": ["string"]
|
|
71
|
+
}
|
|
72
|
+
],
|
|
73
|
+
"recommendations": [
|
|
74
|
+
{
|
|
75
|
+
"id": "string (rec_1, rec_2, ...)",
|
|
76
|
+
"type": "visualization | further_analysis | data_collection | action",
|
|
77
|
+
"title": "string (recommendation title)",
|
|
78
|
+
"description": "string (detailed recommendation)",
|
|
79
|
+
"priority": "high | medium | low"
|
|
80
|
+
}
|
|
81
|
+
],
|
|
82
|
+
"visualizations": [
|
|
83
|
+
{
|
|
84
|
+
"id": "string (viz_1, viz_2, ...)",
|
|
85
|
+
"type": "histogram | scatter | line | bar | heatmap | box | pie",
|
|
86
|
+
"title": "string (chart title)",
|
|
87
|
+
"config": {
|
|
88
|
+
"x": "string (X-axis field name)",
|
|
89
|
+
"y": "string (Y-axis field name, optional)",
|
|
90
|
+
"group_by": "string (grouping field, optional)",
|
|
91
|
+
"aggregation": "string (aggregation method: sum/mean/count, optional)"
|
|
92
|
+
},
|
|
93
|
+
"insight_refs": ["string (associated insight id)"]
|
|
94
|
+
}
|
|
95
|
+
]
|
|
96
|
+
}
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
## CRITICAL: JSON Validity Rules
|
|
100
|
+
|
|
101
|
+
Python `json.dumps()` defaults to `allow_nan=True`, which outputs `NaN` and `Infinity` as bare tokens — **these are NOT valid JSON** and will cause TypeScript build failures (TS1328).
|
|
102
|
+
|
|
103
|
+
| Python value | json.dumps output | Valid JSON? | Fix |
|
|
104
|
+
|-------------|-------------------|-------------|-----|
|
|
105
|
+
| `float('nan')` | `NaN` | **NO** | Replace with `null` |
|
|
106
|
+
| `float('inf')` | `Infinity` | **NO** | Replace with `null` or max value |
|
|
107
|
+
| `float('-inf')` | `-Infinity` | **NO** | Replace with `null` or min value |
|
|
108
|
+
| `None` | `null` | Yes | OK |
|
|
109
|
+
|
|
110
|
+
**MUST** do one of:
|
|
111
|
+
1. Call `json.dump(..., allow_nan=False)` (raises error if NaN exists, forcing explicit handling)
|
|
112
|
+
2. Sanitize all float values before dumping: replace `NaN`/`Infinity` with `None`
|
|
113
|
+
|
|
114
|
+
Common sources of NaN in pandas: `df.mean()` on empty column, `df.std()` on single value, `df.skew()` on constant data, division by zero.
|
|
115
|
+
|
|
116
|
+
## Field Description
|
|
117
|
+
|
|
118
|
+
| Field Path | Required | Description |
|
|
119
|
+
| ---------- | -------- | ----------- |
|
|
120
|
+
| `analysis_id` | Yes | Unique identifier for analysis task, use UUID v4 |
|
|
121
|
+
| `timestamp` | Yes | Analysis completion time, ISO 8601 format |
|
|
122
|
+
| `metadata.*` | Yes | Data source metadata |
|
|
123
|
+
| `data_quality.*` | Yes | Data quality assessment results |
|
|
124
|
+
| `statistics.numeric_summary` | Conditional | Required when numeric columns exist |
|
|
125
|
+
| `statistics.categorical_summary` | Conditional | Required when categorical columns exist |
|
|
126
|
+
| `statistics.correlations` | No | Only populate when significant correlations found |
|
|
127
|
+
| `insights` | Yes | At least 1 insight required |
|
|
128
|
+
| `recommendations` | Yes | At least 1 recommendation required |
|
|
129
|
+
| `visualizations` | Yes | At least 1 visualization config required |
|
|
130
|
+
|
|
131
|
+
## Example Output
|
|
132
|
+
|
|
133
|
+
```json
|
|
134
|
+
{
|
|
135
|
+
"analysis_id": "a1b2c3d4-e5f6-7890-abcd-ef1234567890",
|
|
136
|
+
"timestamp": "2024-01-15T10:30:00Z",
|
|
137
|
+
"metadata": {
|
|
138
|
+
"data_source": "/data/sales_2023.csv",
|
|
139
|
+
"row_count": 10000,
|
|
140
|
+
"column_count": 12,
|
|
141
|
+
"analysis_type": "exploratory",
|
|
142
|
+
"business_question": "了解2023年销售数据的整体情况"
|
|
143
|
+
},
|
|
144
|
+
"data_quality": {
|
|
145
|
+
"completeness": 0.95,
|
|
146
|
+
"missing_values": {
|
|
147
|
+
"customer_age": 320,
|
|
148
|
+
"region": 180
|
|
149
|
+
},
|
|
150
|
+
"duplicates": 45,
|
|
151
|
+
"outliers": {
|
|
152
|
+
"order_amount": 23
|
|
153
|
+
},
|
|
154
|
+
"issues": [
|
|
155
|
+
"customer_age 列有 3.2% 缺失值",
|
|
156
|
+
"order_amount 存在 23 个异常高值(超出 IQR 上界)"
|
|
157
|
+
]
|
|
158
|
+
},
|
|
159
|
+
"statistics": {
|
|
160
|
+
"numeric_summary": {
|
|
161
|
+
"order_amount": {
|
|
162
|
+
"count": 10000,
|
|
163
|
+
"mean": 156.78,
|
|
164
|
+
"std": 89.34,
|
|
165
|
+
"min": 5.00,
|
|
166
|
+
"max": 2500.00,
|
|
167
|
+
"quartiles": [85.00, 135.00, 210.00],
|
|
168
|
+
"skew": 2.34,
|
|
169
|
+
"kurtosis": 8.91
|
|
170
|
+
},
|
|
171
|
+
"customer_age": {
|
|
172
|
+
"count": 9680,
|
|
173
|
+
"mean": 35.6,
|
|
174
|
+
"std": 12.3,
|
|
175
|
+
"min": 18,
|
|
176
|
+
"max": 75,
|
|
177
|
+
"quartiles": [26, 34, 44]
|
|
178
|
+
}
|
|
179
|
+
},
|
|
180
|
+
"categorical_summary": {
|
|
181
|
+
"product_category": {
|
|
182
|
+
"unique_count": 5,
|
|
183
|
+
"top_values": [
|
|
184
|
+
{"value": "Electronics", "count": 3500, "percentage": 35.0},
|
|
185
|
+
{"value": "Clothing", "count": 2800, "percentage": 28.0},
|
|
186
|
+
{"value": "Home", "count": 2000, "percentage": 20.0}
|
|
187
|
+
],
|
|
188
|
+
"mode": "Electronics"
|
|
189
|
+
}
|
|
190
|
+
},
|
|
191
|
+
"correlations": [
|
|
192
|
+
{
|
|
193
|
+
"column1": "order_amount",
|
|
194
|
+
"column2": "customer_age",
|
|
195
|
+
"coefficient": 0.72,
|
|
196
|
+
"method": "pearson"
|
|
197
|
+
}
|
|
198
|
+
]
|
|
199
|
+
},
|
|
200
|
+
"insights": [
|
|
201
|
+
{
|
|
202
|
+
"id": "insight_1",
|
|
203
|
+
"type": "correlation",
|
|
204
|
+
"title": "年龄与消费正相关",
|
|
205
|
+
"description": "客户年龄与订单金额呈较强正相关(r=0.72),年长客户倾向于更高消费。",
|
|
206
|
+
"evidence": {
|
|
207
|
+
"metric": "pearson_correlation",
|
|
208
|
+
"value": 0.72,
|
|
209
|
+
"p_value": 0.001
|
|
210
|
+
},
|
|
211
|
+
"significance": "high",
|
|
212
|
+
"affected_columns": ["customer_age", "order_amount"]
|
|
213
|
+
},
|
|
214
|
+
{
|
|
215
|
+
"id": "insight_2",
|
|
216
|
+
"type": "distribution",
|
|
217
|
+
"title": "订单金额右偏分布",
|
|
218
|
+
"description": "订单金额呈明显右偏(skew=2.34),大部分订单集中在低价区间,少量高价订单拉高均值。",
|
|
219
|
+
"evidence": {
|
|
220
|
+
"metric": "skewness",
|
|
221
|
+
"value": 2.34
|
|
222
|
+
},
|
|
223
|
+
"significance": "medium",
|
|
224
|
+
"affected_columns": ["order_amount"]
|
|
225
|
+
}
|
|
226
|
+
],
|
|
227
|
+
"recommendations": [
|
|
228
|
+
{
|
|
229
|
+
"id": "rec_1",
|
|
230
|
+
"type": "further_analysis",
|
|
231
|
+
"title": "深入分析高价值客户",
|
|
232
|
+
"description": "建议对年龄>40且订单金额>500的客户群体进行细分分析,挖掘其消费特征。",
|
|
233
|
+
"priority": "high"
|
|
234
|
+
},
|
|
235
|
+
{
|
|
236
|
+
"id": "rec_2",
|
|
237
|
+
"type": "visualization",
|
|
238
|
+
"title": "绘制年龄-消费散点图",
|
|
239
|
+
"description": "使用散点图可视化年龄与订单金额的关系,并按产品类别着色。",
|
|
240
|
+
"priority": "medium"
|
|
241
|
+
}
|
|
242
|
+
],
|
|
243
|
+
"visualizations": [
|
|
244
|
+
{
|
|
245
|
+
"id": "viz_1",
|
|
246
|
+
"type": "scatter",
|
|
247
|
+
"title": "客户年龄 vs 订单金额",
|
|
248
|
+
"config": {
|
|
249
|
+
"x": "customer_age",
|
|
250
|
+
"y": "order_amount",
|
|
251
|
+
"group_by": "product_category"
|
|
252
|
+
},
|
|
253
|
+
"insight_refs": ["insight_1"]
|
|
254
|
+
},
|
|
255
|
+
{
|
|
256
|
+
"id": "viz_2",
|
|
257
|
+
"type": "histogram",
|
|
258
|
+
"title": "订单金额分布",
|
|
259
|
+
"config": {
|
|
260
|
+
"x": "order_amount"
|
|
261
|
+
},
|
|
262
|
+
"insight_refs": ["insight_2"]
|
|
263
|
+
},
|
|
264
|
+
{
|
|
265
|
+
"id": "viz_3",
|
|
266
|
+
"type": "bar",
|
|
267
|
+
"title": "各产品类别销售额",
|
|
268
|
+
"config": {
|
|
269
|
+
"x": "product_category",
|
|
270
|
+
"y": "order_amount",
|
|
271
|
+
"aggregation": "sum"
|
|
272
|
+
},
|
|
273
|
+
"insight_refs": []
|
|
274
|
+
}
|
|
275
|
+
]
|
|
276
|
+
}
|
|
277
|
+
```
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
# Post-Analysis Guide
|
|
2
|
+
|
|
3
|
+
Steps to complete after analysis is done: update AGENTS.md and report output locations to user.
|
|
4
|
+
|
|
5
|
+
## 1. Update AGENTS.md
|
|
6
|
+
|
|
7
|
+
**CRITICAL**: After completing analysis, you **MUST** update AGENTS.md. This is NOT optional.
|
|
8
|
+
|
|
9
|
+
### Steps
|
|
10
|
+
|
|
11
|
+
1. Check if `AGENTS.md` exists in the project root; if not, create it
|
|
12
|
+
2. Append the analysis record to `AGENTS.md` following the format below exactly
|
|
13
|
+
|
|
14
|
+
> **NOTE**: Before writing code that consumes analysis data, always read the corresponding type definition in `shared/static/types.ts` first to ensure correct field access.
|
|
15
|
+
|
|
16
|
+
### AGENTS.md Format (MUST follow exactly)
|
|
17
|
+
|
|
18
|
+
> ### [Date] Analysis of [filename]
|
|
19
|
+
> - **Source file**: `path/to/source/data.csv`
|
|
20
|
+
> - **Analysis type**: exploratory | confirmatory | diagnostic | descriptive
|
|
21
|
+
> - **Key findings**: Brief summary of main insights (1-2 sentences)
|
|
22
|
+
> - **Output location**: `tmp/output/filename_analysis.json`
|
|
23
|
+
> - **Copied to**: `shared/static/filename_analysis.json` (if applicable)
|
|
24
|
+
> - **Type definition**: `shared/static/types.ts` → `FilenameAnalysis`
|
|
25
|
+
>
|
|
26
|
+
> #### Consumption
|
|
27
|
+
>
|
|
28
|
+
> ```typescript
|
|
29
|
+
> import type { FilenameAnalysis } from '@shared/static/types';
|
|
30
|
+
> import rawData from '@shared/static/filename_analysis.json';
|
|
31
|
+
>
|
|
32
|
+
> const data = rawData as FilenameAnalysis;
|
|
33
|
+
> // access fields: data.insights, data.statistics, data.visualizations, etc.
|
|
34
|
+
> ```
|
|
35
|
+
|
|
36
|
+
### Example Entry
|
|
37
|
+
|
|
38
|
+
> ### 2024-01-15 Analysis of sales_2023.csv
|
|
39
|
+
> - **Source file**: `data/sales_2023.csv`
|
|
40
|
+
> - **Analysis type**: exploratory
|
|
41
|
+
> - **Key findings**: Strong correlation between customer age and order amount (r=0.72). Order amounts show right-skewed distribution.
|
|
42
|
+
> - **Output location**: `tmp/output/sales_2023_exploratory.json`
|
|
43
|
+
> - **Copied to**: `shared/static/sales_2023_exploratory.json`
|
|
44
|
+
> - **Type definition**: `shared/static/types.ts` → `Sales2023Exploratory`
|
|
45
|
+
>
|
|
46
|
+
> #### Consumption
|
|
47
|
+
>
|
|
48
|
+
> ```typescript
|
|
49
|
+
> import type { Sales2023Exploratory } from '@shared/static/types';
|
|
50
|
+
> import rawData from '@shared/static/sales_2023_exploratory.json';
|
|
51
|
+
>
|
|
52
|
+
> const data = rawData as Sales2023Exploratory;
|
|
53
|
+
> const highInsights = data.insights.filter(i => i.significance === 'high');
|
|
54
|
+
> ```
|
|
55
|
+
|
|
56
|
+
## 2. Report Output Locations
|
|
57
|
+
|
|
58
|
+
After completing analysis, report to user:
|
|
59
|
+
|
|
60
|
+
```text
|
|
61
|
+
Analysis complete! Artifacts saved to:
|
|
62
|
+
|
|
63
|
+
📁 tmp/
|
|
64
|
+
├── output/
|
|
65
|
+
│ ├── sales_2023_exploratory.json # Structured analysis result
|
|
66
|
+
│ └── visualizations/
|
|
67
|
+
│ ├── sales_2023_age_vs_amount.png # Scatter plot
|
|
68
|
+
│ └── sales_2023_amount_dist.png # Histogram
|
|
69
|
+
└── intermediate/
|
|
70
|
+
└── sales_2023_cleaned.csv # Cleaned data (if applicable)
|
|
71
|
+
|
|
72
|
+
To use in your code, copy the JSON to shared/static/:
|
|
73
|
+
mkdir -p shared/static/ && cp tmp/output/sales_2023_exploratory.json shared/static/
|
|
74
|
+
|
|
75
|
+
Then import directly in your code:
|
|
76
|
+
import analysisData from '@shared/static/sales_2023_exploratory.json'
|
|
77
|
+
```
|
|
@@ -0,0 +1,272 @@
|
|
|
1
|
+
# Python Code Standards & Templates
|
|
2
|
+
|
|
3
|
+
## Recommended Libraries
|
|
4
|
+
|
|
5
|
+
```python
|
|
6
|
+
import pandas as pd # Data processing
|
|
7
|
+
import numpy as np # Numerical computation
|
|
8
|
+
from scipy import stats # Statistical analysis
|
|
9
|
+
from datetime import datetime
|
|
10
|
+
import json
|
|
11
|
+
import uuid
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
## Phase-Specific Code Templates
|
|
15
|
+
|
|
16
|
+
### Phase 2: Data Understanding
|
|
17
|
+
|
|
18
|
+
```python
|
|
19
|
+
import pandas as pd
|
|
20
|
+
|
|
21
|
+
df = pd.read_csv('data.csv') # or pd.read_excel(), pd.read_json()
|
|
22
|
+
|
|
23
|
+
print(f"Shape: {df.shape}")
|
|
24
|
+
print(f"Columns: {df.columns.tolist()}")
|
|
25
|
+
print(f"Dtypes:\n{df.dtypes}")
|
|
26
|
+
print(f"Head:\n{df.head()}")
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
### Phase 3: Data Quality Assessment
|
|
30
|
+
|
|
31
|
+
```python
|
|
32
|
+
# Missing values
|
|
33
|
+
missing = df.isnull().sum()
|
|
34
|
+
missing_pct = (missing / len(df) * 100).round(2)
|
|
35
|
+
print(f"Missing values:\n{missing[missing > 0]}")
|
|
36
|
+
|
|
37
|
+
# Duplicate rows
|
|
38
|
+
duplicates = df.duplicated().sum()
|
|
39
|
+
print(f"Duplicate rows: {duplicates}")
|
|
40
|
+
|
|
41
|
+
# Numeric outliers (IQR)
|
|
42
|
+
def detect_outliers_iqr(series):
|
|
43
|
+
Q1, Q3 = series.quantile([0.25, 0.75])
|
|
44
|
+
IQR = Q3 - Q1
|
|
45
|
+
lower, upper = Q1 - 1.5 * IQR, Q3 + 1.5 * IQR
|
|
46
|
+
return ((series < lower) | (series > upper)).sum()
|
|
47
|
+
|
|
48
|
+
for col in df.select_dtypes(include='number').columns:
|
|
49
|
+
outliers = detect_outliers_iqr(df[col])
|
|
50
|
+
if outliers > 0:
|
|
51
|
+
print(f"{col}: {outliers} outliers")
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
### Phase 4: Descriptive Analysis
|
|
55
|
+
|
|
56
|
+
```python
|
|
57
|
+
# 4.1 Numeric variables
|
|
58
|
+
numeric_stats = df.describe(percentiles=[0.25, 0.5, 0.75]).T
|
|
59
|
+
numeric_stats['skew'] = df.select_dtypes(include='number').skew()
|
|
60
|
+
numeric_stats['kurtosis'] = df.select_dtypes(include='number').kurtosis()
|
|
61
|
+
|
|
62
|
+
# 4.2 Categorical variables
|
|
63
|
+
for col in df.select_dtypes(include='object').columns:
|
|
64
|
+
print(f"\n{col}:")
|
|
65
|
+
print(f" Unique: {df[col].nunique()}")
|
|
66
|
+
print(f" Top values:\n{df[col].value_counts().head()}")
|
|
67
|
+
|
|
68
|
+
# 4.3 Correlation analysis (Pearson)
|
|
69
|
+
import numpy as np
|
|
70
|
+
correlation_matrix = df.select_dtypes(include='number').corr(method='pearson')
|
|
71
|
+
corr_pairs = []
|
|
72
|
+
for i in range(len(correlation_matrix.columns)):
|
|
73
|
+
for j in range(i+1, len(correlation_matrix.columns)):
|
|
74
|
+
r = correlation_matrix.iloc[i, j]
|
|
75
|
+
if abs(r) > 0.7:
|
|
76
|
+
corr_pairs.append({
|
|
77
|
+
'col1': correlation_matrix.columns[i],
|
|
78
|
+
'col2': correlation_matrix.columns[j],
|
|
79
|
+
'correlation': round(r, 3)
|
|
80
|
+
})
|
|
81
|
+
|
|
82
|
+
# 4.4 Group aggregation
|
|
83
|
+
grouped = df.groupby('category_column').agg({
|
|
84
|
+
'numeric_col1': ['mean', 'std', 'count'],
|
|
85
|
+
'numeric_col2': ['mean', 'std', 'count']
|
|
86
|
+
})
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
## Complete Analysis Template
|
|
90
|
+
|
|
91
|
+
```python
|
|
92
|
+
import pandas as pd
|
|
93
|
+
import numpy as np
|
|
94
|
+
from scipy import stats
|
|
95
|
+
import json
|
|
96
|
+
import uuid
|
|
97
|
+
from datetime import datetime
|
|
98
|
+
from pathlib import Path
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def setup_tmp_dirs(project_root: Path = None) -> dict:
|
|
102
|
+
"""Initialize tmp directory structure for analysis artifacts."""
|
|
103
|
+
if project_root is None:
|
|
104
|
+
project_root = Path.cwd()
|
|
105
|
+
|
|
106
|
+
tmp_dir = project_root / "tmp"
|
|
107
|
+
paths = {
|
|
108
|
+
"root": tmp_dir,
|
|
109
|
+
"scripts": tmp_dir / "analysis_scripts",
|
|
110
|
+
"intermediate": tmp_dir / "intermediate",
|
|
111
|
+
"output": tmp_dir / "output",
|
|
112
|
+
"visualizations": tmp_dir / "output" / "visualizations",
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
for path in paths.values():
|
|
116
|
+
path.mkdir(parents=True, exist_ok=True)
|
|
117
|
+
|
|
118
|
+
return paths
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def analyze_data(file_path: str, analysis_type: str = "exploratory",
|
|
122
|
+
project_root: Path = None) -> dict:
|
|
123
|
+
"""
|
|
124
|
+
Execute complete data analysis workflow.
|
|
125
|
+
|
|
126
|
+
Args:
|
|
127
|
+
file_path: Data file path
|
|
128
|
+
analysis_type: exploratory/confirmatory/diagnostic/descriptive
|
|
129
|
+
project_root: Project root for tmp directory. Defaults to cwd.
|
|
130
|
+
|
|
131
|
+
Returns:
|
|
132
|
+
Analysis result JSON conforming to specification
|
|
133
|
+
"""
|
|
134
|
+
# 0. Setup tmp directories
|
|
135
|
+
tmp_paths = setup_tmp_dirs(project_root)
|
|
136
|
+
|
|
137
|
+
# 1. Load data
|
|
138
|
+
if file_path.endswith('.csv'):
|
|
139
|
+
df = pd.read_csv(file_path)
|
|
140
|
+
elif file_path.endswith(('.xls', '.xlsx')):
|
|
141
|
+
df = pd.read_excel(file_path)
|
|
142
|
+
elif file_path.endswith('.json'):
|
|
143
|
+
df = pd.read_json(file_path)
|
|
144
|
+
else:
|
|
145
|
+
raise ValueError(f"Unsupported file format: {file_path}")
|
|
146
|
+
|
|
147
|
+
# 2. Initialize result structure
|
|
148
|
+
result = {
|
|
149
|
+
"analysis_id": str(uuid.uuid4()),
|
|
150
|
+
"timestamp": datetime.utcnow().isoformat() + "Z",
|
|
151
|
+
"metadata": {
|
|
152
|
+
"data_source": file_path,
|
|
153
|
+
"row_count": len(df),
|
|
154
|
+
"column_count": len(df.columns),
|
|
155
|
+
"analysis_type": analysis_type
|
|
156
|
+
},
|
|
157
|
+
"data_quality": {},
|
|
158
|
+
"statistics": {},
|
|
159
|
+
"insights": [],
|
|
160
|
+
"recommendations": [],
|
|
161
|
+
"visualizations": []
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
# 3. Data quality assessment
|
|
165
|
+
missing = df.isnull().sum()
|
|
166
|
+
result["data_quality"] = {
|
|
167
|
+
"completeness": round(1 - df.isnull().sum().sum() / df.size, 4),
|
|
168
|
+
"missing_values": {k: int(v) for k, v in missing[missing > 0].items()},
|
|
169
|
+
"duplicates": int(df.duplicated().sum()),
|
|
170
|
+
"outliers": {},
|
|
171
|
+
"issues": []
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
for col in df.select_dtypes(include='number').columns:
|
|
175
|
+
Q1, Q3 = df[col].quantile([0.25, 0.75])
|
|
176
|
+
IQR = Q3 - Q1
|
|
177
|
+
outlier_count = int(((df[col] < Q1 - 1.5 * IQR) | (df[col] > Q3 + 1.5 * IQR)).sum())
|
|
178
|
+
if outlier_count > 0:
|
|
179
|
+
result["data_quality"]["outliers"][col] = outlier_count
|
|
180
|
+
|
|
181
|
+
# 4. Descriptive statistics - Numeric
|
|
182
|
+
numeric_cols = df.select_dtypes(include='number').columns
|
|
183
|
+
if len(numeric_cols) > 0:
|
|
184
|
+
result["statistics"]["numeric_summary"] = {}
|
|
185
|
+
for col in numeric_cols:
|
|
186
|
+
series = df[col].dropna()
|
|
187
|
+
result["statistics"]["numeric_summary"][col] = {
|
|
188
|
+
"count": int(series.count()),
|
|
189
|
+
"mean": round(float(series.mean()), 4),
|
|
190
|
+
"std": round(float(series.std()), 4),
|
|
191
|
+
"min": round(float(series.min()), 4),
|
|
192
|
+
"max": round(float(series.max()), 4),
|
|
193
|
+
"quartiles": [
|
|
194
|
+
round(float(series.quantile(0.25)), 4),
|
|
195
|
+
round(float(series.quantile(0.50)), 4),
|
|
196
|
+
round(float(series.quantile(0.75)), 4)
|
|
197
|
+
],
|
|
198
|
+
"skew": round(float(series.skew()), 4),
|
|
199
|
+
"kurtosis": round(float(series.kurtosis()), 4)
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
# Categorical
|
|
203
|
+
categorical_cols = df.select_dtypes(include=['object', 'category']).columns
|
|
204
|
+
if len(categorical_cols) > 0:
|
|
205
|
+
result["statistics"]["categorical_summary"] = {}
|
|
206
|
+
for col in categorical_cols:
|
|
207
|
+
value_counts = df[col].value_counts()
|
|
208
|
+
total = len(df[col].dropna())
|
|
209
|
+
result["statistics"]["categorical_summary"][col] = {
|
|
210
|
+
"unique_count": int(df[col].nunique()),
|
|
211
|
+
"top_values": [
|
|
212
|
+
{
|
|
213
|
+
"value": str(idx),
|
|
214
|
+
"count": int(cnt),
|
|
215
|
+
"percentage": round(cnt / total * 100, 2)
|
|
216
|
+
}
|
|
217
|
+
for idx, cnt in value_counts.head(5).items()
|
|
218
|
+
],
|
|
219
|
+
"mode": str(value_counts.index[0]) if len(value_counts) > 0 else None
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
# 5. Correlation analysis
|
|
223
|
+
if len(numeric_cols) > 1:
|
|
224
|
+
corr_matrix = df[numeric_cols].corr()
|
|
225
|
+
correlations = []
|
|
226
|
+
for i in range(len(numeric_cols)):
|
|
227
|
+
for j in range(i + 1, len(numeric_cols)):
|
|
228
|
+
r = corr_matrix.iloc[i, j]
|
|
229
|
+
if abs(r) > 0.5:
|
|
230
|
+
correlations.append({
|
|
231
|
+
"column1": numeric_cols[i],
|
|
232
|
+
"column2": numeric_cols[j],
|
|
233
|
+
"coefficient": round(float(r), 4),
|
|
234
|
+
"method": "pearson"
|
|
235
|
+
})
|
|
236
|
+
if correlations:
|
|
237
|
+
result["statistics"]["correlations"] = correlations
|
|
238
|
+
|
|
239
|
+
# 6. Save output (MUST sanitize NaN/Infinity before dumping)
|
|
240
|
+
base_name = Path(file_path).stem
|
|
241
|
+
output_file = tmp_paths["output"] / f"{base_name}_{analysis_type}.json"
|
|
242
|
+
|
|
243
|
+
def sanitize_for_json(obj):
|
|
244
|
+
"""Replace NaN/Infinity with None to produce valid JSON."""
|
|
245
|
+
if isinstance(obj, float) and (np.isnan(obj) or np.isinf(obj)):
|
|
246
|
+
return None
|
|
247
|
+
if isinstance(obj, dict):
|
|
248
|
+
return {k: sanitize_for_json(v) for k, v in obj.items()}
|
|
249
|
+
if isinstance(obj, list):
|
|
250
|
+
return [sanitize_for_json(v) for v in obj]
|
|
251
|
+
return obj
|
|
252
|
+
|
|
253
|
+
result = sanitize_for_json(result)
|
|
254
|
+
with open(output_file, 'w', encoding='utf-8') as f:
|
|
255
|
+
json.dump(result, f, indent=2, ensure_ascii=False, allow_nan=False)
|
|
256
|
+
print(f"Analysis result saved to: {output_file}")
|
|
257
|
+
|
|
258
|
+
return result
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def save_visualization(fig, viz_id: str, tmp_paths: dict, format: str = "png"):
|
|
262
|
+
"""Save matplotlib/seaborn figure to tmp/output/visualizations."""
|
|
263
|
+
output_path = tmp_paths["visualizations"] / f"{viz_id}.{format}"
|
|
264
|
+
fig.savefig(output_path, dpi=150, bbox_inches='tight')
|
|
265
|
+
print(f"Visualization saved to: {output_path}")
|
|
266
|
+
return output_path
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
# Usage:
|
|
270
|
+
# result = analyze_data("sales_2023.csv", "exploratory")
|
|
271
|
+
# print(json.dumps(result, indent=2, ensure_ascii=False))
|
|
272
|
+
```
|