mindstudio-probe 1.0.1__py3-none-any.whl → 1.0.3__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (249) hide show
  1. {mindstudio_probe-1.0.1.dist-info → mindstudio_probe-1.0.3.dist-info}/METADATA +5 -1
  2. mindstudio_probe-1.0.3.dist-info/RECORD +272 -0
  3. msprobe/README.md +78 -23
  4. msprobe/__init__.py +1 -0
  5. msprobe/config/README.md +182 -40
  6. msprobe/config/config.json +22 -0
  7. msprobe/core/__init__.py +0 -0
  8. msprobe/{pytorch → core}/advisor/advisor.py +3 -3
  9. msprobe/{pytorch → core}/advisor/advisor_result.py +2 -2
  10. msprobe/core/common/const.py +82 -5
  11. msprobe/core/common/exceptions.py +30 -18
  12. msprobe/core/common/file_check.py +19 -1
  13. msprobe/core/common/log.py +15 -1
  14. msprobe/core/common/utils.py +130 -30
  15. msprobe/core/common_config.py +32 -19
  16. msprobe/core/compare/acc_compare.py +299 -0
  17. msprobe/core/compare/check.py +95 -0
  18. msprobe/core/compare/compare_cli.py +49 -0
  19. msprobe/core/compare/highlight.py +222 -0
  20. msprobe/core/compare/multiprocessing_compute.py +149 -0
  21. msprobe/{pytorch → core}/compare/npy_compare.py +55 -4
  22. msprobe/core/compare/utils.py +429 -0
  23. msprobe/core/data_dump/data_collector.py +39 -35
  24. msprobe/core/data_dump/data_processor/base.py +85 -37
  25. msprobe/core/data_dump/data_processor/factory.py +5 -7
  26. msprobe/core/data_dump/data_processor/mindspore_processor.py +198 -0
  27. msprobe/core/data_dump/data_processor/pytorch_processor.py +94 -51
  28. msprobe/core/data_dump/json_writer.py +11 -11
  29. msprobe/core/grad_probe/__init__.py +0 -0
  30. msprobe/core/grad_probe/constant.py +71 -0
  31. msprobe/core/grad_probe/grad_compare.py +175 -0
  32. msprobe/core/grad_probe/utils.py +52 -0
  33. msprobe/doc/grad_probe/grad_probe.md +207 -0
  34. msprobe/doc/grad_probe/img/image-1.png +0 -0
  35. msprobe/doc/grad_probe/img/image-2.png +0 -0
  36. msprobe/doc/grad_probe/img/image-3.png +0 -0
  37. msprobe/doc/grad_probe/img/image-4.png +0 -0
  38. msprobe/doc/grad_probe/img/image.png +0 -0
  39. msprobe/mindspore/api_accuracy_checker/__init__.py +0 -0
  40. msprobe/mindspore/api_accuracy_checker/api_accuracy_checker.py +246 -0
  41. msprobe/mindspore/api_accuracy_checker/api_info.py +69 -0
  42. msprobe/mindspore/api_accuracy_checker/api_runner.py +152 -0
  43. msprobe/mindspore/api_accuracy_checker/base_compare_algorithm.py +197 -0
  44. msprobe/mindspore/api_accuracy_checker/compute_element.py +224 -0
  45. msprobe/mindspore/api_accuracy_checker/main.py +16 -0
  46. msprobe/mindspore/api_accuracy_checker/type_mapping.py +114 -0
  47. msprobe/mindspore/api_accuracy_checker/utils.py +63 -0
  48. msprobe/mindspore/cell_processor.py +34 -0
  49. msprobe/mindspore/common/const.py +87 -0
  50. msprobe/mindspore/common/log.py +38 -0
  51. msprobe/mindspore/common/utils.py +57 -0
  52. msprobe/mindspore/compare/distributed_compare.py +75 -0
  53. msprobe/mindspore/compare/ms_compare.py +117 -0
  54. msprobe/mindspore/compare/ms_graph_compare.py +317 -0
  55. msprobe/mindspore/compare/ms_to_pt_api.yaml +399 -0
  56. msprobe/mindspore/debugger/debugger_config.py +38 -15
  57. msprobe/mindspore/debugger/precision_debugger.py +79 -4
  58. msprobe/mindspore/doc/compare.md +58 -0
  59. msprobe/mindspore/doc/dump.md +158 -6
  60. msprobe/mindspore/dump/dump_tool_factory.py +19 -22
  61. msprobe/mindspore/dump/hook_cell/api_registry.py +104 -0
  62. msprobe/mindspore/dump/hook_cell/hook_cell.py +53 -0
  63. msprobe/mindspore/dump/hook_cell/support_wrap_ops.yaml +925 -0
  64. msprobe/mindspore/dump/hook_cell/wrap_functional.py +91 -0
  65. msprobe/mindspore/dump/hook_cell/wrap_tensor.py +63 -0
  66. msprobe/mindspore/dump/jit_dump.py +56 -0
  67. msprobe/mindspore/dump/kernel_kbyk_dump.py +65 -0
  68. msprobe/mindspore/free_benchmark/__init__.py +0 -0
  69. msprobe/mindspore/free_benchmark/api_pynative_self_check.py +116 -0
  70. msprobe/mindspore/free_benchmark/common/__init__.py +0 -0
  71. msprobe/mindspore/free_benchmark/common/config.py +12 -0
  72. msprobe/mindspore/free_benchmark/common/handler_params.py +17 -0
  73. msprobe/mindspore/free_benchmark/common/utils.py +71 -0
  74. msprobe/mindspore/free_benchmark/data/support_wrap_ops.yaml +842 -0
  75. msprobe/mindspore/free_benchmark/decorator/__init__.py +0 -0
  76. msprobe/mindspore/free_benchmark/decorator/dec_forward.py +42 -0
  77. msprobe/mindspore/free_benchmark/decorator/decorator_factory.py +107 -0
  78. msprobe/mindspore/free_benchmark/handler/__init__.py +0 -0
  79. msprobe/mindspore/free_benchmark/handler/base_handler.py +90 -0
  80. msprobe/mindspore/free_benchmark/handler/check_handler.py +41 -0
  81. msprobe/mindspore/free_benchmark/handler/fix_handler.py +36 -0
  82. msprobe/mindspore/free_benchmark/handler/handler_factory.py +21 -0
  83. msprobe/mindspore/free_benchmark/perturbation/add_noise.py +67 -0
  84. msprobe/mindspore/free_benchmark/perturbation/base_perturbation.py +21 -0
  85. msprobe/mindspore/free_benchmark/perturbation/bit_noise.py +63 -0
  86. msprobe/mindspore/free_benchmark/perturbation/improve_precision.py +34 -0
  87. msprobe/mindspore/free_benchmark/perturbation/no_change.py +12 -0
  88. msprobe/mindspore/free_benchmark/perturbation/perturbation_factory.py +27 -0
  89. msprobe/mindspore/free_benchmark/self_check_tool_factory.py +33 -0
  90. msprobe/mindspore/grad_probe/__init__.py +0 -0
  91. msprobe/mindspore/grad_probe/global_context.py +91 -0
  92. msprobe/mindspore/grad_probe/grad_analyzer.py +231 -0
  93. msprobe/mindspore/grad_probe/grad_monitor.py +27 -0
  94. msprobe/mindspore/grad_probe/grad_stat_csv.py +132 -0
  95. msprobe/mindspore/grad_probe/hook.py +92 -0
  96. msprobe/mindspore/grad_probe/utils.py +29 -0
  97. msprobe/mindspore/ms_config.py +63 -15
  98. msprobe/mindspore/overflow_check/overflow_check_tool_factory.py +17 -15
  99. msprobe/mindspore/runtime.py +4 -0
  100. msprobe/mindspore/service.py +354 -0
  101. msprobe/mindspore/task_handler_factory.py +7 -4
  102. msprobe/msprobe.py +66 -26
  103. msprobe/pytorch/__init__.py +1 -1
  104. msprobe/pytorch/api_accuracy_checker/common/config.py +21 -16
  105. msprobe/pytorch/api_accuracy_checker/common/utils.py +1 -60
  106. msprobe/pytorch/api_accuracy_checker/compare/algorithm.py +2 -5
  107. msprobe/pytorch/api_accuracy_checker/compare/api_precision_compare.py +46 -10
  108. msprobe/pytorch/api_accuracy_checker/compare/compare.py +84 -48
  109. msprobe/pytorch/api_accuracy_checker/compare/compare_utils.py +8 -12
  110. msprobe/pytorch/api_accuracy_checker/config.yaml +7 -1
  111. msprobe/pytorch/api_accuracy_checker/run_ut/data_generate.py +15 -11
  112. msprobe/pytorch/api_accuracy_checker/run_ut/multi_run_ut.py +11 -15
  113. msprobe/pytorch/api_accuracy_checker/run_ut/run_overflow_check.py +16 -9
  114. msprobe/pytorch/api_accuracy_checker/run_ut/run_ut.py +193 -105
  115. msprobe/pytorch/api_accuracy_checker/run_ut/run_ut_utils.py +68 -1
  116. msprobe/pytorch/api_accuracy_checker/tensor_transport_layer/__init__.py +0 -0
  117. msprobe/pytorch/api_accuracy_checker/tensor_transport_layer/attl.py +202 -0
  118. msprobe/pytorch/api_accuracy_checker/tensor_transport_layer/client.py +324 -0
  119. msprobe/pytorch/api_accuracy_checker/tensor_transport_layer/device_dispatch.py +204 -0
  120. msprobe/pytorch/api_accuracy_checker/tensor_transport_layer/server.py +218 -0
  121. msprobe/pytorch/api_accuracy_checker/tensor_transport_layer/ssl_config.py +10 -0
  122. msprobe/pytorch/bench_functions/__init__.py +15 -0
  123. msprobe/pytorch/bench_functions/apply_adam_w.py +28 -0
  124. msprobe/pytorch/bench_functions/confusion_transpose.py +19 -0
  125. msprobe/pytorch/bench_functions/fast_gelu.py +55 -0
  126. msprobe/pytorch/bench_functions/layer_norm_eval.py +6 -0
  127. msprobe/pytorch/bench_functions/linear.py +12 -0
  128. msprobe/pytorch/bench_functions/matmul_backward.py +48 -0
  129. msprobe/pytorch/bench_functions/npu_fusion_attention.py +421 -0
  130. msprobe/pytorch/bench_functions/rms_norm.py +15 -0
  131. msprobe/pytorch/bench_functions/rotary_mul.py +52 -0
  132. msprobe/pytorch/bench_functions/scaled_mask_softmax.py +26 -0
  133. msprobe/pytorch/bench_functions/swiglu.py +55 -0
  134. msprobe/pytorch/common/parse_json.py +3 -1
  135. msprobe/pytorch/common/utils.py +83 -7
  136. msprobe/pytorch/compare/distributed_compare.py +19 -64
  137. msprobe/pytorch/compare/match.py +3 -6
  138. msprobe/pytorch/compare/pt_compare.py +40 -0
  139. msprobe/pytorch/debugger/debugger_config.py +11 -2
  140. msprobe/pytorch/debugger/precision_debugger.py +34 -4
  141. msprobe/pytorch/doc/api_accuracy_checker.md +57 -13
  142. msprobe/pytorch/doc/api_accuracy_checker_online.md +187 -0
  143. msprobe/pytorch/doc/dump.md +73 -20
  144. msprobe/pytorch/doc/ptdbg_ascend_compare.md +75 -11
  145. msprobe/pytorch/doc/ptdbg_ascend_quickstart.md +3 -3
  146. msprobe/pytorch/doc/run_overflow_check.md +1 -1
  147. msprobe/pytorch/doc//321/206/320/247/320/260/321/206/320/260/320/227/321/206/320/255/320/226/321/205/342/225/226/320/265/321/205/320/225/342/225/226/321/205/320/254/342/225/221/321/206/320/251/320/277/321/211/320/272/320/234/321/210/320/277/320/221/321/205/320/242/320/234/321/206/320/220/320/267/321/210/320/223/342/225/234/321/205/320/257/342/225/221/321/207/342/225/221/342/224/220/321/206/320/232/320/265/321/205/320/241/320/232.md +151 -0
  148. msprobe/pytorch/free_benchmark/common/constant.py +3 -0
  149. msprobe/pytorch/free_benchmark/common/utils.py +4 -0
  150. msprobe/pytorch/free_benchmark/compare/grad_saver.py +22 -26
  151. msprobe/pytorch/free_benchmark/main.py +7 -4
  152. msprobe/pytorch/free_benchmark/perturbed_layers/npu/add_noise.py +1 -1
  153. msprobe/pytorch/free_benchmark/perturbed_layers/npu/bit_noise.py +1 -1
  154. msprobe/pytorch/free_benchmark/perturbed_layers/npu/change_value.py +1 -1
  155. msprobe/pytorch/free_benchmark/perturbed_layers/npu/improve_precision.py +3 -3
  156. msprobe/pytorch/free_benchmark/perturbed_layers/npu/no_change.py +1 -1
  157. msprobe/pytorch/free_benchmark/perturbed_layers/run_cpu.py +1 -1
  158. msprobe/pytorch/free_benchmark/result_handlers/base_handler.py +43 -29
  159. msprobe/pytorch/free_benchmark/result_handlers/handler_factory.py +0 -1
  160. msprobe/pytorch/function_factory.py +75 -0
  161. msprobe/pytorch/functional/dump_module.py +4 -4
  162. msprobe/pytorch/grad_probe/__init__.py +0 -0
  163. msprobe/pytorch/grad_probe/grad_monitor.py +90 -0
  164. msprobe/pytorch/grad_probe/grad_stat_csv.py +129 -0
  165. msprobe/pytorch/hook_module/hook_module.py +14 -3
  166. msprobe/pytorch/hook_module/support_wrap_ops.yaml +2 -1
  167. msprobe/pytorch/hook_module/utils.py +9 -9
  168. msprobe/pytorch/hook_module/wrap_aten.py +20 -10
  169. msprobe/pytorch/hook_module/wrap_distributed.py +10 -7
  170. msprobe/pytorch/hook_module/wrap_functional.py +4 -7
  171. msprobe/pytorch/hook_module/wrap_npu_custom.py +21 -10
  172. msprobe/pytorch/hook_module/wrap_tensor.py +5 -6
  173. msprobe/pytorch/hook_module/wrap_torch.py +5 -7
  174. msprobe/pytorch/hook_module/wrap_vf.py +6 -8
  175. msprobe/pytorch/module_processer.py +53 -13
  176. msprobe/pytorch/online_dispatch/compare.py +4 -4
  177. msprobe/pytorch/online_dispatch/dispatch.py +39 -41
  178. msprobe/pytorch/online_dispatch/dump_compare.py +17 -47
  179. msprobe/pytorch/online_dispatch/single_compare.py +5 -5
  180. msprobe/pytorch/online_dispatch/utils.py +2 -43
  181. msprobe/pytorch/parse_tool/lib/compare.py +31 -19
  182. msprobe/pytorch/parse_tool/lib/config.py +2 -1
  183. msprobe/pytorch/parse_tool/lib/parse_tool.py +4 -4
  184. msprobe/pytorch/parse_tool/lib/utils.py +34 -80
  185. msprobe/pytorch/parse_tool/lib/visualization.py +4 -3
  186. msprobe/pytorch/pt_config.py +100 -6
  187. msprobe/pytorch/service.py +104 -19
  188. mindstudio_probe-1.0.1.dist-info/RECORD +0 -228
  189. msprobe/mindspore/dump/api_kbk_dump.py +0 -55
  190. msprobe/pytorch/compare/acc_compare.py +0 -1024
  191. msprobe/pytorch/compare/highlight.py +0 -100
  192. msprobe/test/core_ut/common/test_utils.py +0 -345
  193. msprobe/test/core_ut/data_dump/test_data_collector.py +0 -47
  194. msprobe/test/core_ut/data_dump/test_json_writer.py +0 -183
  195. msprobe/test/core_ut/data_dump/test_scope.py +0 -151
  196. msprobe/test/core_ut/test_common_config.py +0 -152
  197. msprobe/test/core_ut/test_file_check.py +0 -218
  198. msprobe/test/core_ut/test_log.py +0 -109
  199. msprobe/test/mindspore_ut/test_api_kbk_dump.py +0 -51
  200. msprobe/test/mindspore_ut/test_debugger_config.py +0 -42
  201. msprobe/test/mindspore_ut/test_dump_tool_factory.py +0 -51
  202. msprobe/test/mindspore_ut/test_kernel_graph_dump.py +0 -66
  203. msprobe/test/mindspore_ut/test_kernel_graph_overflow_check.py +0 -63
  204. msprobe/test/mindspore_ut/test_ms_config.py +0 -69
  205. msprobe/test/mindspore_ut/test_overflow_check_tool_factory.py +0 -51
  206. msprobe/test/mindspore_ut/test_precision_debugger.py +0 -56
  207. msprobe/test/mindspore_ut/test_task_handler_factory.py +0 -58
  208. msprobe/test/pytorch_ut/advisor/test_advisor.py +0 -83
  209. msprobe/test/pytorch_ut/api_accuracy_checker/common/test_common_utils.py +0 -108
  210. msprobe/test/pytorch_ut/api_accuracy_checker/common/test_config.py +0 -39
  211. msprobe/test/pytorch_ut/api_accuracy_checker/compare/test_algorithm.py +0 -112
  212. msprobe/test/pytorch_ut/api_accuracy_checker/compare/test_api_precision_compare.py +0 -77
  213. msprobe/test/pytorch_ut/api_accuracy_checker/compare/test_compare.py +0 -125
  214. msprobe/test/pytorch_ut/api_accuracy_checker/compare/test_compare_column.py +0 -10
  215. msprobe/test/pytorch_ut/api_accuracy_checker/compare/test_compare_utils.py +0 -43
  216. msprobe/test/pytorch_ut/api_accuracy_checker/run_ut/dump.json +0 -179
  217. msprobe/test/pytorch_ut/api_accuracy_checker/run_ut/forward.json +0 -63
  218. msprobe/test/pytorch_ut/api_accuracy_checker/run_ut/test_data_generate.py +0 -99
  219. msprobe/test/pytorch_ut/api_accuracy_checker/run_ut/test_multi_run_ut.py +0 -115
  220. msprobe/test/pytorch_ut/api_accuracy_checker/run_ut/test_run_ut.py +0 -72
  221. msprobe/test/pytorch_ut/compare/test_acc_compare.py +0 -17
  222. msprobe/test/pytorch_ut/free_benchmark/perturbed_layers/test_perturbed_layser.py +0 -105
  223. msprobe/test/pytorch_ut/free_benchmark/result_handlers/test_result_handler.py +0 -121
  224. msprobe/test/pytorch_ut/free_benchmark/test_main.py +0 -101
  225. msprobe/test/pytorch_ut/functional/test_dump_module.py +0 -15
  226. msprobe/test/pytorch_ut/hook_module/test_api_registry.py +0 -130
  227. msprobe/test/pytorch_ut/hook_module/test_hook_module.py +0 -42
  228. msprobe/test/pytorch_ut/hook_module/test_wrap_aten.py +0 -65
  229. msprobe/test/pytorch_ut/hook_module/test_wrap_distributed.py +0 -35
  230. msprobe/test/pytorch_ut/hook_module/test_wrap_functional.py +0 -20
  231. msprobe/test/pytorch_ut/hook_module/test_wrap_tensor.py +0 -35
  232. msprobe/test/pytorch_ut/hook_module/test_wrap_torch.py +0 -43
  233. msprobe/test/pytorch_ut/hook_module/test_wrap_vf.py +0 -11
  234. msprobe/test/pytorch_ut/test_pt_config.py +0 -69
  235. msprobe/test/pytorch_ut/test_service.py +0 -59
  236. msprobe/test/resources/advisor.txt +0 -3
  237. msprobe/test/resources/compare_result_20230703104808.csv +0 -9
  238. msprobe/test/resources/compare_result_without_accuracy.csv +0 -9
  239. msprobe/test/resources/config.yaml +0 -3
  240. msprobe/test/resources/npu_test.pkl +0 -8
  241. msprobe/test/run_test.sh +0 -30
  242. msprobe/test/run_ut.py +0 -58
  243. msprobe/test/test_module_processer.py +0 -64
  244. {mindstudio_probe-1.0.1.dist-info → mindstudio_probe-1.0.3.dist-info}/LICENSE +0 -0
  245. {mindstudio_probe-1.0.1.dist-info → mindstudio_probe-1.0.3.dist-info}/WHEEL +0 -0
  246. {mindstudio_probe-1.0.1.dist-info → mindstudio_probe-1.0.3.dist-info}/entry_points.txt +0 -0
  247. {mindstudio_probe-1.0.1.dist-info → mindstudio_probe-1.0.3.dist-info}/top_level.txt +0 -0
  248. /msprobe/{pytorch → core}/advisor/advisor_const.py +0 -0
  249. /msprobe/pytorch/doc/{atat → msprobe}/321/207/342/226/223/342/225/233/321/205/342/225/221/320/266/321/205/342/225/226/320/265/321/205/320/225/342/225/226/321/206/320/245/342/226/221/321/206/320/235/320/276dump/321/206/320/260/320/227/321/205/320/227/320/226/321/206/320/220/320/267/321/210/320/223/342/225/234/321/205/320/257/342/225/221/321/207/342/225/221/342/224/220/321/206/320/232/320/265/321/205/320/241/320/232.md" +0 -0
@@ -0,0 +1,317 @@
1
+ import csv
2
+ import glob
3
+ import os
4
+ import sys
5
+ import copy
6
+
7
+ import numpy as np
8
+ import pandas as pd
9
+ from msprobe.core.common.const import CompareConst, GraphMode
10
+ from msprobe.core.common.exceptions import FileCheckException
11
+ from msprobe.core.common.file_check import create_directory
12
+ from msprobe.core.common.log import logger
13
+ from msprobe.core.common.utils import add_time_with_xlsx, CompareException
14
+ from msprobe.core.compare.multiprocessing_compute import _ms_graph_handle_multi_process, check_accuracy
15
+ from msprobe.core.compare.npy_compare import npy_data_check, statistics_data_check, reshape_value, compare_ops_apply
16
+ from msprobe.core.common.file_check import FileOpen
17
+
18
+ class row_data:
19
+ def __init__(self, mode):
20
+ self.basic_data = copy.deepcopy(CompareConst.MS_GRAPH_BASE)
21
+ self.npy_data = copy.deepcopy(CompareConst.MS_GRAPH_NPY)
22
+ self.statistic_data = copy.deepcopy(CompareConst.MS_GRAPH_STATISTIC)
23
+ if mode == GraphMode.NPY_MODE:
24
+ self.data = {**self.basic_data, **self.npy_data}
25
+ else:
26
+ self.data = {**self.basic_data, **self.statistic_data}
27
+
28
+ def __call__(self):
29
+ return self.data
30
+
31
+
32
+ def generate_step(npu_path, rank_id):
33
+ step_set = set()
34
+ rank_path = os.path.join(npu_path, f"rank_{rank_id}")
35
+ if not os.path.exists(rank_path):
36
+ return []
37
+ for path in os.listdir(rank_path):
38
+ if path not in ["execution_order", "graphs"]:
39
+ data_path = os.path.join(rank_path, path)
40
+ for graph_path in os.listdir(data_path):
41
+ step_set.update([int(i) for i in os.listdir(os.path.join(data_path, graph_path))])
42
+ return sorted(step_set)
43
+
44
+
45
+ def generate_path_by_rank_step(base_path, rank_id, step_id):
46
+ path_with_rank_id = os.path.join(base_path, f"rank_{rank_id}")
47
+ if not os.path.exists(path_with_rank_id):
48
+ return ''
49
+ for path in os.listdir(path_with_rank_id):
50
+ if path not in ["execution_order", "graphs"]:
51
+
52
+ return os.path.join(path_with_rank_id, path, "*", str(step_id))
53
+ logger.error(f"Data_path {path_with_rank_id} is not exist.")
54
+ return ''
55
+
56
+
57
+ def statistic_data_read(statistic_file_list, statistic_file_path):
58
+ data_list = []
59
+ statistic_data_list = []
60
+ for statistic_file in statistic_file_list:
61
+ with open(statistic_file, "r") as f:
62
+ csv_reader = csv.reader(f, delimiter=",")
63
+ header = next(csv_reader)
64
+ header_index = {'Data Type': None, 'Shape': None, 'Max Value': None, 'Min Value': None,
65
+ 'Avg Value': None, 'L2Norm Value': None}
66
+ for key in header_index.keys():
67
+ for index, value in enumerate(header):
68
+ if key == value:
69
+ header_index[key] = index
70
+ for key in header_index.keys():
71
+ if header_index[key] is None:
72
+ logger.error(f"Data_path {statistic_file_path} has no key {key}")
73
+ raise FileCheckException(f"Data_path {statistic_file_path} has no key {key}")
74
+ statistic_data_list.extend([row for row in csv_reader])
75
+
76
+ for data in statistic_data_list:
77
+ compare_key = f"{data[1]}.{data[2]}.{data[3]}.{data[5]}"
78
+ timestamp = int(data[4])
79
+ data_list.append(
80
+ [statistic_file_path, compare_key, timestamp, data[header_index['Data Type']],
81
+ data[header_index['Shape']], data[header_index['Max Value']], data[header_index['Min Value']],
82
+ data[header_index['Avg Value']], data[header_index['L2Norm Value']]])
83
+ return data_list
84
+
85
+
86
+ def generate_data_name(data_path):
87
+ data_list = []
88
+
89
+ mapping_path = os.path.join(data_path, "mapping.csv")
90
+ statistic_path = os.path.join(data_path, "statistic.csv")
91
+ npy_path = os.path.join(data_path, "*.npy")
92
+
93
+ mapping_file_list = glob.glob(mapping_path)
94
+ statistic_file_list = glob.glob(statistic_path)
95
+ npy_file_list = glob.glob(npy_path)
96
+
97
+ mapping_exist = bool(mapping_file_list)
98
+ statistic_exist = bool(statistic_file_list)
99
+ npy_exist = bool(npy_file_list)
100
+
101
+ mapping_dict = []
102
+ if mapping_exist:
103
+ for mapping_file in mapping_file_list:
104
+ with FileOpen(mapping_file, "r") as f:
105
+ csv_reader = csv.reader(f, delimiter=",")
106
+ header = next(csv_reader)
107
+ for row in csv_reader:
108
+ mapping_dict[row[0]] = row[1]
109
+
110
+ if npy_exist:
111
+ for data in npy_file_list:
112
+ if data in mapping_dict:
113
+ split_list = mapping_dict[data].split(".")
114
+ else:
115
+ split_list = data.split(".")
116
+ compare_key = f"{split_list[1]}.{split_list[2]}.{split_list[3]}.{split_list[5]}.{split_list[6]}"
117
+ timestamp = int(split_list[4])
118
+
119
+ data_list.append([os.path.join(data_path, data), compare_key, timestamp])
120
+ elif statistic_exist:
121
+ data_list = statistic_data_read(statistic_file_list, os.path.join(data_path, statistic_path))
122
+
123
+ if npy_exist:
124
+ mode = GraphMode.NPY_MODE
125
+ elif statistic_exist:
126
+ mode = GraphMode.STATISTIC_MODE
127
+ else:
128
+ mode = GraphMode.ERROR_MODE
129
+ logger.error(f"Error mode.")
130
+ return mode, data_list
131
+
132
+
133
+ def read_npy_data(data_path):
134
+ try:
135
+ data_value = np.load(data_path)
136
+ if data_value.dtype == np.float16:
137
+ data_value = data_value.astype(np.float32)
138
+ except FileNotFoundError as e:
139
+ data_value = None
140
+ except EOFError:
141
+ data_value = None
142
+ return data_value
143
+
144
+
145
+ class GraphMSComparator:
146
+ def __init__(self, input_param, output_path):
147
+ self.output_path = output_path
148
+ self.base_npu_path = input_param.get('npu_path', None)
149
+ self.base_bench_path = input_param.get('bench_path', None)
150
+ self.rank_list = input_param.get('rank_id', [])
151
+ self.step_list = input_param.get('step_id', [])
152
+
153
+ @staticmethod
154
+ def compare_ops(compare_result_db, mode):
155
+
156
+ def npy_mode_compute(row):
157
+ result_dict = row_data(GraphMode.NPY_MODE)()
158
+
159
+ def process_npy_file(file_path, name_prefix, result):
160
+ if os.path.exists(file_path):
161
+ data = read_npy_data(file_path)
162
+ result[f'{name_prefix} Name'] = file_path
163
+ result[f'{name_prefix} Dtype'] = data.dtype
164
+ result[f'{name_prefix} Tensor Shape'] = data.shape
165
+ result[f'{name_prefix} max'] = np.max(data)
166
+ result[f'{name_prefix} min'] = np.min(data)
167
+ result[f'{name_prefix} mean'] = np.mean(data)
168
+ result[f'{name_prefix} l2norm'] = np.linalg.norm(data)
169
+ return data
170
+ return ""
171
+
172
+ n_value = process_npy_file(row[CompareConst.NPU_NAME], 'NPU', result_dict)
173
+ b_value = process_npy_file(row[CompareConst.BENCH_NAME], 'Bench', result_dict)
174
+
175
+ error_flag, error_message = npy_data_check(n_value, b_value)
176
+ result_dict[CompareConst.ERROR_MESSAGE] = error_message
177
+
178
+ if not error_flag:
179
+ n_value, b_value = reshape_value(n_value, b_value)
180
+ result_list, err_msg = compare_ops_apply(n_value, b_value, False, "")
181
+ result_dict[CompareConst.COSINE] = result_list[0]
182
+ result_dict[CompareConst.MAX_ABS_ERR] = result_list[1]
183
+ result_dict[CompareConst.MAX_RELATIVE_ERR] = result_list[2]
184
+ result_dict[CompareConst.ONE_THOUSANDTH_ERR_RATIO] = result_list[3]
185
+ result_dict[CompareConst.FIVE_THOUSANDTHS_ERR_RATIO] = result_list[4]
186
+ result_dict[CompareConst.ACCURACY] = check_accuracy(result_list[0], result_list[1])
187
+ result_dict[CompareConst.ERROR_MESSAGE] = err_msg
188
+
189
+ return pd.Series(result_dict)
190
+
191
+ def statistic_mode_compute(row):
192
+ result_dict = row_data('STATISTIC')()
193
+
194
+ def update_result_dict(result, rows, prefix):
195
+ result[f'{prefix} Name'] = rows[f'{prefix} Name']
196
+ result[f'{prefix} Dtype'] = rows[f'{prefix} Dtype']
197
+ result[f'{prefix} Tensor Shape'] = rows[f'{prefix} Tensor Shape']
198
+ result[f'{prefix} max'] = np.float32(rows[f'{prefix} max'])
199
+ result[f'{prefix} min'] = np.float32(rows[f'{prefix} min'])
200
+ result[f'{prefix} mean'] = np.float32(rows[f'{prefix} mean'])
201
+ result[f'{prefix} l2norm'] = np.float32(rows[f'{prefix} l2norm'])
202
+
203
+ # 使用示例
204
+ update_result_dict(result_dict, row, 'NPU')
205
+ update_result_dict(result_dict, row, 'Bench')
206
+ error_flag, error_message = statistics_data_check(result_dict)
207
+ result_dict[CompareConst.ERROR_MESSAGE] += error_message
208
+ if not error_flag:
209
+ result_dict[CompareConst.MAX_DIFF] = np.abs(
210
+ result_dict[CompareConst.NPU_MAX] - result_dict[CompareConst.BENCH_MAX])
211
+ result_dict[CompareConst.MIN_DIFF] = np.abs(
212
+ result_dict[CompareConst.NPU_MIN] - result_dict[CompareConst.BENCH_MIN])
213
+ result_dict[CompareConst.MEAN_DIFF] = np.abs(
214
+ result_dict[CompareConst.NPU_MEAN] - result_dict[CompareConst.BENCH_MEAN])
215
+ result_dict[CompareConst.NORM_DIFF] = np.abs(
216
+ result_dict[CompareConst.NPU_NORM] - result_dict[CompareConst.BENCH_NORM])
217
+ result_dict[CompareConst.MAX_RELATIVE_ERR] = result_dict[CompareConst.MAX_DIFF] / result_dict[
218
+ CompareConst.BENCH_MAX] if result_dict[CompareConst.BENCH_MAX] > 0 else 0
219
+ result_dict[CompareConst.MAX_RELATIVE_ERR] = str(result_dict[CompareConst.MAX_RELATIVE_ERR] * 100) + "%"
220
+ result_dict[CompareConst.MIN_RELATIVE_ERR] = result_dict[CompareConst.MIN_DIFF] / result_dict[
221
+ CompareConst.BENCH_MIN] if result_dict[CompareConst.BENCH_MIN] > 0 else 0
222
+ result_dict[CompareConst.MIN_RELATIVE_ERR] = str(result_dict[CompareConst.MIN_RELATIVE_ERR] * 100) + "%"
223
+ result_dict[CompareConst.MEAN_RELATIVE_ERR] = result_dict[CompareConst.MEAN_DIFF] / result_dict[
224
+ CompareConst.BENCH_MEAN] if result_dict[CompareConst.BENCH_MEAN] > 0 else 0
225
+ result_dict[CompareConst.MEAN_RELATIVE_ERR] = str(
226
+ result_dict[CompareConst.MEAN_RELATIVE_ERR] * 100) + "%"
227
+ result_dict[CompareConst.NORM_RELATIVE_ERR] = result_dict[CompareConst.NORM_DIFF] / result_dict[
228
+ CompareConst.BENCH_NORM] if result_dict[CompareConst.BENCH_NORM] > 0 else 0
229
+ result_dict[CompareConst.NORM_RELATIVE_ERR] = str(
230
+ result_dict[CompareConst.NORM_RELATIVE_ERR] * 100) + "%"
231
+ magnitude_diff = result_dict[CompareConst.MAX_DIFF] / (
232
+ max(result_dict[CompareConst.NPU_MAX], result_dict[CompareConst.BENCH_MAX]) + 1e-10)
233
+ if magnitude_diff > CompareConst.MAGNITUDE:
234
+ result_dict[CompareConst.ACCURACY] = 'No'
235
+ else:
236
+ result_dict[CompareConst.ACCURACY] = 'Yes'
237
+
238
+ return pd.Series(result_dict)
239
+
240
+ if mode == GraphMode.NPY_MODE:
241
+ compare_result_db = compare_result_db.apply(npy_mode_compute, axis=1)
242
+ else:
243
+ compare_result_db = compare_result_db.apply(statistic_mode_compute, axis=1)
244
+ return compare_result_db
245
+
246
+ def compare_core(self):
247
+ logger.info("Please check whether the input data belongs to you. If not, there may be security risks.")
248
+
249
+ # split by rank and step
250
+ if not self.rank_list:
251
+ self.rank_list = [int(i.split("_")[-1]) for i in os.listdir(self.base_npu_path)]
252
+ for rank_id in self.rank_list:
253
+ if not self.step_list:
254
+ self.step_list = generate_step(self.base_npu_path, rank_id)
255
+ for step_id in self.step_list:
256
+ compare_result_df, mode = self.compare_process(rank_id, step_id)
257
+ if isinstance(compare_result_df, list):
258
+ is_empty = not compare_result_df
259
+ elif isinstance(compare_result_df, pd.DataFrame):
260
+ is_empty = compare_result_df.empty
261
+ else:
262
+ is_empty = True
263
+ if is_empty or not mode:
264
+ continue
265
+ compare_result_df = self._do_multi_process(compare_result_df, mode)
266
+ compare_result_name = add_time_with_xlsx(f"compare_result_{str(rank_id)}_{str(step_id)}")
267
+ compare_result_path = os.path.join(os.path.realpath(self.output_path), f"{compare_result_name}")
268
+ compare_result_df.to_excel(compare_result_path, index=False)
269
+ logger.info(f"Compare rank: {rank_id} step: {step_id} finish. Compare result: {compare_result_path}.")
270
+
271
+ def compare_process(self, rank_id, step_id):
272
+ # generate data_path
273
+ npu_data_path = generate_path_by_rank_step(self.base_npu_path, rank_id, step_id)
274
+ bench_data_path = generate_path_by_rank_step(self.base_bench_path, rank_id, step_id)
275
+ if not npu_data_path or not bench_data_path:
276
+ return [], ''
277
+
278
+ # generate file name
279
+ npu_mode, npu_data_list = generate_data_name(npu_data_path)
280
+ match_mode, match_data_list = generate_data_name(bench_data_path)
281
+
282
+ if npu_mode == "ERROR_MODE" or match_mode == "ERROR_MODE":
283
+ logger.warning(f"Data_path {npu_data_path} or {bench_data_path} is not exist.")
284
+ return [], ''
285
+ if npu_mode != match_mode:
286
+ logger.error(f"NPU mode {npu_mode} not equal to MATCH mode {match_mode}.")
287
+ return [], ''
288
+
289
+ if npu_mode == 'NPY_MODE':
290
+ npu_data_df = pd.DataFrame(npu_data_list, columns=[CompareConst.NPU_NAME, 'Compare Key', 'TimeStamp'])
291
+ bench_data_df = pd.DataFrame(match_data_list, columns=[CompareConst.BENCH_NAME, 'Compare Key', 'TimeStamp'])
292
+ else:
293
+ npu_data_df = pd.DataFrame(npu_data_list,
294
+ columns=[CompareConst.NPU_NAME, 'Compare Key', 'TimeStamp', CompareConst.NPU_DTYPE, CompareConst.NPU_SHAPE,
295
+ CompareConst.NPU_MAX, CompareConst.NPU_MIN, CompareConst.NPU_MEAN, CompareConst.NPU_NORM])
296
+ bench_data_df = pd.DataFrame(match_data_list,
297
+ columns=[CompareConst.BENCH_NAME, 'Compare Key', 'TimeStamp', CompareConst.BENCH_DTYPE,
298
+ CompareConst.BENCH_SHAPE, CompareConst.BENCH_MAX, CompareConst.BENCH_MIN, CompareConst.BENCH_MEAN,
299
+ CompareConst.BENCH_NORM])
300
+
301
+ npu_data_df['Local Index'] = npu_data_df.sort_values('TimeStamp').groupby('Compare Key').cumcount()
302
+ bench_data_df['Local Index'] = bench_data_df.sort_values('TimeStamp').groupby('Compare Key').cumcount()
303
+
304
+ compare_result_df = pd.merge(npu_data_df, bench_data_df, on=['Compare Key', 'Local Index'], how='outer')
305
+
306
+ compare_result_df[CompareConst.NPU_NAME] = compare_result_df[CompareConst.NPU_NAME].fillna('')
307
+ compare_result_df[CompareConst.BENCH_NAME] = compare_result_df[CompareConst.BENCH_NAME].fillna('')
308
+
309
+ return compare_result_df, npu_mode
310
+
311
+ def _do_multi_process(self, result_df, mode):
312
+ try:
313
+ result_df = _ms_graph_handle_multi_process(self.compare_ops, result_df, mode)
314
+ except ValueError as e:
315
+ logger.error('result dataframe is not found.')
316
+ raise CompareException(CompareException.INVALID_DATA_ERROR) from e
317
+ return result_df