clearai-dsh 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (154) hide show
  1. package/CHANGELOG.md +26 -0
  2. package/LICENSE +201 -0
  3. package/README.md +138 -0
  4. package/README.zh-CN.md +138 -0
  5. package/bin/clearai.mjs +224 -0
  6. package/brand/README.md +41 -0
  7. package/brand/logo-512-dark.png +0 -0
  8. package/brand/logo-512.png +0 -0
  9. package/brand/logo-lockup-dark.png +0 -0
  10. package/brand/logo-lockup.png +0 -0
  11. package/brand/logo-lockup.svg +12 -0
  12. package/brand/logo-wordmark.svg +6 -0
  13. package/brand/logo.svg +19 -0
  14. package/cordis.patch.yml +39 -0
  15. package/lib/client.js +3071 -0
  16. package/lib/fold.js +1576 -0
  17. package/lib/host.js +605 -0
  18. package/package.json +65 -0
  19. package/presets/clearai/agent.cordis.yml +226 -0
  20. package/presets/clearai/plugins/brain.js +547 -0
  21. package/presets/clearai/plugins/clearai-kernel.js +5485 -0
  22. package/presets/clearai/plugins/ontology.js +306 -0
  23. package/presets/clearai/plugins/prompts.js +312 -0
  24. package/presets/clearai/preset.yml +5 -0
  25. package/presets/clearai/skills/clearai-loop/SKILL.md +89 -0
  26. package/presets/clearai/template/knowledge/README.md +25 -0
  27. package/presets/clearai/template/memory/README.md +34 -0
  28. package/presets/clearai/template/project.md +49 -0
  29. package/presets/clearai/template/skills/README.md +37 -0
  30. package/presets/clearai/template/skills/chart-diagram-qa/SKILL.md +43 -0
  31. package/presets/clearai/template/skills/citation-management/SKILL.md +73 -0
  32. package/presets/clearai/template/skills/citation-management/references/bibtex_formatting.md +908 -0
  33. package/presets/clearai/template/skills/citation-management/references/citation_validation.md +794 -0
  34. package/presets/clearai/template/skills/citation-management/references/google_scholar_search.md +725 -0
  35. package/presets/clearai/template/skills/citation-management/references/metadata_extraction.md +870 -0
  36. package/presets/clearai/template/skills/citation-management/references/pubmed_search.md +839 -0
  37. package/presets/clearai/template/skills/citation-management/scripts/doi_to_bibtex.py +204 -0
  38. package/presets/clearai/template/skills/citation-management/scripts/extract_metadata.py +569 -0
  39. package/presets/clearai/template/skills/citation-management/scripts/format_bibtex.py +349 -0
  40. package/presets/clearai/template/skills/citation-management/scripts/generate_schematic.py +139 -0
  41. package/presets/clearai/template/skills/citation-management/scripts/generate_schematic_ai.py +817 -0
  42. package/presets/clearai/template/skills/citation-management/scripts/search_google_scholar.py +282 -0
  43. package/presets/clearai/template/skills/citation-management/scripts/search_pubmed.py +398 -0
  44. package/presets/clearai/template/skills/citation-management/scripts/validate_citations.py +497 -0
  45. package/presets/clearai/template/skills/data-analysis/SKILL.md +92 -0
  46. package/presets/clearai/template/skills/data-analysis/checklists/readiness_check.md +23 -0
  47. package/presets/clearai/template/skills/data-analysis/templates/analysis_report.md.tpl +63 -0
  48. package/presets/clearai/template/skills/data-analysis/templates/cleaning_rules_draft.yaml.tpl +32 -0
  49. package/presets/clearai/template/skills/data-analysis/templates/data_dictionary.md.tpl +12 -0
  50. package/presets/clearai/template/skills/data-analysis/templates/domain_knowledge_template.md.tpl +316 -0
  51. package/presets/clearai/template/skills/data-analysis/templates/feature_candidates.json.tpl +20 -0
  52. package/presets/clearai/template/skills/data-analysis/templates/quality_scorecard.md.tpl +30 -0
  53. package/presets/clearai/template/skills/data-analysis/workflows/01-data-profiling.md +42 -0
  54. package/presets/clearai/template/skills/data-analysis/workflows/02-quality-audit.md +36 -0
  55. package/presets/clearai/template/skills/data-analysis/workflows/03-physical-correlation.md +25 -0
  56. package/presets/clearai/template/skills/data-analysis/workflows/04-unstructured-mining.md +26 -0
  57. package/presets/clearai/template/skills/data-qa-analysis/SKILL.md +102 -0
  58. package/presets/clearai/template/skills/data-qa-analysis/checklists/readiness_check.md +62 -0
  59. package/presets/clearai/template/skills/data-qa-analysis/templates/best_in_class_report.md.tpl +56 -0
  60. package/presets/clearai/template/skills/data-qa-analysis/templates/cleaning_rules_draft.yaml.tpl +56 -0
  61. package/presets/clearai/template/skills/data-qa-analysis/templates/data_dictionary.md.tpl +13 -0
  62. package/presets/clearai/template/skills/data-qa-analysis/templates/data_source_inventory_and_lineage.md.tpl +146 -0
  63. package/presets/clearai/template/skills/data-qa-analysis/templates/data_status_report.md.tpl +60 -0
  64. package/presets/clearai/template/skills/data-qa-analysis/templates/steady_state_rules.yaml.tpl +41 -0
  65. package/presets/clearai/template/skills/data-qa-analysis/templates/subsystem_registry.md.tpl +101 -0
  66. package/presets/clearai/template/skills/data-qa-analysis/templates/unified_execution_plan.md.tpl +100 -0
  67. package/presets/clearai/template/skills/data-qa-analysis/workflows/01-data-source-inventory-and-lineage.md +194 -0
  68. package/presets/clearai/template/skills/data-qa-analysis/workflows/02-data-alignment-and-tag-semantics.md +122 -0
  69. package/presets/clearai/template/skills/data-qa-analysis/workflows/03-steady-state-identification.md +126 -0
  70. package/presets/clearai/template/skills/data-qa-analysis/workflows/04-consumption-analysis.md +152 -0
  71. package/presets/clearai/template/skills/data-qa-analysis/workflows/05-best-in-class-and-optimization-space.md +78 -0
  72. package/presets/clearai/template/skills/domain-presearch/SKILL.md +131 -0
  73. package/presets/clearai/template/skills/domain-presearch/checklists/domain_checklist.md +24 -0
  74. package/presets/clearai/template/skills/domain-presearch/references/figure_code.md +78 -0
  75. package/presets/clearai/template/skills/domain-presearch/references/strategic_frameworks.md +38 -0
  76. package/presets/clearai/template/skills/exploration-loop/SKILL.md +81 -0
  77. package/presets/clearai/template/skills/exploratory-data-analysis/SKILL.md +77 -0
  78. package/presets/clearai/template/skills/exploratory-data-analysis/references/bioinformatics_genomics_formats.md +664 -0
  79. package/presets/clearai/template/skills/exploratory-data-analysis/references/chemistry_molecular_formats.md +664 -0
  80. package/presets/clearai/template/skills/exploratory-data-analysis/references/general_scientific_formats.md +518 -0
  81. package/presets/clearai/template/skills/exploratory-data-analysis/references/microscopy_imaging_formats.md +620 -0
  82. package/presets/clearai/template/skills/exploratory-data-analysis/references/proteomics_metabolomics_formats.md +517 -0
  83. package/presets/clearai/template/skills/exploratory-data-analysis/references/spectroscopy_analytical_formats.md +633 -0
  84. package/presets/clearai/template/skills/exploratory-data-analysis/scripts/eda_analyzer.py +547 -0
  85. package/presets/clearai/template/skills/hypothesis-generation/SKILL.md +73 -0
  86. package/presets/clearai/template/skills/hypothesis-generation/references/experimental_design_patterns.md +329 -0
  87. package/presets/clearai/template/skills/hypothesis-generation/references/hypothesis_quality_criteria.md +198 -0
  88. package/presets/clearai/template/skills/hypothesis-generation/references/literature_search_strategies.md +622 -0
  89. package/presets/clearai/template/skills/hypothesis-generation/scripts/generate_schematic.py +139 -0
  90. package/presets/clearai/template/skills/hypothesis-generation/scripts/generate_schematic_ai.py +817 -0
  91. package/presets/clearai/template/skills/literature-review/SKILL.md +72 -0
  92. package/presets/clearai/template/skills/literature-review/references/citation_styles.md +166 -0
  93. package/presets/clearai/template/skills/literature-review/references/database_strategies.md +455 -0
  94. package/presets/clearai/template/skills/literature-review/scripts/generate_pdf.py +176 -0
  95. package/presets/clearai/template/skills/literature-review/scripts/generate_schematic.py +139 -0
  96. package/presets/clearai/template/skills/literature-review/scripts/generate_schematic_ai.py +817 -0
  97. package/presets/clearai/template/skills/literature-review/scripts/search_databases.py +303 -0
  98. package/presets/clearai/template/skills/literature-review/scripts/verify_citations.py +221 -0
  99. package/presets/clearai/template/skills/paper-lookup/SKILL.md +59 -0
  100. package/presets/clearai/template/skills/paper-lookup/references/arxiv.md +161 -0
  101. package/presets/clearai/template/skills/paper-lookup/references/biorxiv.md +118 -0
  102. package/presets/clearai/template/skills/paper-lookup/references/core.md +150 -0
  103. package/presets/clearai/template/skills/paper-lookup/references/crossref.md +181 -0
  104. package/presets/clearai/template/skills/paper-lookup/references/medrxiv.md +104 -0
  105. package/presets/clearai/template/skills/paper-lookup/references/openalex.md +174 -0
  106. package/presets/clearai/template/skills/paper-lookup/references/pmc.md +152 -0
  107. package/presets/clearai/template/skills/paper-lookup/references/pubmed.md +124 -0
  108. package/presets/clearai/template/skills/paper-lookup/references/semantic-scholar.md +203 -0
  109. package/presets/clearai/template/skills/paper-lookup/references/unpaywall.md +127 -0
  110. package/presets/clearai/template/skills/process-presearch/SKILL.md +196 -0
  111. package/presets/clearai/template/skills/process-presearch/checklists/process_checklist.md +18 -0
  112. package/presets/clearai/template/skills/process-presearch/references/figure_code.md +107 -0
  113. package/presets/clearai/template/skills/process-presearch/references/source_attribution_example.md +22 -0
  114. package/presets/clearai/template/skills/process-understanding-extraction/SKILL.md +69 -0
  115. package/presets/clearai/template/skills/process-understanding-extraction/checklists/readiness_check.md +34 -0
  116. package/presets/clearai/template/skills/process-understanding-extraction/templates/docx_raw_dump_extractor.py.tpl +132 -0
  117. package/presets/clearai/template/skills/process-understanding-extraction/templates/entity_map_unit_topology.json.tpl +86 -0
  118. package/presets/clearai/template/skills/process-understanding-extraction/templates/process_brief.md.tpl +89 -0
  119. package/presets/clearai/template/skills/process-understanding-extraction/templates/process_brief_builder_from_raw_dump.py.tpl +203 -0
  120. package/presets/clearai/template/skills/process-understanding-extraction/templates/process_flow_mermaid.md.tpl +41 -0
  121. package/presets/clearai/template/skills/process-understanding-extraction/templates/unified_execution_plan.md.tpl +53 -0
  122. package/presets/clearai/template/skills/process-understanding-extraction/workflows/01-process-doc-discovery.md +173 -0
  123. package/presets/clearai/template/skills/process-understanding-extraction/workflows/02-process-understanding-and-diagramming.md +106 -0
  124. package/presets/clearai/template/skills/scientific-brainstorming/SKILL.md +64 -0
  125. package/presets/clearai/template/skills/scientific-brainstorming/references/brainstorming_methods.md +326 -0
  126. package/presets/clearai/template/skills/scientific-critical-thinking/SKILL.md +72 -0
  127. package/presets/clearai/template/skills/scientific-critical-thinking/references/common_biases.md +364 -0
  128. package/presets/clearai/template/skills/scientific-critical-thinking/references/evidence_hierarchy.md +485 -0
  129. package/presets/clearai/template/skills/scientific-critical-thinking/references/experimental_design.md +496 -0
  130. package/presets/clearai/template/skills/scientific-critical-thinking/references/logical_fallacies.md +478 -0
  131. package/presets/clearai/template/skills/scientific-critical-thinking/references/scientific_method.md +169 -0
  132. package/presets/clearai/template/skills/scientific-critical-thinking/references/statistical_pitfalls.md +506 -0
  133. package/presets/clearai/template/skills/skill-creator/SKILL.md +109 -0
  134. package/presets/clearai/template/skills/skill-creator/references/authoring-guide.md +89 -0
  135. package/presets/clearai/template/skills/statistical-analysis/SKILL.md +79 -0
  136. package/presets/clearai/template/skills/statistical-analysis/references/assumptions_and_diagnostics.md +369 -0
  137. package/presets/clearai/template/skills/statistical-analysis/references/bayesian_statistics.md +653 -0
  138. package/presets/clearai/template/skills/statistical-analysis/references/effect_sizes_and_power.md +578 -0
  139. package/presets/clearai/template/skills/statistical-analysis/references/reporting_standards.md +469 -0
  140. package/presets/clearai/template/skills/statistical-analysis/references/test_selection_guide.md +129 -0
  141. package/presets/clearai/template/skills/statistical-analysis/scripts/assumption_checks.py +538 -0
  142. package/presets/clearai/template/skills/web-artifact/SKILL.md +165 -0
  143. package/presets/clearai/template/skills/web-artifact/assets/renderer/renderer.css +229 -0
  144. package/presets/clearai/template/skills/web-artifact/assets/renderer/renderer.js +373 -0
  145. package/presets/clearai/template/skills/web-artifact/assets/vendor/elkjs/LICENSE +263 -0
  146. package/presets/clearai/template/skills/web-artifact/assets/vendor/elkjs/UPSTREAM.md +26 -0
  147. package/presets/clearai/template/skills/web-artifact/assets/vendor/elkjs/elk.bundled.js +6605 -0
  148. package/presets/clearai/template/skills/web-artifact/references/when-drawing-a-topology.md +150 -0
  149. package/presets/clearai/template/skills/web-artifact/references/when-the-page-must-work-offline.md +62 -0
  150. package/presets/clearai/template/skills/web-artifact/scripts/check_artifact.py +167 -0
  151. package/presets/clearai/template/skills/web-artifact/scripts/render_topology.js +272 -0
  152. package/presets/clearai/template/skills/what-if-oracle/LICENSE.txt +5 -0
  153. package/presets/clearai/template/skills/what-if-oracle/SKILL.md +72 -0
  154. package/presets/clearai/template/skills/what-if-oracle/references/scenario-templates.md +154 -0
@@ -0,0 +1,69 @@
1
+ ---
2
+ name: process-understanding-extraction
3
+ description: |
4
+ 【流程理解·证据抽取】流程叙述定位、边界单元、单元级拓扑骨架(可审计摘录)。适用:对任意流程/系统资料(实验方案、方法论文、业务流程文档、操作规程)的理解与制图。不适用:领域/行业背景预研(用 domain-presearch);下游数据 QA(用 data-qa-analysis)。
5
+ version: 1.0
6
+ metadata:
7
+ tier: system
8
+ origin: template
9
+ created_at: '2026-06-12T02:47:05.409743+00:00'
10
+ ---
11
+
12
+ # 流程理解与提取 Skill (SOP)
13
+
14
+ > 本 Skill 是“流程型系统分析”的上游理解部分:先把系统讲清楚(流程叙述、边界、单元与能量/资源接口、流程图/拓扑),再允许进入任何数据对齐与数据分析。
15
+
16
+ ## 落盘约定
17
+
18
+ - **最终交付**:`products/extracted/`(process_brief、unit_inventory、process_flow、entity_map 等)
19
+ - **过程转储**:`lab/`(raw dump、OCR 中间文本、一次性脚本)
20
+ - 文中若出现 `extracted/` 逻辑别名,**write 必须使用** `products/extracted/` 完整路径
21
+
22
+ ## 1. 核心方法论 (Core Method)
23
+
24
+ - **Evidence-First(证据优先)**:流程必须来自权威资料的“流程叙述”原文;找不到就明确声明未找到,禁止补写流程。
25
+ - **Auditable Extraction(可审计抽取)**:`products/extracted/process_brief.md` 的“原文摘录”必须由代码从 raw dump/OCR 文本生成,避免模型逐字代写。
26
+ - **Boundary-First(边界优先)**:边界口径先行;净输入/净输出/回流(循环)/旁路规则必须写清楚。
27
+ - **Topology Skeleton(拓扑骨架)**:节点以“单元/装置”为主,边为“物料/对象流动 + 能量或资源注入/移除”;不写具体测点/字段名,测点映射后置到数据 skill。
28
+
29
+ ## 2. 决策树 (Decision Tree)
30
+
31
+ 根据用户意图选择工作流:
32
+ 1. **“我需要找到流程介绍/定义系统边界”** -> `workflows/01-process-doc-discovery.md`
33
+ 2. **“我要画流程图/把物流能流讲清楚”** -> `workflows/02-process-understanding-and-diagramming.md`
34
+ 3. **“全面体检(流程部分)”** -> 按顺序执行 01 -> 02。
35
+
36
+ ## 3. 核心指令 (Core Instructions)
37
+
38
+ <instruction>
39
+ <role>
40
+ 你是流程型系统的理解与证据抽取专家。你对“流程叙述主证据”与“补充证据(控制逻辑/启停程序)”严格区分;对边界口径、回流/旁路、能量与资源接口非常敏感;对“按编号推断流向”的错误高度警惕。
41
+ </role>
42
+
43
+ <rule>
44
+ 1. **必须先输出统一 plan 并确认(强制)**:执行本次任务涉及的**第一个 workflow**之前,必须先输出一份覆盖本 skill 全部 workflows(01-02)的统一 plan 供用户确认(确认后再执行)。
45
+ - plan 必须逐 workflow 列出强制产出物清单(逐项列文件名/路径),并声明 `lab/`(过程) vs `products/`(最终)分区与落点。
46
+ - plan 必须包含“产出物对照表”,逐条对照 workflows 文件中的 `## 产出物` 清单逐行复制,禁止概括/省略/合并。
47
+ 2. **资料优先级粘性(强制)**:一旦已找到操作规程/实验方案/流程图等高优先级文件,必须优先尽可能读取与抽取(paragraph+table、转纯文本、OCR),不得因抽取困难而自动下沉用低优先级资料替代流程叙述主证据。
48
+ 3. **流程叙述主证据硬规则(强制)**:未定位到“流程叙述/流程概述/方法描述/Process Description”等段落前,禁止基于猜测补写流程;找不到必须明确写“未找到”,并列出尝试路径。
49
+ 4. **process_brief 原文摘录必须由代码生成(强制)**:禁止模型在对话/笔记中逐字输出原文再手工粘贴;必须先产出 raw dump,再用范围抽取脚本生成 `products/extracted/process_brief.md`。
50
+ 5. **Workflow 闸门(强制)**:每个 workflow 的产出物必须经确认无误并记录确认记录后,才允许进入下一 workflow。
51
+ 6. **拓扑只做骨架(强制)**:`products/extracted/entity_map.json` 仅包含单元/公用支撑系统/边界的 nodes 与 material/energy edges,并带 evidence;禁止写入任何具体测点/字段名(流量计、传感器编号、数据列名等)。
52
+ 7. **流程参数须标注来源**:对话或报告中引用流程参数数值(温度、压力、流量、浓度等)时,须标注 `来源: products/extracted/process_brief.md 原文摘录` 或「推断」。
53
+ </rule>
54
+ </instruction>
55
+
56
+ ## 4. 资源索引 (Resource Index)
57
+
58
+ - **Workflows**:
59
+ - `workflows/01-process-doc-discovery.md`
60
+ - `workflows/02-process-understanding-and-diagramming.md`
61
+ - **Templates**:
62
+ - `templates/unified_execution_plan.md.tpl`(统一 plan 模板:仅覆盖 workflow01-02)
63
+ - `templates/process_brief.md.tpl`(Workflow 01 强制输出骨架)
64
+ - `templates/docx_raw_dump_extractor.py.tpl`(Workflow 01 代码抽取 raw dump)
65
+ - `templates/process_brief_builder_from_raw_dump.py.tpl`(Workflow 01 用代码生成 process_brief)
66
+ - `templates/process_flow_mermaid.md.tpl`(Workflow 02 Mermaid 图模板)
67
+ - `templates/entity_map_unit_topology.json.tpl`(Workflow 02 单元级拓扑骨架)
68
+ - **Checklists**:
69
+ - `checklists/readiness_check.md`
@@ -0,0 +1,34 @@
1
+ # 流程提取交付自检清单(Readiness Checklist:Workflow01-02)
2
+
3
+ > 在宣布“流程提取完成(可进入数据对齐/分析)”前,必须通过以下所有检查。
4
+
5
+ ## 0. 执行计划与交付分区 (Plan & Delivery)
6
+ - [ ] 已在执行第一个 workflow 前输出“覆盖本次所有 workflows(仅限本 skill)的统一 plan”并获得用户确认(确认后再执行)。
7
+ - [ ] plan 已逐 workflow 列出 01-02,且每一步明确标注强制产出物清单(含文件路径)。
8
+ - [ ] plan 已显式声明产物分区:过程文件/中间产物输出至 `lab/`,最终交付输出至 `products/`,且每一步标明输出落点。
9
+ - [ ] plan 包含“产出物对照表”,并已逐条对照所选 workflows 的 `## 产出物` 清单逐行复制(无漏项、无概括省略)。
10
+ - [ ] 最终交付物(workflows 声明的 `products/extracted/*`)已按 plan 约定落在 `products/`(推荐 `products/extracted/*`),过程转储/临时产物不混入最终交付目录。
11
+
12
+ ## 1. 流程资料与证据链 (Process Evidence)
13
+ - [ ] 已遵守“资料优先级粘性”:找到操作规程/实验方案/流程图等高优先级资料后,已尽可能完成鲁棒抽取(paragraph+table、转纯文本、OCR 等),未在高优先级尚可读取时直接降级用低优先级资料替代流程叙述。
14
+ - [ ] `products/extracted/process_brief.md` 的流程叙述段落通过“段落选型闸门”:标题类型匹配、包含流向要素、覆盖性满足;若未找到流程叙述,已明确写出“未找到”并列出已尝试路径(而不是补写流程)。
15
+ - [ ] `products/extracted/process_brief.md` 的“原文摘录”由代码从 raw dump(或 OCR 文本)生成,且填写了 `raw dump 路径 + 定位范围`(可审计、可复查)。
16
+ - [ ] `products/extracted/process_brief.md` 的“补充证据/控制逻辑摘要(可选)”如有填写:已标注来源与性质(补充证据,非流程叙述),且未用其替代流程叙述主证据。
17
+
18
+ ## 2. 单元清单与边界契约 (Inventory & Boundary)
19
+ - [ ] `products/extracted/unit_inventory.md` 已覆盖关键单元(尤其核心装置)的输入/输出与能量/资源接口(蒸汽/冷却水/电/燃气/真空等),并能从单元视角复原流向/能流走向(不要求测点/字段名)。
20
+ - [ ] 存在性三态输出正确:State_A/State_B/State_C 定义清晰,State_B 有证据链(来源/位置/置信度/why_not_high)。
21
+ - [ ] 对 State_C(未确认存在)条目未写成“无计量点/unmetered_stream”(计量点匹配后置到后续数据对齐)。
22
+ - [ ] `products/extracted/segment_boundary.md` 的净输入/净输出/回流/旁路规则与流程叙述一致;无证据时均标注“待确认/待匹配”,不存在强断言测点编号或凭空设备数量。
23
+
24
+ ## 3. 流程图与拓扑骨架 (Diagram & Topology Skeleton)
25
+ - [ ] `products/extracted/process_flow.md` 开头包含“输入去向映射表”,且每股输入的“直接接收单元”都有逐字原文依据(不得按编号顺序推测)。
26
+ - [ ] Mermaid 流程图至少包含:主流程图(净输入/净输出/回流区分清楚)与能量/资源视角图(主要支撑介质表达完整)。
27
+ - [ ] 若存在推断连接:已在图中用 `【推断】` 文本标注,并在文档中给出“推断清单”(理由/依据来源/置信度/验证动作)。
28
+ - [ ] `products/extracted/entity_map.json` 的节点=单元/公用支撑系统/边界;边=流股/能量;每条 edge 有 evidence;不包含任何具体测点/字段名(测点补齐后置)。
29
+
30
+ ---
31
+
32
+ **自检结论**:
33
+ - [ ] 通过 (Ready to Handover)
34
+ - [ ] 需返工 (Needs Rework)
@@ -0,0 +1,132 @@
1
+ """
2
+ DOCX Raw Dump Extractor (Template)
3
+
4
+ 目的:
5
+ - 用“代码”把复杂 DOCX 的段落 + 表格单元格按文档顺序转储为可检索的纯文本/Markdown(raw dump)
6
+ - 为 workflow01 的 `products/extracted/process_brief.md` 提供“逐字粘贴”的来源,避免大模型凭空生成
7
+
8
+ 依赖:
9
+ pip install python-docx
10
+
11
+ 用法(示例):
12
+ python docx_raw_dump_extractor.py \
13
+ --input "/path/to/file.docx" \
14
+ --output "lab/raw_dumps/file.raw_dump.md"
15
+
16
+ 说明:
17
+ - 输出文件会保留 block 顺序(paragraph/table),并标注 style 与索引
18
+ - 你可以在输出里搜索“流程叙述/流程概述/Process Description”等关键词,再逐字复制目标段落到 `products/extracted/process_brief.md`
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ import argparse
24
+ import os
25
+ from typing import Iterator, List, Union
26
+
27
+ from docx import Document
28
+ from docx.oxml.table import CT_Tbl
29
+ from docx.oxml.text.paragraph import CT_P
30
+ from docx.table import Table
31
+ from docx.text.paragraph import Paragraph
32
+
33
+ Block = Union[Paragraph, Table]
34
+
35
+
36
+ def iter_block_items(doc: Document) -> Iterator[Block]:
37
+ """
38
+ Iterate over document blocks (paragraphs and tables) in order.
39
+ Ref: python-docx does not provide a built-in ordered iterator.
40
+ """
41
+ body = doc.element.body
42
+ for child in body.iterchildren():
43
+ if isinstance(child, CT_P):
44
+ yield Paragraph(child, doc)
45
+ elif isinstance(child, CT_Tbl):
46
+ yield Table(child, doc)
47
+
48
+
49
+ def table_to_markdown(table: Table, max_cell_chars: int = 500) -> str:
50
+ rows: List[List[str]] = []
51
+ for r in table.rows:
52
+ row: List[str] = []
53
+ for c in r.cells:
54
+ txt = "\n".join(p.text.strip() for p in c.paragraphs if p.text and p.text.strip())
55
+ txt = txt.replace("|", "\\|")
56
+ if len(txt) > max_cell_chars:
57
+ txt = txt[: max_cell_chars - 3] + "..."
58
+ row.append(txt)
59
+ rows.append(row)
60
+
61
+ if not rows:
62
+ return ""
63
+
64
+ col_count = max(len(r) for r in rows)
65
+ for r in rows:
66
+ if len(r) < col_count:
67
+ r.extend([""] * (col_count - len(r)))
68
+
69
+ header = rows[0]
70
+ sep = ["---"] * col_count
71
+ body = rows[1:] if len(rows) > 1 else []
72
+
73
+ def fmt_row(r: List[str]) -> str:
74
+ return "| " + " | ".join(r) + " |"
75
+
76
+ lines = [fmt_row(header), fmt_row(sep)]
77
+ lines.extend(fmt_row(r) for r in body)
78
+ return "\n".join(lines)
79
+
80
+
81
+ def dump_docx(input_path: str) -> str:
82
+ doc = Document(input_path)
83
+ out: List[str] = []
84
+ out.append("# DOCX Raw Dump\n")
85
+ out.append(f"- source: `{input_path}`")
86
+ out.append("")
87
+
88
+ p_idx = 0
89
+ t_idx = 0
90
+ for b in iter_block_items(doc):
91
+ if isinstance(b, Paragraph):
92
+ text = (b.text or "").rstrip()
93
+ if not text.strip():
94
+ continue
95
+ p_idx += 1
96
+ style = getattr(getattr(b, "style", None), "name", "") or "UnknownStyle"
97
+ out.append(f"## [P{p_idx:05d}] style={style}")
98
+ out.append(text)
99
+ out.append("")
100
+ else:
101
+ t_idx += 1
102
+ out.append(f"## [T{t_idx:05d}] table")
103
+ md = table_to_markdown(b)
104
+ if md.strip():
105
+ out.append(md)
106
+ else:
107
+ out.append("_<empty table>_")
108
+ out.append("")
109
+
110
+ return "\n".join(out).strip() + "\n"
111
+
112
+
113
+ def main() -> None:
114
+ ap = argparse.ArgumentParser()
115
+ ap.add_argument("--input", required=True, help="Input .docx path")
116
+ ap.add_argument("--output", required=True, help="Output .md path")
117
+ args = ap.parse_args()
118
+
119
+ input_path = os.path.abspath(args.input)
120
+ output_path = os.path.abspath(args.output)
121
+ os.makedirs(os.path.dirname(output_path), exist_ok=True)
122
+
123
+ content = dump_docx(input_path)
124
+ with open(output_path, "w", encoding="utf-8") as f:
125
+ f.write(content)
126
+
127
+ print(f"OK: wrote raw dump to {output_path}")
128
+
129
+
130
+ if __name__ == "__main__":
131
+ main()
132
+
@@ -0,0 +1,86 @@
1
+ {
2
+ "schema_version": "1.0",
3
+ "description": "Unit-level topology for process-style systems. Nodes are units/utilities/boundaries; edges are material/energy. No measurement tag or field names in this file; tags are added in the data skill workflow02 enrichment. Each edge should carry an evidence object that points to products/extracted/process_brief.md or products/extracted/unit_inventory.md; inferred edges must be explicitly marked.",
4
+ "nodes": [
5
+ {
6
+ "id": "Unit_S201",
7
+ "type": "unit",
8
+ "label": "S201",
9
+ "notes": "Separation unit (example)"
10
+ },
11
+ {
12
+ "id": "SteamHeader_HP",
13
+ "type": "utility",
14
+ "label": "SteamHeader_HP",
15
+ "notes": "High-pressure steam header (example utility)"
16
+ },
17
+ {
18
+ "id": "Boundary_Feed",
19
+ "type": "boundary",
20
+ "label": "Feed",
21
+ "notes": "Net input boundary"
22
+ },
23
+ {
24
+ "id": "Boundary_Product",
25
+ "type": "boundary",
26
+ "label": "Product",
27
+ "notes": "Net output boundary"
28
+ }
29
+ ],
30
+ "edges": [
31
+ {
32
+ "id": "Edge_Feed_to_S201",
33
+ "kind": "material",
34
+ "from": "Boundary_Feed",
35
+ "to": "Unit_S201",
36
+ "stream_name": "Feed",
37
+ "direction": "in",
38
+ "measurement_slots": [
39
+ "flow"
40
+ ],
41
+ "evidence": {
42
+ "type": "process_brief_quote",
43
+ "source": "products/extracted/process_brief.md",
44
+ "location": "§2 原文摘录, 第x段",
45
+ "excerpt": "在此粘贴支持该连接的逐字引用(或说明来自 products/extracted/unit_inventory.md)"
46
+ },
47
+ "notes": "Data skill workflow02 should enrich this edge with flow_tag and unit."
48
+ },
49
+ {
50
+ "id": "Edge_S201_to_Product",
51
+ "kind": "material",
52
+ "from": "Unit_S201",
53
+ "to": "Boundary_Product",
54
+ "stream_name": "Product",
55
+ "direction": "out",
56
+ "measurement_slots": [
57
+ "flow"
58
+ ],
59
+ "evidence": {
60
+ "type": "process_brief_quote",
61
+ "source": "products/extracted/process_brief.md",
62
+ "location": "§2 原文摘录, 第x段",
63
+ "excerpt": "在此粘贴支持该连接的逐字引用(或说明来自 products/extracted/unit_inventory.md)"
64
+ },
65
+ "notes": "Avoid treating reflux/internal recycle as net product."
66
+ },
67
+ {
68
+ "id": "Edge_HPSteam_to_S201",
69
+ "kind": "energy",
70
+ "from": "SteamHeader_HP",
71
+ "to": "Unit_S201",
72
+ "energy_medium": "steam",
73
+ "direction": "in",
74
+ "measurement_slots": [
75
+ "steam_flow"
76
+ ],
77
+ "evidence": {
78
+ "type": "process_brief_quote",
79
+ "source": "products/extracted/process_brief.md",
80
+ "location": "§2 原文摘录, 第x段",
81
+ "excerpt": "在此粘贴支持该连接的逐字引用;若为推断,type 写 inferred 并说明验证方法"
82
+ },
83
+ "notes": "Data skill workflow02 should enrich with steam_flow_tag, pressure_level, unit."
84
+ }
85
+ ]
86
+ }
@@ -0,0 +1,89 @@
1
+ # 流程简介 (Process Brief)
2
+
3
+ **对象系统**: ${SYSTEM_NAME}
4
+ **分析日期**: ${DATE}
5
+ **负责人**: ${OWNER}
6
+
7
+ ---
8
+
9
+ ## 1. 定位信息
10
+
11
+ - **目标文件**:${SOURCE_FILE_PATH}
12
+ - **目标章节**:${TARGET_SECTION}(例如:5.2.2 流程叙述)
13
+ - **代码抽取产物(raw dump)路径**:${RAW_DUMP_PATH}(建议:`lab/raw_dumps/<doc>.raw_dump.md`)
14
+ - **raw dump 定位范围**:${RAW_DUMP_RANGE}(例如:P00123-P00168 或 “Heading 5.2.2 前后各 2 级标题范围”)
15
+ - **抽取方式**:${EXTRACTION_METHOD}(code_raw_dump / paragraph+table / pandoc / docx2txt / mammoth / OCR)
16
+ - **定位证据**:${LOCATION_EVIDENCE}(标题路径 / 页码 / 表格位置 / 截图说明)
17
+
18
+ > 段落选型闸门是否通过?
19
+ > - [ ] 标题类型匹配(指向"流程叙述/流程说明/方法描述"一类,不是"启动/停止/参数调整/联锁/报警"等排除项)
20
+ > - [ ] 内容包含流向要素(至少 1 个净输入 + 至少 1 个净输出/去向 + 能读出"从 A 到 B"的流向)
21
+ > - [ ] 覆盖性满足(若系统含多个关键单元,原文能覆盖其主流向关系)
22
+
23
+ ---
24
+
25
+ ## 2. 原文摘录(逐字粘贴,必须)
26
+
27
+ > **硬规则**:本区域必须由**代码**从“raw dump / OCR 转文本”自动生成填充,不允许大模型在对话中逐字输出原文后再手工粘贴(避免错贴/漏贴/不可审计)。
28
+ > 若原文过长,可分段粘贴并标注每段的标题路径/页码。
29
+ > 若原文中有表格,请以表格形式复现。
30
+
31
+ ```
32
+ <在此处逐字粘贴原文>
33
+ ```
34
+
35
+ ---
36
+
37
+ ## 3. AI 摘要(可选)
38
+
39
+ > **硬规则**:本区域只能在"原文摘录"完成后才允许填写。必须明确标注为"摘要",禁止与原文混排。
40
+ > 若不需要摘要,删除本区域或留空即可。
41
+
42
+ ${AI_SUMMARY_IF_NEEDED}
43
+
44
+ ---
45
+
46
+ ## 4. 疑点清单
47
+
48
+ > 若原文存在歧义、抽取不完整、或你无法确认某些信息,必须在此列出。
49
+ > 禁止在后续 `products/extracted/unit_inventory.md` / `products/extracted/segment_boundary.md` 中用强断言填空(改用"待确认")。
50
+
51
+ | 序号 | 疑点/缺口 | 影响范围 | 建议下一步 |
52
+ |------|-----------|----------|-----------|
53
+ | 1 | ${GAP_1} | ${IMPACT_1} | ${NEXT_STEP_1} |
54
+
55
+ ---
56
+
57
+ ## 5. 补充证据 / 控制逻辑摘要(可选,但推荐)
58
+
59
+ > 允许收录"启动/停止/参数调整/控制逻辑/动态平衡机理"等段落作为补充证据。
60
+ > 必须标注:章节来源(标题路径/页码)、性质(补充证据,非流程叙述)。
61
+ > 不得据此在 `products/extracted/segment_boundary.md` 中写出无证据的测点编号/设备数量等强断言。
62
+
63
+ ### 5.1 补充段落 1
64
+
65
+ - **来源**:${SUPPLEMENT_SOURCE_1}(标题路径/页码)
66
+ - **性质**:补充证据(非流程叙述)
67
+ - **原文摘录或摘要**:
68
+
69
+ ```
70
+ <在此处粘贴补充段落原文或写摘要(需标明哪种)>
71
+ ```
72
+
73
+ - **对分析的价值**:${VALUE_1}(例如:控制逻辑理解 / 真空建立顺序 / 回流策略 / 稳态判据参考)
74
+
75
+ ---
76
+
77
+ ## 6. 确认记录(闸门,进入下一 Workflow 前必填)
78
+
79
+ - **确认日期**:${CONFIRM_DATE}
80
+ - **确认人**:${CONFIRMER}
81
+ - **强制产出物是否齐全**:
82
+ - [ ] `products/extracted/process_brief.md`(本文件)
83
+ - [ ] `products/extracted/unit_inventory.md`
84
+ - [ ] `products/extracted/segment_boundary.md`
85
+ - **段落选型闸门是否通过**:[ ] 是 / [ ] 否(若否,必须继续回到 Step 2 搜索)
86
+ - **"原文摘录"区域是否为逐字粘贴**:[ ] 是 / [ ] 否(若否,必须回退重做原文摘录)
87
+ - **是否存在无证据强断言**:[ ] 否 / [ ] 是(若是,列出问题并回退或标为"待确认")
88
+ - **结论**:${CONCLUSION}
89
+ - **疑点与后续动作**:${REMAINING_ISSUES}
@@ -0,0 +1,203 @@
1
+ """
2
+ Process Brief Builder (Template)
3
+
4
+ 目标:
5
+ - 从 docx raw dump(由 `lab/scripts/docx_raw_dump_extractor.py` 生成)中按 range 抽取原文块
6
+ - 自动生成 `products/extracted/process_brief.md`,把“原文摘录”区域由代码填充
7
+ - 从机制上避免大模型在对话中“逐字输出原文”
8
+
9
+ 依赖:
10
+ Python 3.x (标准库即可)
11
+
12
+ 用法(示例):
13
+ python process_brief_builder_from_raw_dump.py \
14
+ --raw-dump "lab/raw_dumps/sop.raw_dump.md" \
15
+ --range "P00376-P00420" \
16
+ --source-file "input/xxx.docx" \
17
+ --target-section "5.2.2 流程叙述(第24页)" \
18
+ --output "products/extracted/process_brief.md"
19
+
20
+ 说明:
21
+ - raw dump block 的标题格式来自 `docx_raw_dump_extractor.py`:`## [P00001] style=...` 或 `## [T00001] table`
22
+ - range 支持:
23
+ - P 起止:P00376-P00420
24
+ - T 起止:T00012-T00019
25
+ - 单点:P00376
26
+ - 混合范围暂不支持(需要时可拆两次追加或扩大范围)
27
+ """
28
+
29
+ from __future__ import annotations
30
+
31
+ import argparse
32
+ import os
33
+ import re
34
+ from dataclasses import dataclass
35
+ from typing import List, Optional, Tuple
36
+
37
+
38
+ BLOCK_RE = re.compile(r"^## \[(?P<kind>[PT])(?P<idx>\d{5})\].*$")
39
+
40
+
41
+ @dataclass
42
+ class Block:
43
+ kind: str # 'P' or 'T'
44
+ idx: int
45
+ header_line: str
46
+ lines: List[str]
47
+
48
+
49
+ def parse_raw_dump(md_text: str) -> List[Block]:
50
+ lines = md_text.splitlines()
51
+ blocks: List[Block] = []
52
+ cur: Optional[Block] = None
53
+
54
+ for line in lines:
55
+ m = BLOCK_RE.match(line.strip())
56
+ if m:
57
+ if cur is not None:
58
+ blocks.append(cur)
59
+ cur = Block(
60
+ kind=m.group("kind"),
61
+ idx=int(m.group("idx")),
62
+ header_line=line,
63
+ lines=[],
64
+ )
65
+ continue
66
+ if cur is not None:
67
+ cur.lines.append(line)
68
+
69
+ if cur is not None:
70
+ blocks.append(cur)
71
+ return blocks
72
+
73
+
74
+ def parse_range(rng: str) -> Tuple[str, int, int]:
75
+ s = rng.strip()
76
+ if "-" in s:
77
+ a, b = s.split("-", 1)
78
+ kind_a, idx_a = a[0].upper(), int(a[1:])
79
+ kind_b, idx_b = b[0].upper(), int(b[1:])
80
+ if kind_a != kind_b:
81
+ raise ValueError("Mixed kind range is not supported. Use Pxxxxx-Pyyyyy or Txxxxx-Tyyyyy.")
82
+ if idx_b < idx_a:
83
+ raise ValueError("Range end must be >= start.")
84
+ return kind_a, idx_a, idx_b
85
+ kind, idx = s[0].upper(), int(s[1:])
86
+ return kind, idx, idx
87
+
88
+
89
+ def extract_blocks(blocks: List[Block], kind: str, start: int, end: int) -> List[Block]:
90
+ selected = [b for b in blocks if b.kind == kind and start <= b.idx <= end]
91
+ if not selected:
92
+ raise ValueError(f"No blocks matched range {kind}{start:05d}-{kind}{end:05d}.")
93
+ return selected
94
+
95
+
96
+ def build_process_brief(
97
+ *,
98
+ source_file: str,
99
+ target_section: str,
100
+ raw_dump_path: str,
101
+ raw_dump_range: str,
102
+ extraction_method: str,
103
+ excerpt_blocks: List[Block],
104
+ ) -> str:
105
+ excerpt_lines: List[str] = []
106
+ for b in excerpt_blocks:
107
+ excerpt_lines.append(b.header_line)
108
+ excerpt_lines.extend(b.lines)
109
+ excerpt_lines.append("")
110
+
111
+ excerpt_text = "\n".join(excerpt_lines).strip()
112
+
113
+ return f"""# 流程简介 (Process Brief)
114
+
115
+ <!-- GENERATED_BY: process_brief_builder_from_raw_dump.py -->
116
+
117
+ ## 1. 定位信息
118
+
119
+ - **目标文件**:{source_file}
120
+ - **目标章节**:{target_section}
121
+ - **代码抽取产物(raw dump)路径**:{raw_dump_path}
122
+ - **raw dump 定位范围**:{raw_dump_range}
123
+ - **抽取方式**:{extraction_method}
124
+ - **定位证据**:见 raw dump block 标题(P/T 编号 + style/table)
125
+
126
+ ---
127
+
128
+ ## 2. 原文摘录(由代码填充,禁止手工改写)
129
+
130
+ > **硬规则**:本区域为“代码从 raw dump 抽取的逐字原文”。禁止大模型在对话中逐字输出原文、禁止人工改写替换。
131
+
132
+ ```text
133
+ {excerpt_text}
134
+ ```
135
+
136
+ ---
137
+
138
+ ## 3. AI 摘要(可选)
139
+
140
+ (可选,需明确标注为摘要;禁止与原文混排)
141
+
142
+ ---
143
+
144
+ ## 4. 疑点清单
145
+
146
+ | 序号 | 疑点/缺口 | 影响范围 | 建议下一步 |
147
+ |------|-----------|----------|-----------|
148
+ | 1 | | | |
149
+
150
+ ---
151
+
152
+ ## 5. 补充证据 / 控制逻辑摘要(可选)
153
+
154
+ (允许,但必须标注来源与性质:补充证据,非流程叙述)
155
+
156
+ ---
157
+
158
+ ## 6. 确认记录(闸门)
159
+
160
+ - **确认日期**:
161
+ - **确认人**:
162
+ - **"原文摘录"是否由代码生成**:[ ] 是 / [ ] 否(若否,不合格,必须回退用代码生成)
163
+ """
164
+
165
+
166
+ def main() -> None:
167
+ ap = argparse.ArgumentParser()
168
+ ap.add_argument("--raw-dump", required=True, help="Path to raw dump markdown")
169
+ ap.add_argument("--range", required=True, help="Block range, e.g. P00376-P00420")
170
+ ap.add_argument("--source-file", required=True, help="Original doc path")
171
+ ap.add_argument("--target-section", required=True, help="Target section name/page")
172
+ ap.add_argument("--output", required=True, help="Output process_brief.md path")
173
+ ap.add_argument("--extraction-method", default="code_raw_dump", help="Extraction method label")
174
+ args = ap.parse_args()
175
+
176
+ raw_dump_path = os.path.abspath(args.raw_dump)
177
+ with open(raw_dump_path, "r", encoding="utf-8") as f:
178
+ md_text = f.read()
179
+
180
+ blocks = parse_raw_dump(md_text)
181
+ kind, start, end = parse_range(args.range)
182
+ excerpt_blocks = extract_blocks(blocks, kind, start, end)
183
+
184
+ content = build_process_brief(
185
+ source_file=args.source_file,
186
+ target_section=args.target_section,
187
+ raw_dump_path=args.raw_dump,
188
+ raw_dump_range=args.range,
189
+ extraction_method=args.extraction_method,
190
+ excerpt_blocks=excerpt_blocks,
191
+ )
192
+
193
+ out_path = os.path.abspath(args.output)
194
+ os.makedirs(os.path.dirname(out_path), exist_ok=True)
195
+ with open(out_path, "w", encoding="utf-8") as f:
196
+ f.write(content)
197
+
198
+ print(f"OK: wrote {out_path}")
199
+
200
+
201
+ if __name__ == "__main__":
202
+ main()
203
+
@@ -0,0 +1,41 @@
1
+ # 流程图模板(Mermaid)
2
+
3
+ ## 模板说明
4
+ 建议输出两张图:
5
+ 1. 主流程(只画对象/物料流)
6
+ 2. 能量/资源视角(补充蒸汽/冷却水/电等支撑介质)
7
+
8
+ ---
9
+
10
+ # 1) 主流程(对象/物料流)
11
+
12
+ ```mermaid
13
+ flowchart LR
14
+ Feed["Feed"] --> UnitA["UnitA"]
15
+ UnitA --> UnitB["UnitB"]
16
+ UnitB --> Product["Product"]
17
+ UnitB --> Recycle["Recycle"]
18
+ Recycle --> UnitA
19
+ ```
20
+
21
+ ## 关键流股说明
22
+ - Feed:${FEED_DESC}(代表测点/字段:${FEED_TAG})
23
+ - Product:${PRODUCT_DESC}(代表测点/字段:${PRODUCT_TAG})
24
+ - Recycle:${RECYCLE_DESC}(是否计入净输出:${RECYCLE_POLICY})
25
+
26
+ ---
27
+
28
+ # 2) 能量/资源视角(公用支撑系统)
29
+
30
+ ```mermaid
31
+ flowchart LR
32
+ Steam["Steam"] --> Heater["Heater"]
33
+ Heater --> UnitB["UnitB"]
34
+ UnitB --> Condenser["Condenser"]
35
+ Condenser --> CW["CoolingWater"]
36
+ Vacuum["VacuumSystem"] --> UnitB
37
+ ```
38
+
39
+ ## 能量口径说明
40
+ - 蒸汽计量点:${STEAM_TAGS}(单位:${STEAM_UNIT})
41
+ - 电耗计量点:${POWER_TAGS}(单位:kW)