clearai-dsh 0.3.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +147 -0
- package/README.md +28 -28
- package/README.zh-CN.md +28 -28
- package/lib/client.js +861 -2555
- package/lib/domain-language.js +552 -48
- package/lib/fold.js +351 -1015
- package/lib/host.js +47 -557
- package/lib/invariant.js +9 -12
- package/lib/knowledge-view.js +363 -226
- package/lib/lang.js +81 -0
- package/package.json +1 -1
- package/presets/clearai/agent.cordis.yml +54 -82
- package/presets/clearai/clearai.patch.yml +54 -82
- package/presets/clearai/plugins/clearai-kernel.js +1590 -5437
- package/presets/clearai/plugins/ontology.js +56 -14
- package/presets/clearai/plugins/prompts.js +73 -363
- package/presets/clearai/skills/clearai-loop/SKILL.md +73 -59
- package/presets/clearai/plugins/brain.js +0 -547
- package/presets/clearai/plugins/commands.js +0 -199
- package/presets/clearai/template/knowledge/README.md +0 -25
- package/presets/clearai/template/memory/README.md +0 -34
- package/presets/clearai/template/project.md +0 -49
- package/presets/clearai/template/skills/README.md +0 -37
- package/presets/clearai/template/skills/chart-diagram-qa/SKILL.md +0 -43
- package/presets/clearai/template/skills/citation-management/SKILL.md +0 -73
- package/presets/clearai/template/skills/citation-management/references/bibtex_formatting.md +0 -908
- package/presets/clearai/template/skills/citation-management/references/citation_validation.md +0 -794
- package/presets/clearai/template/skills/citation-management/references/google_scholar_search.md +0 -725
- package/presets/clearai/template/skills/citation-management/references/metadata_extraction.md +0 -870
- package/presets/clearai/template/skills/citation-management/references/pubmed_search.md +0 -839
- package/presets/clearai/template/skills/citation-management/scripts/doi_to_bibtex.py +0 -204
- package/presets/clearai/template/skills/citation-management/scripts/extract_metadata.py +0 -569
- package/presets/clearai/template/skills/citation-management/scripts/format_bibtex.py +0 -349
- package/presets/clearai/template/skills/citation-management/scripts/generate_schematic.py +0 -139
- package/presets/clearai/template/skills/citation-management/scripts/generate_schematic_ai.py +0 -817
- package/presets/clearai/template/skills/citation-management/scripts/search_google_scholar.py +0 -282
- package/presets/clearai/template/skills/citation-management/scripts/search_pubmed.py +0 -398
- package/presets/clearai/template/skills/citation-management/scripts/validate_citations.py +0 -497
- package/presets/clearai/template/skills/data-analysis/SKILL.md +0 -92
- package/presets/clearai/template/skills/data-analysis/checklists/readiness_check.md +0 -23
- package/presets/clearai/template/skills/data-analysis/templates/analysis_report.md.tpl +0 -63
- package/presets/clearai/template/skills/data-analysis/templates/cleaning_rules_draft.yaml.tpl +0 -32
- package/presets/clearai/template/skills/data-analysis/templates/data_dictionary.md.tpl +0 -12
- package/presets/clearai/template/skills/data-analysis/templates/domain_knowledge_template.md.tpl +0 -316
- package/presets/clearai/template/skills/data-analysis/templates/feature_candidates.json.tpl +0 -20
- package/presets/clearai/template/skills/data-analysis/templates/quality_scorecard.md.tpl +0 -30
- package/presets/clearai/template/skills/data-analysis/workflows/01-data-profiling.md +0 -42
- package/presets/clearai/template/skills/data-analysis/workflows/02-quality-audit.md +0 -36
- package/presets/clearai/template/skills/data-analysis/workflows/03-physical-correlation.md +0 -25
- package/presets/clearai/template/skills/data-analysis/workflows/04-unstructured-mining.md +0 -26
- package/presets/clearai/template/skills/data-qa-analysis/SKILL.md +0 -102
- package/presets/clearai/template/skills/data-qa-analysis/checklists/readiness_check.md +0 -62
- package/presets/clearai/template/skills/data-qa-analysis/templates/best_in_class_report.md.tpl +0 -56
- package/presets/clearai/template/skills/data-qa-analysis/templates/cleaning_rules_draft.yaml.tpl +0 -56
- package/presets/clearai/template/skills/data-qa-analysis/templates/data_dictionary.md.tpl +0 -13
- package/presets/clearai/template/skills/data-qa-analysis/templates/data_source_inventory_and_lineage.md.tpl +0 -146
- package/presets/clearai/template/skills/data-qa-analysis/templates/data_status_report.md.tpl +0 -60
- package/presets/clearai/template/skills/data-qa-analysis/templates/steady_state_rules.yaml.tpl +0 -41
- package/presets/clearai/template/skills/data-qa-analysis/templates/subsystem_registry.md.tpl +0 -101
- package/presets/clearai/template/skills/data-qa-analysis/templates/unified_execution_plan.md.tpl +0 -100
- package/presets/clearai/template/skills/data-qa-analysis/workflows/01-data-source-inventory-and-lineage.md +0 -194
- package/presets/clearai/template/skills/data-qa-analysis/workflows/02-data-alignment-and-tag-semantics.md +0 -122
- package/presets/clearai/template/skills/data-qa-analysis/workflows/03-steady-state-identification.md +0 -126
- package/presets/clearai/template/skills/data-qa-analysis/workflows/04-consumption-analysis.md +0 -152
- package/presets/clearai/template/skills/data-qa-analysis/workflows/05-best-in-class-and-optimization-space.md +0 -78
- package/presets/clearai/template/skills/domain-presearch/SKILL.md +0 -131
- package/presets/clearai/template/skills/domain-presearch/checklists/domain_checklist.md +0 -24
- package/presets/clearai/template/skills/domain-presearch/references/figure_code.md +0 -78
- package/presets/clearai/template/skills/domain-presearch/references/strategic_frameworks.md +0 -38
- package/presets/clearai/template/skills/exploration-loop/SKILL.md +0 -81
- package/presets/clearai/template/skills/exploratory-data-analysis/SKILL.md +0 -77
- package/presets/clearai/template/skills/exploratory-data-analysis/references/bioinformatics_genomics_formats.md +0 -664
- package/presets/clearai/template/skills/exploratory-data-analysis/references/chemistry_molecular_formats.md +0 -664
- package/presets/clearai/template/skills/exploratory-data-analysis/references/general_scientific_formats.md +0 -518
- package/presets/clearai/template/skills/exploratory-data-analysis/references/microscopy_imaging_formats.md +0 -620
- package/presets/clearai/template/skills/exploratory-data-analysis/references/proteomics_metabolomics_formats.md +0 -517
- package/presets/clearai/template/skills/exploratory-data-analysis/references/spectroscopy_analytical_formats.md +0 -633
- package/presets/clearai/template/skills/exploratory-data-analysis/scripts/eda_analyzer.py +0 -547
- package/presets/clearai/template/skills/hypothesis-generation/SKILL.md +0 -73
- package/presets/clearai/template/skills/hypothesis-generation/references/experimental_design_patterns.md +0 -329
- package/presets/clearai/template/skills/hypothesis-generation/references/hypothesis_quality_criteria.md +0 -198
- package/presets/clearai/template/skills/hypothesis-generation/references/literature_search_strategies.md +0 -622
- package/presets/clearai/template/skills/hypothesis-generation/scripts/generate_schematic.py +0 -139
- package/presets/clearai/template/skills/hypothesis-generation/scripts/generate_schematic_ai.py +0 -817
- package/presets/clearai/template/skills/literature-review/SKILL.md +0 -72
- package/presets/clearai/template/skills/literature-review/references/citation_styles.md +0 -166
- package/presets/clearai/template/skills/literature-review/references/database_strategies.md +0 -455
- package/presets/clearai/template/skills/literature-review/scripts/generate_pdf.py +0 -176
- package/presets/clearai/template/skills/literature-review/scripts/generate_schematic.py +0 -139
- package/presets/clearai/template/skills/literature-review/scripts/generate_schematic_ai.py +0 -817
- package/presets/clearai/template/skills/literature-review/scripts/search_databases.py +0 -303
- package/presets/clearai/template/skills/literature-review/scripts/verify_citations.py +0 -221
- package/presets/clearai/template/skills/paper-lookup/SKILL.md +0 -59
- package/presets/clearai/template/skills/paper-lookup/references/arxiv.md +0 -161
- package/presets/clearai/template/skills/paper-lookup/references/biorxiv.md +0 -118
- package/presets/clearai/template/skills/paper-lookup/references/core.md +0 -150
- package/presets/clearai/template/skills/paper-lookup/references/crossref.md +0 -181
- package/presets/clearai/template/skills/paper-lookup/references/medrxiv.md +0 -104
- package/presets/clearai/template/skills/paper-lookup/references/openalex.md +0 -174
- package/presets/clearai/template/skills/paper-lookup/references/pmc.md +0 -152
- package/presets/clearai/template/skills/paper-lookup/references/pubmed.md +0 -124
- package/presets/clearai/template/skills/paper-lookup/references/semantic-scholar.md +0 -203
- package/presets/clearai/template/skills/paper-lookup/references/unpaywall.md +0 -127
- package/presets/clearai/template/skills/process-presearch/SKILL.md +0 -196
- package/presets/clearai/template/skills/process-presearch/checklists/process_checklist.md +0 -18
- package/presets/clearai/template/skills/process-presearch/references/figure_code.md +0 -107
- package/presets/clearai/template/skills/process-presearch/references/source_attribution_example.md +0 -22
- package/presets/clearai/template/skills/process-understanding-extraction/SKILL.md +0 -69
- package/presets/clearai/template/skills/process-understanding-extraction/checklists/readiness_check.md +0 -34
- package/presets/clearai/template/skills/process-understanding-extraction/templates/docx_raw_dump_extractor.py.tpl +0 -132
- package/presets/clearai/template/skills/process-understanding-extraction/templates/entity_map_unit_topology.json.tpl +0 -86
- package/presets/clearai/template/skills/process-understanding-extraction/templates/process_brief.md.tpl +0 -89
- package/presets/clearai/template/skills/process-understanding-extraction/templates/process_brief_builder_from_raw_dump.py.tpl +0 -203
- package/presets/clearai/template/skills/process-understanding-extraction/templates/process_flow_mermaid.md.tpl +0 -41
- package/presets/clearai/template/skills/process-understanding-extraction/templates/unified_execution_plan.md.tpl +0 -53
- package/presets/clearai/template/skills/process-understanding-extraction/workflows/01-process-doc-discovery.md +0 -173
- package/presets/clearai/template/skills/process-understanding-extraction/workflows/02-process-understanding-and-diagramming.md +0 -106
- package/presets/clearai/template/skills/scientific-brainstorming/SKILL.md +0 -64
- package/presets/clearai/template/skills/scientific-brainstorming/references/brainstorming_methods.md +0 -326
- package/presets/clearai/template/skills/scientific-critical-thinking/SKILL.md +0 -72
- package/presets/clearai/template/skills/scientific-critical-thinking/references/common_biases.md +0 -364
- package/presets/clearai/template/skills/scientific-critical-thinking/references/evidence_hierarchy.md +0 -485
- package/presets/clearai/template/skills/scientific-critical-thinking/references/experimental_design.md +0 -496
- package/presets/clearai/template/skills/scientific-critical-thinking/references/logical_fallacies.md +0 -478
- package/presets/clearai/template/skills/scientific-critical-thinking/references/scientific_method.md +0 -169
- package/presets/clearai/template/skills/scientific-critical-thinking/references/statistical_pitfalls.md +0 -506
- package/presets/clearai/template/skills/skill-creator/SKILL.md +0 -109
- package/presets/clearai/template/skills/skill-creator/references/authoring-guide.md +0 -89
- package/presets/clearai/template/skills/statistical-analysis/SKILL.md +0 -79
- package/presets/clearai/template/skills/statistical-analysis/references/assumptions_and_diagnostics.md +0 -369
- package/presets/clearai/template/skills/statistical-analysis/references/bayesian_statistics.md +0 -653
- package/presets/clearai/template/skills/statistical-analysis/references/effect_sizes_and_power.md +0 -578
- package/presets/clearai/template/skills/statistical-analysis/references/reporting_standards.md +0 -469
- package/presets/clearai/template/skills/statistical-analysis/references/test_selection_guide.md +0 -129
- package/presets/clearai/template/skills/statistical-analysis/scripts/assumption_checks.py +0 -538
- package/presets/clearai/template/skills/web-artifact/SKILL.md +0 -165
- package/presets/clearai/template/skills/web-artifact/assets/renderer/renderer.css +0 -229
- package/presets/clearai/template/skills/web-artifact/assets/renderer/renderer.js +0 -373
- package/presets/clearai/template/skills/web-artifact/assets/vendor/elkjs/LICENSE +0 -263
- package/presets/clearai/template/skills/web-artifact/assets/vendor/elkjs/UPSTREAM.md +0 -26
- package/presets/clearai/template/skills/web-artifact/assets/vendor/elkjs/elk.bundled.js +0 -6605
- package/presets/clearai/template/skills/web-artifact/references/when-drawing-a-topology.md +0 -150
- package/presets/clearai/template/skills/web-artifact/references/when-the-page-must-work-offline.md +0 -62
- package/presets/clearai/template/skills/web-artifact/scripts/check_artifact.py +0 -167
- package/presets/clearai/template/skills/web-artifact/scripts/render_topology.js +0 -272
- package/presets/clearai/template/skills/what-if-oracle/LICENSE.txt +0 -5
- package/presets/clearai/template/skills/what-if-oracle/SKILL.md +0 -72
- package/presets/clearai/template/skills/what-if-oracle/references/scenario-templates.md +0 -154
package/lib/domain-language.js
CHANGED
|
@@ -20,6 +20,8 @@
|
|
|
20
20
|
* 断言里 `predicate` 与 `subject.type` 都是这个空间里的名字,分开两个空间只会让引用含混。
|
|
21
21
|
*/
|
|
22
22
|
|
|
23
|
+
import { bilingual, tr } from './lang.js'
|
|
24
|
+
|
|
23
25
|
/** 字面值的五种形态。断言客体如果是**值**,必是其中之一。 */
|
|
24
26
|
export const VALUE_FORMS = ['statement', 'quantity', 'formula', 'code', 'reference']
|
|
25
27
|
|
|
@@ -202,42 +204,50 @@ export function subjectKey(assertion) {
|
|
|
202
204
|
return `${text(subject.type)}|${text(subject.id)}`
|
|
203
205
|
}
|
|
204
206
|
|
|
207
|
+
/**
|
|
208
|
+
* 一个类型算不算某个概念:就是它,或它的父链上有它(`jepa` is_a … `method` ⇒ 算 `method`)。
|
|
209
|
+
* 主词域与值域都按这一条判——只认字面相等,模型就得为每个子类另立一条谓词。
|
|
210
|
+
*/
|
|
211
|
+
function isKindOf(lexicon, type, ancestor) {
|
|
212
|
+
return type === ancestor || termChain(lexicon, type).chain.includes(ancestor)
|
|
213
|
+
}
|
|
214
|
+
|
|
205
215
|
/** 客体形态与谓词值域是否相容。 */
|
|
206
|
-
function objectProblems(predicate, object) {
|
|
216
|
+
function objectProblems(lexicon, predicate, object) {
|
|
207
217
|
const problems = []
|
|
208
|
-
if (!isPlainObject(object)) return [problem('object_required', '断言要有宾语')]
|
|
218
|
+
if (!isPlainObject(object)) return [problem('object_required', tr('断言要有宾语', 'An assertion needs an object'))]
|
|
209
219
|
const kind = text(object.kind)
|
|
210
|
-
if (!OBJECT_KINDS.includes(kind)) return [problem('object_kind_unknown', `宾语形态「${kind}」不认识(可用:${OBJECT_KINDS.join(' / ')})`)]
|
|
220
|
+
if (!OBJECT_KINDS.includes(kind)) return [problem('object_kind_unknown', tr(`宾语形态「${kind}」不认识(可用:${OBJECT_KINDS.join(' / ')})`, `Unknown object kind "${kind}" (allowed: ${OBJECT_KINDS.join(' / ')})`))]
|
|
211
221
|
const range = isPlainObject(predicate.range) ? predicate.range : {}
|
|
212
222
|
const rangeTerm = text(range.term)
|
|
213
223
|
const rangeForm = text(range.form)
|
|
214
224
|
if (rangeTerm !== '') {
|
|
215
|
-
if (kind !== 'instance') return [problem('object_form_mismatch', `谓词「${predicate.id}」的宾语是概念「${rangeTerm}」的实例,宾语形态应为 instance`)]
|
|
225
|
+
if (kind !== 'instance') return [problem('object_form_mismatch', tr(`谓词「${predicate.id}」的宾语是概念「${rangeTerm}」的实例,宾语形态应为 instance`, `The object of predicate "${predicate.id}" is an instance of concept "${rangeTerm}"; the object kind should be instance`))]
|
|
216
226
|
const type = text(object.type)
|
|
217
|
-
if (type !== '' && type
|
|
218
|
-
if (text(object.value) === '') return [problem('object_value_required', '宾语实例要有名称')]
|
|
227
|
+
if (type !== '' && !isKindOf(lexicon, type, rangeTerm)) return [problem('object_type_mismatch', tr(`宾语实例的类型「${type}」既不是谓词值域「${rangeTerm}」,也不是它的下位概念`, `The object instance type "${type}" is neither the predicate range "${rangeTerm}" nor one of its subconcepts`))]
|
|
228
|
+
if (text(object.value) === '') return [problem('object_value_required', tr('宾语实例要有名称', 'The object instance needs a name'))]
|
|
219
229
|
return []
|
|
220
230
|
}
|
|
221
|
-
if (rangeForm !== '' && kind !== rangeForm) return [problem('object_form_mismatch', `谓词「${predicate.id}」的值形态是 ${rangeForm},宾语却是 ${kind}`)]
|
|
231
|
+
if (rangeForm !== '' && kind !== rangeForm) return [problem('object_form_mismatch', tr(`谓词「${predicate.id}」的值形态是 ${rangeForm},宾语却是 ${kind}`, `Predicate "${predicate.id}" takes ${rangeForm} values, but the object is ${kind}`))]
|
|
222
232
|
switch (kind) {
|
|
223
233
|
case 'statement':
|
|
224
|
-
if (text(object.value) === '') problems.push(problem('object_value_required', '陈述不能为空'))
|
|
225
|
-
else if (String(object.value).length > 2000) problems.push(problem('object_value_too_long', '陈述超过 2000 字:把它拆成断言,或把长文放证据里'))
|
|
234
|
+
if (text(object.value) === '') problems.push(problem('object_value_required', tr('陈述不能为空', 'The statement cannot be empty')))
|
|
235
|
+
else if (String(object.value).length > 2000) problems.push(problem('object_value_too_long', tr('陈述超过 2000 字:把它拆成断言,或把长文放证据里', 'The statement is over 2000 characters: split it into assertions, or put the long text in evidence')))
|
|
226
236
|
break
|
|
227
237
|
case 'quantity':
|
|
228
|
-
if (typeof object.value !== 'number' || !Number.isFinite(object.value)) problems.push(problem('object_quantity_shape', '量形态的 value 必须是有限数值'))
|
|
229
|
-
if (text(object.unit) === '') problems.push(problem('object_unit_required', '量形态要带单位:没有单位的数不是量'))
|
|
238
|
+
if (typeof object.value !== 'number' || !Number.isFinite(object.value)) problems.push(problem('object_quantity_shape', tr('量形态的 value 必须是有限数值', 'A quantity value must be a finite number')))
|
|
239
|
+
if (text(object.unit) === '') problems.push(problem('object_unit_required', tr('量形态要带单位:没有单位的数不是量', 'A quantity needs a unit: a number without a unit is not a quantity')))
|
|
230
240
|
break
|
|
231
241
|
case 'formula':
|
|
232
|
-
if (text(object.value) === '') problems.push(problem('object_value_required', '公式不能为空'))
|
|
242
|
+
if (text(object.value) === '') problems.push(problem('object_value_required', tr('公式不能为空', 'The formula cannot be empty')))
|
|
233
243
|
break
|
|
234
244
|
case 'code':
|
|
235
|
-
if (text(object.value) === '') problems.push(problem('object_value_required', 'code 形态要指向工作区里的一个文件,不能为空'))
|
|
236
|
-
else if (/^([/\\]|[A-Za-z]:[\\/])/.test(text(object.value))) problems.push(problem('object_code_absolute', 'code 形态要写工作区内的相对路径:绝对路径换个工作区就没了'))
|
|
237
|
-
else if (text(object.value).split(/[/\\]/).includes('..')) problems.push(problem('object_code_escape', 'code 形态不许越出工作区(路径里出现了 ..)'))
|
|
245
|
+
if (text(object.value) === '') problems.push(problem('object_value_required', tr('code 形态要指向工作区里的一个文件,不能为空', 'A code value must point to a file in the workspace and cannot be empty')))
|
|
246
|
+
else if (/^([/\\]|[A-Za-z]:[\\/])/.test(text(object.value))) problems.push(problem('object_code_absolute', tr('code 形态要写工作区内的相对路径:绝对路径换个工作区就没了', 'A code value must be a path relative to the workspace: an absolute path breaks in another workspace')))
|
|
247
|
+
else if (text(object.value).split(/[/\\]/).includes('..')) problems.push(problem('object_code_escape', tr('code 形态不许越出工作区(路径里出现了 ..)', 'A code value cannot leave the workspace (the path contains ..)')))
|
|
238
248
|
break
|
|
239
249
|
case 'reference':
|
|
240
|
-
if (text(object.value) === '') problems.push(problem('object_value_required', '引用不能为空'))
|
|
250
|
+
if (text(object.value) === '') problems.push(problem('object_value_required', tr('引用不能为空', 'The reference cannot be empty')))
|
|
241
251
|
break
|
|
242
252
|
default:
|
|
243
253
|
break
|
|
@@ -251,32 +261,32 @@ function objectProblems(predicate, object) {
|
|
|
251
261
|
*/
|
|
252
262
|
export function validateAssertion(lexicon, assertion) {
|
|
253
263
|
const problems = []
|
|
254
|
-
if (!isPlainObject(assertion)) return [problem('assertion_shape', '断言必须是一个对象')]
|
|
264
|
+
if (!isPlainObject(assertion)) return [problem('assertion_shape', tr('断言必须是一个对象', 'An assertion must be an object'))]
|
|
255
265
|
const predicateId = text(assertion.predicate)
|
|
256
|
-
if (predicateId === '') return [problem('predicate_required', '断言要写谓词')]
|
|
266
|
+
if (predicateId === '') return [problem('predicate_required', tr('断言要写谓词', 'An assertion needs a predicate'))]
|
|
257
267
|
const found = findEntry(lexicon, predicateId)
|
|
258
|
-
if (found === null) return [problem('predicate_unknown', `谓词「${predicateId}
|
|
259
|
-
if (found.kind !== 'predicate') return [problem('predicate_not_predicate', `「${predicateId}
|
|
268
|
+
if (found === null) return [problem('predicate_unknown', tr(`谓词「${predicateId}」还没登记`, `Predicate "${predicateId}" does not exist yet`))]
|
|
269
|
+
if (found.kind !== 'predicate') return [problem('predicate_not_predicate', tr(`「${predicateId}」是个概念,不能当谓词用`, `"${predicateId}" is a concept and cannot be used as a predicate`))]
|
|
260
270
|
const predicate = found.entry
|
|
261
|
-
if (!isUsable(predicate)) problems.push(problem('predicate_deprecated', `谓词「${predicateId}
|
|
271
|
+
if (!isUsable(predicate)) problems.push(problem('predicate_deprecated', tr(`谓词「${predicateId}」已废止:新断言不能再用它`, `Predicate "${predicateId}" is deprecated: new assertions cannot use it`)))
|
|
262
272
|
const subject = assertion.subject
|
|
263
|
-
if (!isPlainObject(subject) || text(subject.id) === '') problems.push(problem('subject_required', '断言要有主体(至少一个名称)'))
|
|
273
|
+
if (!isPlainObject(subject) || text(subject.id) === '') problems.push(problem('subject_required', tr('断言要有主体(至少一个名称)', 'An assertion needs a subject (at least a name)')))
|
|
264
274
|
else {
|
|
265
275
|
const domain = text(predicate.domain)
|
|
266
276
|
const type = text(subject.type)
|
|
267
277
|
if (domain !== '') {
|
|
268
|
-
if (type === '') problems.push(problem('subject_type_required', `谓词「${predicateId}」声明了主词域「${domain}」,主体要写明 type`))
|
|
269
|
-
else if (type
|
|
278
|
+
if (type === '') problems.push(problem('subject_type_required', tr(`谓词「${predicateId}」声明了主词域「${domain}」,主体要写明 type`, `Predicate "${predicateId}" declares the domain "${domain}", so the subject needs a type`)))
|
|
279
|
+
else if (!isKindOf(lexicon, type, domain)) problems.push(problem('subject_type_mismatch', tr(`主体类型「${type}」既不是主词域「${domain}」,也不是它的下位概念`, `Subject type "${type}" is neither the domain "${domain}" nor one of its subconcepts`)))
|
|
270
280
|
}
|
|
271
281
|
if (type !== '') {
|
|
272
282
|
const typeEntry = findEntry(lexicon, type)
|
|
273
|
-
if (typeEntry === null) problems.push(problem('subject_type_unknown', `主体类型「${type}
|
|
274
|
-
else if (typeEntry.kind !== 'term') problems.push(problem('subject_type_not_term', `主体类型「${type}
|
|
275
|
-
else if (!isUsable(typeEntry.entry)) problems.push(problem('subject_type_deprecated', `主体类型「${type}
|
|
283
|
+
if (typeEntry === null) problems.push(problem('subject_type_unknown', tr(`主体类型「${type}」还没登记`, `Subject type "${type}" does not exist yet`)))
|
|
284
|
+
else if (typeEntry.kind !== 'term') problems.push(problem('subject_type_not_term', tr(`主体类型「${type}」是个谓词`, `Subject type "${type}" is a predicate`)))
|
|
285
|
+
else if (!isUsable(typeEntry.entry)) problems.push(problem('subject_type_deprecated', tr(`主体类型「${type}」已废止`, `Subject type "${type}" is deprecated`)))
|
|
276
286
|
}
|
|
277
287
|
}
|
|
278
|
-
problems.push(...objectProblems(predicate, assertion.object))
|
|
279
|
-
if (assertion.qualifiers !== undefined && assertion.qualifiers !== null && !isPlainObject(assertion.qualifiers)) problems.push(problem('qualifiers_shape', 'qualifiers 只能是对象(限定条件:时间、工况、适用范围)'))
|
|
288
|
+
problems.push(...objectProblems(lexicon, predicate, assertion.object))
|
|
289
|
+
if (assertion.qualifiers !== undefined && assertion.qualifiers !== null && !isPlainObject(assertion.qualifiers)) problems.push(problem('qualifiers_shape', tr('qualifiers 只能是对象(限定条件:时间、工况、适用范围)', 'qualifiers must be an object (conditions: time, setting, scope)')))
|
|
280
290
|
return problems
|
|
281
291
|
}
|
|
282
292
|
|
|
@@ -296,7 +306,7 @@ export function validateAssertion(lexicon, assertion) {
|
|
|
296
306
|
*/
|
|
297
307
|
export function validateAssertions(state, assertions, options = {}) {
|
|
298
308
|
if (assertions === undefined || assertions === null) return []
|
|
299
|
-
if (!Array.isArray(assertions)) return [problem('assertions_shape', 'assertions 只能是数组')]
|
|
309
|
+
if (!Array.isArray(assertions)) return [problem('assertions_shape', tr('assertions 只能是数组', 'assertions must be an array'))]
|
|
300
310
|
if (assertions.length === 0) return []
|
|
301
311
|
const stateForm = isPlainObject(state) && isPlainObject(state.lexicon)
|
|
302
312
|
const lexicon = stateForm ? state.lexicon : state
|
|
@@ -313,11 +323,11 @@ export function validateAssertions(state, assertions, options = {}) {
|
|
|
313
323
|
const problems = []
|
|
314
324
|
const seen = new Map()
|
|
315
325
|
for (const [index, assertion] of assertions.entries()) {
|
|
316
|
-
for (const item of validateAssertion(lexicon, assertion)) problems.push(`断言 ${index + 1} · ${item}`)
|
|
326
|
+
for (const item of validateAssertion(lexicon, assertion)) problems.push(tr(`断言 ${index + 1} · ${item}`, `Assertion ${index + 1} · ${item}`))
|
|
317
327
|
const key = `${text(assertion?.predicate)}\u0000${subjectKey(assertion)}`
|
|
318
328
|
const value = objectKey(assertion?.object)
|
|
319
329
|
if (seen.has(key) && seen.get(key) !== value) {
|
|
320
|
-
problems.push(problem('assertion_self_conflict', `同一事实里「${text(assertion?.predicate)}」在主体「${text(assertion?.subject?.id)}」上给了两个值:${seen.get(key)} 与 ${value}`))
|
|
330
|
+
problems.push(problem('assertion_self_conflict', tr(`同一事实里「${text(assertion?.predicate)}」在主体「${text(assertion?.subject?.id)}」上给了两个值:${seen.get(key)} 与 ${value}`, `Within one fact, "${text(assertion?.predicate)}" gives two values for subject "${text(assertion?.subject?.id)}": ${seen.get(key)} and ${value}`)))
|
|
321
331
|
}
|
|
322
332
|
seen.set(key, value)
|
|
323
333
|
if (entities === null || legacy) continue
|
|
@@ -341,7 +351,7 @@ export function validateAssertions(state, assertions, options = {}) {
|
|
|
341
351
|
* 换成一个裸对象会让内核当场渲染出 `[object Object]`——那比不报更坏。
|
|
342
352
|
*/
|
|
343
353
|
function subjectUnknown(subject) {
|
|
344
|
-
const message = new String(problem('assert_subject_unknown', `主体「${subject}」还不在实体图上:先登记这个实例(带类型与出处),或在这一批断言里让某个宾语以 instance
|
|
354
|
+
const message = new String(problem('assert_subject_unknown', tr(`主体「${subject}」还不在实体图上:先登记这个实例(带类型与出处),或在这一批断言里让某个宾语以 instance 形态引出它`, `Subject "${subject}" is not on the entity graph yet: write its entity file (with type and source) first, or introduce it as an instance object in this batch of assertions`)))
|
|
345
355
|
message.code = 'assert_subject_unknown'
|
|
346
356
|
message.subject = subject
|
|
347
357
|
return message
|
|
@@ -405,20 +415,20 @@ export function lexiconHealth(lexicon, facts) {
|
|
|
405
415
|
for (const term of normalized.terms) {
|
|
406
416
|
if (text(term.parent) !== '') usedTerms.add(text(term.parent))
|
|
407
417
|
const walk = termChain(normalized, term.id)
|
|
408
|
-
if (walk.cycle !== null) issues.push({ kind: 'cycle', severity: 'warning', id: term.id, detail: `is_a 父链成环(回到「${walk.cycle}」)` })
|
|
409
|
-
else if (walk.missing !== null) issues.push({ kind: 'dangling_parent', severity: 'warning', id: term.id, detail: `父概念「${walk.missing}
|
|
418
|
+
if (walk.cycle !== null) issues.push({ kind: 'cycle', severity: 'warning', id: term.id, detail: tr(`is_a 父链成环(回到「${walk.cycle}」)`, `The is_a chain loops (back to "${walk.cycle}")`) })
|
|
419
|
+
else if (walk.missing !== null) issues.push({ kind: 'dangling_parent', severity: 'warning', id: term.id, detail: tr(`父概念「${walk.missing}」不在词汇里`, `Parent concept "${walk.missing}" is not in the vocabulary`) })
|
|
410
420
|
}
|
|
411
421
|
for (const predicate of normalized.predicates) {
|
|
412
422
|
const domain = text(predicate.domain)
|
|
413
423
|
const range = isPlainObject(predicate.range) ? predicate.range : {}
|
|
414
424
|
if (domain !== '') {
|
|
415
425
|
usedTerms.add(domain)
|
|
416
|
-
if (findEntry(normalized, domain) === null) issues.push({ kind: 'dangling_domain', severity: 'warning', id: predicate.id, detail: `主词域「${domain}
|
|
426
|
+
if (findEntry(normalized, domain) === null) issues.push({ kind: 'dangling_domain', severity: 'warning', id: predicate.id, detail: tr(`主词域「${domain}」不在词汇里`, `Domain "${domain}" is not in the vocabulary`) })
|
|
417
427
|
}
|
|
418
428
|
const rangeTerm = text(range.term)
|
|
419
429
|
if (rangeTerm !== '') {
|
|
420
430
|
usedTerms.add(rangeTerm)
|
|
421
|
-
if (findEntry(normalized, rangeTerm) === null) issues.push({ kind: 'dangling_range', severity: 'warning', id: predicate.id, detail: `宾语域「${rangeTerm}
|
|
431
|
+
if (findEntry(normalized, rangeTerm) === null) issues.push({ kind: 'dangling_range', severity: 'warning', id: predicate.id, detail: tr(`宾语域「${rangeTerm}」不在词汇里`, `Range "${rangeTerm}" is not in the vocabulary`) })
|
|
422
432
|
}
|
|
423
433
|
}
|
|
424
434
|
for (const fact of Array.isArray(facts) ? facts : []) {
|
|
@@ -431,21 +441,21 @@ export function lexiconHealth(lexicon, facts) {
|
|
|
431
441
|
if (type !== '') usedTerms.add(type)
|
|
432
442
|
const predicate = findEntry(normalized, predicateId)
|
|
433
443
|
if (predicate === null) {
|
|
434
|
-
if (!retracted) issues.push({ kind: 'unknown_predicate_in_fact', severity: 'warning', id: String(fact.id ?? ''), detail: `事实里用了没登记的谓词「${predicateId}
|
|
444
|
+
if (!retracted) issues.push({ kind: 'unknown_predicate_in_fact', severity: 'warning', id: String(fact.id ?? ''), detail: tr(`事实里用了没登记的谓词「${predicateId}」`, `A fact uses the unknown predicate "${predicateId}"`) })
|
|
435
445
|
continue
|
|
436
446
|
}
|
|
437
|
-
if (predicate.entry.status === 'deprecated' && !retracted) issues.push({ kind: 'deprecated_in_use', severity: 'info', id: predicateId, detail: `已废止的「${predicateId}」仍被事实「${String(fact.id ?? '')}」引用(记录保留,不再新增)` })
|
|
447
|
+
if (predicate.entry.status === 'deprecated' && !retracted) issues.push({ kind: 'deprecated_in_use', severity: 'info', id: predicateId, detail: tr(`已废止的「${predicateId}」仍被事实「${String(fact.id ?? '')}」引用(记录保留,不再新增)`, `The deprecated "${predicateId}" is still used by fact "${String(fact.id ?? '')}" (the record stays; no new uses)`) })
|
|
438
448
|
}
|
|
439
449
|
}
|
|
440
450
|
for (const term of normalized.terms) {
|
|
441
451
|
if (term.status === 'deprecated') continue
|
|
442
452
|
if (!usedTerms.has(term.id) && !normalized.predicates.some((item) => text(item.domain) === term.id || text(item.range?.term) === term.id)) {
|
|
443
|
-
issues.push({ kind: 'unused', severity: 'info', id: term.id, detail: '还没有任何谓词或事实引用它' })
|
|
453
|
+
issues.push({ kind: 'unused', severity: 'info', id: term.id, detail: tr('还没有任何谓词或事实引用它', 'No predicate or fact uses it yet') })
|
|
444
454
|
}
|
|
445
455
|
}
|
|
446
456
|
for (const predicate of normalized.predicates) {
|
|
447
457
|
if (predicate.status === 'deprecated') continue
|
|
448
|
-
if (!usedPredicates.has(predicate.id)) issues.push({ kind: 'unused', severity: 'info', id: predicate.id, detail: '还没有任何事实用它' })
|
|
458
|
+
if (!usedPredicates.has(predicate.id)) issues.push({ kind: 'unused', severity: 'info', id: predicate.id, detail: tr('还没有任何事实用它', 'No fact uses it yet') })
|
|
449
459
|
}
|
|
450
460
|
return issues.sort((a, b) => (a.kind === b.kind ? (a.id < b.id ? -1 : 1) : a.kind < b.kind ? -1 : 1))
|
|
451
461
|
}
|
|
@@ -453,9 +463,9 @@ export function lexiconHealth(lexicon, facts) {
|
|
|
453
463
|
/**
|
|
454
464
|
* **每个概念被引用了多少次**:断言主体的类型、谓词声明的主词域 / 值域、以及实体断言的类型。
|
|
455
465
|
*
|
|
456
|
-
*
|
|
457
|
-
*
|
|
458
|
-
* 同一件事两种读数;所以引用面只在这里定义一次,两边都读它。
|
|
466
|
+
* 为什么要单独一份判据:「零引用」这件事有两处读者——货架上「零引用的概念」那一节,
|
|
467
|
+
* 与同一份货架里概念那一列的引用数。两处各算一遍,迟早会出现「引用 0 却没进零引用一节」这种
|
|
468
|
+
* 同一件事两种读数;所以引用面只在这里定义一次,两边都读它。(第四阶段起它不再是缺口,只是货架读数。)
|
|
459
469
|
*/
|
|
460
470
|
export function termUsage(lexicon, facts, entityAssertions = []) {
|
|
461
471
|
const normalized = normalizeLexicon(lexicon)
|
|
@@ -579,6 +589,19 @@ export function graphProjection(state) {
|
|
|
579
589
|
facts: [],
|
|
580
590
|
})
|
|
581
591
|
}
|
|
592
|
+
/**
|
|
593
|
+
* **组成 / 属于**:实体文件放在另一个实体的目录里(`line_a/kiln_3.json`)就是它的一部分。
|
|
594
|
+
* 节点带上 `container`(界面据此画可折叠的子图),同时出一条 `part_of` 边。
|
|
595
|
+
*/
|
|
596
|
+
const byRef = new Map([...instances.values()].map((node) => [node.ref, node]))
|
|
597
|
+
for (const entity of Array.isArray(state?.entities) ? state.entities : []) {
|
|
598
|
+
const parent = text(entity?.parent)
|
|
599
|
+
if (parent === '' || !byRef.has(parent)) continue
|
|
600
|
+
const node = instances.get(`${text(entity.type)}|${text(entity.id)}`)
|
|
601
|
+
if (node === undefined) continue
|
|
602
|
+
node.container = byRef.get(parent).id
|
|
603
|
+
edges.push({ id: `part_of:${node.id}`, kind: 'part_of', layer: 'entity', source: 'registered', predicate: null, label: tr('属于', 'part of'), from: node.id, to: node.container, status: 'asserted', level: null, fact: null, scope: null, claim: null })
|
|
604
|
+
}
|
|
582
605
|
}
|
|
583
606
|
for (const fact of facts) {
|
|
584
607
|
const status = fact?.review?.decision === 'retracted' ? 'retracted' : fact?.refuted === true ? 'refuted' : 'live'
|
|
@@ -828,7 +851,7 @@ export function applyLexiconMutation(lexicon, mutation, at) {
|
|
|
828
851
|
* 读的人分不出哪些是可复用的语言、哪些是一次具体的记录。每节开头一句「这一节是什么」。
|
|
829
852
|
*
|
|
830
853
|
* **零引用的概念单独一节**:注册了却没有任何结论引用它,那它还是约定、不是已知。
|
|
831
|
-
* 这一节是这句话的可见面(
|
|
854
|
+
* 这一节是这句话的可见面(只是货架读数,不是缺口)。
|
|
832
855
|
*
|
|
833
856
|
* 第一个参数收两种形状:整份**状态**(推荐,个体那一节要有实体面)或旧的**词汇**;
|
|
834
857
|
* `options.view` 传 `knowledge-view.js` 的 `knowledgeView()` 输出时,「使用」一节读的就是
|
|
@@ -850,7 +873,7 @@ export function describeDomainShelf(lexiconOrState, facts, hypotheses = [], opti
|
|
|
850
873
|
const entityNodes = state === null ? [] : graphProjection(state).nodes.filter((node) => node.layer === 'entity' && node.kind === 'instance')
|
|
851
874
|
/**
|
|
852
875
|
* **引用面读同一份判据**:概念那一列用 `termUsage`(断言主体 + 主词域 / 值域),
|
|
853
|
-
*
|
|
876
|
+
* 与零引用那一节完全同源——否则会出现「引用 0 却没进零引用一节」
|
|
854
877
|
* 这种同一张表里两种读数打架的形状。谓词那一列数的是断言条数。
|
|
855
878
|
*/
|
|
856
879
|
const termRefs = termUsage(normalized, rows, entityAssertions)
|
|
@@ -866,7 +889,7 @@ export function describeDomainShelf(lexiconOrState, facts, hypotheses = [], opti
|
|
|
866
889
|
if (predicate !== '') predicateRefs.set(predicate, (predicateRefs.get(predicate) ?? 0) + 1)
|
|
867
890
|
}
|
|
868
891
|
const conflicts = deriveConflicts(rows, normalized)
|
|
869
|
-
/**
|
|
892
|
+
/** 零引用的概念:与概念那一列读**同一份**引用面。 */
|
|
870
893
|
const orphans = terms.filter((term) => term.status !== 'deprecated' && (termRefs.get(term.id) ?? 0) === 0)
|
|
871
894
|
const lines = [
|
|
872
895
|
'# 领域本体(项目词汇)',
|
|
@@ -891,7 +914,7 @@ export function describeDomainShelf(lexiconOrState, facts, hypotheses = [], opti
|
|
|
891
914
|
}
|
|
892
915
|
lines.push(`## 个体(实例)(${entityNodes.length})`, '', '> 这一节是**具体物**:某个实例在某出处下成立——它带类型与出处,不是约定,也不能再当概念用。', '')
|
|
893
916
|
if (entityNodes.length === 0) {
|
|
894
|
-
lines.push('(
|
|
917
|
+
lines.push('(没有实例。在 `clear/ontology/entities/` 下写一个实体文件,它的 `relations` 让它在图上长出边——只有节点没有边,图仍然是空的。)', '')
|
|
895
918
|
} else {
|
|
896
919
|
lines.push('| id | 名称 | 类型 | 来源 | 依据 / 出处 |', '|---|---|---|---|---|')
|
|
897
920
|
for (const node of entityNodes) {
|
|
@@ -998,3 +1021,484 @@ export function describeDomainGraph(lexicon) {
|
|
|
998
1021
|
lines.push('```', '')
|
|
999
1022
|
return lines.join('\n')
|
|
1000
1023
|
}
|
|
1024
|
+
|
|
1025
|
+
// ═══ 工作区文件:事实与本体住在项目里 ═══════════════════════════════════════
|
|
1026
|
+
//
|
|
1027
|
+
// 会话账本只活在一次会话里,而研究要跨会话攒下来。所以「攒下来的东西」住在工作区的文件里:
|
|
1028
|
+
// · `clear/knowledge/facts/<id>.json`:升格的事实,一条一个文件,只有系统写;
|
|
1029
|
+
// · `clear/ontology/{concepts,relations,entities}/**.json`:本体,模型用原生文件工具直接写。
|
|
1030
|
+
// 内核每一拍把这两处的变化折成一条 `workspace/synced` 变更,投影从文件内容派生词汇、实体与
|
|
1031
|
+
// 事实——折法仍然只吃账本,文件是账本之外唯一的输入,而且进账本时就是一条可重放的事实。
|
|
1032
|
+
|
|
1033
|
+
/** 事实文件所在的目录(相对工作区)。 */
|
|
1034
|
+
export const FACTS_DIR = 'clear/knowledge/facts'
|
|
1035
|
+
/** 本体文件树的根(相对工作区)。 */
|
|
1036
|
+
export const ONTOLOGY_DIR = 'clear/ontology'
|
|
1037
|
+
/** 本体文件树的三支:目录名 → 文件种类。 */
|
|
1038
|
+
export const ONTOLOGY_BRANCHES = { concepts: 'concept', relations: 'relation', entities: 'entity' }
|
|
1039
|
+
/** 读面有界:文件数与单个文件的字节数都有上限,超出的如实报成问题,不静默截断。 */
|
|
1040
|
+
export const WORKSPACE_LIMITS = { files: 2000, bytes: 65536 }
|
|
1041
|
+
|
|
1042
|
+
/** FNV-1a(32 位):浏览器与宿主两侧都能算的同步散列,只用来比「变没变」。 */
|
|
1043
|
+
function hashText(value) {
|
|
1044
|
+
let h = 0x811c9dc5
|
|
1045
|
+
const source = String(value)
|
|
1046
|
+
for (let index = 0; index < source.length; index += 1) {
|
|
1047
|
+
h ^= source.charCodeAt(index)
|
|
1048
|
+
h = Math.imul(h, 0x01000193) >>> 0
|
|
1049
|
+
}
|
|
1050
|
+
return h.toString(16).padStart(8, '0')
|
|
1051
|
+
}
|
|
1052
|
+
|
|
1053
|
+
/**
|
|
1054
|
+
* 一个词条**含义**的指纹:概念看释义与父概念,谓词看主词域、值域与单值性。
|
|
1055
|
+
* 名字、别名、依据不进指纹——改名不改义,事实不必因此复核。
|
|
1056
|
+
*/
|
|
1057
|
+
export function definitionFingerprint(entry) {
|
|
1058
|
+
if (!isPlainObject(entry)) return null
|
|
1059
|
+
const meaning = 'range' in entry || 'domain' in entry ? { domain: text(entry.domain), range: isPlainObject(entry.range) ? { term: text(entry.range.term), form: text(entry.range.form), unit: text(entry.range.unit) } : null, functional: entry.functional === true } : { gloss: text(entry.gloss), parent: text(entry.parent) }
|
|
1060
|
+
return `fnv:${hashText(JSON.stringify(meaning))}`
|
|
1061
|
+
}
|
|
1062
|
+
|
|
1063
|
+
/** 一组断言用到的词条(谓词、主体类型、宾语类型),各自此刻的含义指纹。 */
|
|
1064
|
+
export function fingerprintDefinitions(lexicon, assertions, options = {}) {
|
|
1065
|
+
const normalized = normalizeLexicon(lexicon)
|
|
1066
|
+
const out = {}
|
|
1067
|
+
const take = (id) => {
|
|
1068
|
+
const key = text(id)
|
|
1069
|
+
if (key === '' || key in out) return
|
|
1070
|
+
const found = findEntry(normalized, key)
|
|
1071
|
+
if (found !== null) out[key] = definitionFingerprint(found.entry)
|
|
1072
|
+
}
|
|
1073
|
+
for (const assertion of Array.isArray(assertions) ? assertions : []) {
|
|
1074
|
+
take(assertion?.predicate)
|
|
1075
|
+
take(assertion?.subject?.type)
|
|
1076
|
+
take(assertion?.object?.type)
|
|
1077
|
+
}
|
|
1078
|
+
/**
|
|
1079
|
+
* **按词面挂上**(`options.text`):真跑里模型几乎不给判断写结构化断言,
|
|
1080
|
+
* 只靠断言的话事实与本体是两张互不引用的表,「定义已变」永远触发不了。
|
|
1081
|
+
* 所以升格时也按主张原文里出现的词(id / 名字 / 别名,去空白、不分大小写、至少两个字)
|
|
1082
|
+
* 把概念与关系挂上——与知识预检同一条词面规则,有界、可复核。
|
|
1083
|
+
*/
|
|
1084
|
+
const said = text(options?.text).replace(/\s+/g, '').toLowerCase()
|
|
1085
|
+
if (said !== '') {
|
|
1086
|
+
for (const entry of [...(normalized.terms ?? []), ...(normalized.predicates ?? [])]) {
|
|
1087
|
+
if (entry?.status === 'deprecated') continue
|
|
1088
|
+
const names = [entry?.id, entry?.label, ...(Array.isArray(entry?.aliases) ? entry.aliases : [])].map((name) => text(name).replace(/\s+/g, '').toLowerCase()).filter((name) => name.length >= 2)
|
|
1089
|
+
if (names.some((name) => said.includes(name))) take(entry.id)
|
|
1090
|
+
}
|
|
1091
|
+
}
|
|
1092
|
+
return out
|
|
1093
|
+
}
|
|
1094
|
+
|
|
1095
|
+
/** 事实升格以后,它用到的词条里**含义改过或已不在**的那些 id。 */
|
|
1096
|
+
export function changedDefinitions(lexicon, definitions) {
|
|
1097
|
+
if (!isPlainObject(definitions)) return []
|
|
1098
|
+
const normalized = normalizeLexicon(lexicon)
|
|
1099
|
+
const changed = []
|
|
1100
|
+
for (const [id, fingerprint] of Object.entries(definitions)) {
|
|
1101
|
+
const found = findEntry(normalized, id)
|
|
1102
|
+
if (found === null || definitionFingerprint(found.entry) !== fingerprint) changed.push(id)
|
|
1103
|
+
}
|
|
1104
|
+
return changed
|
|
1105
|
+
}
|
|
1106
|
+
|
|
1107
|
+
/**
|
|
1108
|
+
* 工作区路径 → 它是哪种文件。认不出的返回 null(不是本体也不是事实,不归投影管)。
|
|
1109
|
+
* `dirs` 是本体文件在它那一支里的上级目录(由近及远的反序:最后一个就是直接上级)。
|
|
1110
|
+
*/
|
|
1111
|
+
export function classifyWorkspacePath(path) {
|
|
1112
|
+
const parts = String(path ?? '').replace(/\\/g, '/').split('/').filter((part) => part !== '' && part !== '.')
|
|
1113
|
+
if (parts.length === 0 || !parts.at(-1).endsWith('.json')) return null
|
|
1114
|
+
const id = parts.at(-1).slice(0, -'.json'.length)
|
|
1115
|
+
const facts = FACTS_DIR.split('/')
|
|
1116
|
+
if (parts.length === facts.length + 1 && facts.every((part, index) => parts[index] === part)) return { kind: 'fact', id, dirs: [] }
|
|
1117
|
+
const root = ONTOLOGY_DIR.split('/')
|
|
1118
|
+
if (parts.length >= root.length + 2 && root.every((part, index) => parts[index] === part)) {
|
|
1119
|
+
const kind = ONTOLOGY_BRANCHES[parts[root.length]]
|
|
1120
|
+
if (kind === undefined) return null
|
|
1121
|
+
return { kind, id, dirs: parts.slice(root.length + 1, -1) }
|
|
1122
|
+
}
|
|
1123
|
+
return null
|
|
1124
|
+
}
|
|
1125
|
+
|
|
1126
|
+
/**
|
|
1127
|
+
* 事实文件 → 事实行(与 `fact/promoted` 折出来的同形)。复核写在文件的 `status` / `review` 上。
|
|
1128
|
+
*/
|
|
1129
|
+
export function factFromFile(data, path) {
|
|
1130
|
+
if (!isPlainObject(data) || text(data.id) === '' || text(data.text) === '') return null
|
|
1131
|
+
const review = isPlainObject(data.review) ? { decision: data.review.decision === 'retracted' ? 'retracted' : 'kept', reason: data.review.reason ?? null, at: data.review.at ?? null, by: data.review.by ?? 'user' } : null
|
|
1132
|
+
return {
|
|
1133
|
+
id: text(data.id),
|
|
1134
|
+
goal: data.goal ?? null,
|
|
1135
|
+
hypothesis: data.hypothesis ?? null,
|
|
1136
|
+
text: String(data.text),
|
|
1137
|
+
scope: data.scope ?? null,
|
|
1138
|
+
level: data.level ?? null,
|
|
1139
|
+
evidence: Array.isArray(data.evidence) ? data.evidence : [],
|
|
1140
|
+
assertions: Array.isArray(data.assertions) ? clone(data.assertions) : null,
|
|
1141
|
+
definitions: isPlainObject(data.definitions) ? clone(data.definitions) : null,
|
|
1142
|
+
path: path ?? null,
|
|
1143
|
+
at: typeof data.at === 'number' ? data.at : null,
|
|
1144
|
+
review,
|
|
1145
|
+
}
|
|
1146
|
+
}
|
|
1147
|
+
|
|
1148
|
+
// ═══ 本体文件树:字段、单文件校验、从文件折出词汇与实体 ═════════════════════════
|
|
1149
|
+
//
|
|
1150
|
+
// 本体由模型用原生文件工具直接写,住在 `clear/ontology/{concepts,relations,entities}/` 下:
|
|
1151
|
+
// · 命名规则只有一条:`X.json` 描述节点 X,它的子节点放在同级的 `X/` 目录里;
|
|
1152
|
+
// · 概念目录嵌套 = is_a,实体目录嵌套 = 组成 / 属于,关系目录只是分组;
|
|
1153
|
+
// · 身份是 id(= 文件名),位置是目录。引用只用 id,所以整支目录挪走就是重新分层,引用不断。
|
|
1154
|
+
// 校验分三道:写入时查**单个文件**(这里的 `checkOntologyFile`,不过就拒写);读取时查**跨文件**
|
|
1155
|
+
// (`materializeOntology` 的 problems,只提示、有问题的节点或边不进图);升格时把断言涉及的
|
|
1156
|
+
// 节点全查一遍(就是 `validateAssertions`,不过就不升格)。
|
|
1157
|
+
|
|
1158
|
+
/** 实体 id 比概念宽:具体物的名字常带大写与短横(`V-JEPA_2`),但仍是一个文件名能装下的键。 */
|
|
1159
|
+
const ENTITY_ID_PATTERN = /^[A-Za-z0-9][A-Za-z0-9_-]{0,59}$/
|
|
1160
|
+
const PROVENANCE_KINDS = ['url', 'named', 'backref']
|
|
1161
|
+
const ONTOLOGY_STATUS = ['active', 'deprecated']
|
|
1162
|
+
|
|
1163
|
+
/**
|
|
1164
|
+
* 三种本体文件的字段定义。内核把它铺成 `clear/ontology/SCHEMA.json`(只读),
|
|
1165
|
+
* 写入时的校验与这份定义是同一份:字段在这里改,两边一起变。
|
|
1166
|
+
*/
|
|
1167
|
+
export const ONTOLOGY_SCHEMA = bilingual({
|
|
1168
|
+
about: [
|
|
1169
|
+
'本体文件的字段定义(系统铺设,只读)。命名规则:X.json 描述节点 X,它的子节点放在同级的 X/ 目录里;id 必须等于文件名。概念目录嵌套 = is_a;实体目录嵌套 = 组成 / 属于;关系可以平铺,也可以分目录(只是分组)。引用一律只写 id,挪目录不断引用。写入时只查单个文件;引用断了、类型对不上等跨文件问题在卡片上提示,升格时才拦。',
|
|
1170
|
+
'Field definitions for ontology files (laid down by the system, read-only). Naming: X.json describes node X, and its children live in the sibling X/ directory; id must equal the file name. Nested concept directories mean is_a; nested entity directories mean part-of; relations can be flat or grouped in directories (grouping only). References use ids only, so moving directories never breaks them. Writes check a single file; cross-file problems such as broken references or type mismatches are flagged on the card and only block at promotion.',
|
|
1171
|
+
],
|
|
1172
|
+
concept: {
|
|
1173
|
+
where: 'clear/ontology/concepts/**/<id>.json',
|
|
1174
|
+
required: {
|
|
1175
|
+
id: ['slug:小写字母开头,字母/数字/下划线,≤40,等于文件名', 'slug: starts with a lowercase letter; letters, digits, underscores; ≤40; equals the file name'],
|
|
1176
|
+
label: ['给人看的名字', 'the human-readable name'],
|
|
1177
|
+
gloss: ['一句话释义:它指什么', 'one-sentence gloss: what it refers to'],
|
|
1178
|
+
},
|
|
1179
|
+
optional: {
|
|
1180
|
+
aliases: ['别名数组', 'array of aliases'],
|
|
1181
|
+
basis: ['依据:哪份材料让这个词成立', 'basis: which material establishes this term'],
|
|
1182
|
+
status: 'active | deprecated',
|
|
1183
|
+
replaced_by: ['废止后由哪个 id 接替', 'the id that replaces it once deprecated'],
|
|
1184
|
+
note: ['备注', 'note'],
|
|
1185
|
+
},
|
|
1186
|
+
example: {
|
|
1187
|
+
id: 'jepa',
|
|
1188
|
+
label: ['联合嵌入预测架构', 'Joint-embedding predictive architecture'],
|
|
1189
|
+
aliases: ['JEPA'],
|
|
1190
|
+
gloss: ['在表示空间里从上下文预测目标的表示,不还原像素', 'predicts the representation of a target from context in representation space, without reconstructing pixels'],
|
|
1191
|
+
basis: 'LeCun 2022',
|
|
1192
|
+
},
|
|
1193
|
+
},
|
|
1194
|
+
relation: {
|
|
1195
|
+
where: 'clear/ontology/relations/**/<id>.json',
|
|
1196
|
+
required: {
|
|
1197
|
+
id: ['slug,同概念(概念与关系共用一个 id 空间)', 'slug, as for concepts (concepts and relations share one id space)'],
|
|
1198
|
+
label: ['给人看的名字', 'the human-readable name'],
|
|
1199
|
+
range: [
|
|
1200
|
+
`宾语是哪种东西:一个概念 id(宾语是该概念的实体),或 {"form": ${VALUE_FORMS.map((form) => `"${form}"`).join(' | ')}, "unit"?: "..."}(宾语是字面值)`,
|
|
1201
|
+
`what the object is: a concept id (the object is an entity of that concept), or {"form": ${VALUE_FORMS.map((form) => `"${form}"`).join(' | ')}, "unit"?: "..."} (the object is a literal value)`,
|
|
1202
|
+
],
|
|
1203
|
+
},
|
|
1204
|
+
optional: {
|
|
1205
|
+
gloss: ['一句话释义', 'one-sentence gloss'],
|
|
1206
|
+
domain: ['主语必须是哪个概念(下位概念也算)', 'which concept the subject must be (subconcepts count)'],
|
|
1207
|
+
functional: ['true = 单值:同一主语只能有一个取值,两个不同取值会被报成冲突', 'true = single-valued: one value per subject; two different values are reported as a conflict'],
|
|
1208
|
+
basis: ['依据', 'basis'],
|
|
1209
|
+
status: 'active | deprecated',
|
|
1210
|
+
replaced_by: ['接替的 id', 'the replacing id'],
|
|
1211
|
+
note: ['备注', 'note'],
|
|
1212
|
+
},
|
|
1213
|
+
example: { id: 'derived_from', label: ['衍生自', 'derived from'], domain: 'method', range: 'method', functional: false },
|
|
1214
|
+
},
|
|
1215
|
+
entity: {
|
|
1216
|
+
where: 'clear/ontology/entities/**/<id>.json',
|
|
1217
|
+
required: {
|
|
1218
|
+
id: ['字母或数字开头,字母/数字/下划线/短横,≤60,等于文件名', 'starts with a letter or digit; letters, digits, underscores, hyphens; ≤60; equals the file name'],
|
|
1219
|
+
label: ['给人看的名字', 'the human-readable name'],
|
|
1220
|
+
type: ['它是哪个概念的实体(概念 id)', 'which concept it is an entity of (concept id)'],
|
|
1221
|
+
basis: ['依据:哪份材料让它可以被指认', 'basis: which material lets it be identified'],
|
|
1222
|
+
provenance: [
|
|
1223
|
+
`出处 {"kind": ${PROVENANCE_KINDS.map((kind) => `"${kind}"`).join(' | ')}, "ref": "链接 / 文献名 / 工作区里的文件"}`,
|
|
1224
|
+
`source {"kind": ${PROVENANCE_KINDS.map((kind) => `"${kind}"`).join(' | ')}, "ref": "link / paper name / file in the workspace"}`,
|
|
1225
|
+
],
|
|
1226
|
+
},
|
|
1227
|
+
optional: {
|
|
1228
|
+
aliases: ['别名数组', 'array of aliases'],
|
|
1229
|
+
relations: [
|
|
1230
|
+
'它对外的关系,每条一个出处:{"predicate": 关系 id, "object": 另一个实体的 id} 或 {"predicate": 关系 id, "value": 字面值, "unit"?: 单位};两种都要带 "evidence": {"kind", "ref"}',
|
|
1231
|
+
'its outgoing relations, each with a source: {"predicate": relation id, "object": another entity id} or {"predicate": relation id, "value": literal, "unit"?: unit}; both need "evidence": {"kind", "ref"}',
|
|
1232
|
+
],
|
|
1233
|
+
note: ['备注', 'note'],
|
|
1234
|
+
},
|
|
1235
|
+
example: {
|
|
1236
|
+
id: 'v_jepa_2_ac',
|
|
1237
|
+
label: 'V-JEPA 2-AC',
|
|
1238
|
+
type: 'world_model',
|
|
1239
|
+
basis: ['动作条件的 V-JEPA 2', 'action-conditioned V-JEPA 2'],
|
|
1240
|
+
provenance: { kind: 'named', ref: 'sources/vjepa2.md' },
|
|
1241
|
+
relations: [
|
|
1242
|
+
{ predicate: 'derived_from', object: 'v_jepa_2', evidence: { kind: 'named', ref: ['V-JEPA 2 论文', 'V-JEPA 2 paper'] } },
|
|
1243
|
+
{ predicate: 'released_year', value: 2025, evidence: { kind: 'named', ref: ['V-JEPA 2 论文', 'V-JEPA 2 paper'] } },
|
|
1244
|
+
],
|
|
1245
|
+
},
|
|
1246
|
+
},
|
|
1247
|
+
})
|
|
1248
|
+
|
|
1249
|
+
const FIELDS = {
|
|
1250
|
+
concept: ['id', 'label', 'gloss', 'aliases', 'basis', 'status', 'replaced_by', 'note'],
|
|
1251
|
+
relation: ['id', 'label', 'gloss', 'domain', 'range', 'functional', 'basis', 'status', 'replaced_by', 'note'],
|
|
1252
|
+
entity: ['id', 'label', 'type', 'basis', 'provenance', 'aliases', 'relations', 'note'],
|
|
1253
|
+
}
|
|
1254
|
+
const RELATION_FIELDS = ['predicate', 'object', 'value', 'unit', 'evidence', 'note']
|
|
1255
|
+
|
|
1256
|
+
/** 一个出处对象的形状问题(没有就返回 null)。 */
|
|
1257
|
+
function evidenceShape(value, at) {
|
|
1258
|
+
if (!isPlainObject(value)) return tr(`${at}要是 {"kind", "ref"} 对象`, `${at} must be a {"kind", "ref"} object`)
|
|
1259
|
+
if (!PROVENANCE_KINDS.includes(text(value.kind))) return tr(`${at}.kind 只能是 ${PROVENANCE_KINDS.join(' / ')}`, `${at}.kind must be ${PROVENANCE_KINDS.join(' / ')}`)
|
|
1260
|
+
if (text(value.ref) === '') return tr(`${at}.ref 不能为空`, `${at}.ref cannot be empty`)
|
|
1261
|
+
return null
|
|
1262
|
+
}
|
|
1263
|
+
|
|
1264
|
+
/**
|
|
1265
|
+
* **第一道校验:单个文件**。给路径与解析好的内容(或原文),返回问题清单(空 = 通过)。
|
|
1266
|
+
* 只查这一个文件自己:JSON 能解析、字段齐且类型对、id 等于文件名、枚举取值合法。
|
|
1267
|
+
* 引用指向的东西在不在**不查**——模型改一组文件时中间状态必然暂时不一致,
|
|
1268
|
+
* 这时就查跨文件引用,等于逼它按固定顺序写。
|
|
1269
|
+
*/
|
|
1270
|
+
export function checkOntologyFile(path, content) {
|
|
1271
|
+
const place = classifyWorkspacePath(path)
|
|
1272
|
+
if (place === null || place.kind === 'fact') return [tr(`不是本体文件的位置:本体文件要放在 ${ONTOLOGY_DIR}/{concepts,relations,entities}/ 下,扩展名 .json`, `Not an ontology file location: ontology files go under ${ONTOLOGY_DIR}/{concepts,relations,entities}/ with a .json extension`)]
|
|
1273
|
+
let data = content
|
|
1274
|
+
if (typeof content === 'string') {
|
|
1275
|
+
try {
|
|
1276
|
+
data = JSON.parse(content)
|
|
1277
|
+
} catch (error) {
|
|
1278
|
+
return [tr(`不是合法 JSON:${String(error?.message ?? error).slice(0, 160)}`, `Not valid JSON: ${String(error?.message ?? error).slice(0, 160)}`)]
|
|
1279
|
+
}
|
|
1280
|
+
}
|
|
1281
|
+
if (!isPlainObject(data)) return [tr('文件内容要是一个 JSON 对象', 'The file content must be a JSON object')]
|
|
1282
|
+
const kind = place.kind
|
|
1283
|
+
const problems = []
|
|
1284
|
+
const pattern = kind === 'entity' ? ENTITY_ID_PATTERN : ID_PATTERN
|
|
1285
|
+
if (!pattern.test(place.id)) problems.push(
|
|
1286
|
+
kind === 'entity'
|
|
1287
|
+
? tr(`文件名「${place.id}」不能当实体 id:字母或数字开头,只用字母/数字/下划线/短横,≤60`, `File name "${place.id}" cannot be an entity id: start with a letter or digit; letters, digits, underscores, hyphens only; ≤60`)
|
|
1288
|
+
: tr(`文件名「${place.id}」不能当 id:小写字母开头,只用小写字母/数字/下划线,≤40`, `File name "${place.id}" cannot be an id: start with a lowercase letter; lowercase letters, digits, underscores only; ≤40`),
|
|
1289
|
+
)
|
|
1290
|
+
if (data.id !== undefined && text(data.id) !== place.id) problems.push(tr(`id「${text(data.id)}」要等于文件名「${place.id}」(身份就是文件名;改 id 就是改文件名)`, `id "${text(data.id)}" must equal the file name "${place.id}" (identity is the file name; changing the id means renaming the file)`))
|
|
1291
|
+
const extra = Object.keys(data).filter((key) => !FIELDS[kind].includes(key))
|
|
1292
|
+
if (extra.length > 0) {
|
|
1293
|
+
const kindWord = kind === 'concept' ? tr('概念', 'concepts') : kind === 'relation' ? tr('关系', 'relations') : tr('实体', 'entities')
|
|
1294
|
+
problems.push(tr(`不认识的字段:${extra.join('、')}(${kindWord}可用:${FIELDS[kind].join('、')};详见 ${ONTOLOGY_DIR}/SCHEMA.json)`, `Unknown fields: ${extra.join(', ')} (${kindWord} allow: ${FIELDS[kind].join(', ')}; see ${ONTOLOGY_DIR}/SCHEMA.json)`))
|
|
1295
|
+
}
|
|
1296
|
+
if (text(data.label) === '') problems.push(tr('label 必填:给人看的名字', 'label is required: the human-readable name'))
|
|
1297
|
+
if (data.aliases !== undefined && (!Array.isArray(data.aliases) || data.aliases.some((alias) => typeof alias !== 'string'))) problems.push(tr('aliases 只能是字符串数组', 'aliases must be an array of strings'))
|
|
1298
|
+
for (const key of ['basis', 'gloss', 'note', 'replaced_by']) if (data[key] !== undefined && data[key] !== null && typeof data[key] !== 'string') problems.push(tr(`${key} 只能是字符串`, `${key} must be a string`))
|
|
1299
|
+
if (data.status !== undefined && !ONTOLOGY_STATUS.includes(text(data.status))) problems.push(tr(`status 只能是 ${ONTOLOGY_STATUS.join(' / ')}`, `status must be ${ONTOLOGY_STATUS.join(' / ')}`))
|
|
1300
|
+
if (kind === 'concept' && text(data.gloss) === '') problems.push(tr('gloss 必填:一句话说清它指什么,不然引用它的人各读各的', 'gloss is required: one sentence on what it refers to, or everyone who cites it reads it differently'))
|
|
1301
|
+
if (kind === 'relation') {
|
|
1302
|
+
if (data.domain !== undefined && data.domain !== null && !ID_PATTERN.test(text(data.domain))) problems.push(tr('domain 要是一个概念 id', 'domain must be a concept id'))
|
|
1303
|
+
if (data.functional !== undefined && typeof data.functional !== 'boolean') problems.push(tr('functional 只能是 true / false', 'functional must be true / false'))
|
|
1304
|
+
const range = data.range
|
|
1305
|
+
if (typeof range === 'string') {
|
|
1306
|
+
if (!ID_PATTERN.test(text(range))) problems.push(tr('range 写成字符串时要是一个概念 id', 'range given as a string must be a concept id'))
|
|
1307
|
+
} else if (isPlainObject(range)) {
|
|
1308
|
+
const form = text(range.form)
|
|
1309
|
+
const term = text(range.term)
|
|
1310
|
+
if (form !== '' && term !== '') problems.push(tr('range 只能二选一:一个概念 id,或 {"form"}', 'range is one or the other: a concept id, or {"form"}'))
|
|
1311
|
+
else if (term !== '') {
|
|
1312
|
+
if (!ID_PATTERN.test(term)) problems.push(tr('range.term 要是一个概念 id', 'range.term must be a concept id'))
|
|
1313
|
+
} else if (!VALUE_FORMS.includes(form)) problems.push(tr(`range.form 只能是 ${VALUE_FORMS.join(' / ')}`, `range.form must be ${VALUE_FORMS.join(' / ')}`))
|
|
1314
|
+
if (range.unit !== undefined && typeof range.unit !== 'string') problems.push(tr('range.unit 只能是字符串', 'range.unit must be a string'))
|
|
1315
|
+
} else problems.push(tr(`range 必填:一个概念 id(宾语是实体),或 {"form": ${VALUE_FORMS.join(' | ')}}(宾语是字面值)`, `range is required: a concept id (the object is an entity), or {"form": ${VALUE_FORMS.join(' | ')}} (the object is a literal value)`))
|
|
1316
|
+
}
|
|
1317
|
+
if (kind === 'entity') {
|
|
1318
|
+
if (!ID_PATTERN.test(text(data.type))) problems.push(tr('type 必填:它是哪个概念的实体(概念 id)', 'type is required: which concept it is an entity of (concept id)'))
|
|
1319
|
+
if (text(data.basis) === '') problems.push(tr('basis 必填:实体是观测,要说清哪份材料让它可以被指认', 'basis is required: an entity is an observation, so say which material identifies it'))
|
|
1320
|
+
const shape = evidenceShape(data.provenance, 'provenance')
|
|
1321
|
+
if (shape !== null) problems.push(tr(`${shape}(实体没有出处就进不了图)`, `${shape} (an entity without a source cannot enter the graph)`))
|
|
1322
|
+
if (data.relations !== undefined) {
|
|
1323
|
+
if (!Array.isArray(data.relations)) problems.push(tr('relations 只能是数组', 'relations must be an array'))
|
|
1324
|
+
else
|
|
1325
|
+
data.relations.forEach((relation, index) => {
|
|
1326
|
+
const at = `relations[${index}]`
|
|
1327
|
+
if (!isPlainObject(relation)) return problems.push(tr(`${at} 要是对象`, `${at} must be an object`))
|
|
1328
|
+
const unknown = Object.keys(relation).filter((key) => !RELATION_FIELDS.includes(key))
|
|
1329
|
+
if (unknown.length > 0) problems.push(tr(`${at} 不认识的字段:${unknown.join('、')}(可用:${RELATION_FIELDS.join('、')})`, `${at} has unknown fields: ${unknown.join(', ')} (allowed: ${RELATION_FIELDS.join(', ')})`))
|
|
1330
|
+
if (!ID_PATTERN.test(text(relation.predicate))) problems.push(tr(`${at}.predicate 要是一个关系 id`, `${at}.predicate must be a relation id`))
|
|
1331
|
+
const hasObject = relation.object !== undefined
|
|
1332
|
+
const hasValue = relation.value !== undefined
|
|
1333
|
+
if (hasObject === hasValue) problems.push(tr(`${at} 要么给 object(另一个实体的 id),要么给 value(字面值),二选一`, `${at} needs exactly one of object (another entity id) or value (a literal)`))
|
|
1334
|
+
else if (hasObject && !ENTITY_ID_PATTERN.test(text(relation.object))) problems.push(tr(`${at}.object 要是一个实体 id`, `${at}.object must be an entity id`))
|
|
1335
|
+
if (relation.unit !== undefined && typeof relation.unit !== 'string') problems.push(tr(`${at}.unit 只能是字符串`, `${at}.unit must be a string`))
|
|
1336
|
+
const evidence = evidenceShape(relation.evidence, `${at}.evidence`)
|
|
1337
|
+
if (evidence !== null) problems.push(tr(`${evidence}(一条关系一个出处;没有出处的话是意见,不是观测)`, `${evidence} (one source per relation; without a source it is an opinion, not an observation)`))
|
|
1338
|
+
})
|
|
1339
|
+
}
|
|
1340
|
+
}
|
|
1341
|
+
return problems
|
|
1342
|
+
}
|
|
1343
|
+
|
|
1344
|
+
/** 关系文件的值域 → 词汇里的值域形状(`{term}` 或 `{form, unit?}`)。 */
|
|
1345
|
+
function rangeOf(range) {
|
|
1346
|
+
if (typeof range === 'string') return { term: text(range) }
|
|
1347
|
+
if (!isPlainObject(range)) return null
|
|
1348
|
+
if (text(range.term) !== '') return { term: text(range.term) }
|
|
1349
|
+
return text(range.unit) === '' ? { form: text(range.form) } : { form: text(range.form), unit: text(range.unit) }
|
|
1350
|
+
}
|
|
1351
|
+
|
|
1352
|
+
/**
|
|
1353
|
+
* **从工作区文件折出本体**(第二道校验在这里):词汇(概念 + 关系)、实体、实体断言、问题清单。
|
|
1354
|
+
*
|
|
1355
|
+
* 读的是 `state.workspace.files`(路径 → `{digest, data|error}`),所以它是纯函数:
|
|
1356
|
+
* 同一批文件永远折出同一张图。跨文件的问题只**提示**,有问题的节点或边不进图——
|
|
1357
|
+
* 图上画出来的每一样东西都是这门语言认得的。
|
|
1358
|
+
*/
|
|
1359
|
+
export function materializeOntology(files) {
|
|
1360
|
+
const problems = []
|
|
1361
|
+
const flag = (path, id, code, detail, severity = 'warning') => problems.push({ path, id, code, detail, severity })
|
|
1362
|
+
const entries = Object.entries(isPlainObject(files) ? files : {})
|
|
1363
|
+
.map(([path, file]) => ({ path, file, place: classifyWorkspacePath(path) }))
|
|
1364
|
+
.filter((item) => item.place !== null && item.place.kind !== 'fact')
|
|
1365
|
+
.sort((a, b) => (a.path < b.path ? -1 : 1))
|
|
1366
|
+
const terms = []
|
|
1367
|
+
const predicates = []
|
|
1368
|
+
const entityRows = []
|
|
1369
|
+
const ids = new Map()
|
|
1370
|
+
for (const { path, file, place } of entries) {
|
|
1371
|
+
if (file?.error !== undefined && file?.error !== null) {
|
|
1372
|
+
flag(path, place.id, 'file_unreadable', String(file.error), 'error')
|
|
1373
|
+
continue
|
|
1374
|
+
}
|
|
1375
|
+
const shape = checkOntologyFile(path, file?.data)
|
|
1376
|
+
if (shape.length > 0) {
|
|
1377
|
+
flag(path, place.id, 'file_invalid', shape.join(';'), 'error')
|
|
1378
|
+
continue
|
|
1379
|
+
}
|
|
1380
|
+
const data = file.data
|
|
1381
|
+
/** 概念与关系共用一个 id 空间(断言里谓词与类型都是这里的名字);实体自成一个空间。 */
|
|
1382
|
+
const space = place.kind === 'entity' ? 'entity' : 'lexicon'
|
|
1383
|
+
const key = `${space}:${place.id}`
|
|
1384
|
+
if (ids.has(key)) {
|
|
1385
|
+
flag(path, place.id, 'duplicate_id', tr(`id「${place.id}」重复了:${ids.get(key)} 已经用了它(这一份不进图)`, `Duplicate id "${place.id}": ${ids.get(key)} already uses it (this one stays off the graph)`), 'error')
|
|
1386
|
+
continue
|
|
1387
|
+
}
|
|
1388
|
+
ids.set(key, path)
|
|
1389
|
+
const status = text(data.status) === 'deprecated' ? 'deprecated' : 'admitted'
|
|
1390
|
+
const parent = place.dirs.length === 0 ? null : place.dirs.at(-1)
|
|
1391
|
+
if (place.kind === 'concept') {
|
|
1392
|
+
terms.push({ id: place.id, label: text(data.label), gloss: text(data.gloss), aliases: Array.isArray(data.aliases) ? data.aliases.map(String) : [], parent, status, basis: text(data.basis) || null, replacedBy: text(data.replaced_by) || null, path })
|
|
1393
|
+
} else if (place.kind === 'relation') {
|
|
1394
|
+
predicates.push({ id: place.id, label: text(data.label), gloss: text(data.gloss), domain: text(data.domain) || null, range: rangeOf(data.range), functional: data.functional === true, status, basis: text(data.basis) || null, replacedBy: text(data.replaced_by) || null, path })
|
|
1395
|
+
} else {
|
|
1396
|
+
entityRows.push({ place, path, data, parent })
|
|
1397
|
+
}
|
|
1398
|
+
}
|
|
1399
|
+
const lexicon = { terms, predicates }
|
|
1400
|
+
for (const term of terms) {
|
|
1401
|
+
if (term.parent === null) continue
|
|
1402
|
+
const found = findEntry(lexicon, term.parent)
|
|
1403
|
+
if (found === null || found.kind !== 'term') flag(term.path, term.id, 'dangling_parent', tr(`它放在目录「${term.parent}/」里,但没有概念文件 ${term.parent}.json:上一层不成立`, `It sits in directory "${term.parent}/" but there is no concept file ${term.parent}.json: the level above does not hold`), 'info')
|
|
1404
|
+
}
|
|
1405
|
+
for (const predicate of predicates) {
|
|
1406
|
+
if (predicate.domain !== null && findEntry(lexicon, predicate.domain)?.kind !== 'term') flag(predicate.path, predicate.id, 'dangling_domain', tr(`主语概念「${predicate.domain}」不存在`, `Subject concept "${predicate.domain}" does not exist`))
|
|
1407
|
+
const term = text(predicate.range?.term)
|
|
1408
|
+
if (term !== '' && findEntry(lexicon, term)?.kind !== 'term') flag(predicate.path, predicate.id, 'dangling_range', tr(`宾语概念「${term}」不存在`, `Object concept "${term}" does not exist`))
|
|
1409
|
+
}
|
|
1410
|
+
const entities = []
|
|
1411
|
+
const entityById = new Map()
|
|
1412
|
+
for (const { place, path, data, parent } of entityRows) {
|
|
1413
|
+
const type = text(data.type)
|
|
1414
|
+
const typeEntry = findEntry(lexicon, type)
|
|
1415
|
+
if (typeEntry === null || typeEntry.kind !== 'term') {
|
|
1416
|
+
flag(path, place.id, 'entity_type_unknown', tr(`类型「${type}」不是已有的概念:先写概念文件,这个实体才进图`, `Type "${type}" is not an existing concept: write the concept file first, then this entity enters the graph`))
|
|
1417
|
+
continue
|
|
1418
|
+
}
|
|
1419
|
+
const entity = { id: place.id, type, label: text(data.label), basis: text(data.basis), provenance: { kind: text(data.provenance.kind), ref: text(data.provenance.ref) }, aliases: Array.isArray(data.aliases) ? data.aliases.map(String) : [], parent, path, relations: Array.isArray(data.relations) ? data.relations : [] }
|
|
1420
|
+
entities.push(entity)
|
|
1421
|
+
entityById.set(entity.id, entity)
|
|
1422
|
+
}
|
|
1423
|
+
const entityAssertions = []
|
|
1424
|
+
const functionalSeen = new Map()
|
|
1425
|
+
for (const entity of entities) {
|
|
1426
|
+
if (entity.parent !== null && !entityById.has(entity.parent)) flag(entity.path, entity.id, 'dangling_container', tr(`它放在目录「${entity.parent}/」里,但没有实体文件 ${entity.parent}.json:「属于」这一层不成立`, `It sits in directory "${entity.parent}/" but there is no entity file ${entity.parent}.json: the part-of level does not hold`), 'info')
|
|
1427
|
+
entity.relations.forEach((relation, index) => {
|
|
1428
|
+
const predicateId = text(relation.predicate)
|
|
1429
|
+
const found = findEntry(lexicon, predicateId)
|
|
1430
|
+
const where = tr(`${entity.id} 的第 ${index + 1} 条关系`, `relation ${index + 1} of ${entity.id}`)
|
|
1431
|
+
if (found === null || found.kind !== 'predicate') {
|
|
1432
|
+
flag(entity.path, entity.id, 'relation_unknown', tr(`${where}:关系「${predicateId}」没有关系文件`, `${where}: relation "${predicateId}" has no relation file`))
|
|
1433
|
+
return
|
|
1434
|
+
}
|
|
1435
|
+
let object
|
|
1436
|
+
if (relation.object !== undefined) {
|
|
1437
|
+
const target = entityById.get(text(relation.object))
|
|
1438
|
+
if (target === undefined) {
|
|
1439
|
+
flag(entity.path, entity.id, 'dangling_object', tr(`${where}:宾语实体「${text(relation.object)}」不存在(或它自己没进图)`, `${where}: object entity "${text(relation.object)}" does not exist (or is itself off the graph)`))
|
|
1440
|
+
return
|
|
1441
|
+
}
|
|
1442
|
+
object = { kind: 'instance', value: target.id, type: target.type }
|
|
1443
|
+
} else {
|
|
1444
|
+
const form = text(found.entry.range?.form) || (typeof relation.value === 'number' ? 'quantity' : 'statement')
|
|
1445
|
+
object = { kind: form, value: relation.value }
|
|
1446
|
+
const unit = text(relation.unit) || text(found.entry.range?.unit)
|
|
1447
|
+
if (unit !== '') object.unit = unit
|
|
1448
|
+
}
|
|
1449
|
+
const assertion = { id: `${entity.id}#${index + 1}`, subject: { id: entity.id, type: entity.type }, predicate: predicateId, object, evidence: { kind: text(relation.evidence.kind), ref: text(relation.evidence.ref) } }
|
|
1450
|
+
const invalid = validateAssertion(lexicon, assertion)
|
|
1451
|
+
if (invalid.length > 0) {
|
|
1452
|
+
flag(entity.path, entity.id, 'relation_invalid', `${where}:${invalid.join(';')}`)
|
|
1453
|
+
return
|
|
1454
|
+
}
|
|
1455
|
+
if (found.entry.functional === true) {
|
|
1456
|
+
const key = `${predicateId}\u0000${entity.id}`
|
|
1457
|
+
const value = objectKey(object)
|
|
1458
|
+
if (functionalSeen.has(key) && functionalSeen.get(key) !== value) flag(entity.path, entity.id, 'functional_conflict', tr(`${where}:「${predicateId}」是单值关系,${entity.id} 却有两个取值(${functionalSeen.get(key)} 与 ${value})`, `${where}: "${predicateId}" is single-valued, but ${entity.id} has two values (${functionalSeen.get(key)} and ${value})`))
|
|
1459
|
+
functionalSeen.set(key, value)
|
|
1460
|
+
}
|
|
1461
|
+
entityAssertions.push(assertion)
|
|
1462
|
+
})
|
|
1463
|
+
}
|
|
1464
|
+
return {
|
|
1465
|
+
lexicon,
|
|
1466
|
+
entities: entities.map(({ relations, ...entity }) => entity),
|
|
1467
|
+
entityAssertions,
|
|
1468
|
+
problems,
|
|
1469
|
+
}
|
|
1470
|
+
}
|
|
1471
|
+
|
|
1472
|
+
/**
|
|
1473
|
+
* **本体大纲**(卡片上那一段):概念树与实体树的前两层,每一支标上它底下有多少个节点;
|
|
1474
|
+
* 关系只报总数。模型一眼看得出哪里深、哪里平、哪些节点还散在顶层——
|
|
1475
|
+
* 「让模型感受到结构」的全部做法就是这一段,有上限,细节它自己去读文件。
|
|
1476
|
+
*/
|
|
1477
|
+
export function ontologyOutline(state, limit = 12) {
|
|
1478
|
+
const lexicon = normalizeLexicon(state?.lexicon)
|
|
1479
|
+
const entities = Array.isArray(state?.entities) ? state.entities : []
|
|
1480
|
+
const tree = (nodes, parentOf) => {
|
|
1481
|
+
const children = new Map()
|
|
1482
|
+
for (const node of nodes) {
|
|
1483
|
+
const parent = parentOf(node)
|
|
1484
|
+
const key = parent !== null && nodes.some((item) => item.id === parent) ? parent : null
|
|
1485
|
+
children.set(key, [...(children.get(key) ?? []), node.id])
|
|
1486
|
+
}
|
|
1487
|
+
const count = (id) => (children.get(id) ?? []).reduce((sum, child) => sum + 1 + count(child), 0)
|
|
1488
|
+
const roots = (children.get(null) ?? []).sort()
|
|
1489
|
+
const shown = roots.slice(0, limit).map((id) => {
|
|
1490
|
+
const kids = (children.get(id) ?? []).sort()
|
|
1491
|
+
const below = count(id)
|
|
1492
|
+
const sub = kids.slice(0, 6).map((kid) => (count(kid) > 0 ? `${kid}(${count(kid)})` : kid))
|
|
1493
|
+
return below === 0 ? id : tr(`${id}(${below}):${sub.join('、')}${kids.length > 6 ? ` 等 ${kids.length} 支` : ''}`, `${id} (${below}): ${sub.join(', ')}${kids.length > 6 ? ` and more, ${kids.length} branches` : ''}`)
|
|
1494
|
+
})
|
|
1495
|
+
return { total: nodes.length, roots: roots.length, lines: shown, more: Math.max(0, roots.length - limit) }
|
|
1496
|
+
}
|
|
1497
|
+
return {
|
|
1498
|
+
concepts: tree(lexicon.terms, (term) => text(term.parent) || null),
|
|
1499
|
+
relations: lexicon.predicates.length,
|
|
1500
|
+
entities: tree(entities, (entity) => text(entity.parent) || null),
|
|
1501
|
+
assertions: Array.isArray(state?.entityAssertions) ? state.entityAssertions.length : 0,
|
|
1502
|
+
problems: Array.isArray(state?.ontologyProblems) ? state.ontologyProblems.length : 0,
|
|
1503
|
+
}
|
|
1504
|
+
}
|