eduevidence 6.0.0 → 6.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONTRIBUTING.md +105 -0
- package/README.md +93 -38
- package/README.zh-CN.md +26 -6
- package/SKILL.md +11 -2
- package/assets/readme/landing-tour.gif +0 -0
- package/assets/readme/studio-tour.gif +0 -0
- package/bin/eduevidence.js +2 -1
- package/docs/architecture.md +319 -43
- package/docs/demo-workplace-ai.md +1 -1
- package/docs/install-guide.md +1 -1
- package/docs/orchestration-role-model.md +1 -1
- package/docs/release-closeout/README.md +1 -1
- package/docs/sciverse-api.md +125 -0
- package/eduevidence_cli.py +10 -0
- package/engine/decision_policy.py +96 -0
- package/engine/evidence_graph.py +14 -10
- package/engine/gaps.py +42 -22
- package/engine/ids.py +2 -0
- package/engine/library.py +6 -2
- package/engine/living.py +34 -4
- package/engine/migration.py +88 -3
- package/engine/orchestration.py +5 -5
- package/engine/paths.py +2 -0
- package/engine/pilot.py +34 -32
- package/engine/taxonomy.py +211 -0
- package/engine/tribunal.py +43 -31
- package/engine/versions.py +1 -1
- package/examples/ai-coding-assistant-evidence/EduEvidence_Report.html +1360 -146
- package/examples/ai-coding-assistant-evidence/artifact_manifest.json +3 -3
- package/examples/ai-coding-assistant-evidence/citation_check.json +1 -1
- package/examples/ai-coding-assistant-evidence/final_verdict.json +107 -0
- package/examples/ai-coding-assistant-evidence/gate_report.json +101 -0
- package/examples/ai-coding-assistant-evidence/report_spec.json +23 -12
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_academic.html +447 -127
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_claude.html +447 -127
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab-dark.html +447 -127
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_datalab.html +447 -127
- package/examples/ai-coding-assistant-evidence/reports-5themes/EduEvidence_Report_presentation.html +447 -127
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_academic.html +1360 -146
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_claude.html +1360 -146
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab-dark.html +1360 -146
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_datalab.html +1360 -146
- package/examples/ai-coding-assistant-evidence/reports-5themes/report_presentation.html +1360 -146
- package/examples/ai-coding-assistant-evidence/result.json +13 -9
- package/examples/ai-coding-assistant-evidence/result.zh.json +45 -41
- package/examples/ai-coding-assistant-evidence/skeptic.json +72 -0
- package/examples/ai-coding-assistant-evidence/verdict.json +6 -2
- package/examples/spaced-retrieval-practice/applicability.json +14 -0
- package/examples/spaced-retrieval-practice/artifact_manifest.json +15 -0
- package/examples/spaced-retrieval-practice/claims.jsonl +3 -0
- package/examples/spaced-retrieval-practice/evidence.jsonl +6 -0
- package/examples/spaced-retrieval-practice/final_verdict.json +93 -0
- package/examples/spaced-retrieval-practice/frame.json +58 -0
- package/examples/spaced-retrieval-practice/gate_report.json +101 -0
- package/examples/spaced-retrieval-practice/methodology.json +78 -0
- package/examples/spaced-retrieval-practice/report_spec.json +212 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/EduEvidence_Report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_academic.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_claude.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab-dark.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_datalab.html +2728 -0
- package/examples/spaced-retrieval-practice/reports-5themes/report_presentation.html +2728 -0
- package/examples/spaced-retrieval-practice/result.json +942 -0
- package/examples/spaced-retrieval-practice/result.zh.json +942 -0
- package/examples/spaced-retrieval-practice/skeptic.json +70 -0
- package/examples/spaced-retrieval-practice/sources.jsonl +7 -0
- package/examples/spaced-retrieval-practice/verdict.json +93 -0
- package/examples/workplace-ai-assistant/artifact_manifest.json +15 -0
- package/examples/workplace-ai-assistant/claims.jsonl +4 -4
- package/examples/workplace-ai-assistant/evidence.jsonl +4 -4
- package/examples/workplace-ai-assistant/evidence_graph.json +15 -15
- package/examples/workplace-ai-assistant/final_verdict.json +78 -0
- package/examples/workplace-ai-assistant/gate_report.json +101 -0
- package/examples/workplace-ai-assistant/report_spec.json +209 -40
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_academic.html +435 -105
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_claude.html +435 -105
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab-dark.html +435 -105
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_datalab.html +435 -105
- package/examples/workplace-ai-assistant/reports-5themes/EduEvidence_Report_presentation.html +435 -105
- package/examples/workplace-ai-assistant/reports-5themes/report_academic.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_claude.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab-dark.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_datalab.html +2814 -0
- package/examples/workplace-ai-assistant/reports-5themes/report_presentation.html +2814 -0
- package/examples/workplace-ai-assistant/result.json +82 -20
- package/examples/workplace-ai-assistant/result.zh.json +82 -20
- package/examples/workplace-ai-assistant/skeptic.json +72 -0
- package/examples/workplace-ai-assistant/verdict.json +36 -10
- package/integrations/agent_mcp.py +2 -2
- package/package.json +12 -3
- package/pyproject.toml +4 -3
- package/references/report-copy-style.md +67 -0
- package/references/retrieval-compliance.md +75 -0
- package/references/retrieval-protocol.md +20 -0
- package/retrieval/audit.py +27 -3
- package/retrieval/fetch.py +96 -0
- package/retrieval/sciverse.py +398 -0
- package/retrieval/search.py +47 -7
- package/schemas/applicability.schema.json +94 -0
- package/schemas/chart-spec.schema.json +10 -3
- package/schemas/evidence.schema.json +316 -43
- package/schemas/fetch-result.schema.json +2 -1
- package/schemas/report-result.schema.json +3 -3
- package/schemas/report-spec.schema.json +98 -100
- package/schemas/skeptic.schema.json +86 -0
- package/schemas/source.schema.json +21 -2
- package/schemas/v2/finding.schema.json +5 -1
- package/schemas/v2/methodology-audit.schema.json +5 -1
- package/schemas/v2/outcome.schema.json +28 -5
- package/schemas/v2/study.schema.json +5 -1
- package/schemas/vNext/autoevolve-session.schema.json +34 -1
- package/schemas/vNext/eval-snapshot.schema.json +77 -1
- package/schemas/vNext/execution-plan.schema.json +50 -1
- package/schemas/vNext/gap-priority.schema.json +54 -1
- package/schemas/vNext/negative-search-record.schema.json +68 -1
- package/schemas/vNext/research-iteration.schema.json +87 -1
- package/schemas/vNext/research-strategy.schema.json +62 -1
- package/schemas/vNext/skill-experiment.schema.json +90 -1
- package/schemas/vNext/task-spec.schema.json +156 -1
- package/schemas/vNext/worker-result.schema.json +60 -1
- package/schemas/verdict.schema.json +164 -28
- package/scripts/build_esl_artifacts.py +2 -2
- package/scripts/build_report_variants.py +18 -2
- package/scripts/build_result.py +74 -9
- package/scripts/check_package_parity.py +85 -0
- package/scripts/check_protocol_alignment.py +375 -0
- package/scripts/check_versioned_schemas.py +254 -0
- package/scripts/claim_audit.py +13 -8
- package/scripts/compute_confidence.py +10 -0
- package/scripts/did_regression.py +12 -2
- package/scripts/evidence_score.py +5 -2
- package/scripts/generate_new_projects.py +4 -4
- package/scripts/orchestrator.py +120 -24
- package/scripts/pre_verdict_gate.py +224 -26
- package/scripts/quickstart.py +18 -2
- package/scripts/run_workspace.py +7 -1
- package/scripts/skill_payload.py +4 -1
- package/scripts/test_adversarial_empirical.py +26 -19
- package/scripts/validate_schema.py +31 -1
- package/skill/agents/evaluation-designer.md +20 -4
- package/skill/agents/evidence-analyst.md +19 -3
- package/skill/agents/evidence-judge.md +50 -2
- package/skill/agents/evidence-retriever.md +20 -3
- package/skill/agents/intervention-designer.md +20 -4
- package/skill/agents/method-reviewer.md +18 -2
- package/skill/agents/{education-planner.md → research-planner.md} +19 -3
- package/skill/agents/skeptic.md +18 -2
- package/skill/roles/registry.yaml +11 -11
- package/skill/sub-skills/aihot-trend-analysis/SKILL.md +28 -9
- package/skill/sub-skills/contradiction-analysis/SKILL.md +31 -11
- package/skill/sub-skills/data-analysis/SKILL.md +34 -15
- package/skill/sub-skills/ethics-review/SKILL.md +33 -10
- package/skill/sub-skills/evidence-extraction/SKILL.md +29 -11
- package/skill/sub-skills/evidence-review/SKILL.md +31 -12
- package/skill/sub-skills/gap-analysis/SKILL.md +31 -9
- package/skill/sub-skills/literature-review/SKILL.md +35 -14
- package/skill/sub-skills/methodology-audit/SKILL.md +29 -12
- package/skill/sub-skills/report-generation/SKILL.md +28 -0
- package/skill/sub-skills/research-planning/SKILL.md +41 -14
- package/skill/sub-skills/study-design/SKILL.md +30 -9
- package/skill/task-briefs/adjudicate.md +32 -7
- package/skill/task-briefs/applicability.md +37 -2
- package/skill/task-briefs/audit.md +32 -7
- package/skill/task-briefs/challenge.md +34 -5
- package/skill/task-briefs/evaluate.md +30 -5
- package/skill/task-briefs/extract.md +31 -8
- package/skill/task-briefs/frame.md +39 -10
- package/skill/task-briefs/intervene.md +32 -6
- package/skill/task-briefs/present.md +32 -8
- package/skill/task-briefs/projection.md +36 -2
- package/skill/task-briefs/retrieve.md +36 -6
- package/skill/workflows/decision-and-pilot.md +76 -1
- package/skill/workflows/evaluate-and-update.md +83 -0
- package/skill/workflows/evidence-review.md +104 -0
- package/visualization/eduevidence-report/scripts/build_figures.py +25 -3
- package/visualization/eduevidence-report/scripts/build_infographics.py +5 -1
- package/visualization/eduevidence-report/scripts/build_report.py +512 -65
- package/visualization/eduevidence-report/scripts/charts_data.py +2 -0
- package/visualization/eduevidence-report/scripts/lieflat_engine.py +349 -38
- package/visualization/eduevidence-report/scripts/zh_labels.py +80 -1
- package/web/architecture.html +14885 -0
- package/web/studio/assets/index-B8tkF44Q.css +1 -0
- package/web/studio/index.html +2 -2
- package/web/studio/assets/index-CzXocaGv.css +0 -1
- /package/web/studio/assets/{index-pa7jD7n4.js → index-CQ6Keoyc.js} +0 -0
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "http://json-schema.org/draft-07/schema#",
|
|
3
|
+
"$id": "https://github.com/37chengshan/eduevidence/schemas/skeptic.schema.json",
|
|
4
|
+
"title": "SkepticFindings",
|
|
5
|
+
"description": "Counter-evidence record produced by the Challenge stage (skill/task-briefs/challenge.md, skill/agents/skeptic.md). The nine checks are mandatory: an empty or partial record used to pass the Pre-Verdict Gate, which made the counter-evidence gate cosmetic.",
|
|
6
|
+
"type": "object",
|
|
7
|
+
"additionalProperties": false,
|
|
8
|
+
"required": [
|
|
9
|
+
"search_performed",
|
|
10
|
+
"skeptic_findings",
|
|
11
|
+
"contradictory_evidence_found"
|
|
12
|
+
],
|
|
13
|
+
"properties": {
|
|
14
|
+
"search_performed": {
|
|
15
|
+
"type": "boolean",
|
|
16
|
+
"const": true
|
|
17
|
+
},
|
|
18
|
+
"method": {
|
|
19
|
+
"type": "string"
|
|
20
|
+
},
|
|
21
|
+
"skeptic_findings": {
|
|
22
|
+
"type": "array",
|
|
23
|
+
"minItems": 9,
|
|
24
|
+
"items": {
|
|
25
|
+
"type": "object",
|
|
26
|
+
"additionalProperties": false,
|
|
27
|
+
"required": [
|
|
28
|
+
"check",
|
|
29
|
+
"status",
|
|
30
|
+
"detail"
|
|
31
|
+
],
|
|
32
|
+
"properties": {
|
|
33
|
+
"check": {
|
|
34
|
+
"type": "string",
|
|
35
|
+
"enum": [
|
|
36
|
+
"1_null_result",
|
|
37
|
+
"2_negative_result",
|
|
38
|
+
"3_contradictory_evidence",
|
|
39
|
+
"4_alternative_explanation",
|
|
40
|
+
"5_measurement_mismatch",
|
|
41
|
+
"6_sampling_bias",
|
|
42
|
+
"7_novelty_effect",
|
|
43
|
+
"8_ai_dependency",
|
|
44
|
+
"9_scope_overreach"
|
|
45
|
+
],
|
|
46
|
+
"description": "The fixed check this finding reports on."
|
|
47
|
+
},
|
|
48
|
+
"status": {
|
|
49
|
+
"type": "string",
|
|
50
|
+
"enum": [
|
|
51
|
+
"found",
|
|
52
|
+
"not_found"
|
|
53
|
+
]
|
|
54
|
+
},
|
|
55
|
+
"detail": {
|
|
56
|
+
"type": "string",
|
|
57
|
+
"minLength": 8,
|
|
58
|
+
"description": "What was looked for and what was found; an empty note is not a check."
|
|
59
|
+
},
|
|
60
|
+
"related_evidence_ids": {
|
|
61
|
+
"type": "array",
|
|
62
|
+
"items": {
|
|
63
|
+
"type": "string"
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
},
|
|
69
|
+
"contradictory_evidence_found": {
|
|
70
|
+
"type": "boolean"
|
|
71
|
+
},
|
|
72
|
+
"no_contradictory_evidence_statement": {
|
|
73
|
+
"type": "string"
|
|
74
|
+
},
|
|
75
|
+
"threats_to_validity": {
|
|
76
|
+
"type": "array",
|
|
77
|
+
"items": {
|
|
78
|
+
"type": "string"
|
|
79
|
+
}
|
|
80
|
+
},
|
|
81
|
+
"extensions": {
|
|
82
|
+
"type": "object",
|
|
83
|
+
"additionalProperties": true
|
|
84
|
+
}
|
|
85
|
+
}
|
|
86
|
+
}
|
|
@@ -212,7 +212,8 @@
|
|
|
212
212
|
"markdown_new",
|
|
213
213
|
"defuddle",
|
|
214
214
|
"raw_html",
|
|
215
|
-
"pdf_parser"
|
|
215
|
+
"pdf_parser",
|
|
216
|
+
"sciverse_content"
|
|
216
217
|
]
|
|
217
218
|
},
|
|
218
219
|
"fetch_status": {
|
|
@@ -304,8 +305,26 @@
|
|
|
304
305
|
}
|
|
305
306
|
}
|
|
306
307
|
}
|
|
308
|
+
},
|
|
309
|
+
"extensions": {
|
|
310
|
+
"type": "object",
|
|
311
|
+
"additionalProperties": true,
|
|
312
|
+
"description": "Structured extension container. Provider-specific locators live here - e.g. sciverse: {doc_id, offset, next_offset, more} - so the closed fetch contract can carry them without new top-level fields."
|
|
313
|
+
},
|
|
314
|
+
"doc_id": {
|
|
315
|
+
"type": "string",
|
|
316
|
+
"description": "Full-text artifact id (Sciverse)."
|
|
317
|
+
},
|
|
318
|
+
"chunk_id": {
|
|
319
|
+
"type": "string",
|
|
320
|
+
"description": "Retrieved chunk id (Sciverse)."
|
|
321
|
+
},
|
|
322
|
+
"offset": {
|
|
323
|
+
"type": "integer",
|
|
324
|
+
"minimum": 0,
|
|
325
|
+
"description": "Unicode code-point offset of the fetched slice (Sciverse /content)."
|
|
307
326
|
}
|
|
308
327
|
}
|
|
309
328
|
}
|
|
310
329
|
}
|
|
311
|
-
}
|
|
330
|
+
}
|
|
@@ -11,7 +11,11 @@
|
|
|
11
11
|
],
|
|
12
12
|
"properties": {
|
|
13
13
|
"finding_id": { "type": "string", "pattern": "^FND-" },
|
|
14
|
-
"study_id": {
|
|
14
|
+
"study_id": {
|
|
15
|
+
"type": "string",
|
|
16
|
+
"pattern": "^(ST-|STU-|STUDY-)",
|
|
17
|
+
"description": "Owning Study. ST- preserves the placeholder prefix used by some legacy/curated packs; migration never renames ids."
|
|
18
|
+
},
|
|
15
19
|
"finding_type": {
|
|
16
20
|
"type": "string",
|
|
17
21
|
"enum": [
|
|
@@ -13,7 +13,11 @@
|
|
|
13
13
|
],
|
|
14
14
|
"properties": {
|
|
15
15
|
"audit_id": { "type": "string", "pattern": "^AUD-" },
|
|
16
|
-
"study_id": {
|
|
16
|
+
"study_id": {
|
|
17
|
+
"type": "string",
|
|
18
|
+
"pattern": "^(ST-|STU-|STUDY-)",
|
|
19
|
+
"description": "Audited Study. ST- preserves the placeholder prefix used by some legacy/curated packs; migration never renames ids."
|
|
20
|
+
},
|
|
17
21
|
"policy_version": { "type": "string", "minLength": 1 },
|
|
18
22
|
"design_quality": { "type": "integer", "minimum": 0, "maximum": 2 },
|
|
19
23
|
"sample_quality": { "type": "integer", "minimum": 0, "maximum": 2 },
|
|
@@ -5,14 +5,37 @@
|
|
|
5
5
|
"description": "A measurable education outcome from the Outcome Taxonomy: learning / task performance / process / risk.",
|
|
6
6
|
"type": "object",
|
|
7
7
|
"additionalProperties": false,
|
|
8
|
-
"required": [
|
|
8
|
+
"required": [
|
|
9
|
+
"outcome_id",
|
|
10
|
+
"name",
|
|
11
|
+
"outcome_type",
|
|
12
|
+
"extensions"
|
|
13
|
+
],
|
|
9
14
|
"properties": {
|
|
10
|
-
"outcome_id": {
|
|
11
|
-
|
|
15
|
+
"outcome_id": {
|
|
16
|
+
"type": "string",
|
|
17
|
+
"pattern": "^OUT-"
|
|
18
|
+
},
|
|
19
|
+
"name": {
|
|
20
|
+
"type": "string",
|
|
21
|
+
"minLength": 1
|
|
22
|
+
},
|
|
12
23
|
"outcome_type": {
|
|
13
24
|
"type": "string",
|
|
14
|
-
"enum": [
|
|
25
|
+
"enum": [
|
|
26
|
+
"learning",
|
|
27
|
+
"task_performance",
|
|
28
|
+
"process",
|
|
29
|
+
"risk",
|
|
30
|
+
"effectiveness",
|
|
31
|
+
"cost",
|
|
32
|
+
"equity",
|
|
33
|
+
"feasibility"
|
|
34
|
+
],
|
|
35
|
+
"description": "Outcome category bucket. The education domain declares learning / task_performance / process / risk; the policy domain declares effectiveness / cost / equity / feasibility / risk. domains/<id>/outcome_taxonomy.json is the authority and scripts/check_protocol_alignment.py fails if this enum drifts from it."
|
|
15
36
|
},
|
|
16
|
-
"extensions": {
|
|
37
|
+
"extensions": {
|
|
38
|
+
"type": "object"
|
|
39
|
+
}
|
|
17
40
|
}
|
|
18
41
|
}
|
|
@@ -10,7 +10,11 @@
|
|
|
10
10
|
"independence_key", "identity_status", "extensions"
|
|
11
11
|
],
|
|
12
12
|
"properties": {
|
|
13
|
-
"study_id": {
|
|
13
|
+
"study_id": {
|
|
14
|
+
"type": "string",
|
|
15
|
+
"pattern": "^(ST-|STU-|STUDY-)",
|
|
16
|
+
"description": "Study identity. ST- is the placeholder prefix some legacy/curated packs used before STU-/STUDY-; migration preserves original ids rather than renaming them."
|
|
17
|
+
},
|
|
14
18
|
"source_ids": {
|
|
15
19
|
"type": "array",
|
|
16
20
|
"items": { "type": "string", "pattern": "^(SRC-|S-)" },
|
|
@@ -1 +1,34 @@
|
|
|
1
|
-
{
|
|
1
|
+
{
|
|
2
|
+
"$schema": "http://json-schema.org/draft-07/schema#",
|
|
3
|
+
"$id": "https://github.com/37chengshan/eduevidence/schemas/vNext/autoevolve-session.schema.json",
|
|
4
|
+
"title": "AutoevolveSessionReport",
|
|
5
|
+
"description": "Daily self-evolution session report (engine/autoevolve/runner.py -> state_root/daily-report.json). Records one bounded branch-only session: how many experiments ran, their promotion statuses, the stopping reason, cost and wall time, and whether runner-owned isolation was verified. Automatic KEEP requires verified OS isolation; evaluator self-attestation is ignored.",
|
|
6
|
+
"type": "object",
|
|
7
|
+
"additionalProperties": false,
|
|
8
|
+
"required": [
|
|
9
|
+
"run_tag", "branch", "experiments", "statuses", "cost", "wall_minutes",
|
|
10
|
+
"plateau", "stop_reason", "promotion", "holdout_isolation_verified", "mutation_view"
|
|
11
|
+
],
|
|
12
|
+
"properties": {
|
|
13
|
+
"run_tag": {"type": "string", "minLength": 1, "description": "Session tag; also the SkillExperiment.session_id of every experiment in this session."},
|
|
14
|
+
"branch": {"type": "string", "minLength": 1, "description": "Branch the session ran on. Promotion is branch-only."},
|
|
15
|
+
"experiments": {"type": "integer", "minimum": 0, "description": "Number of experiments attempted."},
|
|
16
|
+
"statuses": {"type": "array", "items": {"enum": ["created", "KEEP", "REJECT", "RETEST", "HUMAN_REVIEW", "CRASH", "INVALID"]}, "description": "Per-experiment promotion verdict, in attempt order."},
|
|
17
|
+
"best_experiment_id": {"type": ["string", "null"]},
|
|
18
|
+
"best_candidate_commit": {"type": ["string", "null"]},
|
|
19
|
+
"cost": {"type": "number", "minimum": 0, "description": "Cumulative session cost in USD, as reported by the agent and evaluator."},
|
|
20
|
+
"wall_minutes": {"type": "number", "minimum": 0},
|
|
21
|
+
"plateau": {"type": "boolean"},
|
|
22
|
+
"stop_reason": {"type": "string", "minLength": 1, "description": "Why the loop stopped (for example budget_exhausted, plateau, completed)."},
|
|
23
|
+
"promotion": {"const": "branch_only", "description": "Daily mode never promotes directly; KEEP only marks a candidate branch."},
|
|
24
|
+
"branch_push_requested": {"type": "boolean"},
|
|
25
|
+
"branch_pushed": {"type": "boolean"},
|
|
26
|
+
"mutation_view": {"type": "string", "minLength": 1, "description": "Isolation actually used for the mutation view."},
|
|
27
|
+
"holdout_isolation_verified": {"type": "boolean", "description": "Runner-owned OS isolation verification. Automatic KEEP requires true."},
|
|
28
|
+
"isolation_provider": {"type": "string"},
|
|
29
|
+
"isolation_reason": {"type": "string"},
|
|
30
|
+
"eval_suite_hash": {"type": "string", "description": "Trusted evaluation suite hash captured before the session; a mid-session change invalidates the experiment."},
|
|
31
|
+
"security_note": {"type": "string"},
|
|
32
|
+
"candidate_artifacts": {"type": "string", "description": "Where candidate artifacts live; local session state is never auto-pushed."}
|
|
33
|
+
}
|
|
34
|
+
}
|
|
@@ -1 +1,77 @@
|
|
|
1
|
-
{
|
|
1
|
+
{
|
|
2
|
+
"$schema": "http://json-schema.org/draft-07/schema#",
|
|
3
|
+
"title": "EvalSnapshot",
|
|
4
|
+
"description": "The measurement record for one self-evolution candidate (engine/autoevolve/core.py). Carries gate outcomes, science/research scores, robustness, cost, latency, repeats and the trusted eval-suite hash. Runner-owned snapshots ignore evaluator self-attestation; a mid-session suite-hash change invalidates the experiment.",
|
|
5
|
+
"type": "object",
|
|
6
|
+
"required": [
|
|
7
|
+
"eval_id",
|
|
8
|
+
"hard_gates_passed",
|
|
9
|
+
"science_score",
|
|
10
|
+
"research_score",
|
|
11
|
+
"robustness",
|
|
12
|
+
"cost",
|
|
13
|
+
"latency",
|
|
14
|
+
"complexity",
|
|
15
|
+
"repeats",
|
|
16
|
+
"noise_floor",
|
|
17
|
+
"dev_passed",
|
|
18
|
+
"holdout_passed",
|
|
19
|
+
"adversarial_passed",
|
|
20
|
+
"holdout_isolation_verified",
|
|
21
|
+
"eval_suite_hash"
|
|
22
|
+
],
|
|
23
|
+
"properties": {
|
|
24
|
+
"eval_id": {
|
|
25
|
+
"type": "string"
|
|
26
|
+
},
|
|
27
|
+
"hard_gates_passed": {
|
|
28
|
+
"type": "boolean"
|
|
29
|
+
},
|
|
30
|
+
"science_score": {
|
|
31
|
+
"type": "number"
|
|
32
|
+
},
|
|
33
|
+
"research_score": {
|
|
34
|
+
"type": "number"
|
|
35
|
+
},
|
|
36
|
+
"robustness": {
|
|
37
|
+
"type": "number"
|
|
38
|
+
},
|
|
39
|
+
"cost": {
|
|
40
|
+
"type": "number",
|
|
41
|
+
"minimum": 0
|
|
42
|
+
},
|
|
43
|
+
"latency": {
|
|
44
|
+
"type": "number",
|
|
45
|
+
"minimum": 0
|
|
46
|
+
},
|
|
47
|
+
"complexity": {
|
|
48
|
+
"type": "number",
|
|
49
|
+
"minimum": 0
|
|
50
|
+
},
|
|
51
|
+
"repeats": {
|
|
52
|
+
"type": "integer",
|
|
53
|
+
"minimum": 1
|
|
54
|
+
},
|
|
55
|
+
"noise_floor": {
|
|
56
|
+
"type": "number",
|
|
57
|
+
"minimum": 0
|
|
58
|
+
},
|
|
59
|
+
"dev_passed": {
|
|
60
|
+
"type": "boolean"
|
|
61
|
+
},
|
|
62
|
+
"holdout_passed": {
|
|
63
|
+
"type": "boolean"
|
|
64
|
+
},
|
|
65
|
+
"adversarial_passed": {
|
|
66
|
+
"type": "boolean"
|
|
67
|
+
},
|
|
68
|
+
"holdout_isolation_verified": {
|
|
69
|
+
"type": "boolean"
|
|
70
|
+
},
|
|
71
|
+
"eval_suite_hash": {
|
|
72
|
+
"type": "string",
|
|
73
|
+
"minLength": 1
|
|
74
|
+
}
|
|
75
|
+
},
|
|
76
|
+
"additionalProperties": false
|
|
77
|
+
}
|
|
@@ -1 +1,50 @@
|
|
|
1
|
-
{
|
|
1
|
+
{
|
|
2
|
+
"$schema": "http://json-schema.org/draft-07/schema#",
|
|
3
|
+
"title": "ExecutionPlan",
|
|
4
|
+
"description": "The plan for one research run (engine/orchestration.py ExecutionPlanner.plan). Complexity level, the ordered TaskSpecs, the maximum parallel workers and the parallel groups; S delegates zero tasks, M and L are capped by policy. scripts/check_autoresearch_invariants.py enforces those caps.",
|
|
5
|
+
"type": "object",
|
|
6
|
+
"required": [
|
|
7
|
+
"complexity",
|
|
8
|
+
"tasks",
|
|
9
|
+
"max_parallel_workers",
|
|
10
|
+
"parallel_groups",
|
|
11
|
+
"plan_id"
|
|
12
|
+
],
|
|
13
|
+
"properties": {
|
|
14
|
+
"complexity": {
|
|
15
|
+
"enum": [
|
|
16
|
+
"S",
|
|
17
|
+
"M",
|
|
18
|
+
"L"
|
|
19
|
+
]
|
|
20
|
+
},
|
|
21
|
+
"tasks": {
|
|
22
|
+
"type": "array",
|
|
23
|
+
"items": {
|
|
24
|
+
"$ref": "task-spec.schema.json"
|
|
25
|
+
}
|
|
26
|
+
},
|
|
27
|
+
"max_parallel_workers": {
|
|
28
|
+
"type": "integer",
|
|
29
|
+
"minimum": 0,
|
|
30
|
+
"maximum": 6
|
|
31
|
+
},
|
|
32
|
+
"parallel_groups": {
|
|
33
|
+
"type": "array",
|
|
34
|
+
"items": {
|
|
35
|
+
"type": "array",
|
|
36
|
+
"minItems": 1,
|
|
37
|
+
"items": {
|
|
38
|
+
"type": "string"
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
},
|
|
42
|
+
"plan_id": {
|
|
43
|
+
"type": [
|
|
44
|
+
"string",
|
|
45
|
+
"null"
|
|
46
|
+
]
|
|
47
|
+
}
|
|
48
|
+
},
|
|
49
|
+
"additionalProperties": false
|
|
50
|
+
}
|
|
@@ -1 +1,54 @@
|
|
|
1
|
-
{
|
|
1
|
+
{
|
|
2
|
+
"$schema": "http://json-schema.org/draft-07/schema#",
|
|
3
|
+
"title": "GapPriority",
|
|
4
|
+
"description": "Ranking of one knowledge gap for research attention (engine/autoresearch/gap_priority.py). Combines decision sensitivity, current uncertainty, directness deficit, applicability, availability and risk minus cost into a score, and names the next research mode. Produced by controllers.select_gap(), not by an LLM.",
|
|
5
|
+
"type": "object",
|
|
6
|
+
"required": [
|
|
7
|
+
"gap_id",
|
|
8
|
+
"dvi_band",
|
|
9
|
+
"cost_band",
|
|
10
|
+
"decision_material",
|
|
11
|
+
"drivers",
|
|
12
|
+
"next_research_mode",
|
|
13
|
+
"score"
|
|
14
|
+
],
|
|
15
|
+
"properties": {
|
|
16
|
+
"gap_id": {
|
|
17
|
+
"type": "string"
|
|
18
|
+
},
|
|
19
|
+
"dvi_band": {
|
|
20
|
+
"enum": [
|
|
21
|
+
"HIGH",
|
|
22
|
+
"MEDIUM",
|
|
23
|
+
"LOW"
|
|
24
|
+
]
|
|
25
|
+
},
|
|
26
|
+
"cost_band": {
|
|
27
|
+
"enum": [
|
|
28
|
+
"HIGH",
|
|
29
|
+
"MEDIUM",
|
|
30
|
+
"LOW"
|
|
31
|
+
]
|
|
32
|
+
},
|
|
33
|
+
"decision_material": {
|
|
34
|
+
"type": "boolean"
|
|
35
|
+
},
|
|
36
|
+
"drivers": {
|
|
37
|
+
"type": "array",
|
|
38
|
+
"items": {
|
|
39
|
+
"type": "string"
|
|
40
|
+
}
|
|
41
|
+
},
|
|
42
|
+
"next_research_mode": {
|
|
43
|
+
"enum": [
|
|
44
|
+
"secondary_evidence_search",
|
|
45
|
+
"defer",
|
|
46
|
+
"empirical_evidence_needed"
|
|
47
|
+
]
|
|
48
|
+
},
|
|
49
|
+
"score": {
|
|
50
|
+
"type": "integer"
|
|
51
|
+
}
|
|
52
|
+
},
|
|
53
|
+
"additionalProperties": false
|
|
54
|
+
}
|
|
@@ -1 +1,68 @@
|
|
|
1
|
-
{
|
|
1
|
+
{
|
|
2
|
+
"$schema": "http://json-schema.org/draft-07/schema#",
|
|
3
|
+
"title": "NegativeSearchRecord",
|
|
4
|
+
"description": "A search that legitimately returned nothing usable (engine/autoresearch/contracts.py NegativeSearchRecord). Zero-result searches are recorded rather than discarded: without them, absence of evidence is indistinguishable from failure to look, and retrieval saturation cannot be judged. Written by ResearchMemory.append_negative_search().",
|
|
5
|
+
"type": "object",
|
|
6
|
+
"required": [
|
|
7
|
+
"negative_search_id",
|
|
8
|
+
"research_iteration_id",
|
|
9
|
+
"gap_id",
|
|
10
|
+
"queries",
|
|
11
|
+
"providers",
|
|
12
|
+
"candidate_count",
|
|
13
|
+
"fetched_count",
|
|
14
|
+
"eligible_count",
|
|
15
|
+
"conclusion"
|
|
16
|
+
],
|
|
17
|
+
"properties": {
|
|
18
|
+
"negative_search_id": {
|
|
19
|
+
"type": "string"
|
|
20
|
+
},
|
|
21
|
+
"research_iteration_id": {
|
|
22
|
+
"type": "string"
|
|
23
|
+
},
|
|
24
|
+
"gap_id": {
|
|
25
|
+
"type": "string"
|
|
26
|
+
},
|
|
27
|
+
"queries": {
|
|
28
|
+
"type": "array",
|
|
29
|
+
"items": {
|
|
30
|
+
"type": "string"
|
|
31
|
+
}
|
|
32
|
+
},
|
|
33
|
+
"providers": {
|
|
34
|
+
"type": "array",
|
|
35
|
+
"items": {
|
|
36
|
+
"type": "string"
|
|
37
|
+
}
|
|
38
|
+
},
|
|
39
|
+
"candidate_count": {
|
|
40
|
+
"type": "integer",
|
|
41
|
+
"minimum": 0
|
|
42
|
+
},
|
|
43
|
+
"fetched_count": {
|
|
44
|
+
"type": "integer",
|
|
45
|
+
"minimum": 0
|
|
46
|
+
},
|
|
47
|
+
"eligible_count": {
|
|
48
|
+
"const": 0
|
|
49
|
+
},
|
|
50
|
+
"exclusion_reasons": {
|
|
51
|
+
"type": "object",
|
|
52
|
+
"additionalProperties": {
|
|
53
|
+
"type": "integer",
|
|
54
|
+
"minimum": 0
|
|
55
|
+
}
|
|
56
|
+
},
|
|
57
|
+
"scope": {
|
|
58
|
+
"type": "object"
|
|
59
|
+
},
|
|
60
|
+
"searched_at": {
|
|
61
|
+
"type": "string"
|
|
62
|
+
},
|
|
63
|
+
"conclusion": {
|
|
64
|
+
"const": "no_eligible_evidence_found_within_search_scope"
|
|
65
|
+
}
|
|
66
|
+
},
|
|
67
|
+
"additionalProperties": false
|
|
68
|
+
}
|
|
@@ -1 +1,87 @@
|
|
|
1
|
-
{
|
|
1
|
+
{
|
|
2
|
+
"$schema": "http://json-schema.org/draft-07/schema#",
|
|
3
|
+
"title": "ResearchIteration",
|
|
4
|
+
"description": "One autoresearch loop iteration (engine/autoresearch/contracts.py ResearchIteration, persisted by ResearchMemory.append_iteration). Records which knowledge gap was attacked, the strategy and budget used, what was retrieved and validated, whether a new GraphRevision was created, and how the loop stopped. Written by scripts/research_auto_cli.py.",
|
|
5
|
+
"type": "object",
|
|
6
|
+
"required": [
|
|
7
|
+
"iteration_id",
|
|
8
|
+
"project_id",
|
|
9
|
+
"base_graph_revision",
|
|
10
|
+
"gap_id",
|
|
11
|
+
"strategy",
|
|
12
|
+
"status"
|
|
13
|
+
],
|
|
14
|
+
"properties": {
|
|
15
|
+
"iteration_id": {
|
|
16
|
+
"type": "string"
|
|
17
|
+
},
|
|
18
|
+
"project_id": {
|
|
19
|
+
"type": "string"
|
|
20
|
+
},
|
|
21
|
+
"base_graph_revision": {
|
|
22
|
+
"type": "integer",
|
|
23
|
+
"minimum": 0
|
|
24
|
+
},
|
|
25
|
+
"gap_id": {
|
|
26
|
+
"type": "string"
|
|
27
|
+
},
|
|
28
|
+
"gap_lineage_key": {
|
|
29
|
+
"type": [
|
|
30
|
+
"string",
|
|
31
|
+
"null"
|
|
32
|
+
],
|
|
33
|
+
"pattern": "^KGK-"
|
|
34
|
+
},
|
|
35
|
+
"strategy": {
|
|
36
|
+
"type": "object"
|
|
37
|
+
},
|
|
38
|
+
"validated_evidence_ids": {
|
|
39
|
+
"type": "array",
|
|
40
|
+
"items": {
|
|
41
|
+
"type": "string"
|
|
42
|
+
}
|
|
43
|
+
},
|
|
44
|
+
"negative_search_ids": {
|
|
45
|
+
"type": "array",
|
|
46
|
+
"items": {
|
|
47
|
+
"type": "string"
|
|
48
|
+
}
|
|
49
|
+
},
|
|
50
|
+
"evidence_gain": {
|
|
51
|
+
"type": "object"
|
|
52
|
+
},
|
|
53
|
+
"new_graph_revision": {
|
|
54
|
+
"type": [
|
|
55
|
+
"integer",
|
|
56
|
+
"null"
|
|
57
|
+
]
|
|
58
|
+
},
|
|
59
|
+
"decision_snapshot_id": {
|
|
60
|
+
"type": [
|
|
61
|
+
"string",
|
|
62
|
+
"null"
|
|
63
|
+
]
|
|
64
|
+
},
|
|
65
|
+
"status": {
|
|
66
|
+
"enum": [
|
|
67
|
+
"completed_gain",
|
|
68
|
+
"completed_no_gain",
|
|
69
|
+
"search_saturated",
|
|
70
|
+
"empirical_needed",
|
|
71
|
+
"budget_exhausted",
|
|
72
|
+
"tool_failure",
|
|
73
|
+
"invalid"
|
|
74
|
+
]
|
|
75
|
+
},
|
|
76
|
+
"started_at": {
|
|
77
|
+
"type": "string"
|
|
78
|
+
},
|
|
79
|
+
"completed_at": {
|
|
80
|
+
"type": [
|
|
81
|
+
"string",
|
|
82
|
+
"null"
|
|
83
|
+
]
|
|
84
|
+
}
|
|
85
|
+
},
|
|
86
|
+
"additionalProperties": true
|
|
87
|
+
}
|
|
@@ -1 +1,62 @@
|
|
|
1
|
-
{
|
|
1
|
+
{
|
|
2
|
+
"$schema": "http://json-schema.org/draft-07/schema#",
|
|
3
|
+
"title": "ResearchStrategy",
|
|
4
|
+
"description": "The planned attack on one knowledge gap (engine/autoresearch/contracts.py ResearchStrategy). Names the experiment type, the hypothesis it tests, the expected gain, and the research budget. strategy_id/hypothesis/expected_gain are required by ResearchStrategy.validate().",
|
|
5
|
+
"type": "object",
|
|
6
|
+
"required": [
|
|
7
|
+
"strategy_id",
|
|
8
|
+
"experiment_type",
|
|
9
|
+
"hypothesis",
|
|
10
|
+
"expected_gain",
|
|
11
|
+
"budget"
|
|
12
|
+
],
|
|
13
|
+
"properties": {
|
|
14
|
+
"strategy_id": {
|
|
15
|
+
"type": "string",
|
|
16
|
+
"minLength": 1
|
|
17
|
+
},
|
|
18
|
+
"experiment_type": {
|
|
19
|
+
"enum": [
|
|
20
|
+
"TARGETED_RETRIEVAL",
|
|
21
|
+
"COUNTER_EVIDENCE_RETRIEVAL",
|
|
22
|
+
"APPLICABILITY_RETRIEVAL",
|
|
23
|
+
"TEMPORAL_REFRESH",
|
|
24
|
+
"CITATION_CHAINING",
|
|
25
|
+
"SCREENING_PRIORITY",
|
|
26
|
+
"SOURCE_RECOVERY"
|
|
27
|
+
]
|
|
28
|
+
},
|
|
29
|
+
"hypothesis": {
|
|
30
|
+
"type": "string",
|
|
31
|
+
"minLength": 1
|
|
32
|
+
},
|
|
33
|
+
"expected_gain": {
|
|
34
|
+
"type": "string",
|
|
35
|
+
"minLength": 1
|
|
36
|
+
},
|
|
37
|
+
"budget": {
|
|
38
|
+
"type": "object",
|
|
39
|
+
"required": [
|
|
40
|
+
"max_queries",
|
|
41
|
+
"max_candidates",
|
|
42
|
+
"max_fulltext_fetches"
|
|
43
|
+
],
|
|
44
|
+
"properties": {
|
|
45
|
+
"max_queries": {
|
|
46
|
+
"type": "integer",
|
|
47
|
+
"minimum": 0
|
|
48
|
+
},
|
|
49
|
+
"max_candidates": {
|
|
50
|
+
"type": "integer",
|
|
51
|
+
"minimum": 0
|
|
52
|
+
},
|
|
53
|
+
"max_fulltext_fetches": {
|
|
54
|
+
"type": "integer",
|
|
55
|
+
"minimum": 0
|
|
56
|
+
}
|
|
57
|
+
},
|
|
58
|
+
"additionalProperties": false
|
|
59
|
+
}
|
|
60
|
+
},
|
|
61
|
+
"additionalProperties": false
|
|
62
|
+
}
|