@zhuoyuezs/ml-platform 0.1.7 → 0.1.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/checksums.json CHANGED
@@ -2,8 +2,8 @@
2
2
  "files": [
3
3
  {
4
4
  "path": "runtime/business-client/README.md",
5
- "sha256": "sha256:b12f30609c756ac4473723df434e6b1055f8713b7acfc3278b091b52288f6347",
6
- "size_bytes": 2459
5
+ "sha256": "sha256:02e6561b80cbef3cce2fd690750d863d8afb9c45bf5ccfcce3e3e2b082ff0761",
6
+ "size_bytes": 3780
7
7
  },
8
8
  {
9
9
  "path": "runtime/business-client/package-lock.json",
@@ -22,8 +22,8 @@
22
22
  },
23
23
  {
24
24
  "path": "runtime/business-client/src/cli.js",
25
- "sha256": "sha256:ca7bbee61e83f4597c0da75fd3527f1ee43d5257fff07693b581987efeeaafd3",
26
- "size_bytes": 41926
25
+ "sha256": "sha256:6ee4ed29a9df9f462502ccc0e10d5d9109bf00b52b8c27c9808730c7d8520592",
26
+ "size_bytes": 44480
27
27
  },
28
28
  {
29
29
  "path": "runtime/business-client/src/config.js",
@@ -37,8 +37,8 @@
37
37
  },
38
38
  {
39
39
  "path": "skills/feature-management/SKILL.md",
40
- "sha256": "sha256:ba725b6be2d0f8e4d741d3a6f96e64657112c15935f8e9fd2818cd0b17380816",
41
- "size_bytes": 30428
40
+ "sha256": "sha256:50646fd4c5884fe25b9488f223b680ad44417b96da160950b454202ea756e7f8",
41
+ "size_bytes": 30846
42
42
  },
43
43
  {
44
44
  "path": "skills/feature-management/agents/openai.yaml",
@@ -97,8 +97,8 @@
97
97
  },
98
98
  {
99
99
  "path": "skills/feature-management/references/contracts.md",
100
- "sha256": "sha256:913d5f0ad52b5183aa3a4bdeef2b6974f7757842300f547679fee38552792892",
101
- "size_bytes": 27738
100
+ "sha256": "sha256:b705dec6e801f95d0d354a0d0f0d6fb5bd139d5f9938f78d70f1fe108b7c95b9",
101
+ "size_bytes": 27741
102
102
  },
103
103
  {
104
104
  "path": "skills/feature-management/references/operator-authoring.md",
@@ -107,33 +107,48 @@
107
107
  },
108
108
  {
109
109
  "path": "skills/feature-management/references/platform-capability-guide.md",
110
- "sha256": "sha256:40d0f43a30688c5852ef09203ec35b4dc2c6b9db4bc7dd26dfae52e12d10e9ee",
111
- "size_bytes": 4257
110
+ "sha256": "sha256:6acb35f57e1bd6385702b6114e96c0ed1db34c9650d18658f74f212af9f91d14",
111
+ "size_bytes": 4423
112
+ },
113
+ {
114
+ "path": "skills/feature-management/references/supervised-datasets.md",
115
+ "sha256": "sha256:2c7a894330a649248e7f52c944422bcd7a3aba4558f7984d129dfd21c8736cf3",
116
+ "size_bytes": 4540
112
117
  },
113
118
  {
114
119
  "path": "skills/model-lifecycle-management/SKILL.md",
115
- "sha256": "sha256:48c60d20ab9203cde9681fc61a0515ce3f316ad674e4f349aa21e320127a7520",
116
- "size_bytes": 2837
120
+ "sha256": "sha256:98437e919b6d76822b2a3521713400b2c73ae42fbe5086f2e0b864aadd6cbf55",
121
+ "size_bytes": 3464
117
122
  },
118
123
  {
119
124
  "path": "skills/model-lifecycle-management/agents/openai.yaml",
120
125
  "sha256": "sha256:37857849a8baa6b06b6d5d07ae6036080239356fcccd36f77bbdc52c1dfa283f",
121
126
  "size_bytes": 249
122
127
  },
128
+ {
129
+ "path": "skills/model-lifecycle-management/references/discovery.md",
130
+ "sha256": "sha256:c87a9efacb1a18125bc1ee1b3061c5a9b876355e238aa5f7f17a985014413281",
131
+ "size_bytes": 4899
132
+ },
123
133
  {
124
134
  "path": "skills/model-lifecycle-management/references/evaluation.md",
125
- "sha256": "sha256:0139e5fc79beeab84fb31792bdac37c3331676e1f591c57390ecc6336873dcc9",
126
- "size_bytes": 1797
135
+ "sha256": "sha256:099fc54af2a1d8bed287f9e557cee8583558e594880235c32b89276b385cce90",
136
+ "size_bytes": 8858
127
137
  },
128
138
  {
129
139
  "path": "skills/model-lifecycle-management/references/packaging.md",
130
- "sha256": "sha256:c987d4095f23a1f18ba55effb14f946e0f4cdcb661d7af28af560cc8f72d8b2c",
131
- "size_bytes": 1101
140
+ "sha256": "sha256:fd87f0ea6f79bff0498ff1688ab782340d572815872c86520032549910160028",
141
+ "size_bytes": 2858
142
+ },
143
+ {
144
+ "path": "skills/model-lifecycle-management/references/training-contracts.md",
145
+ "sha256": "sha256:e93f5361f5f7cfaef53ba65b2606e58b06585d7de6c966706c66995e844b4804",
146
+ "size_bytes": 6909
132
147
  },
133
148
  {
134
149
  "path": "skills/model-lifecycle-management/references/training.md",
135
- "sha256": "sha256:e6971f2ef9c2b8645388dd4bac4562408f7a7b1adb0b7171c7e6b4912ea48b50",
136
- "size_bytes": 1852
150
+ "sha256": "sha256:10a6694f1904ce210f062221273a0687574354764bafb573318a5f7fcb425e55",
151
+ "size_bytes": 4206
137
152
  }
138
153
  ],
139
154
  "schema_version": "data_platform.ml_platform_checksums/v1"
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@zhuoyuezs/ml-platform",
3
- "version": "0.1.7",
3
+ "version": "0.1.8",
4
4
  "description": "ML 数据平台 business CLI 与 Agent Skills 管理工具",
5
5
  "bin": {
6
6
  "ml-platform": "scripts/main.js"
package/release.json CHANGED
@@ -13,7 +13,7 @@
13
13
  "entrypoint": "src/cli.js",
14
14
  "name": "ml-platform",
15
15
  "path": "runtime/business-client",
16
- "sha256": "sha256:bbb3c59d5d7e16f92e6496e5596e59af3af4fdf4f820487d9ceb6eb85ca0857b",
16
+ "sha256": "sha256:fb2ebe76ddba0ccfcd052741e8e290e25fc2897df7656d48267e82fe6be06d1c",
17
17
  "version": "0.7.0"
18
18
  },
19
19
  "policy_sha256": "sha256:02fbde0134696b78c0c365d6e0b76c9814227208da5b2b3d4e5dbd82dc4a3606",
@@ -25,7 +25,7 @@
25
25
  "ml_data_platform.feature_set/v1",
26
26
  "ml_data_platform.dataset_manifest/v1"
27
27
  ],
28
- "release_version": "0.1.7",
28
+ "release_version": "0.1.8",
29
29
  "runtime_requirements": {
30
30
  "node": ">=18",
31
31
  "os": [
@@ -38,15 +38,15 @@
38
38
  "feature-management": {
39
39
  "path": "skills/feature-management",
40
40
  "requires_cli": ">=0.7.0 <0.8.0",
41
- "revision": "0.1.7",
42
- "sha256": "sha256:15131c1303b143373e4b8671301f5b884013a1906a44cfc85dd6f93f43c5d8c8"
41
+ "revision": "0.1.8",
42
+ "sha256": "sha256:38ec81d73353bf8eb39ae56074c15cc0fcae14701e8aba1462b5cbd155a1fa65"
43
43
  },
44
44
  "model-lifecycle-management": {
45
45
  "path": "skills/model-lifecycle-management",
46
46
  "requires_cli": ">=0.7.0 <0.8.0",
47
- "revision": "0.1.7",
48
- "sha256": "sha256:b6eac3ac75a82cf27db2efe2daf2b5b8777c95a0841d48f93ada29ef08b8d8fe"
47
+ "revision": "0.1.8",
48
+ "sha256": "sha256:e3eefccb76748847975024cc6fc124df1a581d286d8de165195bd708f5dfa00f"
49
49
  }
50
50
  },
51
- "source_commit": "62db427abfd17b9c8dc58964b03aec21927217b5"
51
+ "source_commit": "07afde3b78bcf2fcee220a136f226c0ccdf96e38"
52
52
  }
@@ -54,3 +54,25 @@ uv run python scripts/generate_openapi_schema.py --ref origin/dev
54
54
  uv run python scripts/generate_openapi_schema.py --ref origin/dev --check
55
55
  npm run api:check --prefix business-client-js
56
56
  ```
57
+
58
+ ## Lifecycle discovery and preflight
59
+
60
+ Use `get-model-runtime-release` to inspect approved trainer/device/image bindings
61
+ and the effective evaluation identity. Find immutable resources with
62
+ `list-training-runs`, `list-training-jobs`, `list-model-artifacts`,
63
+ `list-evaluation-configs`, `list-evaluation-runs`, `list-evaluation-jobs`,
64
+ `list-metric-definitions`, `list-executable-packages`, and `list-model-packages`.
65
+ These commands accept `-q`, `--limit` (1–500), and `--offset`; responses contain
66
+ `items`, `total`, `limit`, `offset`. Query matches serialized metadata substrings.
67
+ Use the returned exact IDs with detail commands; Job lists contain immutable
68
+ specs, while Job detail refreshes status.
69
+
70
+ `validate-training-job REQUEST_JSON` and `validate-model-package REQUEST_JSON`
71
+ resolve the same contracts as submission without registering or scheduling
72
+ workloads. Read `submitted=false` and `unchecked`: successful preflight does not
73
+ verify image pulls, cluster resources or runtime execution. Complete request
74
+ examples and identity propagation are in the bundled model lifecycle Skill.
75
+
76
+ Maintainers can export the current API, including uncommitted schema changes,
77
+ with `uv run python scripts/generate_openapi_schema.py --working-tree` from the
78
+ repository root; use `--check` with it to verify the committed snapshot.
@@ -9,7 +9,12 @@ const { PlatformApiClient } = require("./http");
9
9
  const { applyCatalog, loadCatalog, normalizeDataset, normalizeRegistryDataset } = require("./catalog");
10
10
  const { version: CLIENT_VERSION } = require("../package.json");
11
11
 
12
+ const LIFECYCLE_COLLECTIONS = new Set(["training-runs", "training-jobs", "model-artifacts", "evaluation-configs", "evaluation-runs", "evaluation-jobs", "metric-definitions", "executable-packages", "model-packages"]);
13
+
12
14
  const BUSINESS_COMMANDS = new Set([
15
+ ...[...LIFECYCLE_COLLECTIONS].map((name) => `list-${name}`),
16
+ "get-model-runtime-release", "get-training-run", "get-model-artifact", "get-executable-package",
17
+ "validate-training-job", "validate-model-package",
13
18
  "version", "create-project", "list-projects", "get-project", "delete-project",
14
19
  "configure", "show-config", "health", "list-parameters", "list-operators", "list-features",
15
20
  "list-feature-sets", "list-datasets", "get-dataset", "list-jobs", "resolve-manifest",
@@ -33,6 +38,13 @@ const BUSINESS_COMMANDS = new Set([
33
38
  ]);
34
39
 
35
40
  const COMMAND_USAGE = {
41
+ ...Object.fromEntries([...LIFECYCLE_COLLECTIONS].map((name) => [`list-${name}`, `list-${name} [--limit N] [--offset N] [-q QUERY]`])),
42
+ "get-model-runtime-release": "get-model-runtime-release",
43
+ "get-training-run": "get-training-run RUN_ID",
44
+ "get-model-artifact": "get-model-artifact ARTIFACT_ID",
45
+ "get-executable-package": "get-executable-package NAME VERSION",
46
+ "validate-training-job": "validate-training-job REQUEST_JSON",
47
+ "validate-model-package": "validate-model-package REQUEST_JSON",
36
48
  version: "version",
37
49
  configure: "configure --api-url URL",
38
50
  "show-config": "show-config",
@@ -354,7 +366,7 @@ async function runBusinessCli(argv) {
354
366
  if (allPages && (rawLimit !== undefined || rawOffset !== undefined)) throw new Error("--all cannot be combined with --limit or --offset");
355
367
  if (allPages) result = await fetchAll(client(options), `/${endpoint}/${encodeURIComponent(project)}/${encodeURIComponent(name)}`);
356
368
  else { const limit = Number(rawLimit ?? 50); const offset = Number(rawOffset ?? 0); if (!Number.isInteger(limit) || limit < 1 || limit > 500) throw new Error("limit must be between 1 and 500"); if (!Number.isInteger(offset) || offset < 0) throw new Error("offset must be >= 0"); result = await client(options).get(`/${endpoint}/${encodeURIComponent(project)}/${encodeURIComponent(name)}`, { limit, offset }); }
357
- } else if (options.command.startsWith("list-") && !["list-jobs", "list-dataset-artifacts", "list-trainer-definitions"].includes(options.command)) {
369
+ } else if (["list-parameters", "list-operators", "list-features", "list-feature-sets", "list-datasets"].includes(options.command)) {
358
370
  const endpoints = { "list-parameters": "/parameters", "list-operators": "/operators", "list-features": "/features", "list-feature-sets": "/feature-sets", "list-datasets": "/datasets" };
359
371
  const project = take(rest, "--project");
360
372
  const name = take(rest, "--name"); const version = take(rest, "--version");
@@ -439,6 +451,21 @@ async function runBusinessCli(argv) {
439
451
  const request = { manifest: readDataset(manifest), cutoff_time: cutoffTime }; if (Object.keys(realtime).length) request.realtime_fetch = realtime; result = await client(options).post("/inference-data/fetch", request);
440
452
  } else if (options.command === "fetch-inference-context") {
441
453
  result = await client(options).post("/inference-data/context", readSpec(positional(rest, "request JSON")));
454
+ } else if (options.command === "get-model-runtime-release") {
455
+ result = await client(options).get("/model-runtime-release");
456
+ } else if (options.command.startsWith("list-") && LIFECYCLE_COLLECTIONS.has(options.command.slice(5))) {
457
+ const params = {};
458
+ const limit = take(rest, "--limit"); const offset = take(rest, "--offset"); const query = take(rest, "-q");
459
+ if (limit !== undefined) { params.limit = integer(limit, "--limit"); if (params.limit < 1 || params.limit > 500) throw new Error("--limit must be between 1 and 500"); }
460
+ if (offset !== undefined) { params.offset = integer(offset, "--offset"); if (params.offset < 0) throw new Error("--offset must be non-negative"); }
461
+ if (query !== undefined) params.q = query;
462
+ result = await client(options).get(`/${options.command.slice(5)}`, params);
463
+ } else if (options.command === "get-training-run" || options.command === "get-model-artifact") {
464
+ const endpoint = options.command === "get-training-run" ? "training-runs" : "model-artifacts";
465
+ result = await client(options).get(`/${endpoint}/${encodeURIComponent(positional(rest, "resource id"))}`);
466
+ } else if (options.command === "get-executable-package") {
467
+ const name = positional(rest, "package name"); const version = positional(rest, "package version");
468
+ result = await client(options).get(`/executable-packages/${encodeURIComponent(name)}/${encodeURIComponent(version)}`);
442
469
  } else if (options.command === "list-trainer-definitions") {
443
470
  result = await client(options).get("/trainer-definitions");
444
471
  } else if (options.command === "get-trainer-definition") {
@@ -446,11 +473,11 @@ async function runBusinessCli(argv) {
446
473
  result = await client(options).get(`/trainer-definitions/${encodeURIComponent(project)}/${encodeURIComponent(name)}/${encodeURIComponent(version)}`);
447
474
  } else if ({
448
475
  "register-trainer": "/trainer-definitions", "register-training-run": "/training-runs", "validate-training-run": "/training-runs/validate",
449
- "submit-training-job": "/training-jobs", "register-evaluation-config": "/evaluation-configs", "register-metric-definition": "/metric-definitions",
476
+ "validate-training-job": "/training-jobs/validate", "validate-model-package": "/model-packages/validate", "submit-training-job": "/training-jobs", "register-evaluation-config": "/evaluation-configs", "register-metric-definition": "/metric-definitions",
450
477
  "register-executable-package": "/executable-packages", "validate-evaluation-run": "/evaluation-runs/validate", "submit-evaluation": "/evaluation-runs",
451
478
  "create-model-package": "/model-packages",
452
479
  }[options.command]) {
453
- const endpoint = { "register-trainer": "/trainer-definitions", "register-training-run": "/training-runs", "validate-training-run": "/training-runs/validate", "submit-training-job": "/training-jobs", "register-evaluation-config": "/evaluation-configs", "register-metric-definition": "/metric-definitions", "register-executable-package": "/executable-packages", "validate-evaluation-run": "/evaluation-runs/validate", "submit-evaluation": "/evaluation-runs", "create-model-package": "/model-packages" }[options.command];
480
+ const endpoint = { "register-trainer": "/trainer-definitions", "register-training-run": "/training-runs", "validate-training-run": "/training-runs/validate", "validate-training-job": "/training-jobs/validate", "validate-model-package": "/model-packages/validate", "submit-training-job": "/training-jobs", "register-evaluation-config": "/evaluation-configs", "register-metric-definition": "/metric-definitions", "register-executable-package": "/executable-packages", "validate-evaluation-run": "/evaluation-runs/validate", "submit-evaluation": "/evaluation-runs", "create-model-package": "/model-packages" }[options.command];
454
481
  result = await client(options).post(endpoint, readSpec(positional(rest, "request JSON")));
455
482
  } else if (/^(get|retry|cancel)-training-job$/.test(options.command)) {
456
483
  const jobId = positional(rest, "training job id"); const endpoint = `/training-jobs/${encodeURIComponent(jobId)}`;
@@ -24,9 +24,11 @@ It records versions, lineage, quality, missingness, freshness, and execution
24
24
  metadata so an algorithm project can consume data without owning source routing
25
25
  or cache details.
26
26
 
27
- It does not train, evaluate, or serve models; own complete business feature
27
+ This Skill does not train, evaluate, or serve models; own complete business feature
28
28
  engineering; provision arbitrary source tables; or administer Kubernetes and
29
- platform services. Read
29
+ platform services. Training, evaluation and model packaging are handled by the
30
+ separate `model-lifecycle-management` Skill; this scope limit is not a claim that
31
+ the platform lacks those capabilities. Read
30
32
  [references/platform-capability-guide.md](references/platform-capability-guide.md)
31
33
  when the user needs the detailed platform boundary or evidence-level explanation.
32
34
 
@@ -308,6 +310,11 @@ Reference immutable Feature versions in the exact consumer column order. Keep th
308
310
 
309
311
  ### Define The DatasetManifest
310
312
 
313
+ For supervised datasets, read
314
+ [references/supervised-datasets.md](references/supervised-datasets.md) before
315
+ publishing the first version. Settle labels, clocks, split bindings and the final
316
+ DataSchema before expensive training.
317
+
311
318
  Do not use DatasetManifest as a catch-all for unresolved semantics. Confirm the
312
319
  dataset purpose and consumer, then separately confirm the target grid, horizon,
313
320
  read policy, explicit Parameter outputs, rowset or rowset splits, abnormal windows and
@@ -524,7 +524,7 @@ Supported strategies are `event_driven`, `fixed_grid`, and
524
524
  `exact_horizon_measured` requires `horizon` and accepts a `tolerance`. An
525
525
  optional rowset `time_range` overrides the manifest range for that strategy.
526
526
  Use `rowset_splits` instead of `rowset` when training needs named strategies
527
- such as `support`, `validation`, and `test`; when present, it overrides the
527
+ such as `support`, `validation`, and `test`; it is mutually exclusive with the
528
528
  single `rowset`. Split ranges must not overlap and must cover every emitted
529
529
  target row. A split-level `fixed_grid.grid` is an explicit assertion and must
530
530
  equal `DatasetManifest.time_range.grid`; it cannot silently resample one split.
@@ -10,8 +10,10 @@ layer for algorithm projects. It turns confirmed source Parameters and approved
10
10
  deterministic Operators into ordered Features and reproducible DatasetManifests,
11
11
  then either materializes a batch DatasetArtifact or serves one causal realtime
12
12
  read. It owns source adapters, quality/missingness/freshness evidence, lineage,
13
- replay metadata, and Job execution; it does not own model training, model
14
- evaluation, model serving, source-table creation, or Kubernetes administration.
13
+ replay metadata, and data Job execution. This Skill covers that data subsystem;
14
+ use the separate model lifecycle Skill for training, evaluation and packaging.
15
+ Model serving, source-table creation and Kubernetes administration are outside
16
+ this Skill's scope. Do not present a Skill boundary as a platform capability gap.
15
17
 
16
18
  The following distinctions are part of the platform contract:
17
19
 
@@ -0,0 +1,101 @@
1
+ # Supervised dataset preflight
2
+
3
+ Use before the first immutable publication for training/evaluation. Examples
4
+ describe the current contract; deployed capabilities still require validation.
5
+
6
+ ## Labels and clocks
7
+
8
+ Keep labels out of FeatureSet order. Training rejects overlap with `data.labels`
9
+ even when an input adapter would omit that column. Generic supervised labels use
10
+ `label_materializations`, mutually exclusive with legacy `target`.
11
+
12
+ Merge this fragment into a complete manifest, replacing source identities:
13
+
14
+ ```json
15
+ {
16
+ "mode": "training",
17
+ "prediction": {"horizon": "5min"},
18
+ "label_materializations": [{
19
+ "name": "future_pressure",
20
+ "source": {"kind": "parameter", "ref": {
21
+ "name": "pressure", "version": "v1", "project": "default"
22
+ }},
23
+ "output_column": "future_pressure",
24
+ "event_time": {"source_field": "timestamp", "offset": "0min"},
25
+ "alignment": {"method": "exact"},
26
+ "missing_policy": "report",
27
+ "dtype": "float64"
28
+ }]
29
+ }
30
+ ```
31
+
32
+ `source_field` names a column in the normalized Parameter frame, commonly
33
+ `timestamp`, not automatically the output Feature frame's `event_time`. Other
34
+ names need source-frame evidence. The schema lists `dataset_column` and
35
+ `operator_output`, but the current materializer supports Parameter sources;
36
+ resolve before building rather than treating schema acceptance as runtime support.
37
+ Choose `report`, `reject` or `drop` according to the intended label population.
38
+
39
+ - `prediction_time = event_time` is the causal cutoff.
40
+ - `label_time = prediction_time + prediction.horizon` is the forecast target.
41
+ - `event_time.offset` shifts the source observation clock before alignment. At
42
+ exact alignment, source time `s` matches label time `t` when `s + offset = t`.
43
+
44
+ For cutoff 10:00 and horizon 5min, expect label time 10:05 and, with zero offset,
45
+ the source value at 10:05. A positive offset is not a shortcut to future values
46
+ and does not change the forecast horizon. `offset_minutes` belongs to legacy
47
+ `target`, not generic label materialization. Free-form dictionaries can accept
48
+ unused keys: inspect resolved semantics and sampled built rows too.
49
+
50
+ Check every Operator's `input_schema.prediction` before publication. A maximum
51
+ horizon of zero is incompatible with a 5min manifest. Review a new immutable
52
+ Operator version and causal tests, or select a compatible Operator; do not
53
+ compensate with label offsets or silently reduce the requested horizon. Zero
54
+ horizon does not itself disable named splits.
55
+
56
+ ## Splits and schema must precede training
57
+
58
+ `rowset_splits` is a mapping, not an array, and is mutually exclusive with
59
+ `rowset`. Values are RowsetStrategy objects: `strategy`, optional `time_range`,
60
+ and `grid` for `fixed_grid`. See [contracts.md](contracts.md) for the full split
61
+ example. Fixed-grid split grids must match the manifest grid; ranges are
62
+ half-open and must not overlap.
63
+
64
+ Building splits adds `rowset_split` and can change the complete DataSchema hash.
65
+ Settle split/metadata columns, labels, time columns, dtypes, roles and FeatureSet
66
+ order before training. Matching feature values alone does not ensure evaluation
67
+ compatibility after adding columns.
68
+
69
+ Dataset split declaration does not populate the TrainingRun binding. Hand this
70
+ fragment to the lifecycle consumer to merge under `data`:
71
+
72
+ ```json
73
+ {
74
+ "splits": {
75
+ "assignment_column": "rowset_split",
76
+ "train": ["training"],
77
+ "validation": ["validation"],
78
+ "test": ["test"]
79
+ }
80
+ }
81
+ ```
82
+
83
+ Keys are consumer roles; array values are actual artifact assignment values.
84
+ For a manifest split named `train`, use `["train"]`. Hand off the exact artifact
85
+ identity, DataSchema hash, FeatureSet order, label/temporal contracts, observed
86
+ split counts and binding together.
87
+
88
+ ## Evidence checkpoints
89
+
90
+ Before publication, review schema and Operator limits and run catalog dry-run.
91
+ After publication, resolve labels, source fields, horizon and dependencies before
92
+ building. After build, inspect DataSchema and sampled Parquet rows for the clock
93
+ equation, source-label alignment, nulls and split counts. A small authorized
94
+ diagnostic build can test source semantics but cannot replace final artifact
95
+ identity or full-data validation.
96
+
97
+ Before training, validate the TrainingRun with explicit splits and confirm
98
+ trainer shape/device and evaluation runtime support. Do not create successive
99
+ immutable dataset versions or launch training to discover field names. Stop at
100
+ the first unexplained error, preserve the request and evidence outside the
101
+ repository, and correct the draft once the cause is understood.
@@ -14,8 +14,19 @@ DatasetArtifact -> TrainingRun -> TrainingJob -> ModelArtifact
14
14
 
15
15
  Read [references/training.md](references/training.md) for training work, [references/evaluation.md](references/evaluation.md) for governed evaluation, and [references/packaging.md](references/packaging.md) for ModelArtifact validation and packaging. Read only the references required by the request.
16
16
 
17
+ Read [references/discovery.md](references/discovery.md) before locating existing
18
+ resources or selecting runtime bindings. It also defines Job/package preflight
19
+ and version-skew handling.
20
+
17
21
  ## Shared Gates
18
22
 
23
+ Before training a model that will be evaluated, also read
24
+ [references/evaluation.md](references/evaluation.md). Check the final dataset
25
+ schema, training split binding and evaluation runtime before submission.
26
+ OpenAPI free-form dictionaries are not complete nested contracts: use the
27
+ examples in these references and same-release validation, not successive 422s
28
+ to discover fields. Preserve release/error evidence when deployment differs.
29
+
19
30
  - Establish exact immutable inputs, TrainerDefinition `v1`, runtime image digest, model inputs/targets, resources, evaluation policy, and package destination from authoritative contracts. Do not infer them from model names.
20
31
  - Validate, register, submit, retry, cancel, and package creation are separate actions. Read-only discovery and validation do not authorize mutation or workload submission.
21
32
  - Before submission, show the exact JSON path, immutable identities, expected workload, and target API. Obtain explicit authorization. Retry and cancel require separate authorization for the exact Job.
@@ -0,0 +1,89 @@
1
+ # Discover exact contracts before authoring requests
2
+
3
+ Use the same-release CLI/API. Start with `version`, `show-config`, `health` and
4
+ `get-model-runtime-release`. If a documented command is absent from `--help`,
5
+ or a new endpoint returns 404, record client/server release evidence and upgrade
6
+ to the matching release; do not guess another resource name or submit work as a
7
+ capability probe.
8
+
9
+ ## Effective runtime policy
10
+
11
+ ```bash
12
+ ml-platform --profile server get-model-runtime-release
13
+ ml-platform --profile server list-trainer-definitions
14
+ ml-platform --profile server get-trainer-definition NAME v1 --project PROJECT
15
+ ```
16
+
17
+ Discovery returns `configured`, `release_id`, `approval_policy`, and `bindings`.
18
+ Each binding contains the exact `project/name:version:device` trainer key,
19
+ immutable training and Serving base images, input modes and architectures.
20
+ Compare the registered TrainerDefinition image against the binding before Job
21
+ preflight. A binding is release approval, not proof that a matching definition
22
+ is registered. `configured=false` means capability-based validation without a
23
+ release allowlist; it does not mean every trainer/device is approved by a release.
24
+ `evaluation.runtime_identity` and `default_executor` are effective server values.
25
+ `worker_dependencies_verified=false` explicitly means discovery has not tested
26
+ imports, image pulls or model prediction in the Worker.
27
+
28
+ ## Resource lookup and pagination
29
+
30
+ ```bash
31
+ ml-platform --profile server list-training-runs -q pressure --limit 50 --offset 0
32
+ ml-platform --profile server get-training-run RUN_ID
33
+ ml-platform --profile server list-training-jobs -q RUN_ID
34
+ ml-platform --profile server list-model-artifacts -q JOB_ID
35
+ ml-platform --profile server get-model-artifact ARTIFACT_ID
36
+ ml-platform --profile server list-evaluation-configs
37
+ ml-platform --profile server list-evaluation-runs -q ARTIFACT_ID
38
+ ml-platform --profile server list-evaluation-jobs -q RUN_ID
39
+ ml-platform --profile server list-metric-definitions
40
+ ml-platform --profile server list-executable-packages
41
+ ml-platform --profile server get-executable-package NAME VERSION
42
+ ml-platform --profile server list-model-packages -q ARTIFACT_ID
43
+ ```
44
+
45
+ All these list commands return `items`, `total`, `limit`, `offset`; default limit
46
+ is 50, maximum 500. Continue with offset plus returned item count until total is
47
+ reached. `-q` is a case-insensitive substring of serialized metadata, not a query
48
+ language or project authorization filter. Always verify exact identities after
49
+ search. Collections are global metadata views; no project filter is claimed.
50
+ Jobs in lists are immutable specs; use the exact get-job command for refreshed
51
+ status. Lists do not reconcile or schedule workloads. `get-training-run` returns
52
+ `training_run` plus `spec_hash`; artifact/package detail returns the object.
53
+
54
+ ## Read-only preflight
55
+
56
+ ```bash
57
+ ml-platform --profile server validate-training-job training-job-request.json
58
+ ml-platform --profile server validate-model-package model-package-request.json
59
+ ```
60
+
61
+ Training preflight requires an already registered Run with its exact hash and
62
+ resolves the same Job spec used by submission, including device/runtime approval,
63
+ resources and distributed capability. Packaging preflight resolves the same
64
+ artifact eligibility, selected runtime, input modes and immutable-version conflict
65
+ checks used by creation. Both return `status=valid`, the proposed `job_spec` or
66
+ `package`, `submitted=false` and `unchecked` execution checks. They do not register
67
+ Jobs/packages or invoke schedulers. Artifact reads can populate local caches.
68
+
69
+ A valid preview does not prove image availability, cluster capacity, credentials,
70
+ framework imports, training success or packaging test success. Read `unchecked`
71
+ and inspect an existing package's status: an immutable failed version is not
72
+ made runnable by validation. Submission rechecks current state; preflight is not
73
+ a reservation. Preserve the preview alongside the approved submission request.
74
+
75
+ ## Structured request errors
76
+
77
+ Training Run `task`, `data`, `data.artifact`, `features.feature_set`, `splits`,
78
+ `temporal` and `trainer` are explicit objects in OpenAPI. Evaluation artifact,
79
+ config references, members and execution, plus ModelPackage requests, are also
80
+ structured. Unknown keys return 422 with `detail[].loc` identifying the field.
81
+ Keep algorithm-specific `trainer.parameters` and documented plugin configuration
82
+ maps extensible; validate them against the registered plugin schema. Do not treat
83
+ all remaining `additionalProperties` as a platform schema defect.
84
+
85
+ These HTTP checks preserve accepted values and do not rewrite historical stored
86
+ contracts or hashes. Legacy records remain readable. A previously ignored typo
87
+ is now rejected on a new submission; correct the input rather than removing
88
+ validation. Omit evaluation runtime identity for the server default; if execution
89
+ is supplied it must name `executor` explicitly.
@@ -1,5 +1,144 @@
1
1
  # Evaluation Commands
2
2
 
3
+ ## Request bodies and prerequisites
4
+
5
+ Evaluation reuses the model's original TrainingRun, replaces its artifact
6
+ reference and compares the complete evaluation DataSchema hash with the trained
7
+ Job's hash. Matching feature names is insufficient: added split/metadata columns
8
+ or changed roles/dtypes can fail. Settle the final schema before training and
9
+ validate compatibility before concluding that retraining is required.
10
+
11
+ Example `evaluation-config.json` (exploratory, not a release gate):
12
+
13
+ ```json
14
+ {
15
+ "name": "pressure_metrics",
16
+ "version": "v1",
17
+ "task_type": "regression",
18
+ "mode": "exploratory",
19
+ "metrics": [{
20
+ "name": "regression.mae", "version": "v1",
21
+ "targets": ["future_pressure"], "calculation_space": "business"
22
+ }]
23
+ }
24
+ ```
25
+
26
+ Inspect `get-metric-definition regression.mae v1` before relying on this metric.
27
+ Release mode additionally requires at least one required validation rule with a
28
+ business-approved threshold; do not invent thresholds to obtain PASS.
29
+
30
+ Example `evaluation-request.json` shape; replace illustrative identities before
31
+ validation or submission:
32
+
33
+ ```json
34
+ {
35
+ "model_artifact_id": "replace_model_artifact_id",
36
+ "dataset": {
37
+ "project": "default",
38
+ "dataset_id": "replace_dataset_id",
39
+ "manifest_hash": "replace_with_exact_artifact_manifest_hash"
40
+ },
41
+ "rowset": "test",
42
+ "targets": ["future_pressure"],
43
+ "config": {"name": "pressure_metrics", "version": "v1"},
44
+ "execution": {
45
+ "executor": "kubernetes",
46
+ "resources": {"cpu": "2", "memory": "3Gi"},
47
+ "deadline": "90min",
48
+ "retry_limit": 0
49
+ }
50
+ }
51
+ ```
52
+
53
+ `config` is a name/version reference, not an inline policy. `targets` are
54
+ model-bound label column names. The HTTP validation and submission endpoints
55
+ fill omitted `runtime_identity` with the server's supported evaluation identity;
56
+ omit it rather than guessing. Explicit pinning requires that exact identity
57
+ (`sha256:` plus 64 hex digits), not an image digest or trainer hash. Inspect the
58
+ returned frozen Job's `policy.runtime_identity` and retain it as evidence.
59
+ Resources above illustrate shape, not Chronos-2 sizing guidance.
60
+
61
+ If `execution` is omitted entirely, these HTTP endpoints use the configured
62
+ server execution backend. If supplied, include `executor` explicitly: an empty
63
+ execution dictionary falls back to the domain's `local` default. Do not assume
64
+ the domain default is the deployed HTTP default. Kubernetes selection alone does
65
+ not prove Worker dependency availability. Validate before submitting and confirm
66
+ the approved Worker supports the model plugin. Successful training in another
67
+ image does not establish evaluation support.
68
+
69
+ ## Diagnose before another build or training run
70
+
71
+ | Symptom | Actual check and next action |
72
+ | --- | --- |
73
+ | `requires a named split` | Inherited TrainingRun `data.splits` is not a dictionary and requested rowset is not `all`. Inspect the original training request, not only manifest/resolved/Parquet splits. This check has no horizon-zero branch. |
74
+ | `rowset is not declared` | Inspect `data.splits` role keys (`train`, `validation`, `test`), nonempty value arrays, assignment column and observed values. Manifest `training` can map to consumer role `train`. |
75
+ | DataSchema mismatch | Compare the full schema with the trained Job's hash, including split columns, roles and dtypes. Do not remove columns or falsify hashes to pass. |
76
+ | Label leakage | Materialize label columns separately from FeatureSet order; input selection alone cannot repair the contract. |
77
+ | Chronos-2 runtime unavailable | Record executor, error stage/Job, runtime identity and Worker image evidence. The prediction plugin cannot load a dependency; request runtime support from the platform owner. New dataset/model versions do not install dependencies. |
78
+
79
+ Do not change to `rowset=all` to bypass a requested holdout. Adding `data.splits`
80
+ to EvaluationRunRequest is unsupported and cannot repair an immutable original
81
+ TrainingRun. Explain the missing binding and validate a corrected training draft
82
+ before any authorized retraining.
83
+
84
+ `ExecutablePackage` is a custom metric Python wheel with verification and
85
+ publication lifecycle, not a model Serving container. Built-in metrics need no
86
+ user-created wheel. `ModelPackage` is a separate model image packaging object;
87
+ creating one does not configure evaluation Workers or automatically satisfy the
88
+ evaluation runtime identity. Empty package registration cannot repair a
89
+ Chronos-2 import failure.
90
+
91
+ ## Discovery and evidence retention
92
+
93
+ Submission returns `run.run_id`, singular `job.job_id` / `job_status` for the
94
+ first member, and `jobs[].spec` / `jobs[].status` for all members. Track all
95
+ members when present. `get-evaluation-job` returns `spec`, `status`, `attempts`;
96
+ inspect `status.phase` and, on success, `status.result_id`. Top-level
97
+ `status=accepted` is not completion. Read `spec.policy.runtime_identity` for the
98
+ resolved runtime. There is no evaluation wait CLI; poll the get command with a
99
+ bounded interval/deadline and preserve the ID on timeout. Do not use dataset
100
+ `wait-job` for a training or evaluation Job.
101
+
102
+ Evaluation request reuse is normally deduplicated. `force_rerun=true` explicitly
103
+ requests another execution; do not set it to work around an unexplained timeout
104
+ or missing result. Retry keeps the immutable contract; corrections to inputs
105
+ require a corrected request, not a retry with hidden changed semantics.
106
+
107
+ Default coverage requires at least one row and both prediction and label
108
+ coverage of 1.0. Null labels and sequence context loss can prevent a gate pass
109
+ despite successful computation. Do not lower coverage to obtain PASS. If the
110
+ business policy intentionally permits incomplete coverage, declare the approved
111
+ values under `coverage.minimum_rows`, `minimum_prediction_coverage` and
112
+ `minimum_label_coverage`, and report the excluded population.
113
+
114
+ A release-rule shape is shown below; the threshold is illustrative and must be
115
+ replaced by a business-approved value. Merge under the policy and set
116
+ `mode=release`; `metric` and version must also appear in `metrics`.
117
+
118
+ ```json
119
+ {
120
+ "validation": [{
121
+ "metric": "regression.mae", "version": "v1",
122
+ "target": "future_pressure", "operator": "<=", "threshold": 1.0,
123
+ "required": true, "level": "job", "source": "metric", "slice": "overall"
124
+ }]
125
+ }
126
+ ```
127
+
128
+ `slices` names metadata/entity-key columns, not filter expressions or arbitrary
129
+ Feature columns. `SUCCEEDED` describes execution; `PASS`, `FAIL`, `INCONCLUSIVE`
130
+ describe decisions. Inspect required decisions and coverage before reporting
131
+ release readiness. A Result and a Summary are distinct evidence objects;
132
+ record the actual returned IDs rather than deriving one ID from another.
133
+
134
+ Use `list-evaluation-configs`, `list-evaluation-runs`, `list-evaluation-jobs`
135
+ and their get commands to locate prior evidence; see [discovery.md](discovery.md)
136
+ for paging and exact identity recovery. `list-evaluation-attempts JOB_ID` lists
137
+ attempt history, not all runs. Preserve request/config versions and returned
138
+ run/job/result/summary IDs; do not search metric names instead of provenance.
139
+
140
+ ## Commands
141
+
3
142
  ```bash
4
143
  ml-platform --profile server register-evaluation-config evaluation-config.json
5
144
  ml-platform --profile server register-metric-definition metric-definition.json
@@ -1,7 +1,41 @@
1
1
  # Model Packaging Commands
2
2
 
3
+ ## Request shape and managed runtime
4
+
5
+ ```json
6
+ {
7
+ "model_artifact_id": "replace_registered_model_artifact_id",
8
+ "package_version": "1.0.0",
9
+ "input_modes": ["inline"]
10
+ }
11
+ ```
12
+
13
+ The artifact must be `REGISTERED` and pass validation. Creation submits work;
14
+ first use `validate-model-package` for read-only contract/runtime preflight. Version defaults to `1` if omitted;
15
+ use an explicit immutable version (semantic versioning is a convention, not an
16
+ enforced three-component schema). Capture `package.package_id` and
17
+ `package.package_version` from the creation response.
18
+
19
+ The server resolves images and destination from the approved trainer/device
20
+ runtime configuration. Optional request keys `base_image`, `worker_image`,
21
+ `builder_image`, `destination_repository` are equality assertions against that
22
+ configuration, not arbitrary overrides. Omit them when using managed defaults;
23
+ do not guess registries or pass deployment secrets. A mismatch needs the approved
24
+ runtime contract, not repeated package versions. Unknown request keys return 422; do not add `runtime`, `image`, or `resources`
25
+ fields. Inspect the returned resolved package.
26
+
27
+ `input_modes` must be unique, include `inline`, and may additionally include
28
+ `feature_lookup` only when supported by the selected runtime and artifact lineage
29
+ has `online_eligible=true`. Do not infer online eligibility from training success.
30
+ Omit rather than use an empty list to request default inline behavior.
31
+
32
+ Always query the exact package ID and version; states are `PACKAGING`, `READY`,
33
+ `FAILED`, `CANCELLED`. A client timeout does not prove packaging failed. Retain the
34
+ known identity and query it before attempting another creation.
35
+
3
36
  ```bash
4
37
  ml-platform --profile server validate-model-artifact <artifact-id>
38
+ ml-platform --profile server validate-model-package model-package-request.json
5
39
  ml-platform --profile server create-model-package model-package-request.json
6
40
  ml-platform --profile server get-model-package <package-id> --package-version <version>
7
41
  ```
@@ -0,0 +1,139 @@
1
+ # Training request and response contracts
2
+
3
+ These examples require only the released CLI, inspected registry definitions and
4
+ downloaded dataset contracts. Replace example identities and business choices;
5
+ they are not a ready-to-submit trainer recommendation. Never import platform
6
+ Python modules or require a checkout to author these requests.
7
+
8
+ ## TrainingRunSpec
9
+
10
+ Example forecasting sequence run:
11
+
12
+ ```json
13
+ {
14
+ "schema_version": "ml_data_platform.training_run/v2",
15
+ "run_id": "pressure_forecast_run_01",
16
+ "experiment": "pressure_forecast",
17
+ "task": {"kind": "forecasting", "objective": "regression"},
18
+ "data": {
19
+ "artifact": {
20
+ "project": "default", "dataset_id": "pressure_dataset",
21
+ "manifest_hash": "REPLACE_WITH_RETURNED_MANIFEST_HASH"
22
+ },
23
+ "features": {"feature_set": {
24
+ "project": "default", "name": "pressure_features", "version": "v1"
25
+ }},
26
+ "labels": ["future_pressure"],
27
+ "temporal": {
28
+ "prediction_time_column": "event_time", "label_time_column": "label_time",
29
+ "frequency": "5min", "horizon": "5min", "series_keys": []
30
+ },
31
+ "splits": {
32
+ "assignment_column": "rowset_split", "train": ["training"],
33
+ "validation": ["validation"], "test": ["test"]
34
+ }
35
+ },
36
+ "input_adapter": {
37
+ "kind": "sequence", "context_length": 24,
38
+ "context_end": "prediction_time_inclusive", "frequency": "5min",
39
+ "stride": 1, "gap_policy": "reject", "padding_policy": "none"
40
+ },
41
+ "label_transform": {"name": "identity", "version": "v1"},
42
+ "trainer": {
43
+ "project": "default", "name": "replace_approved_trainer", "version": "v1",
44
+ "parameters": {}
45
+ },
46
+ "reproducibility": {"seed": 0, "deterministic": true},
47
+ "evaluation": {"split": "validation", "metrics": ["mae", "rmse", "r2"]}
48
+ }
49
+ ```
50
+
51
+ `task.kind`, objective and input kind must match inspected TrainerCapabilities.
52
+ `trainer.parameters` must satisfy the definition's parameter schema; `{}` only
53
+ works if no parameters are required. Resource/device settings belong to the Job,
54
+ not the Run. `run_id` is the immutable run identity, not a name/version pair.
55
+ Do not add guessed `version`, `model`, `dataset` or `hyperparameters` top-level keys.
56
+
57
+ For tabular input use `input_adapter: {"kind":"tabular"}`; omit sequence context
58
+ fields. Forecasting tasks still require `data.temporal`, even with tabular input.
59
+ Non-temporal regression can omit it. Do not change task kind just to evade checks.
60
+ `series_keys` must identify entity-key columns when multiple series coexist;
61
+ an empty list treats the entire artifact as one series.
62
+
63
+ Inspect `data_schema.json`, `resolved_manifest.json` and the feature Parquet via
64
+ `download-dataset-artifact DATASET_ID MANIFEST_HASH --project PROJECT --out-dir DIR`
65
+ (repeat `--file` for selective downloads). The FeatureSet reference must match
66
+ the artifact; the full ordered FeatureSet determines model channels. There is
67
+ no documented `data.features.columns` shortcut for choosing a subset.
68
+
69
+ `identity:v1` leaves labels unchanged. `difference` requires a baseline Feature
70
+ in `required_inputs` and at least one `online_inputs` entry, which must be a
71
+ subset; it must have an inverse. Do not use target transforms to fix incorrect
72
+ dataset label clocks or to invent unsupported transforms.
73
+
74
+ ## TrainingJobRequest and identity propagation
75
+
76
+ After validation and authorized registration, use the returned `spec_hash`
77
+ unchanged in the Job request. Do not calculate it from raw JSON bytes.
78
+
79
+ ```json
80
+ {
81
+ "run": {
82
+ "run_id": "pressure_forecast_run_01",
83
+ "spec_hash": "REPLACE_WITH_REGISTERED_SPEC_HASH"
84
+ },
85
+ "execution": {
86
+ "executor": "kubernetes", "device": "cpu",
87
+ "resources": {"cpu": "2", "memory": "4Gi"},
88
+ "deadline": "90min", "retry_limit": 0
89
+ }
90
+ }
91
+ ```
92
+
93
+ Sizing is illustrative. `spec_hash` uses `sha256:<64 hex>`. Omit `output_uri`
94
+ to use managed storage unless an approved contract supplies it. Runtime device
95
+ approval occurs during Job resolution; `validate-training-run` has no execution
96
+ device and cannot prove device approval. Use `validate-training-job` after Run registration to validate device and
97
+ execution settings without creating a Job.
98
+
99
+ | Command | Fields to retain / inspect |
100
+ | --- | --- |
101
+ | `validate-training-run` | `status=valid`, `spec_hash`, `data_schema_hash`, `feature_columns`, `label_columns` |
102
+ | `register-training-run` | `training_run.run_id`, top-level `spec_hash`; response `status=validated` also represents registration, not mere dry-run |
103
+ | `submit-training-job` | `job_spec.job_id`, `job_spec.job_spec_hash`, `job.phase`; top-level `status=accepted` is not completion |
104
+ | `get-training-job` | `spec` and `status`; inspect `status.phase`, then successful `status.artifact_id` |
105
+ | `validate-model-artifact` | Validation evidence for the exact returned artifact ID; Job success alone is insufficient |
106
+
107
+ Use `get-training-run`, `get-model-artifact` and their list commands to recover
108
+ identities; see [discovery.md](discovery.md). There is no training wait command.
109
+ Preserve the original Run JSON and IDs.
110
+ Poll `get-training-job` at a bounded interval (for example 15 seconds) with a
111
+ user-appropriate deadline; a polling timeout does not cancel or fail the Job.
112
+ After a submission timeout, retain the request and reconcile any known Job ID
113
+ before retrying. Never create a new run merely because the HTTP response was lost.
114
+
115
+ ## Sequence population semantics
116
+
117
+ Context step is `frequency * stride`. Stride spaces context observations; it
118
+ does not select every Nth target endpoint. Inclusive context with length L covers
119
+ offsets `(L-1)..0`; exclusive covers `L..1`. For 24 points, 5min frequency and
120
+ stride 1, inclusive context starts 115min before the cutoff, exclusive 120min.
121
+ Keep Operator source history and endpoint-policy lookback as separate contracts.
122
+
123
+ The adapter materializes history over the full series, then assigns samples by
124
+ the target endpoint's split. Test endpoints may use causal Feature history from
125
+ earlier splits; splitting does not truncate their context. Duplicate timestamps
126
+ within a series are invalid. Missing early context with no padding is skipped;
127
+ null Features or labels also skip samples. Candidate row counts therefore are
128
+ not training/evaluation sample counts. Report the materialized population and
129
+ coverage instead of assuming all Parquet rows were used.
130
+
131
+ `gap_policy=fill` requires `padding_policy=edge` or `zero`; non-fill requires
132
+ `padding_policy=none`. Filling changes model inputs and needs an intended
133
+ business policy. `skip` and padding are not automatic fixes for rejected grids.
134
+
135
+ Training's `evaluation.metrics` uses `mae`, `rmse`, `r2`. Independent governed
136
+ EvaluationConfig uses versioned names such as `regression.mae`. They are different
137
+ contracts. Training `evaluation.split=auto` prefers nonempty test, then validation,
138
+ then train; choose an explicit split when reserving the test set for final review.
139
+ Training metrics are not a governed EvaluationResult or release gate pass.
@@ -1,5 +1,52 @@
1
1
  # Training Commands
2
2
 
3
+ ## Preflight before expensive training
4
+
5
+ Settle the final DataSchema (including split/metadata and label/time columns),
6
+ FeatureSet order and evaluation runtime before the first Job. See
7
+ [evaluation.md](evaluation.md) for inherited split bindings and runtime failures.
8
+
9
+ - Inspect exact TrainerDefinition `project/name:v1`, entrypoint, image digest,
10
+ capabilities and parameter schema. Similar names in different projects are
11
+ distinct identities; registration does not imply runtime approval.
12
+ - Match definition and device to the active runtime release. On `not approved
13
+ for device`, preserve the rejected identity, device and release ID. Use `get-model-runtime-release` to find the
14
+ approved binding; if the deployed release lacks discovery, request its contract;
15
+ do not try similar names or re-register trainers to bypass approval.
16
+ - The current platform Chronos2Plugin requires `[sample, context, 1]`, exactly
17
+ one channel. This is a platform adapter restriction, not a universal statement
18
+ about Chronos-2. Six-feature sequences are incompatible. Do not silently drop
19
+ business inputs: select a suitable trainer or agree a univariate contract.
20
+ - Keep labels outside FeatureSet order, verify temporal horizon against
21
+ `label_time - prediction_time`, and validate the complete TrainingRun before
22
+ registration/submission. Validation may not catch undeclared plugin limits or
23
+ missing dependencies in the eventual Worker.
24
+
25
+ Merge this fragment under TrainingRun `data`, alongside `artifact`, `features`,
26
+ `labels` and the applicable `temporal` binding:
27
+
28
+ ```json
29
+ {
30
+ "splits": {
31
+ "assignment_column": "rowset_split",
32
+ "train": ["training"],
33
+ "validation": ["validation"],
34
+ "test": ["test"]
35
+ }
36
+ }
37
+ ```
38
+
39
+ `train` is the consumer role; `training` is an example observed artifact value.
40
+ Use the actual assignment values. Evaluation inherits this original binding;
41
+ it does not reconstruct it from manifest `rowset_splits`. Save the submitted
42
+ TrainingRun JSON, validation output, Job ID and ModelArtifact ID in the user's
43
+ working directory so later diagnosis needs no source repository.
44
+
45
+ ## Commands
46
+
47
+ Read [training-contracts.md](training-contracts.md) when authoring a Run or Job
48
+ JSON. It provides complete request shapes, response paths and sequence semantics.
49
+
3
50
  Discover already-registered TrainerDefinitions (read-only) before authoring a
4
51
  TrainingRunSpec or deciding whether registration is needed:
5
52
 
@@ -18,6 +65,7 @@ separate mutation below, and a 404 means the definition (not the model plugin) i
18
65
  ml-platform --profile server register-trainer trainer-definition.json
19
66
  ml-platform --profile server validate-training-run training-run.json
20
67
  ml-platform --profile server register-training-run training-run.json
68
+ ml-platform --profile server validate-training-job training-job-request.json
21
69
  ml-platform --profile server submit-training-job training-job-request.json
22
70
  ml-platform --profile server get-training-job <job-id>
23
71
  ```