ls-algorithm-plugin-sdk 0.3.3__py3-none-any.whl → 0.3.5__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -5,6 +5,7 @@ import logging
5
5
  import sys
6
6
 
7
7
  from .cli_impl import configure as configure_command
8
+ from .cli_impl import cut as cut_command
8
9
  from .cli_impl import run as run_command
9
10
  from .cli_impl import serve as serve_command
10
11
 
@@ -22,6 +23,9 @@ def build_parser() -> argparse.ArgumentParser:
22
23
  configure_command.configure_parser(
23
24
  subs.add_parser("configure", help="configure an existing Algorithm repository for systemd")
24
25
  )
26
+ cut_command.configure_parser(
27
+ subs.add_parser("cut", help="cut each dataset episode to a maximum duration")
28
+ )
25
29
  return parser
26
30
 
27
31
 
@@ -62,6 +62,16 @@ def configure_parser(parser: argparse.ArgumentParser) -> None:
62
62
  help="maximum concurrent executions (default: GPU count, or 1)",
63
63
  )
64
64
  parser.add_argument("--scratch-dir")
65
+ parser.add_argument(
66
+ "--local-scratch",
67
+ action="store_true",
68
+ help="override request scratchRoot with <repository>/.scratch",
69
+ )
70
+ parser.add_argument(
71
+ "--local-output",
72
+ action="store_true",
73
+ help="override request output paths with <repository>/.output/<job-id>",
74
+ )
65
75
  parser.add_argument(
66
76
  "--gpu-ids",
67
77
  type=gpu_ids,
@@ -409,6 +419,8 @@ def generate_config(
409
419
  webui=args.webui,
410
420
  max_concurrency=max_concurrency,
411
421
  scratch_dir=args.scratch_dir,
422
+ local_scratch=args.local_scratch,
423
+ local_output=args.local_output,
412
424
  token=args.service_token,
413
425
  registration=registration_from_args(args, repository),
414
426
  gpu_ids=args.gpu_ids,
@@ -0,0 +1,551 @@
1
+ from __future__ import annotations
2
+
3
+ import argparse
4
+ import csv
5
+ import json
6
+ import re
7
+ import shutil
8
+ import subprocess
9
+ from collections import defaultdict
10
+ from pathlib import Path
11
+ from typing import Any
12
+
13
+ import pyarrow as pa
14
+ import pyarrow.parquet as pq
15
+
16
+
17
+ EPISODE_FILE_PATTERN = re.compile(r"episode[-_](\d+)")
18
+ FILE_PATTERN = re.compile(r"file-(\d+)")
19
+ VIDEO_SUFFIXES = {".avi", ".mkv", ".mov", ".mp4", ".webm"}
20
+
21
+
22
+ def positive_minutes(value: str) -> float:
23
+ try:
24
+ minutes = float(value)
25
+ except ValueError as exc:
26
+ raise argparse.ArgumentTypeError("minutes must be a number") from exc
27
+ if minutes <= 0:
28
+ raise argparse.ArgumentTypeError("minutes must be greater than zero")
29
+ return minutes
30
+
31
+
32
+ def configure_parser(parser: argparse.ArgumentParser) -> None:
33
+ parser.add_argument("input_dataset", type=Path)
34
+ parser.add_argument("output_path", type=Path)
35
+ parser.add_argument(
36
+ "--minutes",
37
+ type=positive_minutes,
38
+ required=True,
39
+ help="maximum duration retained from the start of each episode",
40
+ )
41
+ parser.set_defaults(handler=execute)
42
+
43
+
44
+ def _source_name(path: Path, data_root: Path) -> str:
45
+ relative = path.relative_to(data_root)
46
+ parts = []
47
+ for part in relative.parts[:-1]:
48
+ if part.startswith("chunk-"):
49
+ break
50
+ parts.append(part)
51
+ return "/".join(parts)
52
+
53
+
54
+ def _replace_column(table: pa.Table, name: str, values: list[int]) -> pa.Table:
55
+ index = table.column_names.index(name)
56
+ return table.set_column(
57
+ index,
58
+ name,
59
+ pa.array(values, type=table.schema.field(name).type),
60
+ )
61
+
62
+
63
+ def _read_json_lines(path: Path) -> list[dict[str, Any]]:
64
+ if not path.is_file():
65
+ return []
66
+ return [
67
+ json.loads(line)
68
+ for line in path.read_text(encoding="utf-8").splitlines()
69
+ if line.strip()
70
+ ]
71
+
72
+
73
+ def _write_json_lines(path: Path, rows: list[dict[str, Any]]) -> None:
74
+ path.write_text(
75
+ "".join(json.dumps(row, ensure_ascii=False) + "\n" for row in rows),
76
+ encoding="utf-8",
77
+ )
78
+
79
+
80
+ def _episode_tables(root: Path) -> list[tuple[Path, pa.Table]]:
81
+ result = []
82
+ for path in sorted((root / "meta" / "episodes").glob("**/*.parquet")):
83
+ result.append((path, pq.read_table(path)))
84
+ return result
85
+
86
+
87
+ def _episode_rows(tables: list[tuple[Path, pa.Table]]) -> list[dict[str, Any]]:
88
+ return [row for _, table in tables for row in table.to_pylist()]
89
+
90
+
91
+ def _table_time_values(table: pa.Table, fps: float) -> list[float]:
92
+ if "timestamp" in table.column_names:
93
+ return [float(value) for value in table["timestamp"].to_pylist()]
94
+ if "frame_index" in table.column_names:
95
+ return [float(value) / fps for value in table["frame_index"].to_pylist()]
96
+ return [index / fps for index in range(table.num_rows)]
97
+
98
+
99
+ def _candidate_lengths(
100
+ paths: list[Path], data_root: Path, seconds: float, fps: float
101
+ ) -> dict[int, int]:
102
+ counts: dict[str, dict[int, int]] = defaultdict(lambda: defaultdict(int))
103
+ first_timestamps: dict[tuple[str, int], float] = {}
104
+ for path in paths:
105
+ schema = pq.read_schema(path)
106
+ if "episode_index" not in schema.names:
107
+ raise ValueError(f"data parquet lacks episode_index: {path}")
108
+ columns = ["episode_index"]
109
+ for candidate in ("timestamp", "frame_index"):
110
+ if candidate in schema.names:
111
+ columns.append(candidate)
112
+ break
113
+ table = pq.read_table(path, columns=columns)
114
+ source = _source_name(path, data_root)
115
+ episodes = [int(value) for value in table["episode_index"].to_pylist()]
116
+ timestamps = _table_time_values(table, fps)
117
+ for episode, timestamp in zip(episodes, timestamps):
118
+ key = (source, episode)
119
+ first = first_timestamps.setdefault(key, timestamp)
120
+ if timestamp <= first + seconds + 1e-9:
121
+ counts[source][episode] += 1
122
+
123
+ episodes = {episode for values in counts.values() for episode in values}
124
+ if not episodes:
125
+ raise ValueError("no episode rows found in data parquet files")
126
+ lengths = {
127
+ episode: min(
128
+ values[episode] for values in counts.values() if episode in values
129
+ )
130
+ for episode in episodes
131
+ }
132
+ return lengths
133
+
134
+
135
+ def _probe_frames(path: Path) -> int:
136
+ command = [
137
+ "ffprobe",
138
+ "-v",
139
+ "error",
140
+ "-select_streams",
141
+ "v:0",
142
+ "-show_entries",
143
+ "stream=nb_frames",
144
+ "-of",
145
+ "json",
146
+ str(path),
147
+ ]
148
+ try:
149
+ result = subprocess.run(
150
+ command, check=True, capture_output=True, text=True
151
+ )
152
+ stream = json.loads(result.stdout)["streams"][0]
153
+ value = stream.get("nb_frames")
154
+ if value not in (None, "N/A"):
155
+ return int(value)
156
+
157
+ command.insert(command.index("-show_entries"), "-count_frames")
158
+ command[command.index("stream=nb_frames")] = "stream=nb_read_frames"
159
+ result = subprocess.run(
160
+ command, check=True, capture_output=True, text=True
161
+ )
162
+ value = json.loads(result.stdout)["streams"][0].get("nb_read_frames")
163
+ if value not in (None, "N/A"):
164
+ return int(value)
165
+ except (subprocess.CalledProcessError, KeyError, ValueError, json.JSONDecodeError) as exc:
166
+ raise RuntimeError(f"cannot inspect video frames: {path}") from exc
167
+ raise RuntimeError(f"video does not report a frame count: {path}")
168
+
169
+
170
+ def _chunk_index(path: Path) -> int | None:
171
+ for part in path.parts:
172
+ if part.startswith("chunk-"):
173
+ try:
174
+ return int(part.removeprefix("chunk-"))
175
+ except ValueError:
176
+ return None
177
+ return None
178
+
179
+
180
+ def _media_episodes(path: Path, root: Path, rows: list[dict[str, Any]]) -> list[int]:
181
+ direct = EPISODE_FILE_PATTERN.search(path.stem)
182
+ if direct:
183
+ return [int(direct.group(1))]
184
+
185
+ file_match = FILE_PATTERN.search(path.stem)
186
+ if not file_match:
187
+ raise ValueError(f"cannot determine episode for media file: {path}")
188
+ file_index = int(file_match.group(1))
189
+ chunk_index = _chunk_index(path)
190
+ relative = path.relative_to(root).as_posix()
191
+ kind = relative.split("/", 1)[0]
192
+ matches = []
193
+ for row in rows:
194
+ episode = int(row.get("episode_index", -1))
195
+ for key, value in row.items():
196
+ if not key.startswith(f"{kind}/") or not key.endswith("/file_index"):
197
+ continue
198
+ feature = key[len(kind) + 1 : -len("/file_index")]
199
+ chunk_key = key[: -len("file_index")] + "chunk_index"
200
+ if (
201
+ int(value) == file_index
202
+ and (chunk_index is None or int(row.get(chunk_key, -1)) == chunk_index)
203
+ and feature in relative
204
+ ):
205
+ matches.append(episode)
206
+ break
207
+ return sorted(set(matches)) or [file_index]
208
+
209
+
210
+ def _copy_static_tree(source: Path, output: Path) -> None:
211
+ skipped = {"data", "videos", "depths", "sensors"}
212
+
213
+ def ignore(directory: str, names: list[str]) -> set[str]:
214
+ return skipped.intersection(names) if Path(directory) == source else set()
215
+
216
+ shutil.copytree(source, output, ignore=ignore)
217
+
218
+
219
+ def _write_data(
220
+ paths: list[Path],
221
+ source_root: Path,
222
+ output_root: Path,
223
+ limits: dict[int, int],
224
+ fps: float,
225
+ ) -> tuple[dict[str, dict[int, dict[str, float | int]]], dict[str, int]]:
226
+ seen: dict[str, dict[int, int]] = defaultdict(lambda: defaultdict(int))
227
+ global_index: dict[str, int] = defaultdict(int)
228
+ metrics: dict[str, dict[int, dict[str, float | int]]] = defaultdict(dict)
229
+
230
+ for path in paths:
231
+ table = pq.read_table(path)
232
+ source = _source_name(path, source_root)
233
+ episodes = [int(value) for value in table["episode_index"].to_pylist()]
234
+ timestamps = _table_time_values(table, fps)
235
+ keep = []
236
+ frame_indexes = []
237
+ indexes = []
238
+ for row_index, (episode, timestamp) in enumerate(zip(episodes, timestamps)):
239
+ ordinal = seen[source][episode]
240
+ seen[source][episode] += 1
241
+ if ordinal >= limits[episode]:
242
+ continue
243
+ keep.append(row_index)
244
+ frame_indexes.append(ordinal)
245
+ indexes.append(global_index[source])
246
+ global_index[source] += 1
247
+ episode_metrics = metrics[source].setdefault(
248
+ episode,
249
+ {
250
+ "from_index": indexes[-1],
251
+ "to_index": indexes[-1] + 1,
252
+ "from_timestamp": timestamp,
253
+ "to_timestamp": timestamp,
254
+ "length": 0,
255
+ },
256
+ )
257
+ episode_metrics["to_index"] = indexes[-1] + 1
258
+ episode_metrics["to_timestamp"] = timestamp
259
+ episode_metrics["length"] = int(episode_metrics["length"]) + 1
260
+
261
+ trimmed = table.take(pa.array(keep, type=pa.int64()))
262
+ if "frame_index" in trimmed.column_names:
263
+ trimmed = _replace_column(trimmed, "frame_index", frame_indexes)
264
+ if "index" in trimmed.column_names:
265
+ trimmed = _replace_column(trimmed, "index", indexes)
266
+ destination = output_root / path.relative_to(source_root.parent)
267
+ destination.parent.mkdir(parents=True, exist_ok=True)
268
+ pq.write_table(trimmed, destination)
269
+ return metrics, dict(global_index)
270
+
271
+
272
+ def _copy_or_cut_video(source: Path, destination: Path, frames: int) -> None:
273
+ source_frames = _probe_frames(source)
274
+ destination.parent.mkdir(parents=True, exist_ok=True)
275
+ if source_frames <= frames:
276
+ shutil.copy2(source, destination)
277
+ return
278
+ depth_video = "depth" in source.as_posix().lower()
279
+ codec = ["-c:v", "libx264rgb", "-crf", "0"] if depth_video else [
280
+ "-c:v",
281
+ "libx264",
282
+ "-crf",
283
+ "20",
284
+ ]
285
+ subprocess.run(
286
+ [
287
+ "ffmpeg",
288
+ "-y",
289
+ "-loglevel",
290
+ "error",
291
+ "-i",
292
+ str(source),
293
+ "-map",
294
+ "0:v:0",
295
+ "-frames:v",
296
+ str(frames),
297
+ *codec,
298
+ "-preset",
299
+ "veryfast",
300
+ "-an",
301
+ str(destination),
302
+ ],
303
+ check=True,
304
+ )
305
+ actual = _probe_frames(destination)
306
+ if actual != frames:
307
+ raise RuntimeError(
308
+ f"video frame count mismatch for {destination}: expected {frames}, got {actual}"
309
+ )
310
+
311
+
312
+ def _sensor_episode(path: Path, rows: list[dict[str, Any]], root: Path) -> int:
313
+ match = EPISODE_FILE_PATTERN.search(path.stem)
314
+ if match:
315
+ return int(match.group(1))
316
+ relative = path.relative_to(root).as_posix()
317
+ matches = {
318
+ int(row["episode_index"])
319
+ for row in rows
320
+ for key, value in row.items()
321
+ if key.startswith("sensors/") and key.endswith("/path") and value == relative
322
+ }
323
+ if len(matches) != 1:
324
+ raise ValueError(f"cannot determine episode for sensor file: {path}")
325
+ return matches.pop()
326
+
327
+
328
+ def _copy_sensors(
329
+ source: Path,
330
+ output: Path,
331
+ episode_rows: list[dict[str, Any]],
332
+ end_times: dict[int, float],
333
+ ) -> dict[str, dict[str, float | int]]:
334
+ metrics = {}
335
+ sensor_root = source / "sensors"
336
+ if not sensor_root.exists():
337
+ return metrics
338
+ for path in sorted(sensor_root.glob("**/*")):
339
+ if not path.is_file():
340
+ continue
341
+ destination = output / path.relative_to(source)
342
+ destination.parent.mkdir(parents=True, exist_ok=True)
343
+ if path.suffix.lower() != ".csv":
344
+ shutil.copy2(path, destination)
345
+ continue
346
+ episode = _sensor_episode(path, episode_rows, source)
347
+ with path.open(newline="", encoding="utf-8") as stream:
348
+ reader = csv.DictReader(stream)
349
+ rows = list(reader)
350
+ fieldnames = reader.fieldnames or []
351
+ if "timestamp" in fieldnames and rows:
352
+ first_timestamp = float(rows[0]["timestamp"])
353
+ maximum = first_timestamp + end_times[episode] + 1e-9
354
+ rows = [row for row in rows if float(row["timestamp"]) <= maximum]
355
+ if "imu_index" in fieldnames:
356
+ for index, row in enumerate(rows):
357
+ row["imu_index"] = str(index)
358
+ with destination.open("w", newline="", encoding="utf-8") as stream:
359
+ writer = csv.DictWriter(stream, fieldnames=fieldnames)
360
+ writer.writeheader()
361
+ writer.writerows(rows)
362
+ relative = path.relative_to(source).as_posix()
363
+ metrics[relative] = {
364
+ "from_index": 0,
365
+ "to_index": len(rows),
366
+ "from_timestamp": float(rows[0]["timestamp"]) if rows and "timestamp" in fieldnames else 0.0,
367
+ "to_timestamp": float(rows[-1]["timestamp"]) if rows and "timestamp" in fieldnames else 0.0,
368
+ }
369
+ return metrics
370
+
371
+
372
+ def _update_episode_row(
373
+ row: dict[str, Any],
374
+ limits: dict[int, int],
375
+ offsets: dict[int, int],
376
+ source_metrics: dict[str, dict[int, dict[str, float | int]]],
377
+ sensor_metrics: dict[str, dict[str, float | int]],
378
+ end_times: dict[int, float],
379
+ ) -> None:
380
+ episode = int(row["episode_index"])
381
+ length = limits[episode]
382
+ for key in list(row):
383
+ if key == "length" or key.endswith("_length"):
384
+ row[key] = length
385
+ for key in ("duration", "duration_s"):
386
+ if key in row:
387
+ row[key] = min(float(row[key]), end_times[episode])
388
+ if "dataset_from_index" in row:
389
+ row["dataset_from_index"] = offsets[episode]
390
+ if "dataset_to_index" in row:
391
+ row["dataset_to_index"] = offsets[episode] + length
392
+
393
+ for source, episodes in source_metrics.items():
394
+ values = episodes.get(episode)
395
+ if values is None:
396
+ continue
397
+ prefix = f"data/{source}/" if source else "data/"
398
+ for field in ("from_index", "to_index", "from_timestamp", "to_timestamp"):
399
+ key = prefix + field
400
+ if key in row:
401
+ row[key] = values[field]
402
+
403
+ for key, value in list(row.items()):
404
+ if key.startswith(("videos/", "depths/")) and key.endswith("/to_timestamp"):
405
+ prefix = key[: -len("to_timestamp")]
406
+ row[key] = float(row.get(prefix + "from_timestamp", 0.0)) + end_times[episode]
407
+ if key.startswith("sensors/") and key.endswith("/path") and value in sensor_metrics:
408
+ prefix = key[: -len("path")]
409
+ for field, metric_value in sensor_metrics[value].items():
410
+ metric_key = prefix + field
411
+ if metric_key in row:
412
+ row[metric_key] = metric_value
413
+
414
+
415
+ def cut_dataset(input_dataset: Path, output_path: Path, minutes: float) -> None:
416
+ source = input_dataset.expanduser().resolve()
417
+ output = output_path.expanduser().resolve()
418
+ if not source.is_dir():
419
+ raise NotADirectoryError(f"input dataset does not exist: {source}")
420
+ if output.exists():
421
+ raise FileExistsError(f"output path already exists: {output}")
422
+ if output.is_relative_to(source):
423
+ raise ValueError("output path must be outside the input dataset")
424
+ data_root = source / "data"
425
+ data_paths = sorted(data_root.glob("**/*.parquet"))
426
+ if not data_paths:
427
+ raise ValueError(f"no data parquet files found under {data_root}")
428
+
429
+ info_path = source / "meta" / "info.json"
430
+ info = json.loads(info_path.read_text(encoding="utf-8"))
431
+ fps = float(info.get("fps", 30.0))
432
+ limits = _candidate_lengths(data_paths, data_root, minutes * 60, fps)
433
+ episode_tables = _episode_tables(source)
434
+ episode_rows = _episode_rows(episode_tables)
435
+
436
+ media_paths = sorted(
437
+ path
438
+ for directory in (source / "videos", source / "depths")
439
+ if directory.exists()
440
+ for path in directory.glob("**/*")
441
+ if path.is_file() and path.suffix.lower() in VIDEO_SUFFIXES
442
+ )
443
+ media_episodes = {}
444
+ for path in media_paths:
445
+ episodes = _media_episodes(path, source, episode_rows)
446
+ if len(episodes) != 1:
447
+ raise ValueError(
448
+ f"media file contains multiple episodes and cannot be cut safely: {path}"
449
+ )
450
+ episode = episodes[0]
451
+ if episode not in limits:
452
+ raise ValueError(f"media references unknown episode {episode}: {path}")
453
+ media_episodes[path] = episode
454
+ limits[episode] = min(limits[episode], _probe_frames(path))
455
+
456
+ if any(length <= 0 for length in limits.values()):
457
+ raise ValueError("the requested cut produced an empty episode")
458
+
459
+ try:
460
+ _copy_static_tree(source, output)
461
+ source_metrics, source_totals = _write_data(
462
+ data_paths, data_root, output, limits, fps
463
+ )
464
+ end_times = {
465
+ episode: min(
466
+ float(episodes[episode]["to_timestamp"])
467
+ - float(episodes[episode]["from_timestamp"])
468
+ for episodes in source_metrics.values()
469
+ if episode in episodes
470
+ )
471
+ for episode in limits
472
+ }
473
+
474
+ for path, episode in media_episodes.items():
475
+ _copy_or_cut_video(
476
+ path, output / path.relative_to(source), limits[episode]
477
+ )
478
+ for directory in (source / "videos", source / "depths"):
479
+ if not directory.exists():
480
+ continue
481
+ for path in directory.glob("**/*"):
482
+ if path.is_file() and path not in media_episodes:
483
+ destination = output / path.relative_to(source)
484
+ destination.parent.mkdir(parents=True, exist_ok=True)
485
+ shutil.copy2(path, destination)
486
+ sensor_metrics = _copy_sensors(
487
+ source, output, episode_rows, end_times
488
+ )
489
+
490
+ offsets = {}
491
+ offset = 0
492
+ for episode in sorted(limits):
493
+ offsets[episode] = offset
494
+ offset += limits[episode]
495
+
496
+ for original_path, table in episode_tables:
497
+ rows = table.to_pylist()
498
+ for row in rows:
499
+ _update_episode_row(
500
+ row,
501
+ limits,
502
+ offsets,
503
+ source_metrics,
504
+ sensor_metrics,
505
+ end_times,
506
+ )
507
+ destination = output / original_path.relative_to(source)
508
+ pq.write_table(pa.Table.from_pylist(rows, schema=table.schema), destination)
509
+
510
+ jsonl_path = output / "meta" / "episodes.jsonl"
511
+ jsonl_rows = _read_json_lines(jsonl_path)
512
+ for row in jsonl_rows:
513
+ _update_episode_row(
514
+ row,
515
+ limits,
516
+ offsets,
517
+ source_metrics,
518
+ sensor_metrics,
519
+ end_times,
520
+ )
521
+ if jsonl_path.exists():
522
+ _write_json_lines(jsonl_path, jsonl_rows)
523
+
524
+ info["total_frames"] = sum(limits.values())
525
+ if isinstance(info.get("source_frame_counts"), dict):
526
+ info["source_frame_counts"] = {
527
+ source_name: source_totals.get(source_name, 0)
528
+ for source_name in info["source_frame_counts"]
529
+ }
530
+ (output / "meta" / "info.json").write_text(
531
+ json.dumps(info, ensure_ascii=False, indent=2) + "\n",
532
+ encoding="utf-8",
533
+ )
534
+ except Exception:
535
+ shutil.rmtree(output, ignore_errors=True)
536
+ raise
537
+
538
+
539
+ def execute(args: argparse.Namespace) -> int:
540
+ cut_dataset(args.input_dataset, args.output_path, args.minutes)
541
+ print(
542
+ json.dumps(
543
+ {
544
+ "input": str(args.input_dataset),
545
+ "output": str(args.output_path),
546
+ "minutes": args.minutes,
547
+ },
548
+ ensure_ascii=False,
549
+ )
550
+ )
551
+ return 0
@@ -81,6 +81,8 @@ def configure_parser(parser: argparse.ArgumentParser) -> None:
81
81
  def execute(args: argparse.Namespace) -> int:
82
82
  config: DeploymentConfig | None = None
83
83
  service: dict[str, object] | None = None
84
+ local_scratch = False
85
+ local_output = False
84
86
  if args.config:
85
87
  # Apply the process mask before refreshing/loading the algorithm. A
86
88
  # repository module may import CUDA during module import.
@@ -91,6 +93,8 @@ def execute(args: argparse.Namespace) -> int:
91
93
  args.port = int(service["port"])
92
94
  args.max_concurrent_executions = int(service["maxConcurrency"])
93
95
  args.scratch_dir = service.get("scratchDir")
96
+ local_scratch = bool(service.get("localScratch", False))
97
+ local_output = bool(service.get("localOutput", False))
94
98
  args.gpu_ids = list(service.get("gpuIds", []))
95
99
  args.webui = bool(service.get("webui"))
96
100
  values = config.process_environment()
@@ -120,6 +124,8 @@ def execute(args: argparse.Namespace) -> int:
120
124
  ),
121
125
  max_concurrent_executions=args.max_concurrent_executions,
122
126
  scratch_dir=args.scratch_dir,
127
+ local_scratch=local_scratch,
128
+ local_output=local_output,
123
129
  gpu_ids=args.gpu_ids,
124
130
  runner_factory=create_runner,
125
131
  max_attempts=args.max_attempts,
@@ -146,6 +146,8 @@ class DeploymentConfig:
146
146
  gpu_ids: list[int] | None = None,
147
147
  environment: dict[str, str] | None = None,
148
148
  default_parameters: dict[str, Any] | None = None,
149
+ local_scratch: bool = False,
150
+ local_output: bool = False,
149
151
  ) -> "DeploymentConfig":
150
152
  repository = repository_root(root)
151
153
  _make_repository_importable(repository)
@@ -189,6 +191,8 @@ class DeploymentConfig:
189
191
  "webui": webui,
190
192
  "maxConcurrency": max_concurrency,
191
193
  "scratchDir": _absolute(scratch_dir, repository),
194
+ "localScratch": local_scratch,
195
+ "localOutput": local_output,
192
196
  "gpuIds": configured_gpu_ids,
193
197
  "token": token or None,
194
198
  "defaultParameters": dict(default_parameters or {}),
@@ -268,6 +272,9 @@ class DeploymentConfig:
268
272
  raise RuntimeError("service port must be in [1, 65535]")
269
273
  if concurrency < 1:
270
274
  raise RuntimeError("service maxConcurrency must be positive")
275
+ for name in ("localScratch", "localOutput"):
276
+ if not isinstance(service.get(name, False), bool):
277
+ raise RuntimeError(f"service {name} must be a boolean")
271
278
  gpu_ids = service.get("gpuIds")
272
279
  if not isinstance(gpu_ids, list) or any(
273
280
  isinstance(gpu_id, bool) or not isinstance(gpu_id, int) or gpu_id < 0
@@ -10,6 +10,9 @@ from typing import Any
10
10
  from .context import utc_now
11
11
 
12
12
  _SAFE_NAME = re.compile(r"[^A-Za-z0-9._-]+")
13
+ _TIMESTAMP_GLOB = "[0-9]" * 14
14
+ _FILE_LOCKS_GUARD = threading.Lock()
15
+ _FILE_LOCKS: dict[Path, threading.RLock] = {}
13
16
 
14
17
 
15
18
  def _safe_name(value: str) -> str:
@@ -17,20 +20,61 @@ def _safe_name(value: str) -> str:
17
20
  return cleaned or "job"
18
21
 
19
22
 
23
+ def _file_lock(path: Path) -> threading.RLock:
24
+ with _FILE_LOCKS_GUARD:
25
+ return _FILE_LOCKS.setdefault(path, threading.RLock())
26
+
27
+
28
+ def _timestamp_prefix() -> str:
29
+ return utc_now().replace("-", "").replace(":", "").replace("T", "")[:14]
30
+
31
+
32
+ def _job_log_path(directory: Path, job_id: str) -> Path:
33
+ safe_job_id = _safe_name(job_id)
34
+ pattern = f"{_TIMESTAMP_GLOB}-{safe_job_id}.log"
35
+ with _FILE_LOCKS_GUARD:
36
+ existing = sorted(directory.glob(pattern))
37
+ if existing:
38
+ return existing[0]
39
+ timestamp = _timestamp_prefix()
40
+ path = directory / f"{timestamp}-{safe_job_id}.log"
41
+ path.touch(exist_ok=True)
42
+ return path
43
+
44
+
20
45
  class ExecutionLog:
21
- """Per-execution, line-buffered log sink attached to context.logger."""
46
+ """Per-job, line-buffered log sink attached to context.logger."""
22
47
 
23
- def __init__(self, *, log_dir: str | Path, job_id: str, execution_id: str, logger: logging.Logger) -> None:
48
+ def __init__(
49
+ self,
50
+ *,
51
+ log_dir: str | Path,
52
+ job_id: str,
53
+ execution_id: str,
54
+ logger: logging.Logger,
55
+ task_id: str | None = None,
56
+ ) -> None:
57
+ resolved_task_id = task_id or execution_id
58
+ self._event_context = {
59
+ "jobId": job_id,
60
+ "taskId": resolved_task_id,
61
+ "executionId": execution_id,
62
+ "job_id": job_id,
63
+ "task_id": resolved_task_id,
64
+ "execution_id": execution_id,
65
+ }
24
66
  directory = Path(log_dir).expanduser()
25
67
  directory.mkdir(parents=True, exist_ok=True)
26
- path = directory / f"{_safe_name(job_id)}.log"
27
- if path.exists():
28
- path = directory / f"{_safe_name(job_id)}-{_safe_name(execution_id)}.log"
68
+ path = _job_log_path(directory, job_id)
29
69
  self.path = path
30
70
  self._stream = path.open("a", encoding="utf-8", buffering=1)
31
- self._lock = threading.RLock()
32
- self._handler = _FlushFileHandler(self._stream)
33
- self._handler.setFormatter(logging.Formatter("%(asctime)s %(levelname)s %(message)s"))
71
+ self._lock = _file_lock(path)
72
+ self._handler = _FlushFileHandler(self._stream, self._lock)
73
+ self._handler.setFormatter(
74
+ logging.Formatter(
75
+ f"%(asctime)s %(levelname)s [executionId={execution_id}] %(message)s"
76
+ )
77
+ )
34
78
  self._handler.setLevel(logging.DEBUG)
35
79
  self.logger = logger
36
80
  self.logger.setLevel(logging.DEBUG)
@@ -38,7 +82,12 @@ class ExecutionLog:
38
82
  self._closed = False
39
83
 
40
84
  def write_event(self, event: str, **values: Any) -> None:
41
- payload = {"event": event, "timestamp": utc_now(), **values}
85
+ payload = {
86
+ "event": event,
87
+ "timestamp": utc_now(),
88
+ **self._event_context,
89
+ **values,
90
+ }
42
91
  with self._lock:
43
92
  if self._closed:
44
93
  return
@@ -56,10 +105,10 @@ class ExecutionLog:
56
105
 
57
106
 
58
107
  class _FlushFileHandler(logging.Handler):
59
- def __init__(self, stream: Any) -> None:
108
+ def __init__(self, stream: Any, lock: threading.RLock) -> None:
60
109
  super().__init__()
61
110
  self.stream = stream
62
- self._lock = threading.RLock()
111
+ self._lock = lock
63
112
 
64
113
  def emit(self, record: logging.LogRecord) -> None:
65
114
  try:
@@ -6,6 +6,7 @@ import logging
6
6
  import random
7
7
  import threading
8
8
  import time
9
+ import traceback
9
10
  import uuid
10
11
  from concurrent.futures import Future, ThreadPoolExecutor
11
12
  from contextlib import asynccontextmanager
@@ -123,6 +124,8 @@ class ExecutionManager:
123
124
  scratch_dir: str | None = None,
124
125
  repository: str | Path | None = None,
125
126
  log_dir: str | Path | None = None,
127
+ local_scratch: bool = False,
128
+ local_output: bool = False,
126
129
  gpu_ids: list[int] | None = None,
127
130
  runner_factory: RunnerFactory | None = None,
128
131
  max_attempts: int = MAX_ATTEMPTS,
@@ -138,6 +141,17 @@ class ExecutionManager:
138
141
  raise ValueError("retry_seconds must be positive")
139
142
  if not 0 <= retry_jitter_ratio <= 1:
140
143
  raise ValueError("retry_jitter_ratio must be in [0, 1]")
144
+ if not isinstance(local_scratch, bool) or not isinstance(
145
+ local_output, bool
146
+ ):
147
+ raise ValueError("local_scratch and local_output must be booleans")
148
+ repository_path = (
149
+ Path(repository).expanduser().resolve()
150
+ if repository is not None
151
+ else None
152
+ )
153
+ if (local_scratch or local_output) and repository_path is None:
154
+ raise ValueError("repository is required for local scratch or output")
141
155
  configured_gpu_ids = None if gpu_ids is None else list(gpu_ids)
142
156
  if configured_gpu_ids is not None and (
143
157
  any(
@@ -156,10 +170,19 @@ class ExecutionManager:
156
170
  raise ValueError("default_parameters must be a JSON object")
157
171
  self.default_parameters = dict(default_parameters or {})
158
172
  self.scratch_dir = scratch_dir
173
+ self._local_scratch_root = (
174
+ repository_path / ".scratch" if local_scratch else None
175
+ )
176
+ self._local_output_root = (
177
+ repository_path / ".output" if local_output else None
178
+ )
179
+ for local_root in (self._local_scratch_root, self._local_output_root):
180
+ if local_root is not None:
181
+ local_root.mkdir(parents=True, exist_ok=True)
159
182
  self.log_dir = (
160
183
  Path(log_dir).expanduser()
161
184
  if log_dir is not None
162
- else (Path(repository).expanduser() / "logs" if repository is not None else None)
185
+ else (repository_path / "logs" if repository_path is not None else None)
163
186
  )
164
187
  self.gpu_ids = configured_gpu_ids
165
188
  self.max_concurrent_executions = max_concurrent_executions
@@ -191,6 +214,46 @@ class ExecutionManager:
191
214
  self.runner.start()
192
215
  self._started = True
193
216
 
217
+ def _apply_local_paths(
218
+ self,
219
+ request: AlgorithmRequest,
220
+ job_id: str,
221
+ ) -> AlgorithmRequest:
222
+ workspace = dict(request.workspace or {})
223
+ inputs = request.inputs
224
+ if self._local_scratch_root is not None:
225
+ workspace["scratchRoot"] = str(self._local_scratch_root)
226
+ if self._local_output_root is not None:
227
+ workspace["outputRoot"] = str(self._local_output_root)
228
+ job_output = self._job_output_dir(job_id)
229
+ if len(inputs) == 1:
230
+ output_paths = [job_output]
231
+ else:
232
+ output_paths = [
233
+ job_output / f"{index}-{Path(item.output).name or 'output'}"
234
+ for index, item in enumerate(inputs)
235
+ ]
236
+ for output_path in output_paths:
237
+ output_path.mkdir(parents=True, exist_ok=True)
238
+ inputs = [
239
+ replace(item, output=str(output_path))
240
+ for item, output_path in zip(inputs, output_paths)
241
+ ]
242
+ return replace(request, inputs=inputs, workspace=workspace or None)
243
+
244
+ def _job_output_dir(self, job_id: str) -> Path:
245
+ assert self._local_output_root is not None
246
+ if Path(job_id).name != job_id or job_id in {"", ".", ".."}:
247
+ raise ValueError(f"job ID is unsafe for local output: {job_id!r}")
248
+ output = (self._local_output_root / job_id).resolve()
249
+ if (
250
+ output == self._local_output_root
251
+ or self._local_output_root not in output.parents
252
+ ):
253
+ raise ValueError(f"job output escapes local root: {job_id!r}")
254
+ output.mkdir(parents=True, exist_ok=True)
255
+ return output
256
+
194
257
  def submit(
195
258
  self,
196
259
  request: AlgorithmRequest,
@@ -226,6 +289,7 @@ class ExecutionManager:
226
289
  execution_id = f"exec-{uuid.uuid4().hex}"
227
290
  resolved_job_id = str(job_id or execution_id)
228
291
  resolved_task_id = str(task_id or execution_id)
292
+ request = self._apply_local_paths(request, resolved_job_id)
229
293
  record = ExecutionRecord(
230
294
  execution_id=execution_id,
231
295
  request=request,
@@ -237,10 +301,9 @@ class ExecutionManager:
237
301
  record.context = ExecutionContext(
238
302
  execution_id,
239
303
  [item.input_dataset for item in request.inputs],
240
- scratch_dir=(
241
- record.request.workspace["scratchRoot"]
242
- if record.request.workspace is not None
243
- else self.scratch_dir
304
+ scratch_dir=(record.request.workspace or {}).get(
305
+ "scratchRoot",
306
+ self.scratch_dir,
244
307
  ),
245
308
  progress_callback=record.add_progress,
246
309
  )
@@ -248,6 +311,7 @@ class ExecutionManager:
248
311
  record.log = ExecutionLog(
249
312
  log_dir=self.log_dir,
250
313
  job_id=record.job_id,
314
+ task_id=record.task_id,
251
315
  execution_id=record.execution_id,
252
316
  logger=record.context.logger,
253
317
  )
@@ -430,10 +494,16 @@ class ExecutionManager:
430
494
  record.log.write_event("result", state="cancelled", error=str(exc))
431
495
  except Exception as exc:
432
496
  error = f"{type(exc).__name__}: {exc}"
497
+ stack_trace = traceback.format_exc()
433
498
  record.context.mark_unfinished("failed", error)
434
499
  record.set_state("failed", error=error)
435
500
  if record.log is not None:
436
- record.log.write_event("result", state="failed", error=error)
501
+ record.log.write_event(
502
+ "result",
503
+ state="failed",
504
+ error=error,
505
+ traceback=stack_trace,
506
+ )
437
507
  else:
438
508
  record.set_state(result.status, result=result)
439
509
  if record.log is not None:
@@ -466,7 +536,7 @@ class ExecutionManager:
466
536
  self.retry_seconds * (1 - self.retry_jitter_ratio),
467
537
  self.retry_seconds * (1 + self.retry_jitter_ratio),
468
538
  )
469
- logger.warning(
539
+ record.context.logger.warning(
470
540
  "algorithm execution %s failed on attempt %d/%d; "
471
541
  "releasing and restarting algorithm in %.3f seconds: %s: %s",
472
542
  record.execution_id,
@@ -475,16 +545,18 @@ class ExecutionManager:
475
545
  delay,
476
546
  type(exc).__name__,
477
547
  exc,
548
+ exc_info=True,
478
549
  )
479
550
  else:
480
551
  delay = 0.0
481
- logger.error(
552
+ record.context.logger.error(
482
553
  "algorithm execution %s exhausted %d attempts; "
483
554
  "releasing and restarting algorithm for future requests: %s: %s",
484
555
  record.execution_id,
485
556
  self.max_attempts,
486
557
  type(exc).__name__,
487
558
  exc,
559
+ exc_info=True,
488
560
  )
489
561
  finally:
490
562
  self._release_runner()
@@ -1,11 +1,12 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: ls-algorithm-plugin-sdk
3
- Version: 0.3.3
3
+ Version: 0.3.5
4
4
  Summary: Protocol-independent runtime SDK for dataset algorithms
5
5
  Author: Ling Robotics
6
6
  License: Proprietary
7
7
  Requires-Python: >=3.10
8
8
  Description-Content-Type: text/markdown
9
+ Requires-Dist: pyarrow>=14
9
10
  Provides-Extra: service
10
11
  Requires-Dist: fastapi<1,>=0.110; extra == "service"
11
12
  Requires-Dist: pydantic<3,>=2.0; extra == "service"
@@ -40,6 +41,16 @@ algorithm-plugin run camera-space-mano \
40
41
 
41
42
  `--gpu-ids` 和 `--parameters` 可省略,也可用 `--request-file` 读取完整请求 JSON。
42
43
 
44
+ ## 数据集裁剪
45
+
46
+ 将每个 episode 裁剪为最多指定分钟数,并同步更新 Parquet、视频、传感器 CSV 和元数据:
47
+
48
+ ```bash
49
+ algorithm-plugin cut /data/input_dataset /data/output_dataset --minutes 1.5
50
+ ```
51
+
52
+ 支持 LeRobot v2/v3 的 `episode_*`、`file-*` 和多数据源目录布局。视频裁剪需要系统已安装 `ffmpeg` 与 `ffprobe`。
53
+
43
54
  ## 配置现有仓库
44
55
 
45
56
  目标算法仓库必须满足:
@@ -1,23 +1,24 @@
1
1
  algorithm_plugin_sdk/__init__.py,sha256=OKj2VcNXTwZmRgPlECI6QUBlxW1YLtflr4RAgs6w064,1223
2
2
  algorithm_plugin_sdk/algorithm.py,sha256=INjX-ygMQfHk8D1AZbEEdCw1P4VNQLGF989lP1J1H1k,1025
3
- algorithm_plugin_sdk/cli.py,sha256=V9HFJ98iyhM1q6PZqd-366g1AoKuieC_Fn4u9RJcMNI,1252
3
+ algorithm_plugin_sdk/cli.py,sha256=20cjY2JXNQOFS-7ys5sNi-y9-1CObf-GYOTkjJ0Q6WI,1419
4
4
  algorithm_plugin_sdk/context.py,sha256=Kj-eDZg7PgX6xNzrWrQs37UfepTP8fXv4_8hCgsAIDA,11016
5
- algorithm_plugin_sdk/deployment.py,sha256=nx4-26vs5OqO6miBf6GegZIhdvOgFaw9MdjpTHZNDPo,14583
5
+ algorithm_plugin_sdk/deployment.py,sha256=5WlZfML9ymWrAFsmdX-bBHTP2kuVY4WO5wKHs35Dkd4,14936
6
6
  algorithm_plugin_sdk/errors.py,sha256=DNmI2c36Pd7K6tTaonqfEgpBzwX0o6pXzYrc8r2Z--k,730
7
7
  algorithm_plugin_sdk/gpu_isolation.py,sha256=toXUQ1vfzfLPiideJ8dWkfa9NW2Lc8V0sbNx-75YTjg,794
8
8
  algorithm_plugin_sdk/loader.py,sha256=XKmyqbJMoUMdKoo2eRQFGdbsmlz6NBgtUeIdqzumJFs,2561
9
- algorithm_plugin_sdk/log_manager.py,sha256=ktVVvf4khqXiYENr1QRpSf7yTG4HBzxW2AOHIZElDuE,2457
9
+ algorithm_plugin_sdk/log_manager.py,sha256=fmqSivxi-BCOWs4Qomms9nHZC7MRl5NUVuRiTL0J12Q,3723
10
10
  algorithm_plugin_sdk/models.py,sha256=FIwUm8QDK8L37pV7AcCMnmYlDP-2l93zp8yc0KSu4yo,12196
11
11
  algorithm_plugin_sdk/registration.py,sha256=s5oQkf12_uRjdA4CYMYM8SDQjhsk1rzDDF-3w36U_lA,10474
12
12
  algorithm_plugin_sdk/release.py,sha256=KM-kRM1oFYMgjDFnXdu6BLR1b4dsvWWDCEgTNq8cb3w,9439
13
13
  algorithm_plugin_sdk/runner.py,sha256=583Y9_7efvW5rrmM2nt316NEpCRO6kd8dKYjFRjgtCk,3183
14
- algorithm_plugin_sdk/service.py,sha256=HcO8PmtPczogi0an7irajsRCD-11GAH8pnE7rOHOLSM,28004
14
+ algorithm_plugin_sdk/service.py,sha256=HQLDW5nlGGXJNZ0KMFU4I1rPzWHMf_LlE-tsY42PHnc,31045
15
15
  algorithm_plugin_sdk/webui_app.py,sha256=Gvt2FPt0xZBjYi39sqTQqcosllifYeJam49_mGfkEVw,1530
16
16
  algorithm_plugin_sdk/cli_impl/__init__.py,sha256=uLknQVGipNd7eSJegDHZS0NVowMa0qAaOf63wFI3zHM,78
17
- algorithm_plugin_sdk/cli_impl/configure.py,sha256=OlYuNnPZXHbaoWKfBpFivyP8lZk2QQpvMN572hAjriA,13353
17
+ algorithm_plugin_sdk/cli_impl/configure.py,sha256=hO9vwFfZ4yqeQUyYc3heE1MDLK1lppN69OoF2zeVVaM,13761
18
+ algorithm_plugin_sdk/cli_impl/cut.py,sha256=R3ecRzDHffJZNqbJaKeX4S4hUFdQl2afXAfklkX1Ah0,19764
18
19
  algorithm_plugin_sdk/cli_impl/parsing.py,sha256=TDX45Spu4WMrczuzTMN9ZC2SYyHE11OPBCvN45yPsxs,1399
19
20
  algorithm_plugin_sdk/cli_impl/run.py,sha256=C4raNwhj4KZJc7_aTJeoiwNMQtgeXyMWyLTZEVUL0mM,6041
20
- algorithm_plugin_sdk/cli_impl/serve.py,sha256=BBxjFjOI8bORzNij9DbKccAdt4zNzX_vqZyyKtliGSg,4936
21
+ algorithm_plugin_sdk/cli_impl/serve.py,sha256=1Ma08MHMjKk9wKO8tGdymVvedcGC-jaKdaUCVBTPMro,5187
21
22
  algorithm_plugin_sdk/examples/__init__.py,sha256=WxZZzUzxKhqZg3KibVY-7CB960DNjt6IfjciZ8JXbq0,57
22
23
  algorithm_plugin_sdk/examples/example_algorithm.py,sha256=j2RsVaRPxpJn3kK6_H_JqAceKTpJE8txNm-oTrH2v6E,3787
23
24
  algorithm_plugin_sdk/examples/simulated_algorithm.py,sha256=bIaneSMM09CJfV5-oGuZ5QwBDWV1u-7Fn0WDN1TuwGw,2944
@@ -25,8 +26,8 @@ algorithm_plugin_sdk/webui/__init__.py,sha256=cvtaktJXz_DYG4QbV5ppoqY2bFaMLuiUcT
25
26
  algorithm_plugin_sdk/webui/app.css,sha256=5HGDKlbUFLZMgCknlRuwCjiNcCr8poIF5YGHchyaamE,2965
26
27
  algorithm_plugin_sdk/webui/app.js,sha256=0YsXsocEY3yDseG3jh0RhzI8xo_bYmsbSxdN3FdCswQ,7657
27
28
  algorithm_plugin_sdk/webui/index.html,sha256=Z222OuaacWdbr0s-sjJq6wmg45u96tIneM5-LP55Z8I,2261
28
- ls_algorithm_plugin_sdk-0.3.3.dist-info/METADATA,sha256=iLerkePM3Ff_GG0oA4TAdTsidnxYJxP1pmsSaIARMJw,2317
29
- ls_algorithm_plugin_sdk-0.3.3.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
30
- ls_algorithm_plugin_sdk-0.3.3.dist-info/entry_points.txt,sha256=TU0R_TxuB5OuOiZ5BHOEmUPtHvN3v8ARUW8d3PztHcE,67
31
- ls_algorithm_plugin_sdk-0.3.3.dist-info/top_level.txt,sha256=8lsgxZ8HJGLlzJ5Rt8CiVzw9Ccme286i3akbmeMjBos,21
32
- ls_algorithm_plugin_sdk-0.3.3.dist-info/RECORD,,
29
+ ls_algorithm_plugin_sdk-0.3.5.dist-info/METADATA,sha256=Smgf_xFlGbqL5k73zGtyUKpY7VMvkfDRd-_ZNEFyGWM,2709
30
+ ls_algorithm_plugin_sdk-0.3.5.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
31
+ ls_algorithm_plugin_sdk-0.3.5.dist-info/entry_points.txt,sha256=TU0R_TxuB5OuOiZ5BHOEmUPtHvN3v8ARUW8d3PztHcE,67
32
+ ls_algorithm_plugin_sdk-0.3.5.dist-info/top_level.txt,sha256=8lsgxZ8HJGLlzJ5Rt8CiVzw9Ccme286i3akbmeMjBos,21
33
+ ls_algorithm_plugin_sdk-0.3.5.dist-info/RECORD,,