@camstack/addon-post-analysis 1.2.269 → 1.2.271

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -30,7 +30,7 @@ async function d(e) {
30
30
  }
31
31
  }
32
32
  async function f() {
33
- return l ||= d(() => import("./_virtual_mf-localSharedImportMap___mfe_internal__addon_pipeline_analytics_widgets-D42gEB0t.mjs")).catch((e) => {
33
+ return l ||= d(() => import("./_virtual_mf-localSharedImportMap___mfe_internal__addon_pipeline_analytics_widgets-Bs8m-J-O.mjs")).catch((e) => {
34
34
  throw l = void 0, e;
35
35
  }), l;
36
36
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@camstack/addon-post-analysis",
3
- "version": "1.2.269",
3
+ "version": "1.2.271",
4
4
  "description": "Post-Analysis bundle — enrichment, embedding-encoder, pipeline-analytics. Multi-entry npm package shipping addons that consume pipeline output.",
5
5
  "keywords": [
6
6
  "camstack",
@@ -19,6 +19,7 @@ to stderr (stdout is the binary framing channel only).
19
19
  from __future__ import annotations
20
20
 
21
21
  import argparse
22
+ import os
22
23
  import sys
23
24
 
24
25
  import numpy as np
@@ -31,6 +32,7 @@ from tensor_frames import (
31
32
  read_frame,
32
33
  write_frame,
33
34
  )
35
+ from session_threads import resolve_session_threads
34
36
 
35
37
 
36
38
  def numpy_dtype_for(onnx_type: str):
@@ -46,10 +48,23 @@ def numpy_dtype_for(onnx_type: str):
46
48
  def main() -> None:
47
49
  ap = argparse.ArgumentParser()
48
50
  ap.add_argument("model")
51
+ # A CEILING, not a tuning. This engine serves arbitrary vision models, so
52
+ # unlike the text encoder it keeps some parallelism — but onnxruntime's
53
+ # default is the HOST's core count, held for the life of the process, and
54
+ # that is a fact about the machine rather than about the work. Nobody has
55
+ # measured the right number here because this path runs on no node of the
56
+ # current cluster; an operator who measures raises it with --threads.
57
+ ap.add_argument("--threads", type=int, default=4)
49
58
  args = ap.parse_args()
50
59
 
51
60
  print(f"raw_tensor_inference: loading model {args.model}", file=sys.stderr)
52
- sess = ort.InferenceSession(args.model, providers=["CPUExecutionProvider"])
61
+ intra, inter = resolve_session_threads(args.threads, os.cpu_count())
62
+ sess_options = ort.SessionOptions()
63
+ sess_options.intra_op_num_threads = intra
64
+ sess_options.inter_op_num_threads = inter
65
+ sess = ort.InferenceSession(
66
+ args.model, sess_options=sess_options, providers=["CPUExecutionProvider"]
67
+ )
53
68
  inp = sess.get_inputs()[0]
54
69
  input_name = inp.name
55
70
  input_dtype = numpy_dtype_for(inp.type)
@@ -0,0 +1,33 @@
1
+ #!/usr/bin/env python3
2
+ """The thread ceiling for an embedding engine's onnxruntime session.
3
+
4
+ Deliberately free of third-party imports so the decision can be tested without
5
+ onnxruntime or numpy installed — see `test_session_threads.py` for why that
6
+ matters and what this prevents.
7
+ """
8
+ from __future__ import annotations
9
+
10
+
11
+ def resolve_session_threads(
12
+ requested: int, cpu_count: int | None
13
+ ) -> tuple[int, int]:
14
+ """(intra_op_num_threads, inter_op_num_threads) for an engine session.
15
+
16
+ `requested` is a CEILING the caller states, not a tuning: it exists so the
17
+ permanent thread cost of a long-lived engine stops scaling with whatever
18
+ host it lands on. The machine's own core count caps it in turn, because
19
+ asking for more op threads than there are cores only buys coordination.
20
+
21
+ An unknown core count (`os.cpu_count()` answers None when it cannot tell)
22
+ means "no cap available", NOT "no cores": the request stands. Folding an
23
+ unreadable measurement into zero is the mistake D393 names.
24
+
25
+ inter-op is always 1. Both engines serve exactly one request at a time by
26
+ construction — read a framed request, infer, write a framed response — so
27
+ there is never a second branch of the graph in flight for inter-op
28
+ parallelism to run.
29
+ """
30
+ ceiling = max(1, requested)
31
+ if cpu_count is not None and cpu_count > 0:
32
+ ceiling = min(ceiling, cpu_count)
33
+ return ceiling, 1
@@ -0,0 +1,65 @@
1
+ #!/usr/bin/env python3
2
+ """How many threads an embedding engine's onnxruntime session may hold.
3
+
4
+ Both embedding subprocesses — `raw_tensor_inference.py` and
5
+ `text_encoder_inference.py` — build an `ort.InferenceSession` with no
6
+ `SessionOptions`, so onnxruntime takes `intra_op_num_threads` = the HOST's core
7
+ count. These are not one-shot scripts: each writes a READY frame and then sits
8
+ in a `while True` serving requests, so those threads are held for the life of
9
+ the process. On this cluster's hub that is 20 threads per engine, and it grows
10
+ with the machine — a fact about the host, not about the work.
11
+
12
+ The sibling `yamnet_audio.py` already learned this the expensive way
13
+ (~1000% CPU across a handful of audio streams) and caps to one op thread with a
14
+ `--threads` override. This is the same rule, made testable: the caller states a
15
+ ceiling, the machine's core count caps it, and the result is never zero.
16
+
17
+ Kept free of numpy and onnxruntime on purpose — neither installs on the
18
+ operator's Mac, and a decision worth testing should not need a GPU runtime to
19
+ be checked.
20
+ """
21
+ from __future__ import annotations
22
+
23
+ import unittest
24
+
25
+ from session_threads import resolve_session_threads
26
+
27
+
28
+ class TestResolveSessionThreads(unittest.TestCase):
29
+ def test_a_request_larger_than_the_machine_is_capped_by_the_machine(self) -> None:
30
+ intra, _ = resolve_session_threads(8, cpu_count=4)
31
+ self.assertEqual(intra, 4)
32
+
33
+ def test_a_smaller_request_is_honoured_on_a_big_host(self) -> None:
34
+ # The whole point: a 20-core hub must not hand 20 permanent threads to
35
+ # an engine that asked for 4.
36
+ intra, _ = resolve_session_threads(4, cpu_count=20)
37
+ self.assertEqual(intra, 4)
38
+
39
+ def test_it_never_resolves_to_zero_or_fewer_threads(self) -> None:
40
+ for requested in (0, -3):
41
+ with self.subTest(requested=requested):
42
+ intra, inter = resolve_session_threads(requested, cpu_count=20)
43
+ self.assertEqual(intra, 1)
44
+ self.assertGreaterEqual(inter, 1)
45
+
46
+ def test_an_unknown_core_count_does_not_silently_become_no_threads(self) -> None:
47
+ # os.cpu_count() returns None when it cannot tell. Folding that into 0
48
+ # would hand onnxruntime an invalid session; the request stands instead.
49
+ for unknown in (None, 0):
50
+ with self.subTest(cpu_count=unknown):
51
+ intra, _ = resolve_session_threads(4, cpu_count=unknown)
52
+ self.assertEqual(intra, 4)
53
+
54
+ def test_inter_op_is_always_one(self) -> None:
55
+ # These engines serve ONE request at a time by construction (a framed
56
+ # read, an inference, a framed write). There is never a second graph
57
+ # branch in flight for inter-op parallelism to run.
58
+ for cpu in (1, 4, 20, 64):
59
+ with self.subTest(cpu_count=cpu):
60
+ _, inter = resolve_session_threads(4, cpu_count=cpu)
61
+ self.assertEqual(inter, 1)
62
+
63
+
64
+ if __name__ == "__main__":
65
+ unittest.main()
@@ -20,6 +20,7 @@ Diagnostics go to stderr (stdout is the binary framing channel only).
20
20
  from __future__ import annotations
21
21
 
22
22
  import argparse
23
+ import os
23
24
  import sys
24
25
 
25
26
  import numpy as np
@@ -27,6 +28,7 @@ import onnxruntime as ort
27
28
  from tokenizers import Tokenizer
28
29
 
29
30
  from tensor_frames import READY_FRAME, encode_tensor, read_frame, write_frame
31
+ from session_threads import resolve_session_threads
30
32
 
31
33
  # CLIP fixed context length — token ids are truncated/padded to exactly this.
32
34
  CONTEXT_LENGTH = 77
@@ -44,6 +46,12 @@ def main() -> None:
44
46
  ap = argparse.ArgumentParser()
45
47
  ap.add_argument("model")
46
48
  ap.add_argument("tokenizer")
49
+ # A CLIP text encoder is one pass over a fixed 77-token window. There is
50
+ # nothing here worth splitting across threads, and onnxruntime's default is
51
+ # the HOST's core count held for the life of this process — 20 on this
52
+ # cluster's hub. `yamnet_audio.py` measured what that costs (~1000% CPU on
53
+ # a handful of streams) and landed on the same default.
54
+ ap.add_argument("--threads", type=int, default=1)
47
55
  args = ap.parse_args()
48
56
 
49
57
  print(
@@ -51,7 +59,13 @@ def main() -> None:
51
59
  f"tokenizer {args.tokenizer}",
52
60
  file=sys.stderr,
53
61
  )
54
- sess = ort.InferenceSession(args.model, providers=["CPUExecutionProvider"])
62
+ intra, inter = resolve_session_threads(args.threads, os.cpu_count())
63
+ sess_options = ort.SessionOptions()
64
+ sess_options.intra_op_num_threads = intra
65
+ sess_options.inter_op_num_threads = inter
66
+ sess = ort.InferenceSession(
67
+ args.model, sess_options=sess_options, providers=["CPUExecutionProvider"]
68
+ )
55
69
  input_name = sess.get_inputs()[0].name
56
70
  output_name = sess.get_outputs()[0].name
57
71
  tokenizer = build_tokenizer(args.tokenizer)