@camstack/addon-post-analysis 1.2.269 → 1.2.271
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{dist-B78I61U1.js → dist-DTeAk1Ce.js} +101 -7
- package/dist/{dist-ABnG0S3f.mjs → dist-K6zDULiD.mjs} +101 -7
- package/dist/embedding-encoder/index.js +1 -1
- package/dist/embedding-encoder/index.mjs +1 -1
- package/dist/pipeline-analytics/_stub.js +1 -1
- package/dist/pipeline-analytics/{_virtual_mf-localSharedImportMap___mfe_internal__addon_pipeline_analytics_widgets-D42gEB0t.mjs → _virtual_mf-localSharedImportMap___mfe_internal__addon_pipeline_analytics_widgets-Bs8m-J-O.mjs} +2 -2
- package/dist/pipeline-analytics/{_virtual_mf___mfe_internal__addon_pipeline_analytics_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-C8ZDqFq3.mjs → _virtual_mf___mfe_internal__addon_pipeline_analytics_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-B0sS9pCe.mjs} +1 -1
- package/dist/pipeline-analytics/{hostInit-fxXc6Mk3.mjs → hostInit-jMBJBSqj.mjs} +2 -2
- package/dist/pipeline-analytics/index.js +395 -77
- package/dist/pipeline-analytics/index.mjs +395 -77
- package/dist/pipeline-analytics/remoteEntry.js +1 -1
- package/package.json +1 -1
- package/python/raw_tensor_inference.py +16 -1
- package/python/session_threads.py +33 -0
- package/python/test_session_threads.py +65 -0
- package/python/text_encoder_inference.py +15 -1
|
@@ -30,7 +30,7 @@ async function d(e) {
|
|
|
30
30
|
}
|
|
31
31
|
}
|
|
32
32
|
async function f() {
|
|
33
|
-
return l ||= d(() => import("./_virtual_mf-localSharedImportMap___mfe_internal__addon_pipeline_analytics_widgets-
|
|
33
|
+
return l ||= d(() => import("./_virtual_mf-localSharedImportMap___mfe_internal__addon_pipeline_analytics_widgets-Bs8m-J-O.mjs")).catch((e) => {
|
|
34
34
|
throw l = void 0, e;
|
|
35
35
|
}), l;
|
|
36
36
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@camstack/addon-post-analysis",
|
|
3
|
-
"version": "1.2.
|
|
3
|
+
"version": "1.2.271",
|
|
4
4
|
"description": "Post-Analysis bundle — enrichment, embedding-encoder, pipeline-analytics. Multi-entry npm package shipping addons that consume pipeline output.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"camstack",
|
|
@@ -19,6 +19,7 @@ to stderr (stdout is the binary framing channel only).
|
|
|
19
19
|
from __future__ import annotations
|
|
20
20
|
|
|
21
21
|
import argparse
|
|
22
|
+
import os
|
|
22
23
|
import sys
|
|
23
24
|
|
|
24
25
|
import numpy as np
|
|
@@ -31,6 +32,7 @@ from tensor_frames import (
|
|
|
31
32
|
read_frame,
|
|
32
33
|
write_frame,
|
|
33
34
|
)
|
|
35
|
+
from session_threads import resolve_session_threads
|
|
34
36
|
|
|
35
37
|
|
|
36
38
|
def numpy_dtype_for(onnx_type: str):
|
|
@@ -46,10 +48,23 @@ def numpy_dtype_for(onnx_type: str):
|
|
|
46
48
|
def main() -> None:
|
|
47
49
|
ap = argparse.ArgumentParser()
|
|
48
50
|
ap.add_argument("model")
|
|
51
|
+
# A CEILING, not a tuning. This engine serves arbitrary vision models, so
|
|
52
|
+
# unlike the text encoder it keeps some parallelism — but onnxruntime's
|
|
53
|
+
# default is the HOST's core count, held for the life of the process, and
|
|
54
|
+
# that is a fact about the machine rather than about the work. Nobody has
|
|
55
|
+
# measured the right number here because this path runs on no node of the
|
|
56
|
+
# current cluster; an operator who measures raises it with --threads.
|
|
57
|
+
ap.add_argument("--threads", type=int, default=4)
|
|
49
58
|
args = ap.parse_args()
|
|
50
59
|
|
|
51
60
|
print(f"raw_tensor_inference: loading model {args.model}", file=sys.stderr)
|
|
52
|
-
|
|
61
|
+
intra, inter = resolve_session_threads(args.threads, os.cpu_count())
|
|
62
|
+
sess_options = ort.SessionOptions()
|
|
63
|
+
sess_options.intra_op_num_threads = intra
|
|
64
|
+
sess_options.inter_op_num_threads = inter
|
|
65
|
+
sess = ort.InferenceSession(
|
|
66
|
+
args.model, sess_options=sess_options, providers=["CPUExecutionProvider"]
|
|
67
|
+
)
|
|
53
68
|
inp = sess.get_inputs()[0]
|
|
54
69
|
input_name = inp.name
|
|
55
70
|
input_dtype = numpy_dtype_for(inp.type)
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""The thread ceiling for an embedding engine's onnxruntime session.
|
|
3
|
+
|
|
4
|
+
Deliberately free of third-party imports so the decision can be tested without
|
|
5
|
+
onnxruntime or numpy installed — see `test_session_threads.py` for why that
|
|
6
|
+
matters and what this prevents.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def resolve_session_threads(
|
|
12
|
+
requested: int, cpu_count: int | None
|
|
13
|
+
) -> tuple[int, int]:
|
|
14
|
+
"""(intra_op_num_threads, inter_op_num_threads) for an engine session.
|
|
15
|
+
|
|
16
|
+
`requested` is a CEILING the caller states, not a tuning: it exists so the
|
|
17
|
+
permanent thread cost of a long-lived engine stops scaling with whatever
|
|
18
|
+
host it lands on. The machine's own core count caps it in turn, because
|
|
19
|
+
asking for more op threads than there are cores only buys coordination.
|
|
20
|
+
|
|
21
|
+
An unknown core count (`os.cpu_count()` answers None when it cannot tell)
|
|
22
|
+
means "no cap available", NOT "no cores": the request stands. Folding an
|
|
23
|
+
unreadable measurement into zero is the mistake D393 names.
|
|
24
|
+
|
|
25
|
+
inter-op is always 1. Both engines serve exactly one request at a time by
|
|
26
|
+
construction — read a framed request, infer, write a framed response — so
|
|
27
|
+
there is never a second branch of the graph in flight for inter-op
|
|
28
|
+
parallelism to run.
|
|
29
|
+
"""
|
|
30
|
+
ceiling = max(1, requested)
|
|
31
|
+
if cpu_count is not None and cpu_count > 0:
|
|
32
|
+
ceiling = min(ceiling, cpu_count)
|
|
33
|
+
return ceiling, 1
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""How many threads an embedding engine's onnxruntime session may hold.
|
|
3
|
+
|
|
4
|
+
Both embedding subprocesses — `raw_tensor_inference.py` and
|
|
5
|
+
`text_encoder_inference.py` — build an `ort.InferenceSession` with no
|
|
6
|
+
`SessionOptions`, so onnxruntime takes `intra_op_num_threads` = the HOST's core
|
|
7
|
+
count. These are not one-shot scripts: each writes a READY frame and then sits
|
|
8
|
+
in a `while True` serving requests, so those threads are held for the life of
|
|
9
|
+
the process. On this cluster's hub that is 20 threads per engine, and it grows
|
|
10
|
+
with the machine — a fact about the host, not about the work.
|
|
11
|
+
|
|
12
|
+
The sibling `yamnet_audio.py` already learned this the expensive way
|
|
13
|
+
(~1000% CPU across a handful of audio streams) and caps to one op thread with a
|
|
14
|
+
`--threads` override. This is the same rule, made testable: the caller states a
|
|
15
|
+
ceiling, the machine's core count caps it, and the result is never zero.
|
|
16
|
+
|
|
17
|
+
Kept free of numpy and onnxruntime on purpose — neither installs on the
|
|
18
|
+
operator's Mac, and a decision worth testing should not need a GPU runtime to
|
|
19
|
+
be checked.
|
|
20
|
+
"""
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import unittest
|
|
24
|
+
|
|
25
|
+
from session_threads import resolve_session_threads
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class TestResolveSessionThreads(unittest.TestCase):
|
|
29
|
+
def test_a_request_larger_than_the_machine_is_capped_by_the_machine(self) -> None:
|
|
30
|
+
intra, _ = resolve_session_threads(8, cpu_count=4)
|
|
31
|
+
self.assertEqual(intra, 4)
|
|
32
|
+
|
|
33
|
+
def test_a_smaller_request_is_honoured_on_a_big_host(self) -> None:
|
|
34
|
+
# The whole point: a 20-core hub must not hand 20 permanent threads to
|
|
35
|
+
# an engine that asked for 4.
|
|
36
|
+
intra, _ = resolve_session_threads(4, cpu_count=20)
|
|
37
|
+
self.assertEqual(intra, 4)
|
|
38
|
+
|
|
39
|
+
def test_it_never_resolves_to_zero_or_fewer_threads(self) -> None:
|
|
40
|
+
for requested in (0, -3):
|
|
41
|
+
with self.subTest(requested=requested):
|
|
42
|
+
intra, inter = resolve_session_threads(requested, cpu_count=20)
|
|
43
|
+
self.assertEqual(intra, 1)
|
|
44
|
+
self.assertGreaterEqual(inter, 1)
|
|
45
|
+
|
|
46
|
+
def test_an_unknown_core_count_does_not_silently_become_no_threads(self) -> None:
|
|
47
|
+
# os.cpu_count() returns None when it cannot tell. Folding that into 0
|
|
48
|
+
# would hand onnxruntime an invalid session; the request stands instead.
|
|
49
|
+
for unknown in (None, 0):
|
|
50
|
+
with self.subTest(cpu_count=unknown):
|
|
51
|
+
intra, _ = resolve_session_threads(4, cpu_count=unknown)
|
|
52
|
+
self.assertEqual(intra, 4)
|
|
53
|
+
|
|
54
|
+
def test_inter_op_is_always_one(self) -> None:
|
|
55
|
+
# These engines serve ONE request at a time by construction (a framed
|
|
56
|
+
# read, an inference, a framed write). There is never a second graph
|
|
57
|
+
# branch in flight for inter-op parallelism to run.
|
|
58
|
+
for cpu in (1, 4, 20, 64):
|
|
59
|
+
with self.subTest(cpu_count=cpu):
|
|
60
|
+
_, inter = resolve_session_threads(4, cpu_count=cpu)
|
|
61
|
+
self.assertEqual(inter, 1)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
if __name__ == "__main__":
|
|
65
|
+
unittest.main()
|
|
@@ -20,6 +20,7 @@ Diagnostics go to stderr (stdout is the binary framing channel only).
|
|
|
20
20
|
from __future__ import annotations
|
|
21
21
|
|
|
22
22
|
import argparse
|
|
23
|
+
import os
|
|
23
24
|
import sys
|
|
24
25
|
|
|
25
26
|
import numpy as np
|
|
@@ -27,6 +28,7 @@ import onnxruntime as ort
|
|
|
27
28
|
from tokenizers import Tokenizer
|
|
28
29
|
|
|
29
30
|
from tensor_frames import READY_FRAME, encode_tensor, read_frame, write_frame
|
|
31
|
+
from session_threads import resolve_session_threads
|
|
30
32
|
|
|
31
33
|
# CLIP fixed context length — token ids are truncated/padded to exactly this.
|
|
32
34
|
CONTEXT_LENGTH = 77
|
|
@@ -44,6 +46,12 @@ def main() -> None:
|
|
|
44
46
|
ap = argparse.ArgumentParser()
|
|
45
47
|
ap.add_argument("model")
|
|
46
48
|
ap.add_argument("tokenizer")
|
|
49
|
+
# A CLIP text encoder is one pass over a fixed 77-token window. There is
|
|
50
|
+
# nothing here worth splitting across threads, and onnxruntime's default is
|
|
51
|
+
# the HOST's core count held for the life of this process — 20 on this
|
|
52
|
+
# cluster's hub. `yamnet_audio.py` measured what that costs (~1000% CPU on
|
|
53
|
+
# a handful of streams) and landed on the same default.
|
|
54
|
+
ap.add_argument("--threads", type=int, default=1)
|
|
47
55
|
args = ap.parse_args()
|
|
48
56
|
|
|
49
57
|
print(
|
|
@@ -51,7 +59,13 @@ def main() -> None:
|
|
|
51
59
|
f"tokenizer {args.tokenizer}",
|
|
52
60
|
file=sys.stderr,
|
|
53
61
|
)
|
|
54
|
-
|
|
62
|
+
intra, inter = resolve_session_threads(args.threads, os.cpu_count())
|
|
63
|
+
sess_options = ort.SessionOptions()
|
|
64
|
+
sess_options.intra_op_num_threads = intra
|
|
65
|
+
sess_options.inter_op_num_threads = inter
|
|
66
|
+
sess = ort.InferenceSession(
|
|
67
|
+
args.model, sess_options=sess_options, providers=["CPUExecutionProvider"]
|
|
68
|
+
)
|
|
55
69
|
input_name = sess.get_inputs()[0].name
|
|
56
70
|
output_name = sess.get_outputs()[0].name
|
|
57
71
|
tokenizer = build_tokenizer(args.tokenizer)
|