echoact 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. echoact/__init__.py +3 -0
  2. echoact/__main__.py +117 -0
  3. echoact/app.py +315 -0
  4. echoact/audio/__init__.py +0 -0
  5. echoact/audio/devices.py +192 -0
  6. echoact/audio/player.py +611 -0
  7. echoact/audio/wav.py +854 -0
  8. echoact/config/__init__.py +0 -0
  9. echoact/config/budget.py +370 -0
  10. echoact/config/settings.py +1244 -0
  11. echoact/db/__init__.py +0 -0
  12. echoact/db/backup.py +2429 -0
  13. echoact/db/migrations.py +434 -0
  14. echoact/db/schema.sql +214 -0
  15. echoact/db/store.py +2062 -0
  16. echoact/diagnostics.py +902 -0
  17. echoact/domain.py +487 -0
  18. echoact/engine/__init__.py +0 -0
  19. echoact/engine/container.py +843 -0
  20. echoact/engine/protocol.py +241 -0
  21. echoact/engine/runtime.py +324 -0
  22. echoact/engine/supervisor.py +961 -0
  23. echoact/engine/worker.py +659 -0
  24. echoact/errors.py +281 -0
  25. echoact/instance.py +172 -0
  26. echoact/jobs/__init__.py +0 -0
  27. echoact/jobs/engine.py +776 -0
  28. echoact/jobs/request.py +300 -0
  29. echoact/mcp/__init__.py +0 -0
  30. echoact/mcp/__main__.py +50 -0
  31. echoact/mcp/client.py +202 -0
  32. echoact/mcp/config.py +112 -0
  33. echoact/mcp/server.py +340 -0
  34. echoact/models/__init__.py +0 -0
  35. echoact/models/catalog.py +273 -0
  36. echoact/models/manifest.py +278 -0
  37. echoact/models/registry.py +1551 -0
  38. echoact/paths.py +93 -0
  39. echoact/policy.py +189 -0
  40. echoact/security/__init__.py +0 -0
  41. echoact/security/credentials.py +930 -0
  42. echoact/security/ratelimit.py +534 -0
  43. echoact/service/__init__.py +20 -0
  44. echoact/service/app.py +182 -0
  45. echoact/service/deps.py +563 -0
  46. echoact/service/errors.py +241 -0
  47. echoact/service/routes.py +1125 -0
  48. echoact/service/schemas.py +509 -0
  49. echoact/service/server.py +270 -0
  50. echoact/text/__init__.py +0 -0
  51. echoact/text/language.py +44 -0
  52. echoact/text/loader.py +577 -0
  53. echoact/text/normalize.py +924 -0
  54. echoact/text/segment.py +499 -0
  55. echoact/text/sniff.py +1202 -0
  56. echoact/ui/__init__.py +0 -0
  57. echoact/ui/bridge.py +50 -0
  58. echoact/ui/controls.py +360 -0
  59. echoact/ui/credential_dialog.py +131 -0
  60. echoact/ui/fonts.py +94 -0
  61. echoact/ui/i18n.py +260 -0
  62. echoact/ui/icons.py +440 -0
  63. echoact/ui/library.py +1642 -0
  64. echoact/ui/licence.py +162 -0
  65. echoact/ui/main_window.py +1202 -0
  66. echoact/ui/mcp_setup.py +494 -0
  67. echoact/ui/models_view.py +1142 -0
  68. echoact/ui/notifications.py +202 -0
  69. echoact/ui/reading.py +494 -0
  70. echoact/ui/settings_view.py +2258 -0
  71. echoact/ui/status_view.py +1193 -0
  72. echoact/ui/theme.py +579 -0
  73. echoact/util/__init__.py +0 -0
  74. echoact/util/ids.py +62 -0
  75. echoact/util/logging.py +127 -0
  76. echoact-0.1.0.dist-info/METADATA +162 -0
  77. echoact-0.1.0.dist-info/RECORD +80 -0
  78. echoact-0.1.0.dist-info/WHEEL +4 -0
  79. echoact-0.1.0.dist-info/entry_points.txt +3 -0
  80. echoact-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,843 @@
1
+ """The operating-system resource container that holds the synthesis worker.
2
+
3
+ N-03 is the requirement this module exists for, and it is deliberately
4
+ uneven: on Windows the CPU and memory of the generation job are *enforced*,
5
+ while on macOS "an enforced ceiling that blocks even momentary usage spikes
6
+ is not guaranteed". So the container does not present one fiction with a
7
+ different implementation underneath. It applies the strongest facility the
8
+ platform has and then says, per limit, whether that facility is ``ENFORCED``
9
+ (the kernel refuses to let the child exceed it) or ``MONITORED`` (we sample,
10
+ and a spike between samples passes unseen). F-22 shows the answer, which is
11
+ why :class:`ContainerLimits` is part of the public surface rather than an
12
+ implementation note.
13
+
14
+ The container is also what makes N-21 measurable. A Job Object accounts for
15
+ its members and nothing else, so the CPU time and peak memory reported here
16
+ are the generation job's alone -- not the GUI's, the database's, or the
17
+ service's. The polling monitor exists for the same reason on platforms with
18
+ no such accounting, and on Windows it supplies the current working set and a
19
+ CPU rate, which the Job Object does not keep.
20
+
21
+ **Which memory figure is which.** N-03 warns that "the RAM usage shown on
22
+ screen and the memory value used for limit decisions may differ", and on
23
+ Windows they genuinely do: ``ProcessMemoryLimit`` is enforced against
24
+ *committed* memory, while the figure a user recognises as RAM usage is the
25
+ working set. A process can commit 3 GiB and hold 800 MiB resident. So
26
+ :class:`ContainerUsage` reports both and labels them: ``rss_bytes`` is the
27
+ on-screen figure, ``peak_commit_bytes`` is the quantity the Windows limit is
28
+ enforced against, and :attr:`ContainerLimits.memory_basis` names which one
29
+ this platform's limit actually tests.
30
+ """
31
+
32
+ from __future__ import annotations
33
+
34
+ import ctypes
35
+ import os
36
+ import sys
37
+ import threading
38
+ from dataclasses import dataclass
39
+ from enum import StrEnum
40
+ from typing import Any, Final
41
+
42
+ import psutil
43
+
44
+ from ..domain import Budget
45
+ from ..policy import CPU_PERCENT_MAX, CPU_PERCENT_MIN, RESOURCE_SAMPLE_INTERVAL_S
46
+ from ..util.ids import monotonic
47
+ from ..util.logging import get_logger
48
+
49
+ log = get_logger("engine.container")
50
+
51
+ _LOGICAL_CPUS: Final = os.cpu_count() or 1
52
+
53
+
54
+ class Enforcement(StrEnum):
55
+ """How seriously a limit is meant. N-03 requires this distinction to
56
+ reach the user, so it is a value, not a comment."""
57
+
58
+ #: The kernel refuses the excess. The child cannot exceed the limit.
59
+ ENFORCED = "enforced"
60
+ #: Sampled every ``RESOURCE_SAMPLE_INTERVAL_S`` and acted on after the
61
+ #: fact. A spike shorter than the interval is not caught.
62
+ MONITORED = "monitored"
63
+ #: Neither. The figure is reported and nothing is applied.
64
+ UNAVAILABLE = "unavailable"
65
+
66
+
67
+ class LimitBasis(StrEnum):
68
+ """Which memory quantity a platform's memory limit is tested against.
69
+
70
+ N-03 allows the on-screen figure and the limit figure to differ; this
71
+ names the second one so a display can say which is which.
72
+ """
73
+
74
+ COMMIT = "commit"
75
+ RESIDENT = "resident"
76
+ ADDRESS_SPACE = "address_space"
77
+ NONE = "none"
78
+
79
+
80
+ @dataclass(frozen=True, slots=True)
81
+ class ContainerLimits:
82
+ """What was actually applied, for F-22's display and N-03's honesty."""
83
+
84
+ memory: Enforcement
85
+ cpu: Enforcement
86
+ #: Whether closing the container is by itself enough to kill the worker.
87
+ kill_on_close: Enforcement
88
+ memory_basis: LimitBasis
89
+ memory_bytes: int
90
+ cpu_percent: int
91
+ #: Names the facility, e.g. "windows job object, hard CPU cap". Safe for
92
+ #: a log and for the GUI: it never contains a path or user text.
93
+ facility: str = ""
94
+
95
+ @property
96
+ def fully_enforced(self) -> bool:
97
+ return self.memory is Enforcement.ENFORCED and self.cpu is Enforcement.ENFORCED
98
+
99
+
100
+ @dataclass(frozen=True, slots=True)
101
+ class ContainerUsage:
102
+ """One sample of the contained job's resource use.
103
+
104
+ ``rss_bytes`` is the on-screen figure (F-22). ``peak_commit_bytes`` is
105
+ the peak of the quantity Windows enforces against; it is zero where the
106
+ platform does not account for it. N-03 permits the two to differ, and
107
+ this type is where that difference is visible rather than hidden.
108
+ """
109
+
110
+ rss_bytes: int
111
+ peak_rss_bytes: int
112
+ peak_commit_bytes: int
113
+ user_seconds: float
114
+ kernel_seconds: float
115
+ #: Percentage of *total logical CPU capacity*, matching Section 4.1's
116
+ #: unit rule and F-20's setting -- not percentage of one core.
117
+ cpu_percent: float
118
+ process_count: int
119
+ #: ``monotonic()`` at the sample, never wall clock, so a clock change
120
+ #: cannot make a rate negative.
121
+ sampled_at: float
122
+ source: str
123
+
124
+
125
+ class ResourceContainer:
126
+ """Common interface. A container is created per budget, holds exactly
127
+ one worker process tree, and is single-use: :meth:`close` ends it.
128
+
129
+ The lifecycle is fixed by the Windows requirement that a child must be
130
+ inside the job before it runs anything, or F-20 and F-21 would be
131
+ advisory for the first instants of the process. Callers therefore do
132
+ ``popen_kwargs()`` -> spawn -> :meth:`adopt` -> :meth:`start_child`, and
133
+ a platform with no such hazard makes ``start_child`` a no-op.
134
+ """
135
+
136
+ facility = "none"
137
+
138
+ def __init__(self, budget: Budget, *, name: str = "echoact-worker") -> None:
139
+ self.budget = budget
140
+ self.name = name
141
+ self._pid: int | None = None
142
+ self._proc: psutil.Process | None = None
143
+ self._closed = False
144
+ self._monitor: threading.Thread | None = None
145
+ self._stop = threading.Event()
146
+ self._sample_lock = threading.Lock()
147
+ self._usage: ContainerUsage | None = None
148
+ self._peak_rss = 0
149
+ self._prev_cpu_seconds: float | None = None
150
+ self._prev_sampled_at: float | None = None
151
+ self._memory_exceeded = False
152
+
153
+ # -- construction-time facts ------------------------------------------
154
+
155
+ @property
156
+ def limits(self) -> ContainerLimits:
157
+ raise NotImplementedError # pragma: no cover - abstract
158
+
159
+ def popen_kwargs(self) -> dict[str, Any]:
160
+ """Extra ``subprocess.Popen`` arguments this container needs."""
161
+ return {}
162
+
163
+ # -- lifecycle ---------------------------------------------------------
164
+
165
+ def adopt(self, pid: int) -> None:
166
+ """Place an already-created process under this container.
167
+
168
+ Raises ``OSError`` if the platform facility rejects the process; the
169
+ caller must then abandon the child rather than run it unconstrained.
170
+ """
171
+ self._pid = pid
172
+ try:
173
+ self._proc = psutil.Process(pid)
174
+ except psutil.Error as exc: # pragma: no cover - died immediately
175
+ raise OSError(f"cannot observe pid {pid}") from exc
176
+ self._start_monitor()
177
+
178
+ def start_child(self, pid: int) -> None:
179
+ """Let an adopted child begin executing. A no-op where the child was
180
+ never suspended."""
181
+
182
+ def terminate(self) -> None:
183
+ """Kill what is inside without releasing the container.
184
+
185
+ The whole tree, not just the process we spawned: the interpreter a
186
+ virtual environment hands out can be a launcher that runs the real
187
+ one as a child, and killing only the launcher would leave the worker
188
+ holding the model -- exactly what N-22 says must not survive.
189
+ """
190
+ p = self._proc
191
+ if p is None:
192
+ return
193
+ try:
194
+ victims = [*p.children(recursive=True), p]
195
+ except psutil.Error:
196
+ victims = [p]
197
+ for victim in victims:
198
+ try:
199
+ victim.kill()
200
+ except psutil.Error:
201
+ continue
202
+
203
+ def close(self) -> None:
204
+ """Release the container. Where the platform supports it this also
205
+ kills anything still inside, which is what N-22's five-second
206
+ deadline ultimately rests on."""
207
+ self._closed = True
208
+ self._stop.set()
209
+ m = self._monitor
210
+ if m is not None and m is not threading.current_thread():
211
+ m.join(timeout=1.0)
212
+ self._monitor = None
213
+
214
+ @property
215
+ def closed(self) -> bool:
216
+ return self._closed
217
+
218
+ def __enter__(self) -> ResourceContainer:
219
+ return self
220
+
221
+ def __exit__(self, *exc: object) -> None:
222
+ self.close()
223
+
224
+ # -- accounting --------------------------------------------------------
225
+
226
+ def usage(self) -> ContainerUsage | None:
227
+ """The most recent sample, or ``None`` before the first one."""
228
+ with self._sample_lock:
229
+ return self._usage
230
+
231
+ def sample(self) -> ContainerUsage | None:
232
+ """Take a sample now. The monitor thread calls this on a timer;
233
+ callers may force one to refresh a display."""
234
+ u = self._collect()
235
+ if u is not None:
236
+ with self._sample_lock:
237
+ self._usage = u
238
+ return u
239
+
240
+ @property
241
+ def memory_exceeded(self) -> bool:
242
+ """True once a *monitored* memory limit was seen to be exceeded.
243
+
244
+ Where memory is ENFORCED this stays false: the kernel fails the
245
+ allocation instead, so F-23's report comes from the worker's own
246
+ failure rather than from a sample.
247
+ """
248
+ return self._memory_exceeded
249
+
250
+ def contains(self, pid: int) -> bool:
251
+ """Whether the platform facility currently accounts for this pid."""
252
+ return self._pid == pid and self._alive()
253
+
254
+ # -- internals ---------------------------------------------------------
255
+
256
+ def _alive(self) -> bool:
257
+ p = self._proc
258
+ try:
259
+ return p is not None and p.is_running() and p.status() != psutil.STATUS_ZOMBIE
260
+ except psutil.Error:
261
+ return False
262
+
263
+ def _start_monitor(self) -> None:
264
+ if self._monitor is not None:
265
+ return
266
+ self.sample()
267
+ t = threading.Thread(target=self._monitor_loop, name=f"{self.name}-monitor", daemon=True)
268
+ self._monitor = t
269
+ t.start()
270
+
271
+ def _monitor_loop(self) -> None:
272
+ while not self._stop.wait(RESOURCE_SAMPLE_INTERVAL_S):
273
+ if not self._alive():
274
+ continue
275
+ try:
276
+ self.sample()
277
+ self._enforce_by_polling()
278
+ except Exception: # pragma: no cover - a sample must never crash
279
+ log.debug("resource sample failed", exc_info=True)
280
+
281
+ def _enforce_by_polling(self) -> None:
282
+ """Where memory is MONITORED, act on a sample. Overridden to do
283
+ nothing where the kernel already enforces the ceiling."""
284
+
285
+ def _process_sample(self) -> tuple[int, float, int] | None:
286
+ """``(rss, cpu_seconds, process_count)`` over the worker's tree."""
287
+ p = self._proc
288
+ if p is None:
289
+ return None
290
+ try:
291
+ with p.oneshot():
292
+ rss = int(p.memory_info().rss)
293
+ times = p.cpu_times()
294
+ cpu = float(times.user + times.system)
295
+ count = 1
296
+ for child in p.children(recursive=True):
297
+ try:
298
+ rss += int(child.memory_info().rss)
299
+ ct = child.cpu_times()
300
+ cpu += float(ct.user + ct.system)
301
+ count += 1
302
+ except psutil.Error:
303
+ continue
304
+ return rss, cpu, count
305
+ except psutil.Error:
306
+ return None
307
+
308
+ def _rate(self, cpu_seconds: float, at: float) -> float:
309
+ prev_cpu, prev_at = self._prev_cpu_seconds, self._prev_sampled_at
310
+ self._prev_cpu_seconds, self._prev_sampled_at = cpu_seconds, at
311
+ if prev_cpu is None or prev_at is None or at <= prev_at:
312
+ return 0.0
313
+ used = max(0.0, cpu_seconds - prev_cpu)
314
+ return 100.0 * used / ((at - prev_at) * _LOGICAL_CPUS)
315
+
316
+ def _collect(self) -> ContainerUsage | None:
317
+ raise NotImplementedError # pragma: no cover - abstract
318
+
319
+ def _psutil_collect(self) -> ContainerUsage | None:
320
+ """The sample every container without kernel accounting takes."""
321
+ at = monotonic()
322
+ proc = self._process_sample()
323
+ if proc is None:
324
+ return None
325
+ rss, cpu_seconds, count = proc
326
+ self._peak_rss = max(self._peak_rss, rss)
327
+ return ContainerUsage(
328
+ rss_bytes=rss,
329
+ peak_rss_bytes=self._peak_rss,
330
+ peak_commit_bytes=0,
331
+ user_seconds=cpu_seconds,
332
+ kernel_seconds=0.0,
333
+ cpu_percent=self._rate(cpu_seconds, at),
334
+ process_count=count,
335
+ sampled_at=at,
336
+ source="psutil",
337
+ )
338
+
339
+
340
+ # ======================================================================
341
+ # Windows: a Job Object
342
+ # ======================================================================
343
+
344
+ # JOBOBJECT_CPU_RATE_CONTROL_INFORMATION is not wrapped by pywin32, so the
345
+ # CPU half of F-20 goes through ctypes. The information class number and the
346
+ # flags are from the Win32 headers and are part of the stable ABI.
347
+ _JOB_CPU_RATE_CONTROL_INFO_CLASS: Final = 15
348
+ _CPU_RATE_CONTROL_ENABLE: Final = 0x1
349
+ _CPU_RATE_CONTROL_HARD_CAP: Final = 0x4
350
+ #: ``CpuRate`` is in hundredths of a percent of *total* machine CPU, which is
351
+ #: exactly F-20's unit, so the only conversion is a factor of 100.
352
+ _CPU_RATE_SCALE: Final = 100
353
+
354
+ _CREATE_SUSPENDED: Final = 0x00000004
355
+ _CREATE_NO_WINDOW: Final = 0x08000000
356
+ _THREAD_SUSPEND_RESUME: Final = 0x0002
357
+ _BELOW_NORMAL_PRIORITY_CLASS: Final = 0x00004000
358
+ _RESUME_THREAD_FAILED: Final = 0xFFFFFFFF
359
+
360
+
361
+ class _CpuRateControl(ctypes.Structure):
362
+ _fields_ = (("ControlFlags", ctypes.c_uint32), ("CpuRate", ctypes.c_uint32))
363
+
364
+
365
+ class WindowsJobContainer(ResourceContainer):
366
+ """N-03's enforced case.
367
+
368
+ A Job Object gives all three things the requirements ask for at once: a
369
+ memory ceiling the kernel refuses to exceed (F-21), a CPU rate the
370
+ scheduler enforces (F-20), and ``KILL_ON_JOB_CLOSE`` so the worker cannot
371
+ outlive the container -- the last being what lets N-22 promise release
372
+ within five seconds even if the worker ignores every polite request.
373
+
374
+ Limiting the process directly was rejected: a working-set limit only
375
+ trims, it does not refuse, and an affinity mask fixes *which* cores are
376
+ used rather than how much of them, so neither can express F-20's "20% of
377
+ total CPU".
378
+ """
379
+
380
+ facility = "windows job object"
381
+
382
+ def __init__(self, budget: Budget, *, name: str = "echoact-worker") -> None:
383
+ super().__init__(budget, name=name)
384
+ import win32job
385
+
386
+ self._win32job = win32job
387
+ self._k32 = ctypes.WinDLL("kernel32", use_last_error=True)
388
+ # Unnamed: a name would be a machine-wide handle another process
389
+ # could open, and nothing needs to find this one.
390
+ self._job: Any = win32job.CreateJobObject(None, "")
391
+ self._cpu_enforcement = Enforcement.UNAVAILABLE
392
+ self._cpu_facility = "none"
393
+ self._apply_memory_limit()
394
+ self._apply_cpu_limit()
395
+
396
+ # -- limit application -------------------------------------------------
397
+
398
+ def _apply_memory_limit(self) -> None:
399
+ wj = self._win32job
400
+ info = wj.QueryInformationJobObject(self._job, wj.JobObjectExtendedLimitInformation)
401
+ info["BasicLimitInformation"]["LimitFlags"] = (
402
+ wj.JOB_OBJECT_LIMIT_KILL_ON_JOB_CLOSE
403
+ | wj.JOB_OBJECT_LIMIT_PROCESS_MEMORY
404
+ | wj.JOB_OBJECT_LIMIT_JOB_MEMORY
405
+ | wj.JOB_OBJECT_LIMIT_DIE_ON_UNHANDLED_EXCEPTION
406
+ )
407
+ # Both limits carry the same figure: one worker process is expected,
408
+ # and the job-wide limit catches anything the worker itself spawns,
409
+ # which N-21 counts against the generation job either way.
410
+ info["ProcessMemoryLimit"] = int(self.budget.memory_bytes)
411
+ info["JobMemoryLimit"] = int(self.budget.memory_bytes)
412
+ wj.SetInformationJobObject(self._job, wj.JobObjectExtendedLimitInformation, info)
413
+
414
+ def _apply_cpu_limit(self) -> None:
415
+ percent = max(CPU_PERCENT_MIN, min(CPU_PERCENT_MAX, int(self.budget.cpu_percent)))
416
+ rate = max(1, min(10_000, percent * _CPU_RATE_SCALE))
417
+ self._k32.SetInformationJobObject.argtypes = [
418
+ ctypes.c_void_p,
419
+ ctypes.c_int,
420
+ ctypes.c_void_p,
421
+ ctypes.c_uint32,
422
+ ]
423
+ self._k32.SetInformationJobObject.restype = ctypes.c_int
424
+
425
+ for flags, label in (
426
+ (_CPU_RATE_CONTROL_ENABLE | _CPU_RATE_CONTROL_HARD_CAP, "hard CPU cap"),
427
+ (_CPU_RATE_CONTROL_ENABLE, "CPU rate target"),
428
+ ):
429
+ data = _CpuRateControl(flags, rate)
430
+ ok = self._k32.SetInformationJobObject(
431
+ int(self._job),
432
+ _JOB_CPU_RATE_CONTROL_INFO_CLASS,
433
+ ctypes.byref(data),
434
+ ctypes.sizeof(data),
435
+ )
436
+ if ok:
437
+ self._cpu_enforcement = Enforcement.ENFORCED
438
+ self._cpu_facility = label
439
+ return
440
+ log.debug(
441
+ "job cpu rate control rejected (flags=%#x err=%d)",
442
+ flags,
443
+ ctypes.get_last_error(),
444
+ )
445
+
446
+ # Older or restricted builds reject the information class outright.
447
+ # Say so rather than reporting a limit that is not there: a lowered
448
+ # priority class still yields the machine under contention, but it is
449
+ # a courtesy, not a ceiling, so CPU drops to MONITORED.
450
+ try:
451
+ wj = self._win32job
452
+ # Through the *extended* structure, not the basic one: the job
453
+ # already carries the memory flags, and the basic information
454
+ # class rejects a flag set that contains them.
455
+ info = wj.QueryInformationJobObject(self._job, wj.JobObjectExtendedLimitInformation)
456
+ basic = info["BasicLimitInformation"]
457
+ basic["LimitFlags"] = basic["LimitFlags"] | wj.JOB_OBJECT_LIMIT_PRIORITY_CLASS
458
+ basic["PriorityClass"] = _BELOW_NORMAL_PRIORITY_CLASS
459
+ wj.SetInformationJobObject(self._job, wj.JobObjectExtendedLimitInformation, info)
460
+ self._cpu_facility = "below-normal priority class"
461
+ except Exception: # pragma: no cover - depends on the Windows build
462
+ log.warning("no CPU limiting facility available on this build")
463
+ self._cpu_facility = "none"
464
+ self._cpu_enforcement = Enforcement.MONITORED
465
+
466
+ @property
467
+ def limits(self) -> ContainerLimits:
468
+ return ContainerLimits(
469
+ memory=Enforcement.ENFORCED,
470
+ cpu=self._cpu_enforcement,
471
+ kill_on_close=Enforcement.ENFORCED,
472
+ memory_basis=LimitBasis.COMMIT,
473
+ memory_bytes=int(self.budget.memory_bytes),
474
+ cpu_percent=int(self.budget.cpu_percent),
475
+ facility=f"{self.facility}, {self._cpu_facility}",
476
+ )
477
+
478
+ # -- lifecycle ---------------------------------------------------------
479
+
480
+ def popen_kwargs(self) -> dict[str, Any]:
481
+ """Create the child suspended.
482
+
483
+ A child assigned after it has begun running has already had time to
484
+ allocate outside the limit. ``CREATE_SUSPENDED`` closes that window
485
+ completely: nothing in the child executes until :meth:`start_child`,
486
+ so F-20 and F-21 hold from its first instruction rather than from a
487
+ few milliseconds in.
488
+ """
489
+ return {"creationflags": _CREATE_SUSPENDED | _CREATE_NO_WINDOW}
490
+
491
+ def adopt(self, pid: int) -> None:
492
+ import win32api
493
+ import win32con
494
+
495
+ access = (
496
+ win32con.PROCESS_SET_QUOTA
497
+ | win32con.PROCESS_TERMINATE
498
+ | win32con.PROCESS_QUERY_INFORMATION
499
+ )
500
+ # pywin32 raises ``pywintypes.error``, which is not an OSError, so it
501
+ # is translated here rather than leaking a Windows-only type into a
502
+ # caller that has to work on three platforms.
503
+ try:
504
+ handle = win32api.OpenProcess(access, False, pid)
505
+ except Exception as exc:
506
+ raise OSError(f"cannot open pid {pid} for job assignment: {exc}") from exc
507
+ try:
508
+ self._win32job.AssignProcessToJobObject(self._job, handle)
509
+ except Exception as exc:
510
+ raise OSError(f"cannot assign pid {pid} to the job object: {exc}") from exc
511
+ finally:
512
+ handle.Close()
513
+ super().adopt(pid)
514
+
515
+ def start_child(self, pid: int) -> None:
516
+ """Resume the suspended child, now that it is inside the job.
517
+
518
+ ``subprocess`` closes the primary thread handle it got from
519
+ ``CreateProcess``, so the thread is reopened by id instead. A freshly
520
+ created suspended process has exactly one thread.
521
+ """
522
+ k32 = self._k32
523
+ k32.OpenThread.argtypes = [ctypes.c_uint32, ctypes.c_int, ctypes.c_uint32]
524
+ k32.OpenThread.restype = ctypes.c_void_p
525
+ k32.ResumeThread.argtypes = [ctypes.c_void_p]
526
+ k32.ResumeThread.restype = ctypes.c_uint32
527
+ k32.CloseHandle.argtypes = [ctypes.c_void_p]
528
+
529
+ try:
530
+ threads = psutil.Process(pid).threads()
531
+ except psutil.Error as exc:
532
+ raise OSError(f"cannot enumerate threads of pid {pid}") from exc
533
+ if not threads: # pragma: no cover - a live process always has one
534
+ raise OSError(f"pid {pid} has no threads to resume")
535
+
536
+ resumed = False
537
+ for t in threads:
538
+ handle = k32.OpenThread(_THREAD_SUSPEND_RESUME, False, int(t.id))
539
+ if not handle:
540
+ continue
541
+ try:
542
+ if k32.ResumeThread(handle) != _RESUME_THREAD_FAILED:
543
+ resumed = True
544
+ finally:
545
+ k32.CloseHandle(handle)
546
+ if not resumed:
547
+ raise OSError(f"could not resume pid {pid}")
548
+
549
+ def terminate(self) -> None:
550
+ """Kill everything in the job at once, without closing it."""
551
+ if self._job is None:
552
+ return
553
+ try:
554
+ self._win32job.TerminateJobObject(self._job, 1)
555
+ except Exception: # pragma: no cover - already gone
556
+ log.debug("terminate job object failed", exc_info=True)
557
+
558
+ def close(self) -> None:
559
+ super().close()
560
+ job, self._job = self._job, None
561
+ if job is None:
562
+ return
563
+ # KILL_ON_JOB_CLOSE makes this handle the leash: dropping the last
564
+ # handle kills whatever is still inside.
565
+ try:
566
+ job.Close()
567
+ except Exception: # pragma: no cover - double close
568
+ log.debug("closing job object failed", exc_info=True)
569
+
570
+ # -- accounting --------------------------------------------------------
571
+
572
+ def job_accounting(self) -> dict[str, Any] | None:
573
+ """Raw Job Object accounting: the generation job's own totals, which
574
+ is exactly what N-21 needs to separate it from whole-app usage."""
575
+ if self._job is None:
576
+ return None
577
+ wj = self._win32job
578
+ try:
579
+ acct = wj.QueryInformationJobObject(
580
+ self._job, wj.JobObjectBasicAndIoAccountingInformation
581
+ )
582
+ ext = wj.QueryInformationJobObject(self._job, wj.JobObjectExtendedLimitInformation)
583
+ except Exception:
584
+ return None
585
+ basic = acct["BasicInfo"]
586
+ return {
587
+ # 100-nanosecond units, per the Win32 struct.
588
+ "user_seconds": basic["TotalUserTime"] / 1e7,
589
+ "kernel_seconds": basic["TotalKernelTime"] / 1e7,
590
+ "active_processes": int(basic["ActiveProcesses"]),
591
+ "total_processes": int(basic["TotalProcesses"]),
592
+ "peak_process_commit": int(ext["PeakProcessMemoryUsed"]),
593
+ "peak_job_commit": int(ext["PeakJobMemoryUsed"]),
594
+ }
595
+
596
+ def job_pids(self) -> tuple[int, ...]:
597
+ """The pids the kernel currently accounts to this job."""
598
+ if self._job is None:
599
+ return ()
600
+ wj = self._win32job
601
+ try:
602
+ listing = wj.QueryInformationJobObject(self._job, wj.JobObjectBasicProcessIdList)
603
+ except Exception:
604
+ return ()
605
+ return tuple(int(p) for p in listing)
606
+
607
+ def contains(self, pid: int) -> bool:
608
+ return pid in self.job_pids()
609
+
610
+ def _collect(self) -> ContainerUsage | None:
611
+ at = monotonic()
612
+ acct = self.job_accounting()
613
+ proc = self._process_sample()
614
+ if acct is None and proc is None:
615
+ return None
616
+ rss = proc[0] if proc else 0
617
+ self._peak_rss = max(self._peak_rss, rss)
618
+ if acct is not None:
619
+ user = acct["user_seconds"]
620
+ kernel = acct["kernel_seconds"]
621
+ cpu_seconds = user + kernel
622
+ count = acct["active_processes"]
623
+ peak_commit = max(acct["peak_process_commit"], acct["peak_job_commit"])
624
+ source = "job object"
625
+ else: # pragma: no cover - job closed between the two reads
626
+ cpu_seconds = proc[1] if proc else 0.0
627
+ user = kernel = 0.0
628
+ count = proc[2] if proc else 0
629
+ peak_commit = 0
630
+ source = "psutil"
631
+ return ContainerUsage(
632
+ rss_bytes=rss,
633
+ peak_rss_bytes=self._peak_rss,
634
+ peak_commit_bytes=peak_commit,
635
+ user_seconds=user,
636
+ kernel_seconds=kernel,
637
+ cpu_percent=self._rate(cpu_seconds, at),
638
+ process_count=count,
639
+ sampled_at=at,
640
+ source=source,
641
+ )
642
+
643
+
644
+ # ======================================================================
645
+ # POSIX: rlimits, niceness, a process group, and honest labelling
646
+ # ======================================================================
647
+
648
+
649
+ class PosixContainer(ResourceContainer):
650
+ """N-03's unenforced case, stated as such.
651
+
652
+ macOS has no Job Object. What it does have is applied here -- an address
653
+ space rlimit, a nice value, and a process group so :meth:`close` can
654
+ reach the whole tree -- but none of it is the ceiling Windows gets:
655
+
656
+ * ``RLIMIT_AS`` bounds address space, not resident memory, and a 64-bit
657
+ process reserves far more than it touches, so a limit tight enough to
658
+ mean anything for RSS would refuse allocations the worker never faults
659
+ in. It is applied with headroom on Linux, where refusing an allocation
660
+ is a genuine ceiling, and as a coarse backstop on Darwin, where memory
661
+ is reported MONITORED because N-03 explicitly declines to promise an
662
+ enforced ceiling there.
663
+ * Niceness changes scheduling order, not share. An idle machine will
664
+ happily give a niced process everything, so F-20's percentage is a
665
+ target here and never a cap.
666
+
667
+ The polling monitor therefore does real work on this platform: it is what
668
+ turns F-23's "if usage exceeds the limit ... the job is halted" into
669
+ behaviour rather than a hope.
670
+ """
671
+
672
+ facility = "posix rlimit + nice"
673
+
674
+ #: Address space may exceed the RSS budget by this factor before the
675
+ #: rlimit refuses, because reserved-but-untouched mappings are normal and
676
+ #: a 1:1 limit would kill a healthy worker.
677
+ _AS_HEADROOM: Final = 4
678
+ #: The most yielding nice value, used at F-20's lowest CPU setting.
679
+ _MAX_NICE: Final = 19
680
+
681
+ def __init__(self, budget: Budget, *, name: str = "echoact-worker") -> None:
682
+ super().__init__(budget, name=name)
683
+ self._enforced_memory = sys.platform.startswith("linux")
684
+
685
+ @property
686
+ def limits(self) -> ContainerLimits:
687
+ return ContainerLimits(
688
+ memory=Enforcement.ENFORCED if self._enforced_memory else Enforcement.MONITORED,
689
+ cpu=Enforcement.MONITORED,
690
+ kill_on_close=Enforcement.MONITORED,
691
+ memory_basis=(
692
+ LimitBasis.ADDRESS_SPACE if self._enforced_memory else LimitBasis.RESIDENT
693
+ ),
694
+ memory_bytes=int(self.budget.memory_bytes),
695
+ cpu_percent=int(self.budget.cpu_percent),
696
+ facility=self.facility,
697
+ )
698
+
699
+ def popen_kwargs(self) -> dict[str, Any]:
700
+ """Limits are applied in the child before ``exec``.
701
+
702
+ That is the POSIX equivalent of the Windows suspended-create: the
703
+ image never runs a single instruction outside its limits.
704
+ """
705
+ limit = int(self.budget.memory_bytes) * self._AS_HEADROOM
706
+ nice = self._nice_value()
707
+
708
+ def _apply() -> None: # pragma: no cover - runs in the forked child
709
+ import resource
710
+
711
+ for which in ("RLIMIT_AS", "RLIMIT_DATA"):
712
+ res = getattr(resource, which, None)
713
+ if res is None:
714
+ continue
715
+ try:
716
+ _soft, hard = resource.getrlimit(res)
717
+ ceiling = limit if hard == resource.RLIM_INFINITY else min(limit, hard)
718
+ resource.setrlimit(res, (ceiling, hard))
719
+ break
720
+ except (ValueError, OSError):
721
+ continue
722
+ try:
723
+ os.nice(nice)
724
+ except OSError:
725
+ pass
726
+
727
+ return {"preexec_fn": _apply, "start_new_session": True}
728
+
729
+ def _nice_value(self) -> int:
730
+ span = CPU_PERCENT_MAX - CPU_PERCENT_MIN
731
+ frac = (int(self.budget.cpu_percent) - CPU_PERCENT_MIN) / (span or 1)
732
+ return int(round(self._MAX_NICE * (1.0 - max(0.0, min(1.0, frac)))))
733
+
734
+ def terminate(self) -> None:
735
+ self._signal_group()
736
+
737
+ def close(self) -> None:
738
+ super().close()
739
+ # No KILL_ON_JOB_CLOSE here; the process group is the nearest thing,
740
+ # and a child that leaves the group survives. ``limits`` says so.
741
+ self._signal_group()
742
+
743
+ def _signal_group(self) -> None:
744
+ pid = self._pid
745
+ if pid is None:
746
+ return
747
+ try:
748
+ os.killpg(os.getpgid(pid), 9)
749
+ except (OSError, AttributeError):
750
+ pass
751
+
752
+ def _enforce_by_polling(self) -> None:
753
+ if self._enforced_memory:
754
+ return
755
+ u = self.usage()
756
+ if u is None or u.rss_bytes <= int(self.budget.memory_bytes):
757
+ return
758
+ self._memory_exceeded = True
759
+ log.warning(
760
+ "worker exceeded the monitored memory budget (%d > %d); halting it",
761
+ u.rss_bytes,
762
+ self.budget.memory_bytes,
763
+ )
764
+ self.terminate()
765
+
766
+ def _collect(self) -> ContainerUsage | None:
767
+ return self._psutil_collect()
768
+
769
+
770
+ class NullContainer(ResourceContainer):
771
+ """No limits at all: for tests, and for a platform whose facilities we
772
+ cannot use.
773
+
774
+ It still samples, because F-22's display and N-21's separate measurement
775
+ are worth having even where no ceiling can be applied, and because a test
776
+ of the supervisor should exercise the same sampling path the product
777
+ uses. Every limit reports UNAVAILABLE, so nothing can mistake it for
778
+ protection.
779
+ """
780
+
781
+ facility = "none"
782
+
783
+ @property
784
+ def limits(self) -> ContainerLimits:
785
+ return ContainerLimits(
786
+ memory=Enforcement.UNAVAILABLE,
787
+ cpu=Enforcement.UNAVAILABLE,
788
+ kill_on_close=Enforcement.UNAVAILABLE,
789
+ memory_basis=LimitBasis.NONE,
790
+ memory_bytes=int(self.budget.memory_bytes),
791
+ cpu_percent=int(self.budget.cpu_percent),
792
+ facility=self.facility,
793
+ )
794
+
795
+ def close(self) -> None:
796
+ super().close()
797
+ # Nothing enforces anything here, but a test container that leaked a
798
+ # process would be worse than useless, so the child is still killed.
799
+ self.terminate()
800
+
801
+ def _collect(self) -> ContainerUsage | None:
802
+ return self._psutil_collect()
803
+
804
+
805
+ #: Chosen once, at import, by platform -- not per call, so a test can see
806
+ #: which implementation this machine will really use.
807
+ if sys.platform == "win32":
808
+ PlatformContainer: type[ResourceContainer] = WindowsJobContainer
809
+ elif os.name == "posix":
810
+ PlatformContainer = PosixContainer
811
+ else: # pragma: no cover - no third kind exists today
812
+ PlatformContainer = NullContainer
813
+
814
+
815
+ def make_container(budget: Budget, *, name: str = "echoact-worker") -> ResourceContainer:
816
+ """The container this platform can actually provide.
817
+
818
+ Falls back to :class:`NullContainer` if the platform facility cannot be
819
+ created. Refusing to synthesise because a Job Object could not be made
820
+ would be worse than saying the limits are unavailable, which
821
+ :attr:`ResourceContainer.limits` makes visible and F-22 shows.
822
+ """
823
+ try:
824
+ return PlatformContainer(budget, name=name)
825
+ except Exception:
826
+ log.warning(
827
+ "resource container unavailable; running without enforced limits", exc_info=True
828
+ )
829
+ return NullContainer(budget, name=name)
830
+
831
+
832
+ __all__ = [
833
+ "ContainerLimits",
834
+ "ContainerUsage",
835
+ "Enforcement",
836
+ "LimitBasis",
837
+ "NullContainer",
838
+ "PlatformContainer",
839
+ "PosixContainer",
840
+ "ResourceContainer",
841
+ "WindowsJobContainer",
842
+ "make_container",
843
+ ]