anbaric 1.58.3 → 1.60.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/api/cli.md +12 -4
- package/docs/api/state-machine.md +27 -3
- package/docs/api/web.md +1 -1
- package/docs/features/actions-and-actors.md +22 -0
- package/package.json +5 -5
package/docs/api/cli.md
CHANGED
|
@@ -23,11 +23,19 @@ project — they walk up to the nearest `package.json`.
|
|
|
23
23
|
| --- | --- |
|
|
24
24
|
| `anbaric apps` | List deployed apps (name, status, port). |
|
|
25
25
|
| `anbaric app configure` | Create or update `.anbaric/app-config.json` (`name`, `internalPort`). |
|
|
26
|
-
| `anbaric app deploy` | Deploy the app and wait until it is live (prompts before replacing a running one). |
|
|
26
|
+
| `anbaric app deploy` | Deploy the app and wait until it is live (prompts before replacing a running one). A running version is drained first — see below. |
|
|
27
27
|
| `anbaric app update` | Deploy, replacing a running app **without** prompting. |
|
|
28
|
-
| `anbaric app status [name]` | Show deploy state, liveness, and recent logs. |
|
|
28
|
+
| `anbaric app status [name]` | Show deploy state (`building`, `draining`, `running`, `failed`, `stopped`), liveness, and recent logs. |
|
|
29
29
|
| `anbaric app tail [name]` | Stream the app's runtime logs. |
|
|
30
|
-
| `anbaric app tear-down [name]` | Stop and remove the app (prompts unless `--yes`). |
|
|
30
|
+
| `anbaric app tear-down [name]` | Stop and remove the app (prompts unless `--yes`). A running app is drained first. |
|
|
31
|
+
|
|
32
|
+
**Draining.** Replacing or removing an app never cuts a job off mid-step. The
|
|
33
|
+
running version is told to stop taking on new work — messages the platform
|
|
34
|
+
offers it are refused and wait for the new version — and given up to **five
|
|
35
|
+
minutes** to finish the steps it has in hand; `app deploy`, `app update` and
|
|
36
|
+
`app tear-down` show `draining (finishing n steps in flight)` while it does.
|
|
37
|
+
Only then is it stopped. A step still running at the deadline is stopped
|
|
38
|
+
regardless: once an app has taken a job on, what becomes of it is the app's.
|
|
31
39
|
|
|
32
40
|
The name is optional, because an `app` command is normally about the app you
|
|
33
41
|
are standing in: the CLI walks up from the working directory to the nearest
|
|
@@ -67,7 +75,7 @@ ask for, so the CLI runs unattended in scripts and CI.
|
|
|
67
75
|
| `--name <name>` | `app configure`/`deploy`/`update` | App name; prompts if omitted (suggested from `package.json`). |
|
|
68
76
|
| `--port <port>` | `app configure`/`deploy`/`update` | Internal port (1–65535); prompts if omitted. |
|
|
69
77
|
| `--yes` | `app deploy`, `app tear-down`, `jobs kill-old` | Skip confirmation. (`app update` implies it.) |
|
|
70
|
-
| `--state <state>`, `--status <status>`, `--app <app>` | `jobs list` | Only jobs in that state / with that status (`active`, `Awaiting input`, `Failed`) / belonging to that app. |
|
|
78
|
+
| `--state <state>`, `--status <status>`, `--app <app>` | `jobs list` | Only jobs in that state / with that status (`active`, `Awaiting input`, `Failed`, `Stalled`) / belonging to that app. |
|
|
71
79
|
| `--page <n>`, `--page-size <n>` | `jobs list` | Which page (from 0) and how many per page (default 100). |
|
|
72
80
|
| `--oldest` | `jobs list` | Oldest first instead of newest. |
|
|
73
81
|
| `--help`, `-h` | all | Print usage. |
|
|
@@ -156,6 +156,7 @@ readonly id : string
|
|
|
156
156
|
name : string
|
|
157
157
|
description : string
|
|
158
158
|
actor : Actor
|
|
159
|
+
longRunning : boolean = false
|
|
159
160
|
|
|
160
161
|
constructor(name : string, actor : Actor, description : string = "", id? : string)
|
|
161
162
|
|
|
@@ -177,6 +178,10 @@ sendWelcome.run = async (job) => new Map([["welcomeSent", true]]);
|
|
|
177
178
|
- **`run`** — return a `Map` of the properties to change. Only properties in the
|
|
178
179
|
machine's schema are applied; others are ignored with a warning. Only the
|
|
179
180
|
properties that actually changed are written back.
|
|
181
|
+
- **`longRunning`** — set it on a step that takes minutes rather than seconds.
|
|
182
|
+
Nothing constrains how long a step may take either way; what this changes is
|
|
183
|
+
that the job is [heartbeated](#stalled-jobs) while the step runs, so a step
|
|
184
|
+
that dies is visible rather than silent.
|
|
180
185
|
|
|
181
186
|
See [Actions and actors](../features/actions-and-actors.md).
|
|
182
187
|
|
|
@@ -269,22 +274,24 @@ One instance moving through a machine. You mostly **read** jobs (returned by
|
|
|
269
274
|
```ts
|
|
270
275
|
readonly id : string
|
|
271
276
|
readonly state : string
|
|
272
|
-
readonly properties :
|
|
277
|
+
readonly properties : JobProperties // read on demand; see below
|
|
273
278
|
readonly workflowId? : string // the machine's id …
|
|
274
279
|
readonly appId? : string // … and the app it runs in (its composite identity)
|
|
275
280
|
readonly startedAt : Date
|
|
276
281
|
readonly startedBy : string
|
|
277
282
|
readonly lastUpdated : Date
|
|
278
283
|
readonly killed : boolean
|
|
279
|
-
status : string // "active", "Awaiting input" or "
|
|
284
|
+
status : string // "active", "Awaiting input", "Failed" or "Stalled"
|
|
280
285
|
waitingFor? : string
|
|
281
286
|
awaitMetadata? : WaitForInput // present while parked on an Await
|
|
287
|
+
heartbeatAt? : Date // when a long-running step last said it was going
|
|
282
288
|
|
|
283
289
|
namespace Job {
|
|
284
290
|
const Status = {
|
|
285
291
|
ACTIVE: "active",
|
|
286
292
|
AWAITING_INPUT: "Awaiting input",
|
|
287
293
|
FAILED: "Failed",
|
|
294
|
+
STALLED: "Stalled",
|
|
288
295
|
} as const
|
|
289
296
|
}
|
|
290
297
|
```
|
|
@@ -296,6 +303,23 @@ A job whose action threw is `FAILED`, with the reason in its audit trail. It
|
|
|
296
303
|
stays in its state rather than transitioning, and an update that moves it on
|
|
297
304
|
returns it to `ACTIVE` — so a failure is recoverable, not terminal.
|
|
298
305
|
|
|
306
|
+
### Stalled jobs
|
|
307
|
+
|
|
308
|
+
While an action marked [`longRunning`](#action) runs, the machine heartbeats
|
|
309
|
+
its job once a minute. If those heartbeats stop for five minutes — the process
|
|
310
|
+
running the step died, in a crash or a restart — the job is marked `STALLED`.
|
|
311
|
+
|
|
312
|
+
That is all that happens. The platform does not re-run the step and does not
|
|
313
|
+
requeue the job: only the app knows what a half-finished long step already did,
|
|
314
|
+
so replaying it is not the platform's call. A stalled job is a job somebody
|
|
315
|
+
should look at — list them with `anbaric jobs list --status Stalled` — and
|
|
316
|
+
resume, by updating it or setting its state, if that is the right thing to do.
|
|
317
|
+
A heartbeat arriving from a job already marked stalled returns it to `ACTIVE`,
|
|
318
|
+
so a step that was merely slower than the sweep corrects itself.
|
|
319
|
+
|
|
320
|
+
Ordinary actions are not heartbeated: they are expected to be short, and the
|
|
321
|
+
[drain](cli.md#apps) on deploy is what protects them.
|
|
322
|
+
|
|
299
323
|
### `JobProperties`
|
|
300
324
|
|
|
301
325
|
`job.properties` is not a `Map`: it reads on demand, so a job may hold a great
|
|
@@ -339,7 +363,7 @@ type JobPersistence.Query = {
|
|
|
339
363
|
workflowId? : string,
|
|
340
364
|
appId? : string,
|
|
341
365
|
state? : string,
|
|
342
|
-
status? : string, // "active", "Awaiting input", "Failed"
|
|
366
|
+
status? : string, // "active", "Awaiting input", "Failed", "Stalled"
|
|
343
367
|
killed? : boolean,
|
|
344
368
|
order? : "oldest" | "newest" // by when the job was started; oldest by default
|
|
345
369
|
}
|
package/docs/api/web.md
CHANGED
|
@@ -109,7 +109,7 @@ job was started: `oldest` (the default, for compatibility) or `newest`.
|
|
|
109
109
|
The other parameters are filters, applied on the platform **before** paging, so
|
|
110
110
|
a filter sees every job and page numbers count matching jobs only. All are
|
|
111
111
|
optional and combine with AND: `workflowId` (the state machine), `appId`,
|
|
112
|
-
`state`, `status` (`active`, `Awaiting input`, `Failed`) and `killed`
|
|
112
|
+
`state`, `status` (`active`, `Awaiting input`, `Failed`, `Stalled`) and `killed`
|
|
113
113
|
(`true`/`false`).
|
|
114
114
|
|
|
115
115
|
In code the same listing is `JobPersistenceFactory.instance().list(actor,
|
|
@@ -41,6 +41,28 @@ the example above, and why a transition's guard may be too. Afterwards the
|
|
|
41
41
|
machine writes back only the properties that changed — never the ones the pass
|
|
42
42
|
read, let alone the ones it never loaded.
|
|
43
43
|
|
|
44
|
+
### Say when a step is long-running
|
|
45
|
+
|
|
46
|
+
Nothing limits how long an action may take — a mailbox import that runs for
|
|
47
|
+
twenty minutes is a perfectly ordinary action. But a step that long is worth
|
|
48
|
+
declaring:
|
|
49
|
+
|
|
50
|
+
```ts
|
|
51
|
+
importMailbox.longRunning = true;
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
While it runs, the machine tells the platform once a minute that the job is
|
|
55
|
+
still going. If that stops for five minutes — the process died, in a crash or
|
|
56
|
+
a restart — the job is marked **`Stalled`**. Nothing else happens: the platform
|
|
57
|
+
never re-runs a long step on its own, because only your app knows what a
|
|
58
|
+
half-finished one already did. A stalled job is one for a person to look at
|
|
59
|
+
(`anbaric jobs list --status Stalled`) and resume if it should be. A heartbeat
|
|
60
|
+
from a stalled job puts it back to `active`, so a step slower than the check
|
|
61
|
+
corrects itself.
|
|
62
|
+
|
|
63
|
+
Leave it off for ordinary steps: they are short, and a
|
|
64
|
+
[drain](../api/cli.md#apps) on deploy is what keeps those from being cut off.
|
|
65
|
+
|
|
44
66
|
### Prewarm what a state needs
|
|
45
67
|
|
|
46
68
|
Each property read that wasn't already held is a fetch. A state can name what
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "anbaric",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.60.0",
|
|
4
4
|
"description": "Everything needed to write an Anbaric app: state machines, jobs, document and secret stores, local in-memory implementations and the Anbaric Cloud clients",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"type": "module",
|
|
@@ -24,9 +24,9 @@
|
|
|
24
24
|
"prepublishOnly": "npm run build"
|
|
25
25
|
},
|
|
26
26
|
"dependencies": {
|
|
27
|
-
"anbaric-impl-cloud": "^1.
|
|
28
|
-
"anbaric-data-store": "^1.
|
|
29
|
-
"anbaric-state-machine": "^1.
|
|
30
|
-
"anbaric-tsapi": "^1.
|
|
27
|
+
"anbaric-impl-cloud": "^1.60.0",
|
|
28
|
+
"anbaric-data-store": "^1.60.0",
|
|
29
|
+
"anbaric-state-machine": "^1.60.0",
|
|
30
|
+
"anbaric-tsapi": "^1.60.0"
|
|
31
31
|
}
|
|
32
32
|
}
|