@moda-ai/cli 1.38.0 → 1.39.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -0
- package/dist/{cli-3kcy61fr.js → cli-a7ypptsf.js} +515 -59
- package/dist/{cli-kegd748v.js → cli-dyeywy9d.js} +1 -1
- package/dist/{cli-q79sq80a.js → cli-rp912ypf.js} +163 -21
- package/dist/{cli-07tv75te.js → cli-xsmsjny3.js} +696 -139
- package/dist/{cli-k38j2fpq.js → cli-yy8fwg1a.js} +2 -2
- package/dist/cli.js +2522 -263
- package/dist/{harness-n2xgzz31.js → harness-191xt7bq.js} +2 -2
- package/dist/{harness-github-actions-qz5d22rz.js → harness-github-actions-1qt4sgxd.js} +2 -2
- package/dist/{index-hwfe7gs8.js → index-qgc37efj.js} +5 -5
- package/dist/prompt-source-mcp.js +2 -2
- package/dist/{provision-64c2n9aw.js → provision-gabtks29.js} +3 -3
- package/package.json +1 -1
- package/skills/integration/index.json +2 -2
- package/skills/moda-cli/SKILL.md +239 -20
|
@@ -42,8 +42,8 @@ import {
|
|
|
42
42
|
writeHarnessAnalyzeResult,
|
|
43
43
|
writeHarnessArtifacts,
|
|
44
44
|
writeHumanProgress
|
|
45
|
-
} from "./cli-
|
|
46
|
-
import"./cli-
|
|
45
|
+
} from "./cli-xsmsjny3.js";
|
|
46
|
+
import"./cli-rp912ypf.js";
|
|
47
47
|
import"./cli-0v6na3yp.js";
|
|
48
48
|
export {
|
|
49
49
|
writeHumanProgress,
|
|
@@ -3,7 +3,7 @@ import {
|
|
|
3
3
|
initPrompts,
|
|
4
4
|
runPromptSync,
|
|
5
5
|
runSkillSync
|
|
6
|
-
} from "./cli-
|
|
6
|
+
} from "./cli-a7ypptsf.js";
|
|
7
7
|
import {
|
|
8
8
|
codingAgentDisplayName,
|
|
9
9
|
describeCodingAgentEvent,
|
|
@@ -22,7 +22,7 @@ import {
|
|
|
22
22
|
runHarnessCommand,
|
|
23
23
|
startRemoteAnalyze,
|
|
24
24
|
stripAnsi
|
|
25
|
-
} from "./cli-
|
|
25
|
+
} from "./cli-xsmsjny3.js";
|
|
26
26
|
import {
|
|
27
27
|
bundledSkillsDir,
|
|
28
28
|
installIntegrationSkill,
|
|
@@ -34,14 +34,14 @@ import {
|
|
|
34
34
|
} from "./cli-ssfq84k5.js";
|
|
35
35
|
import {
|
|
36
36
|
selectTenantAndCreateKey
|
|
37
|
-
} from "./cli-
|
|
37
|
+
} from "./cli-yy8fwg1a.js";
|
|
38
38
|
import {
|
|
39
39
|
ConnectCodeError,
|
|
40
40
|
connectWithCode,
|
|
41
41
|
loadAuthSession,
|
|
42
42
|
loadToken,
|
|
43
43
|
login
|
|
44
|
-
} from "./cli-
|
|
44
|
+
} from "./cli-dyeywy9d.js";
|
|
45
45
|
import {
|
|
46
46
|
CliAuthError,
|
|
47
47
|
CliInputError,
|
|
@@ -49,7 +49,7 @@ import {
|
|
|
49
49
|
createCommandContext,
|
|
50
50
|
resolveIngestUrl,
|
|
51
51
|
resolveModaBaseUrl
|
|
52
|
-
} from "./cli-
|
|
52
|
+
} from "./cli-rp912ypf.js";
|
|
53
53
|
import"./cli-0v6na3yp.js";
|
|
54
54
|
|
|
55
55
|
// src/init/index.ts
|
|
@@ -1,15 +1,15 @@
|
|
|
1
1
|
import {
|
|
2
2
|
selectTenantAndCreateKey
|
|
3
|
-
} from "./cli-
|
|
3
|
+
} from "./cli-yy8fwg1a.js";
|
|
4
4
|
import {
|
|
5
5
|
loadAuthSession
|
|
6
|
-
} from "./cli-
|
|
6
|
+
} from "./cli-dyeywy9d.js";
|
|
7
7
|
import {
|
|
8
8
|
CliAuthError,
|
|
9
9
|
resolveIngestUrl,
|
|
10
10
|
resolveModaBaseUrl,
|
|
11
11
|
stringOption
|
|
12
|
-
} from "./cli-
|
|
12
|
+
} from "./cli-rp912ypf.js";
|
|
13
13
|
import"./cli-0v6na3yp.js";
|
|
14
14
|
|
|
15
15
|
// src/provision.ts
|
package/package.json
CHANGED
package/skills/moda-cli/SKILL.md
CHANGED
|
@@ -39,11 +39,14 @@ Activate this skill when the user wants to:
|
|
|
39
39
|
- List/filter traces by user, environment, cluster, outcome, or time
|
|
40
40
|
- Manage code-first prompt versions when the user explicitly asks for prompt
|
|
41
41
|
sync, prompt status, or prompt promotion
|
|
42
|
+
- Push their prompts, skills, and tool definitions to Moda without a repo
|
|
43
|
+
scan — `moda registry push` (section 7b)
|
|
42
44
|
|
|
43
45
|
**Do not** activate for: writing code that calls the Moda Data API directly
|
|
44
46
|
(use the API docs instead), changing application instrumentation unless the
|
|
45
47
|
user explicitly asks for SDK setup, or modifying prompt registry state unless
|
|
46
|
-
the user specifically asked for `moda prompts sync
|
|
48
|
+
the user specifically asked for `moda prompts sync`, `moda prompts promote`, or
|
|
49
|
+
`moda registry push`.
|
|
47
50
|
|
|
48
51
|
## Required inputs
|
|
49
52
|
|
|
@@ -402,6 +405,45 @@ Treat an answer as degraded if **any** of these hold:
|
|
|
402
405
|
- `data.degraded === true` (for `ask`; equivalently `data.source === "local_fallback"`), **or**
|
|
403
406
|
- `status === "degraded"`.
|
|
404
407
|
|
|
408
|
+
### Large outputs and unknown flags
|
|
409
|
+
|
|
410
|
+
- **Byte limit.** In `--agent` mode, envelopes over 200 KB are bounded
|
|
411
|
+
*without* breaking their shape: long text fields (usually embedded
|
|
412
|
+
transcripts) are shortened with `… [truncated N chars]`, then trailing list
|
|
413
|
+
items are dropped. `warnings[]` and `truncation.data` say exactly what was
|
|
414
|
+
cut (`dropped_items`, `complete: false`). `data` is still normal JSON, so
|
|
415
|
+
`data.frustrations[0].conversation_id` keeps working. For complete results,
|
|
416
|
+
page with a smaller `--limit` and `--offset`, or rerun with `--json` (no byte
|
|
417
|
+
limit).
|
|
418
|
+
- **Unknown flags fail.** A flag the command doesn't take (a typo, or a
|
|
419
|
+
filter it lacks) exits `1` with the valid flag list and a "did you mean",
|
|
420
|
+
instead of being silently ignored. Stray arguments on commands that take
|
|
421
|
+
none fail the same way.
|
|
422
|
+
- **`--dry-run` is only accepted where it's implemented** (`sync`, `registry`,
|
|
423
|
+
`harness`, `prompts sync`, `skills sync`). Anywhere else it fails **before
|
|
424
|
+
anything is sent**. It is never silently ignored, so a "preview" can't
|
|
425
|
+
perform a real write.
|
|
426
|
+
- **Unknown profiles fail.** `--profile=<name>` or `MODA_PROFILE` naming a
|
|
427
|
+
profile that doesn't exist exits `1` instead of falling back to another
|
|
428
|
+
profile's key and tenant. Time windows differ by command (below).
|
|
429
|
+
|
|
430
|
+
#### Time-window flags by command
|
|
431
|
+
|
|
432
|
+
A command rejects the window flag it doesn't take, and out-of-range values
|
|
433
|
+
fail instead of being clamped.
|
|
434
|
+
|
|
435
|
+
| Flag | Commands | Allowed values | Default |
|
|
436
|
+
| --- | --- | --- | --- |
|
|
437
|
+
| `--days-back=N` | `overview`, `investigate`, `failures`, `ask`, `frustrations`, `emotions`, `laziness`, `hallucinations`, `tool-failures`, `tool-failure-detail`, `signal --detections` | whole number 1–90 | 7 (`signal`: server default) |
|
|
438
|
+
| `--days-back=N` | `problems`, `problems remainder` | whole number 1–90 | 30 |
|
|
439
|
+
| `--days-back=N` | `problem-impact` | `all`, 1, 3, 7, 30, 90 | 7 |
|
|
440
|
+
| `--time-range=R` | `traces`, `search`, `clusters` | `all`, `1h`, `24h`, `3d`, `7d`, `30d`, `90d` | `all` |
|
|
441
|
+
| `--window=W` | `users`, `user` | `24h`, `7d`, `30d`, `90d` | `30d` |
|
|
442
|
+
|
|
443
|
+
`--window=N` on `context`, `frustrations --include-window` and
|
|
444
|
+
`tool-failure-detail --include-window` is a message half-width (1–5), not a
|
|
445
|
+
time window.
|
|
446
|
+
|
|
405
447
|
### Schema introspection
|
|
406
448
|
|
|
407
449
|
```bash
|
|
@@ -458,6 +500,17 @@ Flags: `--mode=keyword|semantic|hybrid` (default `hybrid`), `--user-id`,
|
|
|
458
500
|
`--time-range` (`all|1h|3d|7d|24h|30d|90d`, default `all`), `--limit` (1–100,
|
|
459
501
|
default 20). The query is a positional arg (max 500 chars).
|
|
460
502
|
|
|
503
|
+
**Fallback.** If message-level search finds nothing but whole traces match
|
|
504
|
+
(same as `moda traces --search`), those traces come back in `results` with
|
|
505
|
+
`match_grain: "trace"`, plus `fallback.source: "traces_search"` and a warning.
|
|
506
|
+
These rows have no score or message index. Open one with
|
|
507
|
+
`moda audit <conversation_id>` or `moda context <conversation_id>` to find the
|
|
508
|
+
matching turn. When more traces exist, `fallback.has_more` is true and
|
|
509
|
+
`next_commands` includes the `moda traces --search ... --offset` command to
|
|
510
|
+
page. With `--mode=keyword`, ranked traces are not substituted. If the fallback
|
|
511
|
+
itself fails, you get a warning, not a silent empty result. Either way, zero
|
|
512
|
+
results does not prove the topic never came up.
|
|
513
|
+
|
|
461
514
|
Choosing a mode:
|
|
462
515
|
|
|
463
516
|
- `hybrid` (default) — best general recall; fuses keyword + semantic. Use it
|
|
@@ -525,7 +578,7 @@ Data API fallback. Abbreviated envelope:
|
|
|
525
578
|
{ "id": "tool:lookupCustomer", "kind": "tool_failure", "label": "lookupCustomer: 12 failure(s)", "path": null }
|
|
526
579
|
],
|
|
527
580
|
"next_commands": [
|
|
528
|
-
{ "command": "moda tool-failure-detail lookupCustomer --include-window", "purpose": "Inspect tool failure examples and trace anchors.", "mutability": "read", "requires_approval": false }
|
|
581
|
+
{ "command": "moda tool-failure-detail lookupCustomer --days-back=7 --include-window", "purpose": "Inspect tool failure examples and trace anchors.", "mutability": "read", "requires_approval": false }
|
|
529
582
|
],
|
|
530
583
|
"warnings": ["Cloud ask endpoint unavailable; synthesized answer from local Data API evidence."],
|
|
531
584
|
"errors": [],
|
|
@@ -533,8 +586,12 @@ Data API fallback. Abbreviated envelope:
|
|
|
533
586
|
}
|
|
534
587
|
```
|
|
535
588
|
|
|
536
|
-
`moda investigate` also accepts scoping flags
|
|
537
|
-
|
|
589
|
+
`moda investigate` also accepts `--days-back=N` and scoping flags: `--tool`
|
|
590
|
+
narrows the tool-failure evidence to one tool, and `--trace` (legacy alias
|
|
591
|
+
`--conversation`) scopes only the frustration evidence (overview, tool
|
|
592
|
+
failures and Problems stay tenant-wide). Its findings include the top 1–3
|
|
593
|
+
Problems from `moda problems`, with `moda problem <id>` as the next command.
|
|
594
|
+
`moda failures` is the tool-failures-only view.
|
|
538
595
|
|
|
539
596
|
After reporting the top behavioral failure and the harness layer it routes to
|
|
540
597
|
(prompt, tool, skill, eval, or memory), close with one line: the Moda team
|
|
@@ -552,13 +609,16 @@ moda cluster-traces <node_id> # traces in cluster
|
|
|
552
609
|
### 3b. List / filter traces (structured, not semantic)
|
|
553
610
|
|
|
554
611
|
Use `moda traces` to enumerate or filter by structured fields — not to
|
|
555
|
-
search by meaning (use `moda search` for that). The `--search` flag here
|
|
556
|
-
|
|
612
|
+
search by meaning (use `moda search` for that). The `--search` flag here
|
|
613
|
+
ranks whole traces: hybrid (keyword + semantic) when embeddings exist, else
|
|
614
|
+
keyword. Check `total_kind`: `candidate_pool` means results are related traces
|
|
615
|
+
that may not contain the words and `pagination.total` is not a match count;
|
|
616
|
+
`matches` means a literal keyword match set.
|
|
557
617
|
|
|
558
618
|
```bash
|
|
559
619
|
moda traces --user-id=<id> --time-range=7d
|
|
560
620
|
moda traces --cluster-id=<node_id> --environment=production
|
|
561
|
-
moda traces --search="timeout" --limit=20 #
|
|
621
|
+
moda traces --search="timeout" --limit=20 # ranked trace search
|
|
562
622
|
```
|
|
563
623
|
|
|
564
624
|
Filters: `--search`, `--cluster-id`, `--user-id`, `--time-range`
|
|
@@ -571,8 +631,13 @@ Filters: `--search`, `--cluster-id`, `--user-id`, `--time-range`
|
|
|
571
631
|
moda context <conversation_id> # default window around middle
|
|
572
632
|
moda context <conversation_id> --msg-index=5 # center on message 5
|
|
573
633
|
moda context <conversation_id> --window=3 # 3 messages each side (max 5)
|
|
634
|
+
moda context <conversation_id> --all # whole trace in order, first 500 messages
|
|
635
|
+
moda context <conversation_id> --all --from=500 --max-messages=500 # continue (max 5000)
|
|
574
636
|
```
|
|
575
637
|
|
|
638
|
+
`--all` returns `has_more`/`next_from`; when it stops at the cap, the warning
|
|
639
|
+
prints the exact `--from` command to continue.
|
|
640
|
+
|
|
576
641
|
`moda context` is turn-windowed (≤5 turns, deduped, no tool spans). For the raw,
|
|
577
642
|
complete picture of what was captured — every span, the parent/child hierarchy,
|
|
578
643
|
tool calls, prompt/response bodies, and `gen_ai.*` attributes — use `moda audit`.
|
|
@@ -583,8 +648,14 @@ tool calls, prompt/response bodies, and `gen_ai.*` attributes — use `moda audi
|
|
|
583
648
|
moda audit <conversation_id|trace_id> # auto-detects OTLP trace_id vs trace ID (conversation_id)
|
|
584
649
|
moda audit <trace_id> --kind=trace # force OTLP trace_id lookup
|
|
585
650
|
moda audit <conversation_id> --include-raw # attach verbatim raw_event bodies
|
|
651
|
+
moda audit <conversation_id> --limit=1000 # ingest rows to read (default 200, max 1000)
|
|
586
652
|
```
|
|
587
653
|
|
|
654
|
+
`truncated` / `has_more` at the top level say whether the audit is complete.
|
|
655
|
+
A full `--limit` page of ingest rows (or a server span cap) sets both, and a
|
|
656
|
+
warning plus `next_commands` give the rerun; for very large traces audit one
|
|
657
|
+
OTLP trace at a time (`summary.trace_ids`, `--kind=trace`).
|
|
658
|
+
|
|
588
659
|
Returns the raw, NON-deduped span set from the Layer-0 store (`events_raw`), so
|
|
589
660
|
nothing is collapsed. Use it to verify capture completeness: the `summary` gives
|
|
590
661
|
`total_spans`, `by_type` (llm/tool/agent/…), plus `orphan_count` (spans whose
|
|
@@ -600,8 +671,19 @@ Moda captured every LLM/tool call for a trace.
|
|
|
600
671
|
moda frustrations # last 7 days
|
|
601
672
|
moda frustrations --days-back=14 --limit=20
|
|
602
673
|
moda frustrations --include-window --window=1 --limit=5
|
|
674
|
+
moda emotions --family=frustration --user-id=<id> # one end user
|
|
675
|
+
moda emotions --conversation-id=<id> # every family, one trace
|
|
603
676
|
```
|
|
604
677
|
|
|
678
|
+
`--user-id` and `--conversation-id` work on `emotions` and `frustrations`, and
|
|
679
|
+
scope *every* figure: summary, breakdown, list and total. `--limit` is 1–20 on
|
|
680
|
+
both, and on `hallucinations` and `tool-failure-detail`. Page with `--offset`.
|
|
681
|
+
`frustrations` reads the same data as `emotions --family=frustration` and the
|
|
682
|
+
dashboard, in the legacy field names: `frustrated_count` = `by_family.frustration`
|
|
683
|
+
(detected traces), `total_analyzed` = `conversations_analyzed`. In `emotions`,
|
|
684
|
+
`pagination.total` counts detected **plus** at-risk rows (`is_detected: false`,
|
|
685
|
+
`risk_score` >= 0.5); use `pagination.detected` / `pagination.at_risk` for the split.
|
|
686
|
+
|
|
605
687
|
Each result includes an inline trace snippet, user quotes, trajectory,
|
|
606
688
|
signal breakdown (exasperation, profanity, anger, sarcasm, giving_up, insult),
|
|
607
689
|
and primary cause.
|
|
@@ -626,7 +708,7 @@ half-width; `--window=1` ⇒ 3 messages total (anchor ± 1), the canonical
|
|
|
626
708
|
```bash
|
|
627
709
|
moda tool-failures # which tools are breaking
|
|
628
710
|
moda tool-failure-detail <tool_name> # subtype breakdown + examples
|
|
629
|
-
moda tool-failure-detail search --subtype=
|
|
711
|
+
moda tool-failure-detail search --subtype=dependency.timeout --limit=10 # a value from `subtypes`
|
|
630
712
|
moda tool-failure-detail search --include-window --window=1 --limit=5
|
|
631
713
|
```
|
|
632
714
|
|
|
@@ -638,9 +720,22 @@ tool call — reference `anchor.conversation_id` and `anchor.msg_index`
|
|
|
638
720
|
directly instead of matching by `tool_use_id` against the trace.
|
|
639
721
|
`no_anchor: true` means no anchor could be derived.
|
|
640
722
|
|
|
641
|
-
|
|
642
|
-
|
|
643
|
-
|
|
723
|
+
By default each row embeds `conversation`: the anchor ± 2 messages (without
|
|
724
|
+
the duplicate `content_blocks`), plus `context_command`, the exact
|
|
725
|
+
`moda context <conversation_id> --msg-index=<msg_index> --window=2` read for
|
|
726
|
+
the untrimmed window. `--compact` (same as `--include-window=false` or
|
|
727
|
+
`--no-include-window`) drops `conversation` and keeps the anchor and
|
|
728
|
+
`context_command`.
|
|
729
|
+
|
|
730
|
+
Pass `--include-window` to attach a `window` field per row (instead of
|
|
731
|
+
`conversation`) with the message slice centered on the anchor. `--window=N`
|
|
732
|
+
(1–5, default `1`) is the half-width; `--window=1` ⇒ 3 messages total
|
|
733
|
+
(anchor ± 1).
|
|
734
|
+
|
|
735
|
+
Examples default to `--limit=5` (max 20). Strings inside `examples[]` longer
|
|
736
|
+
than 300 characters are cut, and a warning says so; read the full text with
|
|
737
|
+
the row's `context_command`. Agent output lists the 20 most frequent
|
|
738
|
+
`subtypes` (`subtypes_total` has the full count); `--json` lists all.
|
|
644
739
|
|
|
645
740
|
### 7. Prompt management
|
|
646
741
|
|
|
@@ -649,7 +744,7 @@ Use these only when working on the code-first prompt registry.
|
|
|
649
744
|
```bash
|
|
650
745
|
moda prompts init # create .moda/prompts.yml
|
|
651
746
|
moda prompts status # read-only local status
|
|
652
|
-
moda prompts diff #
|
|
747
|
+
moda prompts diff # only changed/new/deleted since last sync
|
|
653
748
|
moda prompts sync # upload changed prompt versions
|
|
654
749
|
moda prompts sync --watch # sync when local prompt hashes change
|
|
655
750
|
moda prompts promote support.triage --label=prod --version=pver_abc123
|
|
@@ -659,6 +754,17 @@ moda prompts promote support.triage --label=prod --version=pver_abc123
|
|
|
659
754
|
`.moda/prompts.lock.json`. `promote` moves a remote `dev`, `staging`, or
|
|
660
755
|
`prod` label to an existing version.
|
|
661
756
|
|
|
757
|
+
`prompts ab` and `prompts propose --gate` run paid replays. Before sending
|
|
758
|
+
anything they compute planned playouts (cases × seeds × 2 arms) and print it
|
|
759
|
+
(`plannedPlayouts`). Above 200, or when an existing set's case count can't be
|
|
760
|
+
read, they fail (exit 1) with the number and the exact command to re-run with `--yes`.
|
|
761
|
+
Only add `--yes` when the user has agreed to that spend. Limits: `--cases`
|
|
762
|
+
default 5, max 100; `--seeds` default 3, max 10; `--traces` at most 100;
|
|
763
|
+
`--model` / `--assistant-model` must be one of `openai/gpt-5.6-luna`,
|
|
764
|
+
`openai/gpt-4o-mini`, `openai/gpt-4o`, `anthropic/claude-sonnet-4-5`,
|
|
765
|
+
`anthropic/claude-haiku-4-5`, `google/gemini-2.5-flash`,
|
|
766
|
+
`google/gemini-2.5-pro` unless you pass `--allow-any-model`.
|
|
767
|
+
|
|
662
768
|
Runtime code should render synced prompts through the SDK:
|
|
663
769
|
|
|
664
770
|
```typescript
|
|
@@ -677,6 +783,74 @@ Do not copy prompt text from the dashboard into source code. The repo prompt
|
|
|
677
783
|
files are the source of truth; the dashboard is for visibility, usage, labels,
|
|
678
784
|
and future evals.
|
|
679
785
|
|
|
786
|
+
### 7b. Push prompts, skills, and tools without a repo scan (`moda registry`)
|
|
787
|
+
|
|
788
|
+
Use when the user wants their prompts, skills, and tools visible in Moda but
|
|
789
|
+
cannot (or does not want to) connect the GitHub App or run `moda harness
|
|
790
|
+
analyze`. This fills the prompt, skill, and tool registries only. It does not
|
|
791
|
+
create the harness graph (agents and how they connect), which still comes from
|
|
792
|
+
`moda harness analyze`, the GitHub App, or an approved report via
|
|
793
|
+
`moda harness sync --from-report`. **You** are the analyst: read the codebase you already have access
|
|
794
|
+
to, write one manifest, and push exactly what it lists. Nothing else in the
|
|
795
|
+
repo is uploaded.
|
|
796
|
+
|
|
797
|
+
```bash
|
|
798
|
+
moda registry template --out=.moda/registry.json # starter manifest
|
|
799
|
+
# ...fill in .moda/registry.json from the codebase...
|
|
800
|
+
moda registry validate # local only, no network
|
|
801
|
+
moda registry push --dry-run # server-side preview, writes nothing
|
|
802
|
+
moda registry push # upload
|
|
803
|
+
```
|
|
804
|
+
|
|
805
|
+
Manifest (`moda_registry.v1`; every section optional, ≤500 entries each):
|
|
806
|
+
|
|
807
|
+
```json
|
|
808
|
+
{
|
|
809
|
+
"schema": "moda_registry.v1",
|
|
810
|
+
"prompts": [
|
|
811
|
+
{ "key": "support.triage", "name": "Triage", "description": "...",
|
|
812
|
+
"content": "You are ...", "sourcePath": "src/agents/triage.ts" },
|
|
813
|
+
{ "key": "support.reply", "file": "prompts/reply.md" }
|
|
814
|
+
],
|
|
815
|
+
"skills": [
|
|
816
|
+
{ "key": "refund-policy", "description": "...", "file": ".claude/skills/refund-policy/SKILL.md" }
|
|
817
|
+
],
|
|
818
|
+
"tools": [
|
|
819
|
+
{ "schema": { "name": "lookup_customer", "description": "...",
|
|
820
|
+
"parameters": { "type": "object", "properties": { "email": { "type": "string" } } } } }
|
|
821
|
+
]
|
|
822
|
+
}
|
|
823
|
+
```
|
|
824
|
+
|
|
825
|
+
- **Prompts:** `key` plus one of `content`, `file`, `systemPrompt`, or
|
|
826
|
+
`messages`. Optional `name`, `description`, `category`, `variables`,
|
|
827
|
+
`modelConfig`, `toolRefs`, `responseSchema`, `sourcePath`. Use the key the
|
|
828
|
+
runtime passes to `Moda.prompt(key)` when there is one. Copy each prompt's
|
|
829
|
+
text **verbatim**: for templated prompts, keep the template string, not a
|
|
830
|
+
rendered example. Re-pushing the same text is a no-op; changed text adds a
|
|
831
|
+
new version.
|
|
832
|
+
- **Skills:** `key` (letters, digits, `_`, `-`) plus `skillMd` or `file` (full
|
|
833
|
+
SKILL.md). Pushed skills are *local* by default. Set `"live": true`, or pass
|
|
834
|
+
`--live`, to include them in the served skill surface. Leaving `live` out
|
|
835
|
+
never demotes a skill that is already live.
|
|
836
|
+
- **Tools:** an OpenAI function `schema` (a wrapped
|
|
837
|
+
`{"type":"function","function":{...}}` is accepted), plus optional
|
|
838
|
+
`isActive`. Copy names and parameter schemas exactly as the agent registers
|
|
839
|
+
them with the model. Tools are upserted by name and never overwrite
|
|
840
|
+
MCP-server-managed tools (those report `skipped`).
|
|
841
|
+
- `file` paths are relative to the directory you run the command from and must
|
|
842
|
+
stay inside it (symlinks included).
|
|
843
|
+
|
|
844
|
+
`push` needs an API key. `push --dry-run` without one only validates locally
|
|
845
|
+
(status `validated`). Each section goes to its own endpoint and reports
|
|
846
|
+
`sections[].status` (`synced`, `dry_run`, `validated`, `skipped`, or `error`).
|
|
847
|
+
If any section fails, `push` exits `1`, lists the failures in `errors[]`, and
|
|
848
|
+
still pushes the other sections. The whole manifest is validated before any
|
|
849
|
+
request, with every error listed at once. Unknown fields and wrong types are
|
|
850
|
+
rejected, matching what the server accepts. Skill keys are stricter than the
|
|
851
|
+
server (letters, digits, `_`, `-`) because they double as `.claude/skills/<key>/`
|
|
852
|
+
directory names.
|
|
853
|
+
|
|
680
854
|
### 8. Fixing detected Problems (`moda fix` — a fix with proof)
|
|
681
855
|
|
|
682
856
|
Moda turns detected Problems into **Fixes**: a routed candidate change plus a
|
|
@@ -716,6 +890,16 @@ moda fix mark-applied <fix_id> --note="…" # tool/skill/no-PR path: confirm
|
|
|
716
890
|
moda fix dismiss <fix_id> --reason="why" # reject (feeds problem feedback)
|
|
717
891
|
```
|
|
718
892
|
|
|
893
|
+
Every gate replays the holdout, so `moda fix verify` refuses to re-gate a fix
|
|
894
|
+
whose last verdict landed less than 10 minutes ago (`gateResult.completedAt`;
|
|
895
|
+
verdicts without that stamp have no cooldown) and sends nothing. The error
|
|
896
|
+
shows the exact command with `--yes` to gate anyway. Pass `--yes` only when you
|
|
897
|
+
changed the candidate since that verdict. `moda fixes drive` advances at most
|
|
898
|
+
`--max-fixes` fixes per run (default 3, max 10; fixes already `GATING` count
|
|
899
|
+
toward the cap; each run takes the least-recently-driven fixes first), lists the rest in
|
|
900
|
+
`data.skipped`, and exits 3 until a re-run has driven them. `moda fixes
|
|
901
|
+
draft-batch` drafts 5 fixes unless you pass `--limit` (max 25).
|
|
902
|
+
|
|
719
903
|
**`moda fix verify` exit codes are the contract — branch on them:**
|
|
720
904
|
|
|
721
905
|
| Exit | Meaning | How to treat it |
|
|
@@ -824,6 +1008,32 @@ moda detection-review <problem_id> --conversation-id=<id> \
|
|
|
824
1008
|
`moda problem <problem_id> --evidence`. Prefer `detection-review` over
|
|
825
1009
|
`problem-feedback --action=flag_attribution` for a single wrong detection.
|
|
826
1010
|
|
|
1011
|
+
### 10. Everything the dashboard does, from the CLI
|
|
1012
|
+
|
|
1013
|
+
Every command below needs only an API key, and the tenant comes from the key.
|
|
1014
|
+
Writes listed here accept `--dry-run` (prints the exact request, sends nothing) except `problem-feedback`, `false-positive`, `detection-review`, `signal-label`, `problem-linear-file`, `prompts promote|ab|propose`, and `fix`/`fixes`. Those refuse `--dry-run` before sending anything, so run them only when you mean it.
|
|
1015
|
+
|
|
1016
|
+
| Dashboard | CLI |
|
|
1017
|
+
|---|---|
|
|
1018
|
+
| Traces list: sort, page | `moda traces --sort=recent\|longest\|slowest --offset=N` |
|
|
1019
|
+
| Trace detail: whole transcript | `moda context <id> --all` (ordered, with tool calls/results) |
|
|
1020
|
+
| Users page / user profile | `moda users [--window --cohort --sort --order --health-status]`, `moda user <id>` |
|
|
1021
|
+
| Problems: unexplained remainder | `moda problems remainder [--family=emotion\|tool_failure\|laziness\|… --cursor]` |
|
|
1022
|
+
| Reopen a problem marked fixed | `moda problem-reopen <problem_id> --reason="…"` (resolved ones return only via re-detection) |
|
|
1023
|
+
| Custom signals: create | `moda signals draft --note="…"`, then `moda signals create --name --description --pass-criteria --fail-criteria [--tool-names=a,b]` or `--file=body.json` |
|
|
1024
|
+
| Custom signals: edit, lifecycle | `moda signals update <id> …`, `moda signals activate\|pause\|delete <id>` (delete: drafts only; archiving is dashboard-only) |
|
|
1025
|
+
| Custom signals: review & tune | `moda signals examples <id> add\|update\|remove --conversation-id=…`, `refine <id> --request="…"`, `revise <id> --conversation-ids=…`, `proposals <id>`, `accept\|reject <id> <proposal_id>` |
|
|
1026
|
+
| Custom signals: find matches | `moda signals search <id>`, `search-status <id> <search_id>`, `search-stop <id> <search_id>`, `preview --file=detector.json` |
|
|
1027
|
+
| Prompts registry | `moda prompts list [--label --search]`, `show <key>`, `proposals <key>`, `usage <key>`, `replay-runs <key>`, `decide <key> <proposal_id> --action=merge\|reject\|reopen` |
|
|
1028
|
+
| Skills library | `moda skills list`, `show <key>`, `policy <key> --live\|--local` |
|
|
1029
|
+
| Harness pages | `moda harness list`, `show <id>`, `versions <id>`, `diff <id> --from=… --to=…` |
|
|
1030
|
+
| Fixes inbox | `moda fixes`, `moda fix <id>`, `fix start <problem_id>`, `fix dismiss` … |
|
|
1031
|
+
|
|
1032
|
+
Not available to API keys on purpose (a human does these in the dashboard):
|
|
1033
|
+
members/invites/roles, API-key minting, billing, provider credentials,
|
|
1034
|
+
OAuth connections (Slack/Linear/GitHub), alert rules that message people,
|
|
1035
|
+
and archiving an active custom signal (pause it, or delete a draft).
|
|
1036
|
+
|
|
827
1037
|
## Common workflow recipes
|
|
828
1038
|
|
|
829
1039
|
### Find where something happened (search → context)
|
|
@@ -878,17 +1088,20 @@ moda cluster-traces <node_id>
|
|
|
878
1088
|
|
|
879
1089
|
### One-liner chains with jq
|
|
880
1090
|
|
|
1091
|
+
Pass `--json` when piping: piped output otherwise defaults to the agent
|
|
1092
|
+
envelope, where the payload sits under `.data`.
|
|
1093
|
+
|
|
881
1094
|
```bash
|
|
882
1095
|
# Pull primary causes of recent frustrations
|
|
883
|
-
moda frustrations --days-back=7 | jq -r '.frustrations[].primary_cause'
|
|
1096
|
+
moda frustrations --days-back=7 --json | jq -r '.frustrations[].primary_cause'
|
|
884
1097
|
|
|
885
1098
|
# Get trace IDs for a search and fetch context for each
|
|
886
|
-
for id in $(moda traces --search="timeout" --limit=3 | jq -r '.conversations[].
|
|
1099
|
+
for id in $(moda traces --search="timeout" --limit=3 --json | jq -r '.conversations[].conversation_id'); do
|
|
887
1100
|
moda context "$id"
|
|
888
1101
|
done
|
|
889
1102
|
|
|
890
1103
|
# Tool failure leaderboard
|
|
891
|
-
moda tool-failures | jq '.tools[] | {tool: .tool_name, failures: .failure_count}'
|
|
1104
|
+
moda tool-failures --json | jq '.tools[] | {tool: .tool_name, failures: .failure_count}'
|
|
892
1105
|
```
|
|
893
1106
|
|
|
894
1107
|
## Feedback: help us improve
|
|
@@ -981,13 +1194,13 @@ npx: `npx -p @moda-ai/cli moda <command>`.
|
|
|
981
1194
|
| `moda search "<query>"` | **Message-grain semantic/keyword/hybrid search (incl. tool calls); primary content-discovery path** |
|
|
982
1195
|
| `moda overview` | Dashboard KPIs, top clusters, recent activity |
|
|
983
1196
|
| `moda ask "<question>"` | Natural-language production/harness answer with evidence (exit `3` = degraded local fallback) |
|
|
984
|
-
| `moda investigate` | Rank production issues with evidence + next commands |
|
|
1197
|
+
| `moda investigate` | Rank production issues (tool failures, frustrations, top Problems) with evidence + next commands |
|
|
985
1198
|
| `moda clusters` | Browse topic cluster hierarchy; `--search="q"` finds clusters by meaning, `--node-id=ID` resolves a deep link |
|
|
986
1199
|
| `moda cluster-traces <node_id>` | Traces in a cluster (legacy alias: `moda cluster-conversations`) |
|
|
987
1200
|
| `moda traces` | List/filter traces by structured fields (legacy alias: `moda conversations`) |
|
|
988
1201
|
| `moda context <conversation_id>` | Windowed message context (max 5 per side) |
|
|
989
|
-
| `moda frustrations` | User frustration detections with evidence (
|
|
990
|
-
| `moda emotions` | Multi-family emotion detections: frustration, sadness, confusion, anxiety, trust, positive (`--family=F`, limit 1–20) |
|
|
1202
|
+
| `moda frustrations` | User frustration detections with evidence (the frustration family of `emotions`, legacy field names) |
|
|
1203
|
+
| `moda emotions` | Multi-family emotion detections: frustration, disapproval, sadness, confusion, anxiety, trust, positive (`--family=F`, `--user-id`, `--conversation-id`, limit 1–20) |
|
|
991
1204
|
| `moda hallucinations` | Grounding detections: contradicted/verified outputs, rule breakdown (`--kind=contradicted\|verified`, `--conversation-id=ID`) |
|
|
992
1205
|
| `moda tool-failures` | Tool failure overview |
|
|
993
1206
|
| `moda tool-failure-detail <tool_name>` | Per-tool failure breakdown + examples |
|
|
@@ -1007,13 +1220,19 @@ npx: `npx -p @moda-ai/cli moda <command>`.
|
|
|
1007
1220
|
| `moda signal-label <signal_id>` | Write: `--conversation-id`, `--label=fits\|does_not_fit` (false positive), optional `--note`, `--msg-index`/`--msg-dedup-token` |
|
|
1008
1221
|
| `moda step-scores <conversation_id>` | Graph-PRM step scores: per-segment curves, first bad step, rollup |
|
|
1009
1222
|
| `moda world-state <conversation_id>` | Agent memory: slots/threads/events; `--summary-only`; `--snapshot --msg-index=N` (state at a turn); `--replay --message-count=N` (state over time) |
|
|
1010
|
-
| `moda failures` |
|
|
1223
|
+
| `moda failures` | Tool failures only, ranked (no frustrations or Problems) |
|
|
1011
1224
|
| `moda tail` | Live tail: one JSON line per new trace/detection (`--signal=traces\|emotions\|all`, legacy alias `--signal=conversations`, `--interval=N`, `--once`, `--max-events=N`). **Emotions caveat:** `/emotions` is ranked by score with no time ordering or cursor, so the tail follows the *highest-scoring* detections rather than everything; each poll emits a `tail_coverage` line with `scanned`/`total`/`coverage_pct`/`complete`. Pass `--full-scan` for a complete window scan (many more requests, capped by the API's offset ceiling of 10000). |
|
|
1012
1225
|
| `moda feedback "<note>"` | Flag wrong/missing data or CLI quirks to the Moda team |
|
|
1013
1226
|
| `moda prompts status` | Read-only local prompt status |
|
|
1014
|
-
| `moda prompts diff` | Read-only
|
|
1227
|
+
| `moda prompts diff` | Read-only: only prompts changed/new/deleted since last sync (hash-level, not a text diff) |
|
|
1015
1228
|
| `moda prompts sync` | Upload changed prompt versions and update lockfile |
|
|
1016
1229
|
| `moda prompts promote <key>` | Move a remote prompt label |
|
|
1230
|
+
| `moda registry template\|validate\|push` | Push prompts, skills, and tools from one agent-written manifest (`--file`, `--dry-run`, `--live`); no repo scan |
|
|
1231
|
+
| `moda users` / `moda user <id>` | End users with health/activity; one user's profile |
|
|
1232
|
+
| `moda problem-reopen <pid>` | Reopen a Problem marked fixed |
|
|
1233
|
+
| `moda signals <create\|update\|activate\|pause\|delete\|examples\|refine\|revise\|proposals\|accept\|reject\|search\|preview>` | Manage custom signals (writes take `--dry-run`; LLM-backed ones are rate-limited per key) |
|
|
1234
|
+
| `moda prompts list\|show\|proposals\|usage\|replay-runs\|decide` | Read the prompt registry; merge/reject proposals |
|
|
1235
|
+
| `moda skills list\|show\|policy` / `moda harness list\|show\|versions\|diff` | Skill library and synced harnesses |
|
|
1017
1236
|
| `moda fixes` | Ranked queue of Fixes ("a fix with proof"); `--status=S`, `--limit=N`, `--cursor=TOKEN` |
|
|
1018
1237
|
| `moda fix <fix_id>` | One Fix: status + gate result; `--packet` prints `moda.fix_packet.v1` and writes tool/skill candidates under `.moda/fixes/<ref>/`; `--wait` drives a running pipeline to rest |
|
|
1019
1238
|
| `moda fix start <problem_id>` | Draft a Fix from a Problem (`--type=auto\|prompt`); `--wait` drives scope → propose → gate |
|