lazycodex-ai 5.0.0-beta.6 → 5.0.0-beta.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +12 -1
- package/dist/cli/index.js +274 -190
- package/dist/cli-node/index.js +274 -190
- package/package.json +1 -1
- package/packages/omo-codex/plugin/.codex-plugin/plugin.json +1 -1
- package/packages/omo-codex/plugin/components/bootstrap/hooks/hooks.json +1 -1
- package/packages/omo-codex/plugin/components/bootstrap/package.json +1 -1
- package/packages/omo-codex/plugin/components/codegraph/dist/cli.js +76 -3
- package/packages/omo-codex/plugin/components/codegraph/dist/serve.js +76 -3
- package/packages/omo-codex/plugin/components/codegraph/package.json +1 -1
- package/packages/omo-codex/plugin/components/comment-checker/hooks/hooks.json +1 -1
- package/packages/omo-codex/plugin/components/comment-checker/package.json +1 -1
- package/packages/omo-codex/plugin/components/git-bash/hooks/hooks.json +2 -2
- package/packages/omo-codex/plugin/components/git-bash/package.json +1 -1
- package/packages/omo-codex/plugin/components/lazycodex-executor-verify/hooks/hooks.json +1 -1
- package/packages/omo-codex/plugin/components/lazycodex-executor-verify/package.json +1 -1
- package/packages/omo-codex/plugin/components/lsp/dist/.omo-runtime-manifest.json +2 -2
- package/packages/omo-codex/plugin/components/lsp/hooks/hooks.json +2 -2
- package/packages/omo-codex/plugin/components/lsp/package.json +1 -1
- package/packages/omo-codex/plugin/components/rules/hooks/hooks.json +4 -4
- package/packages/omo-codex/plugin/components/rules/package.json +1 -1
- package/packages/omo-codex/plugin/components/start-work-continuation/hooks/hooks.json +2 -2
- package/packages/omo-codex/plugin/components/start-work-continuation/package.json +1 -1
- package/packages/omo-codex/plugin/components/teammode/hooks/hooks.json +1 -1
- package/packages/omo-codex/plugin/components/teammode/package.json +1 -1
- package/packages/omo-codex/plugin/components/telemetry/hooks/hooks.json +1 -1
- package/packages/omo-codex/plugin/components/telemetry/package.json +1 -1
- package/packages/omo-codex/plugin/components/ultrawork/directive.md +6 -0
- package/packages/omo-codex/plugin/components/ultrawork/hooks/hooks.json +1 -1
- package/packages/omo-codex/plugin/components/ultrawork/package.json +1 -1
- package/packages/omo-codex/plugin/components/ultrawork/skills/ultrawork/SKILL.md +6 -0
- package/packages/omo-codex/plugin/components/ulw-loop/directive.md +6 -0
- package/packages/omo-codex/plugin/components/ulw-loop/hooks/hooks.json +4 -4
- package/packages/omo-codex/plugin/components/ulw-loop/package.json +1 -1
- package/packages/omo-codex/plugin/components/ulw-loop/skills/ulw-loop/SKILL.md +3 -2
- package/packages/omo-codex/plugin/components/ulw-loop/skills/ulw-loop/references/define-goal.md +106 -0
- package/packages/omo-codex/plugin/components/ulw-loop/skills/ulw-loop/references/full-workflow.md +1 -0
- package/packages/omo-codex/plugin/hooks/post-compact-resetting-git-bash-mcp-reminder.json +1 -1
- package/packages/omo-codex/plugin/hooks/post-compact-resetting-lsp-diagnostics-cache.json +1 -1
- package/packages/omo-codex/plugin/hooks/post-compact-resetting-project-rule-cache.json +1 -1
- package/packages/omo-codex/plugin/hooks/post-tool-use-checking-codegraph-init-guidance.json +1 -1
- package/packages/omo-codex/plugin/hooks/post-tool-use-checking-comments.json +1 -1
- package/packages/omo-codex/plugin/hooks/post-tool-use-checking-lsp-diagnostics.json +1 -1
- package/packages/omo-codex/plugin/hooks/post-tool-use-checking-thread-title-hygiene.json +1 -1
- package/packages/omo-codex/plugin/hooks/post-tool-use-matching-project-rules.json +1 -1
- package/packages/omo-codex/plugin/hooks/pre-tool-use-enforcing-unlimited-goal-budget.json +1 -1
- package/packages/omo-codex/plugin/hooks/pre-tool-use-guarding-ulw-loop-spawns.json +1 -1
- package/packages/omo-codex/plugin/hooks/pre-tool-use-recommending-git-bash-mcp.json +1 -1
- package/packages/omo-codex/plugin/hooks/session-start-checking-auto-update.json +1 -1
- package/packages/omo-codex/plugin/hooks/session-start-checking-bootstrap-provisioning.json +1 -1
- package/packages/omo-codex/plugin/hooks/session-start-checking-codegraph-bootstrap.json +1 -1
- package/packages/omo-codex/plugin/hooks/session-start-loading-project-rules.json +1 -1
- package/packages/omo-codex/plugin/hooks/session-start-recording-session-telemetry.json +1 -1
- package/packages/omo-codex/plugin/hooks/stop-checking-start-work-continuation.json +1 -1
- package/packages/omo-codex/plugin/hooks/stop-checking-ulw-loop-resume.json +1 -1
- package/packages/omo-codex/plugin/hooks/subagent-stop-checking-start-work-continuation.json +1 -1
- package/packages/omo-codex/plugin/hooks/subagent-stop-verifying-lazycodex-executor-evidence.json +1 -1
- package/packages/omo-codex/plugin/hooks/user-prompt-submit-checking-ultrawork-trigger.json +1 -1
- package/packages/omo-codex/plugin/hooks/user-prompt-submit-checking-ulw-loop-steering.json +1 -1
- package/packages/omo-codex/plugin/hooks/user-prompt-submit-loading-project-rules.json +1 -1
- package/packages/omo-codex/plugin/package-lock.json +13 -13
- package/packages/omo-codex/plugin/package.json +1 -1
- package/packages/omo-codex/plugin/skills/ultrawork/SKILL.md +6 -0
- package/packages/omo-codex/plugin/skills/ulw-loop/SKILL.md +3 -2
- package/packages/omo-codex/plugin/skills/ulw-loop/references/define-goal.md +106 -0
- package/packages/omo-codex/plugin/skills/ulw-loop/references/full-workflow.md +1 -0
- package/packages/omo-codex/scripts/install-dist/install-local.mjs +2 -2
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
# Define Goal
|
|
2
|
+
|
|
3
|
+
How to turn a brief into a registered goal the run can be held to. Read this BEFORE calling `create_goal`: the objective you register is the binding contract for the whole run, and the run's quality is capped by the quality of this objective.
|
|
4
|
+
|
|
5
|
+
A goal is a prompt to the agent that executes it, including future-you after compaction. It earns its tokens the way any prompt does: it carries only what the run cannot re-derive later, the outcome, the proof, the bounds, and the stop state. Everything else is noise that steals attention from the parts that decide completion.
|
|
6
|
+
|
|
7
|
+
## The quality bar
|
|
8
|
+
|
|
9
|
+
Before registering, the objective must answer all five:
|
|
10
|
+
|
|
11
|
+
1. What concrete thing will be TRUE when this is done? An outcome, never an activity.
|
|
12
|
+
2. What evidence will prove it? Commands, validators, artifacts someone can open.
|
|
13
|
+
3. What quantitative or binary threshold defines success?
|
|
14
|
+
4. What scope boundaries matter? What is in, and what is explicitly out.
|
|
15
|
+
5. What should make the agent stop and ask instead of grinding?
|
|
16
|
+
|
|
17
|
+
An objective that cannot answer one of these is not ready. Repair it (below) before calling the tool.
|
|
18
|
+
|
|
19
|
+
## Objective anatomy
|
|
20
|
+
|
|
21
|
+
Write the objective outcome-first, in this order:
|
|
22
|
+
|
|
23
|
+
1. **Outcome**: one sentence stating what will be true, naming the artifact, system, repo, or user-facing behavior involved.
|
|
24
|
+
2. **Deliverables**: the named surfaces the work lands on (files, endpoints, packages, environments). Use literal paths and names: the executing agent interprets the objective literally and will not infer surfaces you did not name.
|
|
25
|
+
3. **Success criteria**: sized by tier (below), each one a binary observable with its scenario and evidence named upfront.
|
|
26
|
+
4. **Scope bounds**: what is out of scope, stated wherever ambiguity would let the run expand. Unstated bounds do not exist.
|
|
27
|
+
5. **WHEN TO STOP**: one line, "I'll stop right away when <the exact observable state that ends this run>". This line is binding: the moment it holds, the run delivers and stops. Work past it is a defect, not diligence.
|
|
28
|
+
|
|
29
|
+
State the motivation when it changes execution ("p95 matters because the checkout SLA is 300ms") and omit it when it does not. Positive statements beat prohibitions: "verify against staging" carries more signal than "do not touch production".
|
|
30
|
+
|
|
31
|
+
## Success criteria construction
|
|
32
|
+
|
|
33
|
+
Count by tier, mirroring the run's tier triage:
|
|
34
|
+
|
|
35
|
+
- LIGHT (known pattern, no open design decisions): 1-2 criteria, happy path plus the riskiest edge.
|
|
36
|
+
- HEAVY (new module or abstraction, auth or security, external integration, schema or migration, concurrency, cross-domain refactor, or the user demanded care): 3+ criteria covering happy path, edge (boundary, empty, malformed, concurrent), adjacent-surface regression named by file and function, and the adversarial risk the change actually creates.
|
|
37
|
+
|
|
38
|
+
Every criterion carries, at definition time, not after the work:
|
|
39
|
+
|
|
40
|
+
- a binary pass condition ("returns 200 and the body matches the schema", never "works correctly");
|
|
41
|
+
- the exact scenario: the literal command, request, page action, or payload that will prove it;
|
|
42
|
+
- the evidence artifact it will capture: transcript, status plus body, screenshot path, diff, parsed dump;
|
|
43
|
+
- the failing-first proof (test id or scenario) that will be captured RED before implementation.
|
|
44
|
+
|
|
45
|
+
A criterion that cannot fail is not a criterion. If no input could make the scenario fail, it measures nothing; rewrite it until failure is possible.
|
|
46
|
+
|
|
47
|
+
## Make it quantitative
|
|
48
|
+
|
|
49
|
+
Prefer numbers that represent real success over decorative precision. A threshold nobody would act on differently is noise.
|
|
50
|
+
|
|
51
|
+
| Domain | Quantify as |
|
|
52
|
+
| --- | --- |
|
|
53
|
+
| Bug fix | reproduction first, fix second: the failing case captured RED, then the same validator green |
|
|
54
|
+
| Tests | the exact command and required pass condition, plus run count for flake-sensitive suites |
|
|
55
|
+
| Performance | metric, target threshold, measurement method, and run count ("p95 under 250ms across 3 consecutive local runs") |
|
|
56
|
+
| Quality work | the observable acceptance bar: lint, typecheck, and test pass; reviewed examples; a user-approved artifact |
|
|
57
|
+
| Research | the decision the research must enable, the sources or systems in scope, and the evidence standard per claim |
|
|
58
|
+
| Operations | healthy state, monitoring window, failure threshold, and the rollback or escalation trigger |
|
|
59
|
+
|
|
60
|
+
## Repair weak goals
|
|
61
|
+
|
|
62
|
+
Reject pure activity objectives: "make progress", "keep investigating", "improve things", "work on X". They cannot fail, so they cannot finish.
|
|
63
|
+
|
|
64
|
+
Rewrite vague goals into measurable ones when local context makes the rewrite safe. Ask ONE narrow question only when the missing detail changes the intended outcome or its validation, shaped around the missing validator or bound:
|
|
65
|
+
|
|
66
|
+
- "What metric defines success here: latency, cost, accuracy, or user-visible behavior?"
|
|
67
|
+
- "Which environment do I verify against: local, staging, or production?"
|
|
68
|
+
- "What is the minimum evidence you want before this goal is marked complete?"
|
|
69
|
+
|
|
70
|
+
When the user cannot provide a metric, propose the most honest binary validator available and proceed with it stated in the objective.
|
|
71
|
+
|
|
72
|
+
Weak: "Make checkout faster."
|
|
73
|
+
Repaired: "Reduce checkout API p95 below 250ms on the documented slow path with the smallest safe server-side change; prove it with `npm run test:checkout` green plus the local latency benchmark showing p95 under 250ms across 3 consecutive runs; out of scope: client-side changes and new caching layers."
|
|
74
|
+
|
|
75
|
+
Weak: "Keep investigating the PR comments."
|
|
76
|
+
Repaired: "Resolve every open change-requesting review comment on PR 123 touching only the affected auth files and their tests; prove it with the targeted auth test command green plus `gh pr view 123` showing zero unresolved change-request threads."
|
|
77
|
+
|
|
78
|
+
## Registration protocol
|
|
79
|
+
|
|
80
|
+
1. Call `get_goal` first, then act by state:
|
|
81
|
+
|
|
82
|
+
| get_goal shows | Action |
|
|
83
|
+
| --- | --- |
|
|
84
|
+
| no active goal | Register with `create_goal`, passing exactly `objective`. Never include lifecycle fields such as `status`; never register a goal in prose, a notepad, or a plan instead of the tool. |
|
|
85
|
+
| an active goal matching this intent | Continue it. Never register a duplicate. |
|
|
86
|
+
| an active goal conflicting with this intent | Stop and surface the conflict; the user decides whether to finish it, complete it, or branch. |
|
|
87
|
+
|
|
88
|
+
2. Goals are unlimited. Never invent a numeric budget, token limit, or deadline the user did not state.
|
|
89
|
+
3. In a ulw-loop run, the loop CLI owns per-goal state (`.omo/ulw-loop/goals.json`): `create_goal` registers the aggregate objective from the printed handoff, and this reference shapes both that objective and every goal's `successCriteria` at `create-goals` time.
|
|
90
|
+
|
|
91
|
+
## Completion honesty
|
|
92
|
+
|
|
93
|
+
- Report `update_goal` complete only after auditing every criterion against evidence captured in this run. A green suite is supporting evidence, never completion proof by itself.
|
|
94
|
+
- Waiting is not blocked: while a monitor, background child, or scheduled continuation can wake the run, end the turn and let it fire. Blocked requires a true impasse: no live resumption channel, and the same block recurring across consecutive turns.
|
|
95
|
+
- The moment the WHEN TO STOP line holds with evidence in hand, deliver and stop.
|
|
96
|
+
|
|
97
|
+
## Anti-patterns
|
|
98
|
+
|
|
99
|
+
| Anti-pattern | Why it fails | Instead |
|
|
100
|
+
| --- | --- | --- |
|
|
101
|
+
| Activity objective ("investigate X") | Cannot fail, so cannot finish; the run wanders | Name the outcome the activity must produce and its evidence |
|
|
102
|
+
| Criteria added after implementation | The contract bent to fit the work; nothing was proven | Write criteria and scenarios at registration, before any edit |
|
|
103
|
+
| Decorative precision ("99.97% uptime" nobody measures) | A threshold no validator checks is noise wearing a suit | Only thresholds a named validator will actually check |
|
|
104
|
+
| Padded objective (role prose, restated context, filler) | Every extra token competes with the criteria for attention | Outcome, deliverables, criteria, bounds, stop line; nothing else |
|
|
105
|
+
| Goal registered in prose or a notepad | Nothing binds the run; completion becomes a vibe | `create_goal` with the objective, every time the tool exists |
|
|
106
|
+
| Duplicate goal for the same intent | Two contracts, neither authoritative | Continue the active goal or surface the conflict |
|
|
@@ -121,6 +121,7 @@ only when deliberately overwriting completed evidence.
|
|
|
121
121
|
Write state through the CLI path. Do not hand-edit state files.
|
|
122
122
|
|
|
123
123
|
### 2. Refine success criteria + a Prometheus-grade QA and parallelism plan per goal
|
|
124
|
+
Shape every goal's objective and `successCriteria` by `references/define-goal.md`: its quality bar, objective anatomy, and criterion construction govern this step.
|
|
124
125
|
Gather context BEFORE planning with parallel `explorer` / `librarian` workers plus your own read-only tools.
|
|
125
126
|
First survey available skills: read every loosely-relevant skill's description, deliberately choose which this work uses, and prefer applying genuinely-relevant skills over working raw.
|
|
126
127
|
Then run tier triage per goal — rigor (LIGHT/HEAVY below) and shape (`delivery` default, or `research` when the deliverable is a cited answer, not an artifact) — and record both in an `annotate_ledger` steering entry. Default is LIGHT — a narrow change inside existing layers. Take HEAVY only on a fact you can point to: a new module / abstraction / domain model; auth, security, or session; an external integration; a DB schema or migration; concurrency, transaction boundaries, or cache invalidation; a cross-domain refactor; or the user signaled care or demanded review. When unsure, take HEAVY; upgrade the moment a HEAVY fact surfaces, never downgrade mid-run.
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
// omo-codex-install:7bc4b0f020f0a1b4448cc75385a7e3c60042e3b270e304d4c1a43a859e81a4c2:
|
|
2
|
+
// omo-codex-install:7bc4b0f020f0a1b4448cc75385a7e3c60042e3b270e304d4c1a43a859e81a4c2:b7c9fa9c8d92e9e8082bed55796691f3ce7feb478dc3826605dde249c3937b4e
|
|
3
3
|
var __defProp = Object.defineProperty;
|
|
4
4
|
var __returnValue = (v) => v;
|
|
5
5
|
function __exportSetter(name, newValue) {
|
|
@@ -5927,7 +5927,7 @@ var package_default;
|
|
|
5927
5927
|
var init_package = __esm(() => {
|
|
5928
5928
|
package_default = {
|
|
5929
5929
|
name: "@oh-my-opencode/omo-codex",
|
|
5930
|
-
version: "5.0.0-beta.
|
|
5930
|
+
version: "5.0.0-beta.7",
|
|
5931
5931
|
type: "module",
|
|
5932
5932
|
private: true,
|
|
5933
5933
|
description: "Codex harness adapter for oh-my-openagent. Vendored Codex plugin namespace (omo) + TypeScript installer + telemetry.",
|