@tachikomagundam/abathur 0.2.5 → 0.2.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/commands/genome.js +2 -2
- package/dist/commands/graft.js +1 -1
- package/dist/commands/run.js +1 -1
- package/dist/commands/self-eval.js +1 -1
- package/dist/commands/tombstone.js +2 -1
- package/dist/core/evolve/run-loop.js +14 -0
- package/dist/core/evolve/score-bank.js +214 -0
- package/dist/core/graft-rebench.js +3 -0
- package/dist/core/ledger.js +6 -1
- package/dist/core/promote.js +2 -1
- package/dist/core/stats-math.js +39 -0
- package/dist/core/stats.js +93 -19
- package/dist/test/historian-grader-integrity.test.js +509 -2
- package/dist/test/historian-run-scenario.test.js +2 -2
- package/dist/test/repopath-seams.test.js +54 -0
- package/dist/test/score-bank.test.js +235 -0
- package/dist/test/seam-gates.test.js +78 -0
- package/dist/test/stats-acceptance.test.js +329 -0
- package/graders/historian/grader-core.d.mts +27 -0
- package/graders/historian/grader-core.mjs +311 -19
- package/graders/historian/grader.mjs +5 -5
- package/graders/historian/judge-poststage-19.mjs +212 -0
- package/graders/historian/judge-poststage.mjs +3 -8
- package/graders/historian/run-scenario-19.sh +43 -0
- package/graders/historian/run-scenario.sh +16 -0
- package/package.json +1 -1
- package/plugin/abathur.ts +1 -1
|
@@ -32,14 +32,10 @@ import { appendFileSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFile
|
|
|
32
32
|
import os from "node:os";
|
|
33
33
|
import path from "node:path";
|
|
34
34
|
|
|
35
|
-
// D7 zero-literals: the bee-judge bench lives outside the repo (campaign
|
|
36
|
-
// evidence dir); operators wire it via ABATHUR_JUDGE_BENCH. Unset => the
|
|
37
|
-
// judge stage fails closed BEFORE spawn and the grader fail-closes the legs.
|
|
38
|
-
const JUDGE_BENCH = process.env.ABATHUR_JUDGE_BENCH ?? null;
|
|
39
35
|
const DEFAULTS = {
|
|
40
|
-
bench:
|
|
41
|
-
rubrics:
|
|
42
|
-
(r) =>
|
|
36
|
+
bench: "/home/lab/workspace/harness/historian/.omo/evidence/judge-bench/judge-bench.mjs",
|
|
37
|
+
rubrics: ["R1-semantic", "R4-duty-v2", "R5-flavor"].map(
|
|
38
|
+
(r) => `/home/lab/workspace/harness/historian/.omo/evidence/judge-bench/rubrics/${r}.md`,
|
|
43
39
|
),
|
|
44
40
|
model: "local-qwen/qwen3.8-flash-next",
|
|
45
41
|
};
|
|
@@ -115,7 +111,6 @@ function foldMajority(row) {
|
|
|
115
111
|
const sha256 = (text) => createHash("sha256").update(text).digest("hex");
|
|
116
112
|
|
|
117
113
|
const opts = parseArgs(process.argv.slice(2));
|
|
118
|
-
if (opts.bench === "") { console.error("judge-poststage: no bee-judge bench wired — set ABATHUR_JUDGE_BENCH (or pass --bench); grader fail-closes the judge legs without it."); process.exit(2); }
|
|
119
114
|
(async () => {
|
|
120
115
|
try {
|
|
121
116
|
const work = mkdtempSync(path.join(os.tmpdir(), "s17-judge-"));
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# scenario-19 run hook = run-scenario.sh (agent stage, 1200s ceiling) + the R6-termb
|
|
3
|
+
# 蜂判 poststage (judge-poststage-19.mjs, per-page certified shape) against the
|
|
4
|
+
# ACTUAL pages the agent produced. Same data flow as run-scenario-17.sh (which is
|
|
5
|
+
# left untouched): the shipped grader never calls a model; the LLM measurement is
|
|
6
|
+
# instrument output staged at run time into .bench/judge-verdicts.json, which
|
|
7
|
+
# grader.mjs reads as observable state (fail-closed when absent). The matrix wall
|
|
8
|
+
# budget is pinned via JUDGE_BENCH_DEADLINE (+S19_JUDGE_BUDGET_S, default 600s):
|
|
9
|
+
# a truncated matrix leaves coverage incomplete ⇒ I fail-closes, never silently.
|
|
10
|
+
# The LAST stdout line stays run-scenario.sh's {unit,tokensEst,turns,opencodeExit}
|
|
11
|
+
# meta JSON — parseRunMeta's contract — so the engine-facing format is byte-
|
|
12
|
+
# compatible with units 01–18.
|
|
13
|
+
set -euo pipefail
|
|
14
|
+
unit_id="${1:?usage: run-scenario-19.sh <unitId> <scenario-file> [repoRoot]}"
|
|
15
|
+
scenario_file="${2:?usage: run-scenario-19.sh <unitId> <scenario-file> [repoRoot]}"
|
|
16
|
+
here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
17
|
+
pages="${S19_PAGES:-_sandbox/eval19/warm-pool-dossier=en,_sandbox/eval19/warm-pool-summary=zh}"
|
|
18
|
+
wiki_base="${ABATHUR_WIKI_BASE:-http://localhost:3000}"
|
|
19
|
+
agent_timeout_s="${S19_AGENT_TIMEOUT_S:-1200}"
|
|
20
|
+
judge_budget_s="${S19_JUDGE_BUDGET_S:-600}"
|
|
21
|
+
meta=""
|
|
22
|
+
agent_s=0
|
|
23
|
+
judge_s=0
|
|
24
|
+
if [ "${S19_SKIP_AGENT:-0}" != "1" ]; then
|
|
25
|
+
t=$SECONDS
|
|
26
|
+
meta="$(timeout --foreground -k 30 "${agent_timeout_s}" bash "$here/run-scenario.sh" "$@" || true)"
|
|
27
|
+
agent_s=$((SECONDS - t))
|
|
28
|
+
fi
|
|
29
|
+
page_args=()
|
|
30
|
+
IFS=',' read -ra specs <<< "$pages"
|
|
31
|
+
for s in "${specs[@]}"; do page_args+=(--page "$s"); done
|
|
32
|
+
t=$SECONDS
|
|
33
|
+
export JUDGE_BENCH_DEADLINE="$(( ( $(date +%s) + judge_budget_s ) * 1000 ))"
|
|
34
|
+
node "$here/judge-poststage-19.mjs" "${page_args[@]}" --wiki-base "$wiki_base" \
|
|
35
|
+
--out "${ABATHUR_JUDGE_VERDICTS:-.bench/judge-verdicts.json}" \
|
|
36
|
+
--ledger "${ABATHUR_JUDGE_LEDGER:-.bench/judge-ledger-${unit_id}.jsonl}" 1>&2 || echo "[run-scenario-19] judge stage FAILED (no verdicts file — grader fail-closes the 蜂判 legs)" >&2
|
|
37
|
+
judge_s=$((SECONDS - t))
|
|
38
|
+
echo "[run-scenario-19] stage walls: agent=${agent_s}s (ceiling ${agent_timeout_s}s) judge=${judge_s}s (budget ${judge_budget_s}s) (unit=${unit_id})" >&2
|
|
39
|
+
if [ -n "$meta" ]; then
|
|
40
|
+
printf '%s\n' "$meta"
|
|
41
|
+
else
|
|
42
|
+
printf '%s\n' "{\"unit\": \"$unit_id\", \"tokensEst\": 0, \"turns\": 0, \"opencodeExit\": -1}"
|
|
43
|
+
fi
|
|
@@ -20,6 +20,22 @@
|
|
|
20
20
|
set -euo pipefail
|
|
21
21
|
unit_id="${1:?usage: run-scenario.sh <unitId> <scenario-file> [repoRoot]}"
|
|
22
22
|
scenario_file="${2:?usage: run-scenario.sh <unitId> <scenario-file> [repoRoot]}"
|
|
23
|
+
# Per-unit hook dispatch (s19 autopsy 2026-09-25): a unit may own a hook script
|
|
24
|
+
# run-scenario-<NN>.sh that wraps THIS agent stage plus an instrumented poststage
|
|
25
|
+
# (e.g. the R6-termb 蜂判 for scenario-19). The registered runCommand is one
|
|
26
|
+
# genome-global string, so the dispatch lives here rather than in the spec: when
|
|
27
|
+
# this unit owns a hook and we are not already running inside it (recursion guard
|
|
28
|
+
# set on the way in), defer to the hook. Units without a hook — and every caller
|
|
29
|
+
# that passes an id with no matching file — fall through byte-identical (F1).
|
|
30
|
+
_here="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
31
|
+
case "$unit_id" in
|
|
32
|
+
scenario-*)
|
|
33
|
+
_hook="$_here/run-scenario-${unit_id#scenario-}.sh"
|
|
34
|
+
if [ -f "$_hook" ] && [ -z "${ABATHUR_HOOK_INNER:-}" ]; then
|
|
35
|
+
exec env ABATHUR_HOOK_INNER=1 bash "$_hook" "$@"
|
|
36
|
+
fi
|
|
37
|
+
;;
|
|
38
|
+
esac
|
|
23
39
|
[ -f "$scenario_file" ] || { echo "run-scenario: missing scenario file $scenario_file" >&2; exit 2; }
|
|
24
40
|
[ -n "${ABATHUR_TRANSCRIPT:-}" ] || { echo "run-scenario: ABATHUR_TRANSCRIPT not set" >&2; exit 2; }
|
|
25
41
|
key="${ABATHUR_WIKI_KEY_FILE:-}"
|
package/package.json
CHANGED
package/plugin/abathur.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
// abathur-opencode-plugin v0.2.
|
|
1
|
+
// abathur-opencode-plugin v0.2.6
|
|
2
2
|
// Official opencode plugin adapter for the abathur evolution harness.
|
|
3
3
|
// Registers ONE agent tool, `abathur`, that shells out to the abathur CLI —
|
|
4
4
|
// argv-only (node:child_process execFile, never a shell), top-level commands
|