chofex-cli 0.1.189 → 0.1.190
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/index.js +146 -14
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -27,7 +27,7 @@ chofex challenge ranking --challenge black-box
|
|
|
27
27
|
chofex challenge init --challenge broken-agent
|
|
28
28
|
cd broken-agent && npm test
|
|
29
29
|
chofex challenge test --challenge broken-agent --source ./scheduler.js
|
|
30
|
-
chofex challenge evaluate --challenge broken-agent --source ./scheduler.js
|
|
30
|
+
chofex challenge evaluate --challenge broken-agent --source ./scheduler.js --review ./review.json
|
|
31
31
|
chofex challenge ranking --challenge broken-agent
|
|
32
32
|
chofex confirm
|
|
33
33
|
chofex badge
|
package/dist/index.js
CHANGED
|
@@ -58710,6 +58710,33 @@ var JavascriptSourceSolutionSchema = exports_Schema.Struct({
|
|
|
58710
58710
|
kind: exports_Schema.Literal("javascript_source"),
|
|
58711
58711
|
source: exports_Schema.String.pipe(exports_Schema.check(exports_Schema.isMinLength(1), exports_Schema.isMaxLength(32768)))
|
|
58712
58712
|
});
|
|
58713
|
+
var HumanReviewAnswerSchema = exports_Schema.Trim.pipe(exports_Schema.check(exports_Schema.isMinLength(20), exports_Schema.isMaxLength(1000)));
|
|
58714
|
+
var BrokenAgentHumanReviewSchema = exports_Schema.Struct({
|
|
58715
|
+
sourceDigest: exports_Schema.String.pipe(exports_Schema.check(exports_Schema.isPattern(/^[a-f0-9]{64}$/, {
|
|
58716
|
+
identifier: "SHA-256 source digest"
|
|
58717
|
+
}))),
|
|
58718
|
+
focus: exports_Schema.Literals([
|
|
58719
|
+
"concurrency",
|
|
58720
|
+
"persistence",
|
|
58721
|
+
"lease_recovery",
|
|
58722
|
+
"retry_idempotency",
|
|
58723
|
+
"regression_safety",
|
|
58724
|
+
"performance"
|
|
58725
|
+
]),
|
|
58726
|
+
failureScenario: HumanReviewAnswerSchema,
|
|
58727
|
+
evidence: HumanReviewAnswerSchema,
|
|
58728
|
+
decision: exports_Schema.Literals(["ship", "block"]),
|
|
58729
|
+
confidence: exports_Schema.Int.pipe(exports_Schema.check(exports_Schema.isBetween({ minimum: 0, maximum: 100 }))),
|
|
58730
|
+
remainingRisk: HumanReviewAnswerSchema
|
|
58731
|
+
});
|
|
58732
|
+
var BrokenAgentEvaluationSolutionSchema = exports_Schema.Struct({
|
|
58733
|
+
...JavascriptSourceSolutionSchema.fields,
|
|
58734
|
+
review: BrokenAgentHumanReviewSchema
|
|
58735
|
+
});
|
|
58736
|
+
var ChallengeSolutionSchema = exports_Schema.Union([
|
|
58737
|
+
JavascriptSourceSolutionSchema,
|
|
58738
|
+
BrokenAgentEvaluationSolutionSchema
|
|
58739
|
+
]);
|
|
58713
58740
|
var ChallengeObservationSchema = exports_Schema.Struct({
|
|
58714
58741
|
sequence: exports_Schema.Number,
|
|
58715
58742
|
input: exports_Schema.Unknown,
|
|
@@ -59770,6 +59797,7 @@ var evaluateChallenge = (options, slug, input) => request3(options, `/api/v1/cha
|
|
|
59770
59797
|
var getChallengeRanking = (options, slug) => publicRequest(options, `/api/v1/challenges/${slug}/ranking`, { method: "GET" }, decodeChallengeRanking);
|
|
59771
59798
|
|
|
59772
59799
|
// src/challenge-input.ts
|
|
59800
|
+
import { createHash as createHash3 } from "node:crypto";
|
|
59773
59801
|
import { readFile as readFile3 } from "node:fs/promises";
|
|
59774
59802
|
var defaultChallengeSlug = "broken-agent";
|
|
59775
59803
|
var readStdin = async () => {
|
|
@@ -59796,6 +59824,50 @@ var javascriptSourceFromPath = (path4, challenge = defaultChallengeSlug) => {
|
|
|
59796
59824
|
}
|
|
59797
59825
|
return readTextFile(path4).pipe(exports_Effect.map((source) => ({ kind: "javascript_source", source })));
|
|
59798
59826
|
};
|
|
59827
|
+
var humanReviewRequiredDetails = {
|
|
59828
|
+
requiredFields: [
|
|
59829
|
+
"sourceDigest",
|
|
59830
|
+
"focus",
|
|
59831
|
+
"failureScenario",
|
|
59832
|
+
"evidence",
|
|
59833
|
+
"decision",
|
|
59834
|
+
"confidence",
|
|
59835
|
+
"remainingRisk"
|
|
59836
|
+
],
|
|
59837
|
+
next: "Ask the participant to reason about the patch and provide these answers in review.json, then pass --review review.json."
|
|
59838
|
+
};
|
|
59839
|
+
var brokenAgentReviewFromPath = (path4) => {
|
|
59840
|
+
if (!path4) {
|
|
59841
|
+
return exports_Effect.fail(cliError("HUMAN_REVIEW_REQUIRED", "Broken Agent requires the participant's engineering review before an official evaluation", false, humanReviewRequiredDetails));
|
|
59842
|
+
}
|
|
59843
|
+
return readTextFile(path4).pipe(exports_Effect.flatMap((contents) => exports_Effect.try({
|
|
59844
|
+
try: () => JSON.parse(contents),
|
|
59845
|
+
catch: (error2) => cliError("INVALID_REVIEW_FILE", `Could not parse JSON review: ${String(error2)}`)
|
|
59846
|
+
})), exports_Effect.flatMap((input) => exports_Schema.decodeUnknownEffect(BrokenAgentHumanReviewSchema, {
|
|
59847
|
+
onExcessProperty: "error"
|
|
59848
|
+
})(input).pipe(exports_Effect.mapError((error2) => cliError("INVALID_HUMAN_REVIEW", error2.message, false, {
|
|
59849
|
+
...humanReviewRequiredDetails,
|
|
59850
|
+
issues: String(error2)
|
|
59851
|
+
})))));
|
|
59852
|
+
};
|
|
59853
|
+
var challengeEvaluationInput = (sourcePath, reviewPath, challenge) => exports_Effect.gen(function* () {
|
|
59854
|
+
const solution = yield* javascriptSourceFromPath(sourcePath, challenge);
|
|
59855
|
+
if (challenge !== "broken-agent")
|
|
59856
|
+
return solution;
|
|
59857
|
+
const sourceDigest = createHash3("sha256").update(solution.source).digest("hex");
|
|
59858
|
+
if (!reviewPath) {
|
|
59859
|
+
return yield* exports_Effect.fail(cliError("HUMAN_REVIEW_REQUIRED", "Broken Agent requires the participant's engineering review before an official evaluation", false, { ...humanReviewRequiredDetails, sourceDigest }));
|
|
59860
|
+
}
|
|
59861
|
+
const review = yield* brokenAgentReviewFromPath(reviewPath);
|
|
59862
|
+
if (review.sourceDigest !== sourceDigest) {
|
|
59863
|
+
return yield* exports_Effect.fail(cliError("STALE_HUMAN_REVIEW", "review.json does not match the current scheduler.js", false, {
|
|
59864
|
+
expectedSourceDigest: sourceDigest,
|
|
59865
|
+
reviewSourceDigest: review.sourceDigest,
|
|
59866
|
+
next: "Show the participant the updated evidence, obtain a fresh decision, and replace review.json."
|
|
59867
|
+
}));
|
|
59868
|
+
}
|
|
59869
|
+
return { ...solution, review };
|
|
59870
|
+
});
|
|
59799
59871
|
var requiredInteger = (message) => exports_Prompt.text({
|
|
59800
59872
|
message,
|
|
59801
59873
|
validate: (value3) => {
|
|
@@ -59910,7 +59982,7 @@ var challengeShowText = (attempt) => {
|
|
|
59910
59982
|
`ESTADO DEL CASO: ${caseStatus2}`,
|
|
59911
59983
|
"",
|
|
59912
59984
|
"Los tests públicos están verdes. Tu trabajo es hacer que el scheduler sea confiable bajo condiciones de producción.",
|
|
59913
|
-
"Las herramientas de AI están permitidas.",
|
|
59985
|
+
"Las herramientas de AI están permitidas: el agente implementa; tú eliges el riesgo, revisas la evidencia y decides ship o block.",
|
|
59914
59986
|
"",
|
|
59915
59987
|
`Evaluaciones ${evaluationsRemaining} / ${progress.evaluationsLimit} restantes`
|
|
59916
59988
|
];
|
|
@@ -59921,7 +59993,7 @@ var challengeShowText = (attempt) => {
|
|
|
59921
59993
|
if (progress.shareCode)
|
|
59922
59994
|
lines3.push(`Código #${progress.shareCode}`);
|
|
59923
59995
|
}
|
|
59924
|
-
lines3.push("", "SIGUIENTE PASO", "Lee el contrato,
|
|
59996
|
+
lines3.push("", "SIGUIENTE PASO", "Lee el contrato. Antes de editar, discute tres trazas de falla y elige cuál investigar primero.", " cd broken-agent && npm test", " chofex challenge test --challenge broken-agent --source ./scheduler.js");
|
|
59925
59997
|
return lines3.join(`
|
|
59926
59998
|
`);
|
|
59927
59999
|
}
|
|
@@ -60093,7 +60165,7 @@ var challengeTestText = (result3) => {
|
|
|
60093
60165
|
`${result3.matchedObservations} / ${result3.observationCount} comportamientos visibles pasan`
|
|
60094
60166
|
];
|
|
60095
60167
|
if (result3.accuracy === 1) {
|
|
60096
|
-
lines3.push("", "Todo pasa.", "Eso todavía no significa que el scheduler sea correcto en producción.", "
|
|
60168
|
+
lines3.push("", "Todo pasa.", "Eso todavía no significa que el scheduler sea correcto en producción.", "Antes de evaluar, el participante debe revisar la evidencia y tomar la decisión de release.", "", "Siguiente: discutan una traza de falla concreta y creen review.json con las palabras del participante.", "Luego usa una evaluación oficial solo si su decisión lo permite", " chofex challenge evaluate --challenge broken-agent --source ./scheduler.js --review ./review.json");
|
|
60097
60169
|
return lines3.join(`
|
|
60098
60170
|
`);
|
|
60099
60171
|
}
|
|
@@ -60420,11 +60492,54 @@ Repara \`scheduler.js\` sin cambiar la interfaz exportada
|
|
|
60420
60492
|
\`\`\`sh
|
|
60421
60493
|
npm test
|
|
60422
60494
|
chofex challenge test --challenge broken-agent --source ./scheduler.js
|
|
60423
|
-
chofex challenge evaluate --challenge broken-agent --source ./scheduler.js
|
|
60495
|
+
chofex challenge evaluate --challenge broken-agent --source ./scheduler.js --review ./review.json
|
|
60424
60496
|
\`\`\`
|
|
60425
60497
|
|
|
60426
60498
|
Los tests locales y públicos son ilimitados. Tienes **5 evaluaciones oficiales**
|
|
60427
|
-
contra escenarios ocultos. Las herramientas de AI están permitidas
|
|
60499
|
+
contra escenarios ocultos. Las herramientas de AI están permitidas, pero este es
|
|
60500
|
+
un challenge de colaboración: el agente implementa y el participante toma las
|
|
60501
|
+
decisiones de ingeniería.
|
|
60502
|
+
|
|
60503
|
+
## Protocolo humano–agente
|
|
60504
|
+
|
|
60505
|
+
No le pidas al agente que resuelva todo en silencio. Antes de modificar el
|
|
60506
|
+
scheduler, el agente debe presentarte al menos tres trazas de falla concretas del
|
|
60507
|
+
starter. Tú eliges cuál investigar primero y explicas qué resultado nunca debería
|
|
60508
|
+
ocurrir. El agente convierte esa decisión en una prueba y luego implementa.
|
|
60509
|
+
|
|
60510
|
+
Cuando los tests estén verdes, el agente debe enseñarte el cambio, la evidencia y
|
|
60511
|
+
los supuestos que todavía no verificó. La evaluación oficial requiere
|
|
60512
|
+
\`review.json\` con tus propias palabras:
|
|
60513
|
+
|
|
60514
|
+
- \`focus\`: \`concurrency\`, \`persistence\`, \`lease_recovery\`,
|
|
60515
|
+
\`retry_idempotency\`, \`regression_safety\` o \`performance\`;
|
|
60516
|
+
- \`sourceDigest\`: SHA-256 de \`scheduler.js\` para vincular el juicio al cambio
|
|
60517
|
+
exacto (el agente puede calcular este valor mecánico);
|
|
60518
|
+
- \`failureScenario\`: una secuencia concreta de eventos y su resultado
|
|
60519
|
+
incorrecto;
|
|
60520
|
+
- \`evidence\`: la prueba o inspección que revisaste y qué demostró;
|
|
60521
|
+
- \`decision\`: \`ship\` o \`block\`;
|
|
60522
|
+
- \`confidence\`: un entero de 0 a 100;
|
|
60523
|
+
- \`remainingRisk\`: el riesgo que aceptas o que todavía bloquea el release.
|
|
60524
|
+
|
|
60525
|
+
Forma del archivo (reemplaza cada valor entre <...>):
|
|
60526
|
+
|
|
60527
|
+
\`\`\`json
|
|
60528
|
+
{
|
|
60529
|
+
"sourceDigest": "<sha256 de scheduler.js>",
|
|
60530
|
+
"focus": "<área elegida>",
|
|
60531
|
+
"failureScenario": "<tu traza concreta>",
|
|
60532
|
+
"evidence": "<la evidencia que revisaste>",
|
|
60533
|
+
"decision": "<ship o block>",
|
|
60534
|
+
"confidence": 0,
|
|
60535
|
+
"remainingRisk": "<el riesgo que queda>"
|
|
60536
|
+
}
|
|
60537
|
+
\`\`\`
|
|
60538
|
+
|
|
60539
|
+
Cada respuesta de texto debe tener entre 20 y 1,000 caracteres. El agente puede
|
|
60540
|
+
explicar, debatir y guardar tus respuestas, pero no puede elegir el foco, inventar
|
|
60541
|
+
tu razonamiento ni tomar la decisión de release por ti. Si cambias
|
|
60542
|
+
\`scheduler.js\`, vuelve a revisar la evidencia antes de reemplazar el review.
|
|
60428
60543
|
|
|
60429
60544
|
## Contrato normativo
|
|
60430
60545
|
|
|
@@ -60848,7 +60963,7 @@ var challengeQuickstart = {
|
|
|
60848
60963
|
"Tienes 5 evaluaciones oficiales contra variantes ocultas y determinísticas por participante.",
|
|
60849
60964
|
"Los tests locales y públicos son ilimitados.",
|
|
60850
60965
|
"Gana el puntaje total; los empates usan menos evaluaciones, costo determinístico y hora del mejor envío.",
|
|
60851
|
-
"Las herramientas de AI están permitidas."
|
|
60966
|
+
"Las herramientas de AI están permitidas, pero el participante toma las decisiones de ingeniería."
|
|
60852
60967
|
],
|
|
60853
60968
|
workflow: [
|
|
60854
60969
|
{
|
|
@@ -60877,24 +60992,36 @@ var challengeQuickstart = {
|
|
|
60877
60992
|
},
|
|
60878
60993
|
{
|
|
60879
60994
|
step: 5,
|
|
60995
|
+
action: "Elige el riesgo con el participante",
|
|
60996
|
+
command: "Discute tres trazas de falla concretas",
|
|
60997
|
+
note: "El participante elige cuál investigar primero y explica qué resultado nunca debería ocurrir. El agente no puede decidirlo por su cuenta."
|
|
60998
|
+
},
|
|
60999
|
+
{
|
|
61000
|
+
step: 6,
|
|
60880
61001
|
action: "Audita y repara",
|
|
60881
61002
|
command: "$EDITOR scheduler.js",
|
|
60882
|
-
note: "
|
|
61003
|
+
note: "Convierte la traza elegida en evidencia reproducible y luego endurece claims, leases, reinicios, carreras, cancelación, reintentos e idempotencia."
|
|
60883
61004
|
},
|
|
60884
61005
|
{
|
|
60885
|
-
step:
|
|
61006
|
+
step: 7,
|
|
60886
61007
|
action: "Ejecuta los tests públicos",
|
|
60887
61008
|
command: "chofex challenge test --challenge broken-agent --source ./scheduler.js",
|
|
60888
61009
|
note: "Es seguro repetirlos y no consumen evaluaciones oficiales."
|
|
60889
61010
|
},
|
|
60890
61011
|
{
|
|
60891
|
-
step:
|
|
61012
|
+
step: 8,
|
|
61013
|
+
action: "Obtén el juicio del participante",
|
|
61014
|
+
command: "Crea review.json con sus propias palabras",
|
|
61015
|
+
note: "Debe describir una traza de falla, la evidencia revisada, ship o block, confianza y riesgo restante. Cambiar el código exige revisar de nuevo."
|
|
61016
|
+
},
|
|
61017
|
+
{
|
|
61018
|
+
step: 9,
|
|
60892
61019
|
action: "Solicita un veredicto oculto",
|
|
60893
|
-
command: "chofex challenge evaluate --challenge broken-agent --source ./scheduler.js",
|
|
61020
|
+
command: "chofex challenge evaluate --challenge broken-agent --source ./scheduler.js --review ./review.json",
|
|
60894
61021
|
note: "Úsalo solo cuando enviarías la implementación a producción."
|
|
60895
61022
|
},
|
|
60896
61023
|
{
|
|
60897
|
-
step:
|
|
61024
|
+
step: 10,
|
|
60898
61025
|
action: "Consulta el ranking",
|
|
60899
61026
|
command: "chofex challenge ranking --challenge broken-agent",
|
|
60900
61027
|
note: "Se revela el 1 de octubre a las 15:00, hora de Perú."
|
|
@@ -60946,6 +61073,7 @@ var challengeQuickstartText = (participationState, launchNotice) => {
|
|
|
60946
61073
|
var optionalString = (name, description) => exports_Flag.string(name).pipe(exports_Flag.optional, exports_Flag.withDescription(description));
|
|
60947
61074
|
var challengeFlag = exports_Flag.string("challenge").pipe(exports_Flag.withDefault(defaultChallengeSlug), exports_Flag.withDescription("Challenge slug (default: broken-agent)"));
|
|
60948
61075
|
var sourceFlag = optionalString("source", "JavaScript challenge solution file");
|
|
61076
|
+
var reviewFlag = optionalString("review", "Participant-authored Broken Agent engineering review JSON");
|
|
60949
61077
|
var inputFlag = optionalString("input", "JSON file, or - for stdin");
|
|
60950
61078
|
var formatFlag = exports_Flag.choice("format", ["table", "json", "csv"]).pipe(exports_Flag.withDefault("table"), exports_Flag.withDescription("Notebook format"));
|
|
60951
61079
|
var numberFromOption = (value3) => {
|
|
@@ -61131,17 +61259,21 @@ var testCommand = exports_Command.make("test", { challenge: challengeFlag, sourc
|
|
|
61131
61259
|
description: "Ejecuta los tests públicos de Broken Agent"
|
|
61132
61260
|
}
|
|
61133
61261
|
]));
|
|
61134
|
-
var evaluateCommand = exports_Command.make("evaluate", { challenge: challengeFlag, source: sourceFlag }, exports_Effect.fn("challengeEvaluateCommand")(function* ({
|
|
61262
|
+
var evaluateCommand = exports_Command.make("evaluate", { challenge: challengeFlag, source: sourceFlag, review: reviewFlag }, exports_Effect.fn("challengeEvaluateCommand")(function* ({
|
|
61263
|
+
challenge,
|
|
61264
|
+
source,
|
|
61265
|
+
review
|
|
61266
|
+
}) {
|
|
61135
61267
|
const options = yield* root;
|
|
61136
61268
|
const token = exports_Option.getOrUndefined(options.token);
|
|
61137
61269
|
const operation = exports_Effect.gen(function* () {
|
|
61138
|
-
const solution = yield*
|
|
61270
|
+
const solution = yield* challengeEvaluationInput(exports_Option.getOrUndefined(source), exports_Option.getOrUndefined(review), challenge);
|
|
61139
61271
|
return yield* evaluateChallenge({ apiUrl: options.apiUrl, token }, challenge, solution);
|
|
61140
61272
|
});
|
|
61141
61273
|
yield* execute(options.output, operation, challengeEvaluateText);
|
|
61142
61274
|
})).pipe(exports_Command.withDescription("Score a solution on hidden cases. This consumes one limited official evaluation. Requires sign-in."), exports_Command.withExamples([
|
|
61143
61275
|
{
|
|
61144
|
-
command: "chofex challenge evaluate --challenge broken-agent --source ./scheduler.js",
|
|
61276
|
+
command: "chofex challenge evaluate --challenge broken-agent --source ./scheduler.js --review ./review.json",
|
|
61145
61277
|
description: "Consume una evaluación oficial cuando tu solución esté lista"
|
|
61146
61278
|
}
|
|
61147
61279
|
]));
|