@warmdrift/kgauto-compiler 2.0.0-alpha.45 → 2.0.0-alpha.47

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,4 +1,15 @@
1
1
  // src/profiles.ts
2
+ var LATENCY_TIER_MS = {
3
+ fast: 4e3,
4
+ medium: 11e3,
5
+ slow: 24e3
6
+ };
7
+ function latencyTierOf(profile) {
8
+ if (profile.latencyTier) return profile.latencyTier;
9
+ if (profile.weaknesses.includes("latency")) return "slow";
10
+ if (profile.strengths.includes("speed")) return "fast";
11
+ return "medium";
12
+ }
2
13
  var ANTHROPIC_LOWERING_BASE = {
3
14
  system: { mode: "inline" },
4
15
  cache: {
@@ -545,6 +556,14 @@ var PROFILES_RAW = [
545
556
  ],
546
557
  strengths: ["cost", "1m_context", "json_output", "code", "reasoning"],
547
558
  weaknesses: ["parallel_tools", "large_tool_sets"],
559
+ // alpha.47 — explicit slow override. Tag derivation would say 'medium'
560
+ // (no 'latency' weakness, no 'speed' strength), but the alpha.46 shadow
561
+ // probe MEASURED deepseek-v4-flash at 20485ms served (2026-06-03) — ~2.3×
562
+ // gemini-2.5-flash on PB's synchronous /api/analyze path. The 'flash' name
563
+ // is DeepSeek's, not a speed promise. This is the row that, scoring 0.85
564
+ // baseQuality (it carries 'reasoning') with no latency counterweight,
565
+ // leapfrogged gemini-2.5-flash as PB's summarize leader once reachable.
566
+ latencyTier: "slow",
548
567
  notes: "Cheap workhorse. 1M context, 384k max output. Cache-hit input $0.0028/M (1/50\xD7 of miss). Aliased as `deepseek-chat` (non-thinking) and `deepseek-reasoner` (thinking) \u2014 see ALIASES.",
549
568
  // Master plan §6.2 anchor. Brain-validated tier 1 cross-provider for
550
569
  // classify (169 rows, 0% empty). Tier 0 for summarize-with-no-tools.
@@ -612,6 +631,11 @@ var PROFILES_RAW = [
612
631
  ],
613
632
  strengths: ["quality", "reasoning", "1m_context", "json_output", "code", "extended_thinking"],
614
633
  weaknesses: ["parallel_tools", "large_tool_sets"],
634
+ // alpha.47 — explicit slow override. Measured 47722ms on the alpha.46
635
+ // shadow probe (2026-06-03) — it's an extended-thinking reasoner, slowest
636
+ // of the served set. Tag derivation would say 'medium'; the measurement says
637
+ // otherwise.
638
+ latencyTier: "slow",
615
639
  notes: "Pro tier. 1M context, 384k max output. Regular pricing $1.74/$3.48; 75% promo through 2026-05-31 ($0.435/$0.87). Default mode = thinking.",
616
640
  // Master plan §3.3: tier 3 cross-provider for plan chain. Reasoning
617
641
  // bumped one notch over V4-Flash; same parallel-tool ceiling.
@@ -1264,6 +1288,8 @@ function profilesByProvider(provider) {
1264
1288
  }
1265
1289
 
1266
1290
  export {
1291
+ LATENCY_TIER_MS,
1292
+ latencyTierOf,
1267
1293
  ALIASES,
1268
1294
  _setProfileBrainHook,
1269
1295
  getProfile,
@@ -1,6 +1,6 @@
1
1
  import {
2
2
  tryGetProfile
3
- } from "./chunk-ZHUD3I52.mjs";
3
+ } from "./chunk-PKOFXEB3.mjs";
4
4
 
5
5
  // src/brain-query.ts
6
6
  var FRESH_SNAPSHOT = {
@@ -610,6 +610,14 @@ var PROFILES_RAW = [
610
610
  ],
611
611
  strengths: ["cost", "1m_context", "json_output", "code", "reasoning"],
612
612
  weaknesses: ["parallel_tools", "large_tool_sets"],
613
+ // alpha.47 — explicit slow override. Tag derivation would say 'medium'
614
+ // (no 'latency' weakness, no 'speed' strength), but the alpha.46 shadow
615
+ // probe MEASURED deepseek-v4-flash at 20485ms served (2026-06-03) — ~2.3×
616
+ // gemini-2.5-flash on PB's synchronous /api/analyze path. The 'flash' name
617
+ // is DeepSeek's, not a speed promise. This is the row that, scoring 0.85
618
+ // baseQuality (it carries 'reasoning') with no latency counterweight,
619
+ // leapfrogged gemini-2.5-flash as PB's summarize leader once reachable.
620
+ latencyTier: "slow",
613
621
  notes: "Cheap workhorse. 1M context, 384k max output. Cache-hit input $0.0028/M (1/50\xD7 of miss). Aliased as `deepseek-chat` (non-thinking) and `deepseek-reasoner` (thinking) \u2014 see ALIASES.",
614
622
  // Master plan §6.2 anchor. Brain-validated tier 1 cross-provider for
615
623
  // classify (169 rows, 0% empty). Tier 0 for summarize-with-no-tools.
@@ -677,6 +685,11 @@ var PROFILES_RAW = [
677
685
  ],
678
686
  strengths: ["quality", "reasoning", "1m_context", "json_output", "code", "extended_thinking"],
679
687
  weaknesses: ["parallel_tools", "large_tool_sets"],
688
+ // alpha.47 — explicit slow override. Measured 47722ms on the alpha.46
689
+ // shadow probe (2026-06-03) — it's an extended-thinking reasoner, slowest
690
+ // of the served set. Tag derivation would say 'medium'; the measurement says
691
+ // otherwise.
692
+ latencyTier: "slow",
680
693
  notes: "Pro tier. 1M context, 384k max output. Regular pricing $1.74/$3.48; 75% promo through 2026-05-31 ($0.435/$0.87). Default mode = thinking.",
681
694
  // Master plan §3.3: tier 3 cross-provider for plan chain. Reasoning
682
695
  // bumped one notch over V4-Flash; same parallel-tool ceiling.
@@ -1,10 +1,10 @@
1
1
  import {
2
2
  ARCHETYPE_FLOOR_DEFAULT,
3
3
  getDefaultFallbackChain
4
- } from "../chunk-6MSJQCAP.mjs";
4
+ } from "../chunk-SAWTKMD4.mjs";
5
5
  import {
6
6
  tryGetProfile
7
- } from "../chunk-ZHUD3I52.mjs";
7
+ } from "../chunk-PKOFXEB3.mjs";
8
8
  import {
9
9
  subscribe,
10
10
  subscribeApp
package/dist/index.d.mts CHANGED
@@ -1,7 +1,7 @@
1
1
  import { C as CompilePolicy, N as NormalizedResponse, A as ApiKeys, P as ProviderOverrides, a as CompiledRequest, b as PromptIR, c as CallOptions, d as CallResult, S as SystemModelMessage, e as CompileResult, f as SectionRewrite, R as RecordInput, g as RecordOutcomeInput, O as OutcomeResult, h as OracleScore, B as BestPracticeAdvisory, i as Adapter, j as PerAxisMetrics, k as Provider, l as ChainEntry, G as Grounding } from './ir-BBKgsUX7.mjs';
2
2
  export { m as CallAttempt, n as CallError, o as ChainModelEntry, p as ChainWithGrounding, q as Constraints, F as FallbackReason, H as HistoryCachePolicy, I as IntentDeclaration, M as Message, r as MutationApplied, s as NormalizedTokens, t as OutcomeKind, u as PerAxisMetricsByModel, v as PromptSection, w as SectionKind, x as ShadowProbeConfig, T as ToolCall, y as ToolDefinition } from './ir-BBKgsUX7.mjs';
3
3
  import { ModelProfile, ArchetypeConvention } from './profiles.mjs';
4
- export { ALIASES, CacheStrategy, CliffRule, LoweringSpec, RecoveryRule, StructuredOutputCapability, SystemPromptMode, allProfiles, getProfile, profilesByProvider, tryGetProfile } from './profiles.mjs';
4
+ export { ALIASES, CacheStrategy, CliffRule, LATENCY_TIER_MS, LatencyTier, LoweringSpec, RecoveryRule, StructuredOutputCapability, SystemPromptMode, allProfiles, getProfile, latencyTierOf, profilesByProvider, tryGetProfile } from './profiles.mjs';
5
5
  import { IntentArchetypeName } from './dialect.mjs';
6
6
  export { ALL_ARCHETYPES, ContextBucket, DIALECT_VERSION, HistoryDepth, INTENT_ARCHETYPES, OutputMode, ShapeSignature, ToolCountBucket, bucketContext, bucketHistory, bucketToolCount, hashShape, isArchetype, learningKey } from './dialect.mjs';
7
7
 
@@ -363,6 +363,15 @@ interface ShadowProbeRecordInput {
363
363
  tokensCurrentOut?: number;
364
364
  tokensCandidateIn?: number;
365
365
  tokensCandidateOut?: number;
366
+ /**
367
+ * Wall-clock latency of the served call, ms (alpha.46). The third swap-decision
368
+ * axis alongside quality + cost — a verdict needs to know the candidate is
369
+ * slower, not just better/cheaper. Served latency mirrors
370
+ * `compile_outcomes.latency_ms`; carried here so a probe row is self-contained.
371
+ */
372
+ latencyCurrentMs?: number;
373
+ /** Wall-clock latency of the candidate replay call, ms (alpha.46). */
374
+ latencyCandidateMs?: number;
366
375
  }
367
376
  /**
368
377
  * Pure builder for the `probe_outcomes` row written by the inline shadow-probe.
@@ -390,6 +399,8 @@ declare function buildShadowProbeRow(input: ShadowProbeRecordInput): {
390
399
  tokens_current_out: number | null;
391
400
  tokens_candidate_in: number | null;
392
401
  tokens_candidate_out: number | null;
402
+ latency_current_ms: number | null;
403
+ latency_candidate_ms: number | null;
393
404
  prompt_fidelity: number;
394
405
  replay_source: 'inline-full-ir';
395
406
  };
package/dist/index.d.ts CHANGED
@@ -1,7 +1,7 @@
1
1
  import { C as CompilePolicy, N as NormalizedResponse, A as ApiKeys, P as ProviderOverrides, a as CompiledRequest, b as PromptIR, c as CallOptions, d as CallResult, S as SystemModelMessage, e as CompileResult, f as SectionRewrite, R as RecordInput, g as RecordOutcomeInput, O as OutcomeResult, h as OracleScore, B as BestPracticeAdvisory, i as Adapter, j as PerAxisMetrics, k as Provider, l as ChainEntry, G as Grounding } from './ir-D3n-pBYI.js';
2
2
  export { m as CallAttempt, n as CallError, o as ChainModelEntry, p as ChainWithGrounding, q as Constraints, F as FallbackReason, H as HistoryCachePolicy, I as IntentDeclaration, M as Message, r as MutationApplied, s as NormalizedTokens, t as OutcomeKind, u as PerAxisMetricsByModel, v as PromptSection, w as SectionKind, x as ShadowProbeConfig, T as ToolCall, y as ToolDefinition } from './ir-D3n-pBYI.js';
3
3
  import { ModelProfile, ArchetypeConvention } from './profiles.js';
4
- export { ALIASES, CacheStrategy, CliffRule, LoweringSpec, RecoveryRule, StructuredOutputCapability, SystemPromptMode, allProfiles, getProfile, profilesByProvider, tryGetProfile } from './profiles.js';
4
+ export { ALIASES, CacheStrategy, CliffRule, LATENCY_TIER_MS, LatencyTier, LoweringSpec, RecoveryRule, StructuredOutputCapability, SystemPromptMode, allProfiles, getProfile, latencyTierOf, profilesByProvider, tryGetProfile } from './profiles.js';
5
5
  import { IntentArchetypeName } from './dialect.js';
6
6
  export { ALL_ARCHETYPES, ContextBucket, DIALECT_VERSION, HistoryDepth, INTENT_ARCHETYPES, OutputMode, ShapeSignature, ToolCountBucket, bucketContext, bucketHistory, bucketToolCount, hashShape, isArchetype, learningKey } from './dialect.js';
7
7
 
@@ -363,6 +363,15 @@ interface ShadowProbeRecordInput {
363
363
  tokensCurrentOut?: number;
364
364
  tokensCandidateIn?: number;
365
365
  tokensCandidateOut?: number;
366
+ /**
367
+ * Wall-clock latency of the served call, ms (alpha.46). The third swap-decision
368
+ * axis alongside quality + cost — a verdict needs to know the candidate is
369
+ * slower, not just better/cheaper. Served latency mirrors
370
+ * `compile_outcomes.latency_ms`; carried here so a probe row is self-contained.
371
+ */
372
+ latencyCurrentMs?: number;
373
+ /** Wall-clock latency of the candidate replay call, ms (alpha.46). */
374
+ latencyCandidateMs?: number;
366
375
  }
367
376
  /**
368
377
  * Pure builder for the `probe_outcomes` row written by the inline shadow-probe.
@@ -390,6 +399,8 @@ declare function buildShadowProbeRow(input: ShadowProbeRecordInput): {
390
399
  tokens_current_out: number | null;
391
400
  tokens_candidate_in: number | null;
392
401
  tokens_candidate_out: number | null;
402
+ latency_current_ms: number | null;
403
+ latency_candidate_ms: number | null;
393
404
  prompt_fidelity: number;
394
405
  replay_source: 'inline-full-ir';
395
406
  };
package/dist/index.js CHANGED
@@ -30,6 +30,7 @@ __export(index_exports, {
30
30
  DIALECT_VERSION: () => DIALECT_VERSION,
31
31
  FamilyResolutionError: () => FamilyResolutionError,
32
32
  INTENT_ARCHETYPES: () => INTENT_ARCHETYPES,
33
+ LATENCY_TIER_MS: () => LATENCY_TIER_MS,
33
34
  MEASURED_GROUNDING_MIN_N: () => MEASURED_GROUNDING_MIN_N,
34
35
  PRODUCER_OWNED_RULE_CODES: () => PRODUCER_OWNED_RULE_CODES,
35
36
  PROVIDER_ENV_KEYS: () => PROVIDER_ENV_KEYS,
@@ -76,6 +77,7 @@ __export(index_exports, {
76
77
  isExclusionFindingsBrainActive: () => isExclusionFindingsBrainActive,
77
78
  isModelReachable: () => isModelReachable,
78
79
  isProviderReachable: () => isProviderReachable,
80
+ latencyTierOf: () => latencyTierOf,
79
81
  learningKey: () => learningKey,
80
82
  loadAliasesFromBrain: () => loadAliasesFromBrain,
81
83
  loadArchetypePerfFromBrain: () => loadArchetypePerfFromBrain,
@@ -274,6 +276,17 @@ function mapPerAxisMetrics(raw, fallbackAppId, fallbackArchetype, fallbackModel,
274
276
  }
275
277
 
276
278
  // src/profiles.ts
279
+ var LATENCY_TIER_MS = {
280
+ fast: 4e3,
281
+ medium: 11e3,
282
+ slow: 24e3
283
+ };
284
+ function latencyTierOf(profile) {
285
+ if (profile.latencyTier) return profile.latencyTier;
286
+ if (profile.weaknesses.includes("latency")) return "slow";
287
+ if (profile.strengths.includes("speed")) return "fast";
288
+ return "medium";
289
+ }
277
290
  var ANTHROPIC_LOWERING_BASE = {
278
291
  system: { mode: "inline" },
279
292
  cache: {
@@ -820,6 +833,14 @@ var PROFILES_RAW = [
820
833
  ],
821
834
  strengths: ["cost", "1m_context", "json_output", "code", "reasoning"],
822
835
  weaknesses: ["parallel_tools", "large_tool_sets"],
836
+ // alpha.47 — explicit slow override. Tag derivation would say 'medium'
837
+ // (no 'latency' weakness, no 'speed' strength), but the alpha.46 shadow
838
+ // probe MEASURED deepseek-v4-flash at 20485ms served (2026-06-03) — ~2.3×
839
+ // gemini-2.5-flash on PB's synchronous /api/analyze path. The 'flash' name
840
+ // is DeepSeek's, not a speed promise. This is the row that, scoring 0.85
841
+ // baseQuality (it carries 'reasoning') with no latency counterweight,
842
+ // leapfrogged gemini-2.5-flash as PB's summarize leader once reachable.
843
+ latencyTier: "slow",
823
844
  notes: "Cheap workhorse. 1M context, 384k max output. Cache-hit input $0.0028/M (1/50\xD7 of miss). Aliased as `deepseek-chat` (non-thinking) and `deepseek-reasoner` (thinking) \u2014 see ALIASES.",
824
845
  // Master plan §6.2 anchor. Brain-validated tier 1 cross-provider for
825
846
  // classify (169 rows, 0% empty). Tier 0 for summarize-with-no-tools.
@@ -887,6 +908,11 @@ var PROFILES_RAW = [
887
908
  ],
888
909
  strengths: ["quality", "reasoning", "1m_context", "json_output", "code", "extended_thinking"],
889
910
  weaknesses: ["parallel_tools", "large_tool_sets"],
911
+ // alpha.47 — explicit slow override. Measured 47722ms on the alpha.46
912
+ // shadow probe (2026-06-03) — it's an extended-thinking reasoner, slowest
913
+ // of the served set. Tag derivation would say 'medium'; the measurement says
914
+ // otherwise.
915
+ latencyTier: "slow",
890
916
  notes: "Pro tier. 1M context, 384k max output. Regular pricing $1.74/$3.48; 75% promo through 2026-05-31 ($0.435/$0.87). Default mode = thinking.",
891
917
  // Master plan §3.3: tier 3 cross-provider for plan chain. Reasoning
892
918
  // bumped one notch over V4-Flash; same parallel-tool ceiling.
@@ -2360,6 +2386,8 @@ function passApplyCliffs(ir, profile, estimatedInputTokens) {
2360
2386
  }
2361
2387
  return { value: { ir: nextIR, loweringHints: hints }, mutations };
2362
2388
  }
2389
+ var LATENCY_OVERAGE_WEIGHT = 0.6;
2390
+ var LATENCY_PENALTY_CAP = 1;
2363
2391
  function passScoreTargets(ir, opts) {
2364
2392
  const constraints = ir.constraints ?? {};
2365
2393
  const policy = opts.policy ?? {};
@@ -2415,7 +2443,16 @@ function passScoreTargets(ir, opts) {
2415
2443
  const callerOrderBoost = (modelIds.length - modelIds.indexOf(modelId)) * 0.1;
2416
2444
  const costPenalty = estimatedCostUsd * 5;
2417
2445
  const preferredBoost = preferredSet.has(modelId) ? 0.5 : 0;
2418
- const rank = qualityScore + callerOrderBoost - costPenalty - reasons.length * 10 + preferredBoost;
2446
+ let latencyPenalty = 0;
2447
+ const maxLatencyMs = constraints.maxLatencyMs;
2448
+ if (typeof maxLatencyMs === "number" && maxLatencyMs > 0) {
2449
+ const tierMs = LATENCY_TIER_MS[latencyTierOf(profile)];
2450
+ if (tierMs > maxLatencyMs) {
2451
+ const overage = (tierMs - maxLatencyMs) / maxLatencyMs;
2452
+ latencyPenalty = Math.min(LATENCY_OVERAGE_WEIGHT * overage, LATENCY_PENALTY_CAP);
2453
+ }
2454
+ }
2455
+ const rank = qualityScore + callerOrderBoost - costPenalty - reasons.length * 10 + preferredBoost - latencyPenalty;
2419
2456
  scores.push({
2420
2457
  modelId,
2421
2458
  estimatedCostUsd,
@@ -2448,6 +2485,15 @@ function passScoreTargets(ir, opts) {
2448
2485
  description: `Model ${modelId} rank boosted by CompilePolicy.preferredModels`
2449
2486
  });
2450
2487
  }
2488
+ if (latencyPenalty > 0) {
2489
+ const tier = latencyTierOf(profile);
2490
+ policyMutations.push({
2491
+ id: `latency-downrank-${modelId}`,
2492
+ source: "latency_guard",
2493
+ passName: "score_targets",
2494
+ description: `Model ${modelId} rank down ${latencyPenalty.toFixed(2)} \u2014 latency tier '${tier}' (~${LATENCY_TIER_MS[tier]}ms) exceeds constraints.maxLatencyMs (${maxLatencyMs}ms)`
2495
+ });
2496
+ }
2451
2497
  }
2452
2498
  return { value: scores, mutations: policyMutations };
2453
2499
  }
@@ -4494,6 +4540,8 @@ function buildShadowProbeRow(input) {
4494
4540
  tokens_current_out: input.tokensCurrentOut ?? null,
4495
4541
  tokens_candidate_in: input.tokensCandidateIn ?? null,
4496
4542
  tokens_candidate_out: input.tokensCandidateOut ?? null,
4543
+ latency_current_ms: input.latencyCurrentMs ?? null,
4544
+ latency_candidate_ms: input.latencyCandidateMs ?? null,
4497
4545
  // Full IR was replayed (not a truncated preview), so fidelity is 1.0 — the
4498
4546
  // prompt-fidelity guard never fires on these rows.
4499
4547
  prompt_fidelity: 1,
@@ -6014,7 +6062,8 @@ async function call(ir, opts = {}) {
6014
6062
  ir,
6015
6063
  opts,
6016
6064
  servedModel: targetModel,
6017
- servedResponse: validated.response
6065
+ servedResponse: validated.response,
6066
+ servedLatencyMs: latencyMs2
6018
6067
  });
6019
6068
  if (isBrainSync()) {
6020
6069
  await probe;
@@ -6127,11 +6176,13 @@ async function runShadowProbe(args) {
6127
6176
  },
6128
6177
  args.opts
6129
6178
  );
6179
+ const candStart = Date.now();
6130
6180
  const exec = await execute(candCompile.request, {
6131
6181
  apiKeys: args.opts.apiKeys,
6132
6182
  fetchImpl: args.opts.fetchImpl,
6133
6183
  providerOverrides: args.opts.providerOverrides
6134
6184
  });
6185
+ const candidateLatencyMs = Date.now() - candStart;
6135
6186
  if (!exec.ok) continue;
6136
6187
  await recordShadowProbe({
6137
6188
  appId: args.ir.appId,
@@ -6145,7 +6196,12 @@ async function runShadowProbe(args) {
6145
6196
  tokensCurrentIn: args.servedResponse.tokens.input,
6146
6197
  tokensCurrentOut: args.servedResponse.tokens.output,
6147
6198
  tokensCandidateIn: exec.response.tokens.input,
6148
- tokensCandidateOut: exec.response.tokens.output
6199
+ tokensCandidateOut: exec.response.tokens.output,
6200
+ // alpha.46 — speed is the third swap-decision axis. Served latency
6201
+ // mirrors the user's actual wait; candidate latency is measured only
6202
+ // here and stored nowhere else (irrecoverable if not captured now).
6203
+ latencyCurrentMs: args.servedLatencyMs,
6204
+ latencyCandidateMs: candidateLatencyMs
6149
6205
  });
6150
6206
  } catch {
6151
6207
  }
@@ -6681,6 +6737,7 @@ function compile2(ir, opts) {
6681
6737
  DIALECT_VERSION,
6682
6738
  FamilyResolutionError,
6683
6739
  INTENT_ARCHETYPES,
6740
+ LATENCY_TIER_MS,
6684
6741
  MEASURED_GROUNDING_MIN_N,
6685
6742
  PRODUCER_OWNED_RULE_CODES,
6686
6743
  PROVIDER_ENV_KEYS,
@@ -6727,6 +6784,7 @@ function compile2(ir, opts) {
6727
6784
  isExclusionFindingsBrainActive,
6728
6785
  isModelReachable,
6729
6786
  isProviderReachable,
6787
+ latencyTierOf,
6730
6788
  learningKey,
6731
6789
  loadAliasesFromBrain,
6732
6790
  loadArchetypePerfFromBrain,
package/dist/index.mjs CHANGED
@@ -33,16 +33,18 @@ import {
33
33
  loadChainsFromBrain,
34
34
  readBrainReadEnv,
35
35
  resolveProviderKey
36
- } from "./chunk-6MSJQCAP.mjs";
36
+ } from "./chunk-SAWTKMD4.mjs";
37
37
  import {
38
38
  ALIASES,
39
+ LATENCY_TIER_MS,
39
40
  _setProfileBrainHook,
40
41
  allProfiles,
41
42
  allProfilesRaw,
42
43
  getProfile,
44
+ latencyTierOf,
43
45
  profilesByProvider,
44
46
  tryGetProfile
45
- } from "./chunk-ZHUD3I52.mjs";
47
+ } from "./chunk-PKOFXEB3.mjs";
46
48
  import {
47
49
  emitAdvisoryFired,
48
50
  emitCompileDone,
@@ -710,6 +712,8 @@ function passApplyCliffs(ir, profile, estimatedInputTokens) {
710
712
  }
711
713
  return { value: { ir: nextIR, loweringHints: hints }, mutations };
712
714
  }
715
+ var LATENCY_OVERAGE_WEIGHT = 0.6;
716
+ var LATENCY_PENALTY_CAP = 1;
713
717
  function passScoreTargets(ir, opts) {
714
718
  const constraints = ir.constraints ?? {};
715
719
  const policy = opts.policy ?? {};
@@ -765,7 +769,16 @@ function passScoreTargets(ir, opts) {
765
769
  const callerOrderBoost = (modelIds.length - modelIds.indexOf(modelId)) * 0.1;
766
770
  const costPenalty = estimatedCostUsd * 5;
767
771
  const preferredBoost = preferredSet.has(modelId) ? 0.5 : 0;
768
- const rank = qualityScore + callerOrderBoost - costPenalty - reasons.length * 10 + preferredBoost;
772
+ let latencyPenalty = 0;
773
+ const maxLatencyMs = constraints.maxLatencyMs;
774
+ if (typeof maxLatencyMs === "number" && maxLatencyMs > 0) {
775
+ const tierMs = LATENCY_TIER_MS[latencyTierOf(profile)];
776
+ if (tierMs > maxLatencyMs) {
777
+ const overage = (tierMs - maxLatencyMs) / maxLatencyMs;
778
+ latencyPenalty = Math.min(LATENCY_OVERAGE_WEIGHT * overage, LATENCY_PENALTY_CAP);
779
+ }
780
+ }
781
+ const rank = qualityScore + callerOrderBoost - costPenalty - reasons.length * 10 + preferredBoost - latencyPenalty;
769
782
  scores.push({
770
783
  modelId,
771
784
  estimatedCostUsd,
@@ -798,6 +811,15 @@ function passScoreTargets(ir, opts) {
798
811
  description: `Model ${modelId} rank boosted by CompilePolicy.preferredModels`
799
812
  });
800
813
  }
814
+ if (latencyPenalty > 0) {
815
+ const tier = latencyTierOf(profile);
816
+ policyMutations.push({
817
+ id: `latency-downrank-${modelId}`,
818
+ source: "latency_guard",
819
+ passName: "score_targets",
820
+ description: `Model ${modelId} rank down ${latencyPenalty.toFixed(2)} \u2014 latency tier '${tier}' (~${LATENCY_TIER_MS[tier]}ms) exceeds constraints.maxLatencyMs (${maxLatencyMs}ms)`
821
+ });
822
+ }
801
823
  }
802
824
  return { value: scores, mutations: policyMutations };
803
825
  }
@@ -2844,6 +2866,8 @@ function buildShadowProbeRow(input) {
2844
2866
  tokens_current_out: input.tokensCurrentOut ?? null,
2845
2867
  tokens_candidate_in: input.tokensCandidateIn ?? null,
2846
2868
  tokens_candidate_out: input.tokensCandidateOut ?? null,
2869
+ latency_current_ms: input.latencyCurrentMs ?? null,
2870
+ latency_candidate_ms: input.latencyCandidateMs ?? null,
2847
2871
  // Full IR was replayed (not a truncated preview), so fidelity is 1.0 — the
2848
2872
  // prompt-fidelity guard never fires on these rows.
2849
2873
  prompt_fidelity: 1,
@@ -3649,7 +3673,8 @@ async function call(ir, opts = {}) {
3649
3673
  ir,
3650
3674
  opts,
3651
3675
  servedModel: targetModel,
3652
- servedResponse: validated.response
3676
+ servedResponse: validated.response,
3677
+ servedLatencyMs: latencyMs2
3653
3678
  });
3654
3679
  if (isBrainSync()) {
3655
3680
  await probe;
@@ -3762,11 +3787,13 @@ async function runShadowProbe(args) {
3762
3787
  },
3763
3788
  args.opts
3764
3789
  );
3790
+ const candStart = Date.now();
3765
3791
  const exec = await execute(candCompile.request, {
3766
3792
  apiKeys: args.opts.apiKeys,
3767
3793
  fetchImpl: args.opts.fetchImpl,
3768
3794
  providerOverrides: args.opts.providerOverrides
3769
3795
  });
3796
+ const candidateLatencyMs = Date.now() - candStart;
3770
3797
  if (!exec.ok) continue;
3771
3798
  await recordShadowProbe({
3772
3799
  appId: args.ir.appId,
@@ -3780,7 +3807,12 @@ async function runShadowProbe(args) {
3780
3807
  tokensCurrentIn: args.servedResponse.tokens.input,
3781
3808
  tokensCurrentOut: args.servedResponse.tokens.output,
3782
3809
  tokensCandidateIn: exec.response.tokens.input,
3783
- tokensCandidateOut: exec.response.tokens.output
3810
+ tokensCandidateOut: exec.response.tokens.output,
3811
+ // alpha.46 — speed is the third swap-decision axis. Served latency
3812
+ // mirrors the user's actual wait; candidate latency is measured only
3813
+ // here and stored nowhere else (irrecoverable if not captured now).
3814
+ latencyCurrentMs: args.servedLatencyMs,
3815
+ latencyCandidateMs: candidateLatencyMs
3784
3816
  });
3785
3817
  } catch {
3786
3818
  }
@@ -4315,6 +4347,7 @@ export {
4315
4347
  DIALECT_VERSION,
4316
4348
  FamilyResolutionError,
4317
4349
  INTENT_ARCHETYPES,
4350
+ LATENCY_TIER_MS,
4318
4351
  MEASURED_GROUNDING_MIN_N,
4319
4352
  PRODUCER_OWNED_RULE_CODES,
4320
4353
  PROVIDER_ENV_KEYS,
@@ -4361,6 +4394,7 @@ export {
4361
4394
  isExclusionFindingsBrainActive,
4362
4395
  isModelReachable,
4363
4396
  isProviderReachable,
4397
+ latencyTierOf,
4364
4398
  learningKey,
4365
4399
  loadAliasesFromBrain,
4366
4400
  loadArchetypePerfFromBrain,
@@ -145,6 +145,14 @@ interface LoweringSpec {
145
145
  default?: number | 'auto' | 'off';
146
146
  };
147
147
  }
148
+ /**
149
+ * Coarse latency bucket for a model. alpha.47 — the third swap axis
150
+ * (cost / quality / SPEED). We bucket rather than store per-model ms because
151
+ * served latency varies with token count; honest precision is the tier, not a
152
+ * false-precise number. See {@link LATENCY_TIER_MS} for the representative ms
153
+ * each tier maps to when compared against `constraints.maxLatencyMs`.
154
+ */
155
+ type LatencyTier = 'fast' | 'medium' | 'slow';
148
156
  interface ModelProfile {
149
157
  id: string;
150
158
  provider: Provider;
@@ -163,6 +171,16 @@ interface ModelProfile {
163
171
  recovery: RecoveryRule[];
164
172
  strengths: string[];
165
173
  weaknesses: string[];
174
+ /**
175
+ * alpha.47 — explicit latency bucket. OPTIONAL: when unset, `latencyTierOf`
176
+ * derives it from the tags above (`weaknesses` includes `'latency'` → slow,
177
+ * `strengths` includes `'speed'` → fast, else medium). Set this explicitly
178
+ * ONLY when measured evidence contradicts the tag derivation — e.g.
179
+ * `deepseek-v4-flash` carries no `'latency'` weakness yet measures ~20s
180
+ * (alpha.46 shadow-probe). Carry provenance in an inline comment when you
181
+ * override, same discipline as capability data (step zero / L-081).
182
+ */
183
+ latencyTier?: LatencyTier;
166
184
  notes?: string;
167
185
  verifiedAgainstDocs?: string;
168
186
  /**
@@ -236,6 +254,26 @@ interface ModelProfile {
236
254
  */
237
255
  archetypeConventions?: ArchetypeConvention[];
238
256
  }
257
+ /**
258
+ * Representative p50 latency (ms) per tier. Coarse on purpose — used only to
259
+ * compare a model against `constraints.maxLatencyMs`, never reported as a
260
+ * per-model number. Grounded: alpha.46 shadow-probe measured
261
+ * `deepseek-v4-flash` at 20485ms and `deepseek-v4-pro` at 47722ms (2026-06-03),
262
+ * both bucket `slow`; gemini-2.5-flash / haiku-4-5 serve in single-digit
263
+ * seconds → `fast`; mid-tier (sonnet, gemini-pro) → `medium`. Phase 2 can swap
264
+ * these buckets for measured per-(archetype,model) p50 from
265
+ * `compile_outcomes.latency_ms` once enough served rows accumulate.
266
+ */
267
+ declare const LATENCY_TIER_MS: Record<LatencyTier, number>;
268
+ /**
269
+ * Resolve a model's latency tier. Explicit `profile.latencyTier` wins;
270
+ * otherwise derive from the tags already on the profile so we don't maintain a
271
+ * second source of truth that can silently disagree (L-073 family):
272
+ * - `weaknesses` includes `'latency'` → `'slow'`
273
+ * - `strengths` includes `'speed'` → `'fast'`
274
+ * - else → `'medium'`
275
+ */
276
+ declare function latencyTierOf(profile: ModelProfile): LatencyTier;
239
277
  declare const ALIASES: Record<string, string>;
240
278
  interface ProfileBrainHook {
241
279
  getProfile?: (canonicalId: string) => ModelProfile | undefined;
@@ -251,4 +289,4 @@ declare function allProfiles(): readonly ModelProfile[];
251
289
  declare function allProfilesRaw(): readonly ModelProfile[];
252
290
  declare function profilesByProvider(provider: Provider): readonly ModelProfile[];
253
291
 
254
- export { ALIASES, type ArchetypeConvention, type CacheStrategy, type CliffRule, type LoweringSpec, type ModelProfile, type RecoveryRule, type StructuredOutputCapability, type SystemPromptMode, _setProfileBrainHook, allProfiles, allProfilesRaw, getProfile, profilesByProvider, tryGetProfile };
292
+ export { ALIASES, type ArchetypeConvention, type CacheStrategy, type CliffRule, LATENCY_TIER_MS, type LatencyTier, type LoweringSpec, type ModelProfile, type RecoveryRule, type StructuredOutputCapability, type SystemPromptMode, _setProfileBrainHook, allProfiles, allProfilesRaw, getProfile, latencyTierOf, profilesByProvider, tryGetProfile };
@@ -145,6 +145,14 @@ interface LoweringSpec {
145
145
  default?: number | 'auto' | 'off';
146
146
  };
147
147
  }
148
+ /**
149
+ * Coarse latency bucket for a model. alpha.47 — the third swap axis
150
+ * (cost / quality / SPEED). We bucket rather than store per-model ms because
151
+ * served latency varies with token count; honest precision is the tier, not a
152
+ * false-precise number. See {@link LATENCY_TIER_MS} for the representative ms
153
+ * each tier maps to when compared against `constraints.maxLatencyMs`.
154
+ */
155
+ type LatencyTier = 'fast' | 'medium' | 'slow';
148
156
  interface ModelProfile {
149
157
  id: string;
150
158
  provider: Provider;
@@ -163,6 +171,16 @@ interface ModelProfile {
163
171
  recovery: RecoveryRule[];
164
172
  strengths: string[];
165
173
  weaknesses: string[];
174
+ /**
175
+ * alpha.47 — explicit latency bucket. OPTIONAL: when unset, `latencyTierOf`
176
+ * derives it from the tags above (`weaknesses` includes `'latency'` → slow,
177
+ * `strengths` includes `'speed'` → fast, else medium). Set this explicitly
178
+ * ONLY when measured evidence contradicts the tag derivation — e.g.
179
+ * `deepseek-v4-flash` carries no `'latency'` weakness yet measures ~20s
180
+ * (alpha.46 shadow-probe). Carry provenance in an inline comment when you
181
+ * override, same discipline as capability data (step zero / L-081).
182
+ */
183
+ latencyTier?: LatencyTier;
166
184
  notes?: string;
167
185
  verifiedAgainstDocs?: string;
168
186
  /**
@@ -236,6 +254,26 @@ interface ModelProfile {
236
254
  */
237
255
  archetypeConventions?: ArchetypeConvention[];
238
256
  }
257
+ /**
258
+ * Representative p50 latency (ms) per tier. Coarse on purpose — used only to
259
+ * compare a model against `constraints.maxLatencyMs`, never reported as a
260
+ * per-model number. Grounded: alpha.46 shadow-probe measured
261
+ * `deepseek-v4-flash` at 20485ms and `deepseek-v4-pro` at 47722ms (2026-06-03),
262
+ * both bucket `slow`; gemini-2.5-flash / haiku-4-5 serve in single-digit
263
+ * seconds → `fast`; mid-tier (sonnet, gemini-pro) → `medium`. Phase 2 can swap
264
+ * these buckets for measured per-(archetype,model) p50 from
265
+ * `compile_outcomes.latency_ms` once enough served rows accumulate.
266
+ */
267
+ declare const LATENCY_TIER_MS: Record<LatencyTier, number>;
268
+ /**
269
+ * Resolve a model's latency tier. Explicit `profile.latencyTier` wins;
270
+ * otherwise derive from the tags already on the profile so we don't maintain a
271
+ * second source of truth that can silently disagree (L-073 family):
272
+ * - `weaknesses` includes `'latency'` → `'slow'`
273
+ * - `strengths` includes `'speed'` → `'fast'`
274
+ * - else → `'medium'`
275
+ */
276
+ declare function latencyTierOf(profile: ModelProfile): LatencyTier;
239
277
  declare const ALIASES: Record<string, string>;
240
278
  interface ProfileBrainHook {
241
279
  getProfile?: (canonicalId: string) => ModelProfile | undefined;
@@ -251,4 +289,4 @@ declare function allProfiles(): readonly ModelProfile[];
251
289
  declare function allProfilesRaw(): readonly ModelProfile[];
252
290
  declare function profilesByProvider(provider: Provider): readonly ModelProfile[];
253
291
 
254
- export { ALIASES, type ArchetypeConvention, type CacheStrategy, type CliffRule, type LoweringSpec, type ModelProfile, type RecoveryRule, type StructuredOutputCapability, type SystemPromptMode, _setProfileBrainHook, allProfiles, allProfilesRaw, getProfile, profilesByProvider, tryGetProfile };
292
+ export { ALIASES, type ArchetypeConvention, type CacheStrategy, type CliffRule, LATENCY_TIER_MS, type LatencyTier, type LoweringSpec, type ModelProfile, type RecoveryRule, type StructuredOutputCapability, type SystemPromptMode, _setProfileBrainHook, allProfiles, allProfilesRaw, getProfile, latencyTierOf, profilesByProvider, tryGetProfile };
package/dist/profiles.js CHANGED
@@ -21,14 +21,27 @@ var __toCommonJS = (mod) => __copyProps(__defProp({}, "__esModule", { value: tru
21
21
  var profiles_exports = {};
22
22
  __export(profiles_exports, {
23
23
  ALIASES: () => ALIASES,
24
+ LATENCY_TIER_MS: () => LATENCY_TIER_MS,
24
25
  _setProfileBrainHook: () => _setProfileBrainHook,
25
26
  allProfiles: () => allProfiles,
26
27
  allProfilesRaw: () => allProfilesRaw,
27
28
  getProfile: () => getProfile,
29
+ latencyTierOf: () => latencyTierOf,
28
30
  profilesByProvider: () => profilesByProvider,
29
31
  tryGetProfile: () => tryGetProfile
30
32
  });
31
33
  module.exports = __toCommonJS(profiles_exports);
34
+ var LATENCY_TIER_MS = {
35
+ fast: 4e3,
36
+ medium: 11e3,
37
+ slow: 24e3
38
+ };
39
+ function latencyTierOf(profile) {
40
+ if (profile.latencyTier) return profile.latencyTier;
41
+ if (profile.weaknesses.includes("latency")) return "slow";
42
+ if (profile.strengths.includes("speed")) return "fast";
43
+ return "medium";
44
+ }
32
45
  var ANTHROPIC_LOWERING_BASE = {
33
46
  system: { mode: "inline" },
34
47
  cache: {
@@ -575,6 +588,14 @@ var PROFILES_RAW = [
575
588
  ],
576
589
  strengths: ["cost", "1m_context", "json_output", "code", "reasoning"],
577
590
  weaknesses: ["parallel_tools", "large_tool_sets"],
591
+ // alpha.47 — explicit slow override. Tag derivation would say 'medium'
592
+ // (no 'latency' weakness, no 'speed' strength), but the alpha.46 shadow
593
+ // probe MEASURED deepseek-v4-flash at 20485ms served (2026-06-03) — ~2.3×
594
+ // gemini-2.5-flash on PB's synchronous /api/analyze path. The 'flash' name
595
+ // is DeepSeek's, not a speed promise. This is the row that, scoring 0.85
596
+ // baseQuality (it carries 'reasoning') with no latency counterweight,
597
+ // leapfrogged gemini-2.5-flash as PB's summarize leader once reachable.
598
+ latencyTier: "slow",
578
599
  notes: "Cheap workhorse. 1M context, 384k max output. Cache-hit input $0.0028/M (1/50\xD7 of miss). Aliased as `deepseek-chat` (non-thinking) and `deepseek-reasoner` (thinking) \u2014 see ALIASES.",
579
600
  // Master plan §6.2 anchor. Brain-validated tier 1 cross-provider for
580
601
  // classify (169 rows, 0% empty). Tier 0 for summarize-with-no-tools.
@@ -642,6 +663,11 @@ var PROFILES_RAW = [
642
663
  ],
643
664
  strengths: ["quality", "reasoning", "1m_context", "json_output", "code", "extended_thinking"],
644
665
  weaknesses: ["parallel_tools", "large_tool_sets"],
666
+ // alpha.47 — explicit slow override. Measured 47722ms on the alpha.46
667
+ // shadow probe (2026-06-03) — it's an extended-thinking reasoner, slowest
668
+ // of the served set. Tag derivation would say 'medium'; the measurement says
669
+ // otherwise.
670
+ latencyTier: "slow",
645
671
  notes: "Pro tier. 1M context, 384k max output. Regular pricing $1.74/$3.48; 75% promo through 2026-05-31 ($0.435/$0.87). Default mode = thinking.",
646
672
  // Master plan §3.3: tier 3 cross-provider for plan chain. Reasoning
647
673
  // bumped one notch over V4-Flash; same parallel-tool ceiling.
@@ -1295,10 +1321,12 @@ function profilesByProvider(provider) {
1295
1321
  // Annotate the CommonJS export names for ESM import in node:
1296
1322
  0 && (module.exports = {
1297
1323
  ALIASES,
1324
+ LATENCY_TIER_MS,
1298
1325
  _setProfileBrainHook,
1299
1326
  allProfiles,
1300
1327
  allProfilesRaw,
1301
1328
  getProfile,
1329
+ latencyTierOf,
1302
1330
  profilesByProvider,
1303
1331
  tryGetProfile
1304
1332
  });
package/dist/profiles.mjs CHANGED
@@ -1,18 +1,22 @@
1
1
  import {
2
2
  ALIASES,
3
+ LATENCY_TIER_MS,
3
4
  _setProfileBrainHook,
4
5
  allProfiles,
5
6
  allProfilesRaw,
6
7
  getProfile,
8
+ latencyTierOf,
7
9
  profilesByProvider,
8
10
  tryGetProfile
9
- } from "./chunk-ZHUD3I52.mjs";
11
+ } from "./chunk-PKOFXEB3.mjs";
10
12
  export {
11
13
  ALIASES,
14
+ LATENCY_TIER_MS,
12
15
  _setProfileBrainHook,
13
16
  allProfiles,
14
17
  allProfilesRaw,
15
18
  getProfile,
19
+ latencyTierOf,
16
20
  profilesByProvider,
17
21
  tryGetProfile
18
22
  };
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@warmdrift/kgauto-compiler",
3
- "version": "2.0.0-alpha.45",
3
+ "version": "2.0.0-alpha.47",
4
4
  "description": "Prompt compiler + central learning brain for multi-model AI apps. Swap models without rewriting prompts.",
5
5
  "main": "./dist/index.js",
6
6
  "module": "./dist/index.mjs",