@ngockhoale/ukit 1.5.24 → 1.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ngockhoale/ukit",
3
- "version": "1.5.24",
3
+ "version": "1.6.1",
4
4
  "description": "Install/update an index-first AI workspace for Claude Code, Antigravity, OpenAI Codex, and OpenCode.",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -6,8 +6,9 @@ import { compactContextBlock } from './index.js';
6
6
  const DEFAULT_MAX_PROMPT_ENTRIES = 12;
7
7
  const DEFAULT_MAX_OUTPUT_ENTRIES = 12;
8
8
  const DEFAULT_HISTORY_LAYER_COUNT = 2;
9
- const SOFT_RELEASE_RATIO = 0.9;
9
+ const SOFT_RELEASE_RATIO = 0.6;
10
10
  const HARD_RELEASE_RATIO = 0.92;
11
+ const COMPACT_RELEASE_TARGET = 30_000;
11
12
  const PHASE_COOLDOWN_MS = {
12
13
  soft: 5 * 60 * 1000,
13
14
  hard: 10 * 60 * 1000,
@@ -414,15 +415,20 @@ function computeEstimatedTotalTokens({
414
415
  recentOutputs = [],
415
416
  estimatedContextTokens = 0,
416
417
  baselineTokens = 0,
418
+ sessionTokens = 0,
417
419
  }) {
418
420
  const promptTokens = recentPrompts.reduce((sum, entry) => sum + finiteNumber(entry.tokens, 0), 0);
419
421
  const outputTokens = recentOutputs.reduce((sum, entry) => sum + finiteNumber(entry.tokensBefore ?? entry.tokens, 0), 0);
420
- return baselineTokens + estimatedContextTokens + promptTokens + outputTokens;
422
+ const windowTokens = promptTokens + outputTokens;
423
+ // sessionTokens accumulates across all prompts/outputs (beyond the recent window)
424
+ // include only the excess over the recent window to avoid double-counting
425
+ const sessionExcess = Math.max(0, finiteNumber(sessionTokens, 0) - windowTokens);
426
+ return baselineTokens + estimatedContextTokens + windowTokens + sessionExcess;
421
427
  }
422
428
 
423
429
  export function buildCompactThresholds(config = {}) {
424
- const softThreshold = Math.max(1, finiteNumber(config?.compact?.tokenThreshold, 100_000));
425
- const hardThreshold = Math.max(softThreshold + 1, Math.round(softThreshold * 1.2));
430
+ const softThreshold = Math.max(1, finiteNumber(config?.compact?.tokenThreshold, 50_000));
431
+ const hardThreshold = Math.max(softThreshold + 1, Math.round(softThreshold * 1.6));
426
432
  const baselineTokens = Math.max(120, Math.min(18_000, Math.round(softThreshold * 0.18)));
427
433
 
428
434
  return {
@@ -551,7 +557,7 @@ export function buildCompactPressureState(rawState = null, config = {}) {
551
557
  thresholds.softThreshold = rawSoftThreshold;
552
558
  thresholds.hardThreshold = rawHardThreshold > rawSoftThreshold
553
559
  ? rawHardThreshold
554
- : Math.max(rawSoftThreshold + 1, Math.round(rawSoftThreshold * 1.2));
560
+ : Math.max(rawSoftThreshold + 1, Math.round(rawSoftThreshold * 1.6));
555
561
  thresholds.baselineTokens = Math.max(120, Math.min(18_000, Math.round(thresholds.softThreshold * 0.18)));
556
562
  }
557
563
  const recentPrompts = (Array.isArray(rawState?.recentPrompts) ? rawState.recentPrompts : [])
@@ -565,6 +571,7 @@ export function buildCompactPressureState(rawState = null, config = {}) {
565
571
  .sort((left, right) => (right.timestamp ?? 0) - (left.timestamp ?? 0))
566
572
  .slice(0, DEFAULT_MAX_OUTPUT_ENTRIES);
567
573
  const estimatedContextTokens = computeEstimatedContextTokens(rawState);
574
+ const sessionTokens = Math.max(0, finiteNumber(rawState?.sessionTokens, 0));
568
575
  const estimatedTotalTokens = finiteNumber(
569
576
  rawState?.estimatedTotalTokens,
570
577
  computeEstimatedTotalTokens({
@@ -572,6 +579,7 @@ export function buildCompactPressureState(rawState = null, config = {}) {
572
579
  recentOutputs,
573
580
  estimatedContextTokens,
574
581
  baselineTokens: thresholds.baselineTokens,
582
+ sessionTokens,
575
583
  }),
576
584
  );
577
585
  const releaseThresholds = buildReleaseThresholds(thresholds);
@@ -589,6 +597,7 @@ export function buildCompactPressureState(rawState = null, config = {}) {
589
597
  hardReleaseThreshold: releaseThresholds.hardReleaseThreshold,
590
598
  baselineTokens: thresholds.baselineTokens,
591
599
  estimatedContextTokens,
600
+ sessionTokens,
592
601
  estimatedTotalTokens,
593
602
  phase: resolveCompactPhase(estimatedTotalTokens, thresholds, rawState),
594
603
  taskMode,
@@ -617,9 +626,10 @@ export function registerPromptPressure(state, {
617
626
  return current;
618
627
  }
619
628
 
629
+ const promptTokenCount = estimateTokenCount(normalizedPrompt);
620
630
  const entry = normalizePromptEntry({
621
631
  timestamp,
622
- tokens: estimateTokenCount(normalizedPrompt),
632
+ tokens: promptTokenCount,
623
633
  summary: summarizePromptText(normalizedPrompt, routingContext),
624
634
  targetFile: routingContext?.targetFile ?? null,
625
635
  taskType: routingContext?.taskType ?? null,
@@ -636,6 +646,7 @@ export function registerPromptPressure(state, {
636
646
  updatedAt: timestamp,
637
647
  routingContext,
638
648
  recentPrompts,
649
+ sessionTokens: current.sessionTokens + promptTokenCount,
639
650
  estimatedTotalTokens: undefined,
640
651
  estimatedContextTokens: estimateTokenCount([
641
652
  routeSummary?.line ?? '',
@@ -666,12 +677,13 @@ export function registerOutputPressure(state, {
666
677
  return current;
667
678
  }
668
679
 
680
+ const normalizedTokensBefore = Math.max(1, finiteNumber(tokensBefore, estimateTokenCount(summary || normalizedSummary)));
669
681
  const entry = normalizeOutputEntry({
670
682
  timestamp,
671
683
  command,
672
684
  profile,
673
685
  exitCode,
674
- tokensBefore: Math.max(1, finiteNumber(tokensBefore, estimateTokenCount(summary || normalizedSummary))),
686
+ tokensBefore: normalizedTokensBefore,
675
687
  tokensAfter: Math.max(1, finiteNumber(tokensAfter, estimateTokenCount(normalizedSummary))),
676
688
  savedTokens: Math.max(0, finiteNumber(savedTokens, 0)),
677
689
  summary: normalizedSummary,
@@ -685,6 +697,7 @@ export function registerOutputPressure(state, {
685
697
  ...current,
686
698
  updatedAt: timestamp,
687
699
  recentOutputs,
700
+ sessionTokens: current.sessionTokens + normalizedTokensBefore,
688
701
  estimatedTotalTokens: undefined,
689
702
  taskMode: deriveTaskMode({
690
703
  routingContext: current.routingContext ?? {},
@@ -755,9 +768,19 @@ export function registerThresholdCompactPlan(state, plan, config = {}) {
755
768
  stats.softHits += 1;
756
769
  }
757
770
 
771
+ // Decay sessionTokens on hard compact plan: cap at COMPACT_RELEASE_TARGET for large sessions,
772
+ // or fall back to recent-window tokens when already below the release target
773
+ const windowTokens = current.recentPrompts.reduce((sum, entry) => sum + finiteNumber(entry.tokens, 0), 0)
774
+ + current.recentOutputs.reduce((sum, entry) => sum + finiteNumber(entry.tokensBefore ?? entry.tokens, 0), 0);
775
+ const decayTarget = current.sessionTokens > COMPACT_RELEASE_TARGET ? COMPACT_RELEASE_TARGET : windowTokens;
776
+ const decayedSessionTokens = normalizePhase(plan.phase) === 'hard'
777
+ ? Math.min(current.sessionTokens, decayTarget)
778
+ : current.sessionTokens;
779
+
758
780
  return buildCompactPressureState({
759
781
  ...current,
760
782
  updatedAt: now,
783
+ sessionTokens: decayedSessionTokens,
761
784
  cooldownUntil: now + resolvePhaseCooldownMs(plan.phase),
762
785
  historyLayers: uniqueStrings(plan.layers ?? []).slice(0, DEFAULT_HISTORY_LAYER_COUNT),
763
786
  latestPlan: {
@@ -58,7 +58,7 @@ export function buildDefaultRuntimeConfig(overrides = {}) {
58
58
  },
59
59
  compact: {
60
60
  enabled: true,
61
- tokenThreshold: 100_000,
61
+ tokenThreshold: 50_000,
62
62
  contextRotDetection: true,
63
63
  askBeforeDrop: true,
64
64
  agentContext: {
@@ -433,6 +433,9 @@ function printRouteState(state) {
433
433
  console.log(`recent-output: ${recentOutputDisplay}`);
434
434
  }
435
435
  console.log(`summary: ${compactSummary}`);
436
+ if (state.routeSummary?.executionContract?.modelTier) {
437
+ console.log(`modelTier: ${state.routeSummary.executionContract.modelTier}`);
438
+ }
436
439
  if (nextDisplay) {
437
440
  console.log(`next: ${nextDisplay}`);
438
441
  }
@@ -1655,6 +1658,7 @@ function buildExecutionContract(executionMode = null) {
1655
1658
 
1656
1659
  const contracts = {
1657
1660
  'tiny-fix': {
1661
+ modelTier: 'lite',
1658
1662
  maxReadPasses: 0,
1659
1663
  maxContextPulls: 0,
1660
1664
  verificationPolicy: 'minimal-or-targeted',
@@ -1663,6 +1667,7 @@ function buildExecutionContract(executionMode = null) {
1663
1667
  completionEvidence: ['write-evidence'],
1664
1668
  },
1665
1669
  'local-fix': {
1670
+ modelTier: 'code',
1666
1671
  maxReadPasses: 1,
1667
1672
  maxContextPulls: 1,
1668
1673
  verificationPolicy: 'targeted-if-covered',
@@ -1671,6 +1676,7 @@ function buildExecutionContract(executionMode = null) {
1671
1676
  completionEvidence: ['write-evidence'],
1672
1677
  },
1673
1678
  'local-build': {
1679
+ modelTier: 'code',
1674
1680
  maxReadPasses: 2,
1675
1681
  maxContextPulls: 1,
1676
1682
  verificationPolicy: 'targeted-if-covered',
@@ -1679,6 +1685,7 @@ function buildExecutionContract(executionMode = null) {
1679
1685
  completionEvidence: ['write-evidence'],
1680
1686
  },
1681
1687
  'find-cause': {
1688
+ modelTier: 'smart',
1682
1689
  maxReadPassesBeforeReassess: 3,
1683
1690
  verificationPolicy: 'root-cause-then-targeted',
1684
1691
  completionRule: 'never-claim-fixed-without-write-and-verification',
@@ -1686,6 +1693,7 @@ function buildExecutionContract(executionMode = null) {
1686
1693
  completionEvidence: ['write-evidence', 'verification-evidence'],
1687
1694
  },
1688
1695
  'shared-edit': {
1696
+ modelTier: 'code',
1689
1697
  maxReadPasses: 2,
1690
1698
  maxContextPulls: 2,
1691
1699
  verificationPolicy: 'targeted-then-widen-on-risk',
@@ -1695,6 +1703,7 @@ function buildExecutionContract(executionMode = null) {
1695
1703
  mirrorConsistencyRequired: true,
1696
1704
  },
1697
1705
  'map-impact': {
1706
+ modelTier: 'smart',
1698
1707
  maxReadPasses: 3,
1699
1708
  maxContextPulls: 3,
1700
1709
  verificationPolicy: 'impact-first-then-targeted-then-widen-on-risk',
@@ -1704,6 +1713,7 @@ function buildExecutionContract(executionMode = null) {
1704
1713
  mirrorConsistencyRequired: true,
1705
1714
  },
1706
1715
  'review-release': {
1716
+ modelTier: 'smart',
1707
1717
  verificationPolicy: 'evidence-first',
1708
1718
  completionRule: 'report-findings-not-implementation',
1709
1719
  delegationPolicy: 'allow-review-sidecar',
@@ -13,8 +13,9 @@ import {
13
13
  const DEFAULT_MAX_PROMPT_ENTRIES = 12;
14
14
  const DEFAULT_MAX_OUTPUT_ENTRIES = 12;
15
15
  const DEFAULT_HISTORY_LAYER_COUNT = 2;
16
- const SOFT_RELEASE_RATIO = 0.9;
16
+ const SOFT_RELEASE_RATIO = 0.6;
17
17
  const HARD_RELEASE_RATIO = 0.92;
18
+ const COMPACT_RELEASE_TARGET = 30_000;
18
19
  const PHASE_COOLDOWN_MS = {
19
20
  soft: 5 * 60 * 1000,
20
21
  hard: 10 * 60 * 1000,
@@ -421,15 +422,20 @@ function computeEstimatedTotalTokens({
421
422
  recentOutputs = [],
422
423
  estimatedContextTokens = 0,
423
424
  baselineTokens = 0,
425
+ sessionTokens = 0,
424
426
  }) {
425
427
  const promptTokens = recentPrompts.reduce((sum, entry) => sum + finiteNumber(entry.tokens, 0), 0);
426
428
  const outputTokens = recentOutputs.reduce((sum, entry) => sum + finiteNumber(entry.tokensBefore ?? entry.tokens, 0), 0);
427
- return baselineTokens + estimatedContextTokens + promptTokens + outputTokens;
429
+ const windowTokens = promptTokens + outputTokens;
430
+ // sessionTokens accumulates across all prompts/outputs (beyond the recent window)
431
+ // include only the excess over the recent window to avoid double-counting
432
+ const sessionExcess = Math.max(0, finiteNumber(sessionTokens, 0) - windowTokens);
433
+ return baselineTokens + estimatedContextTokens + windowTokens + sessionExcess;
428
434
  }
429
435
 
430
436
  export function buildCompactThresholds(config = {}) {
431
- const softThreshold = Math.max(1, finiteNumber(config?.compact?.tokenThreshold, 100_000));
432
- const hardThreshold = Math.max(softThreshold + 1, Math.round(softThreshold * 1.2));
437
+ const softThreshold = Math.max(1, finiteNumber(config?.compact?.tokenThreshold, 50_000));
438
+ const hardThreshold = Math.max(softThreshold + 1, Math.round(softThreshold * 1.6));
433
439
  const baselineTokens = Math.max(120, Math.min(18_000, Math.round(softThreshold * 0.18)));
434
440
 
435
441
  return {
@@ -558,7 +564,7 @@ export function buildCompactPressureState(rawState = null, config = {}) {
558
564
  thresholds.softThreshold = rawSoftThreshold;
559
565
  thresholds.hardThreshold = rawHardThreshold > rawSoftThreshold
560
566
  ? rawHardThreshold
561
- : Math.max(rawSoftThreshold + 1, Math.round(rawSoftThreshold * 1.2));
567
+ : Math.max(rawSoftThreshold + 1, Math.round(rawSoftThreshold * 1.6));
562
568
  thresholds.baselineTokens = Math.max(120, Math.min(18_000, Math.round(thresholds.softThreshold * 0.18)));
563
569
  }
564
570
  const recentPrompts = (Array.isArray(rawState?.recentPrompts) ? rawState.recentPrompts : [])
@@ -572,6 +578,7 @@ export function buildCompactPressureState(rawState = null, config = {}) {
572
578
  .sort((left, right) => (right.timestamp ?? 0) - (left.timestamp ?? 0))
573
579
  .slice(0, DEFAULT_MAX_OUTPUT_ENTRIES);
574
580
  const estimatedContextTokens = computeEstimatedContextTokens(rawState);
581
+ const sessionTokens = Math.max(0, finiteNumber(rawState?.sessionTokens, 0));
575
582
  const estimatedTotalTokens = finiteNumber(
576
583
  rawState?.estimatedTotalTokens,
577
584
  computeEstimatedTotalTokens({
@@ -579,6 +586,7 @@ export function buildCompactPressureState(rawState = null, config = {}) {
579
586
  recentOutputs,
580
587
  estimatedContextTokens,
581
588
  baselineTokens: thresholds.baselineTokens,
589
+ sessionTokens,
582
590
  }),
583
591
  );
584
592
  const releaseThresholds = buildReleaseThresholds(thresholds);
@@ -596,6 +604,7 @@ export function buildCompactPressureState(rawState = null, config = {}) {
596
604
  hardReleaseThreshold: releaseThresholds.hardReleaseThreshold,
597
605
  baselineTokens: thresholds.baselineTokens,
598
606
  estimatedContextTokens,
607
+ sessionTokens,
599
608
  estimatedTotalTokens,
600
609
  phase: resolveCompactPhase(estimatedTotalTokens, thresholds, rawState),
601
610
  taskMode,
@@ -624,9 +633,10 @@ export function registerPromptPressure(state, {
624
633
  return current;
625
634
  }
626
635
 
636
+ const promptTokenCount = estimateTokenCount(normalizedPrompt);
627
637
  const entry = normalizePromptEntry({
628
638
  timestamp,
629
- tokens: estimateTokenCount(normalizedPrompt),
639
+ tokens: promptTokenCount,
630
640
  summary: summarizePromptText(normalizedPrompt, routingContext),
631
641
  targetFile: routingContext?.targetFile ?? null,
632
642
  taskType: routingContext?.taskType ?? null,
@@ -643,6 +653,7 @@ export function registerPromptPressure(state, {
643
653
  updatedAt: timestamp,
644
654
  routingContext,
645
655
  recentPrompts,
656
+ sessionTokens: current.sessionTokens + promptTokenCount,
646
657
  estimatedTotalTokens: undefined,
647
658
  estimatedContextTokens: estimateTokenCount([
648
659
  routeSummary?.line ?? '',
@@ -673,12 +684,13 @@ export function registerOutputPressure(state, {
673
684
  return current;
674
685
  }
675
686
 
687
+ const normalizedTokensBefore = Math.max(1, finiteNumber(tokensBefore, estimateTokenCount(summary || normalizedSummary)));
676
688
  const entry = normalizeOutputEntry({
677
689
  timestamp,
678
690
  command,
679
691
  profile,
680
692
  exitCode,
681
- tokensBefore: Math.max(1, finiteNumber(tokensBefore, estimateTokenCount(summary || normalizedSummary))),
693
+ tokensBefore: normalizedTokensBefore,
682
694
  tokensAfter: Math.max(1, finiteNumber(tokensAfter, estimateTokenCount(normalizedSummary))),
683
695
  savedTokens: Math.max(0, finiteNumber(savedTokens, 0)),
684
696
  summary: normalizedSummary,
@@ -692,6 +704,7 @@ export function registerOutputPressure(state, {
692
704
  ...current,
693
705
  updatedAt: timestamp,
694
706
  recentOutputs,
707
+ sessionTokens: current.sessionTokens + normalizedTokensBefore,
695
708
  estimatedTotalTokens: undefined,
696
709
  taskMode: deriveTaskMode({
697
710
  routingContext: current.routingContext ?? {},
@@ -762,9 +775,19 @@ export function registerThresholdCompactPlan(state, plan, config = {}) {
762
775
  stats.softHits += 1;
763
776
  }
764
777
 
778
+ // Decay sessionTokens on hard compact plan: cap at COMPACT_RELEASE_TARGET for large sessions,
779
+ // or fall back to recent-window tokens when already below the release target
780
+ const windowTokens = current.recentPrompts.reduce((sum, entry) => sum + finiteNumber(entry.tokens, 0), 0)
781
+ + current.recentOutputs.reduce((sum, entry) => sum + finiteNumber(entry.tokensBefore ?? entry.tokens, 0), 0);
782
+ const decayTarget = current.sessionTokens > COMPACT_RELEASE_TARGET ? COMPACT_RELEASE_TARGET : windowTokens;
783
+ const decayedSessionTokens = normalizePhase(plan.phase) === 'hard'
784
+ ? Math.min(current.sessionTokens, decayTarget)
785
+ : current.sessionTokens;
786
+
765
787
  return buildCompactPressureState({
766
788
  ...current,
767
789
  updatedAt: now,
790
+ sessionTokens: decayedSessionTokens,
768
791
  cooldownUntil: now + resolvePhaseCooldownMs(plan.phase),
769
792
  historyLayers: uniqueStrings(plan.layers ?? []).slice(0, DEFAULT_HISTORY_LAYER_COUNT),
770
793
  latestPlan: {
@@ -136,6 +136,32 @@ Khi Handoff mode: đọc `docs/AI_HANDOFF/RULES.md` để biết 4 phase (Idea+P
136
136
  - `autonomy.level` in `.ukit/storage/config.json` controls how much UKit acts without asking first: `conservative` (ask more), `balanced` (default), `free-run` (auto-run more).
137
137
  - End users should not need to change this; maintainers may tune it per-project.
138
138
 
139
+ ## 3-Tier Model Routing
140
+
141
+ **Internal orchestration only — end users still just use natural language. No new commands.**
142
+
143
+ UKit routes tasks to one of three model tiers based on task complexity:
144
+
145
+ | Tier | Generic alias | Claude model | Typical tasks |
146
+ |------|--------------|--------------|---------------|
147
+ | lite | `unic-lite` | claude-haiku | Reads, git queries, bash summaries, small doc edits |
148
+ | code | `unic-code` | claude-sonnet | Normal coding, local fixes, shared edits, builds |
149
+ | smart | `unic-smart` | claude-opus | Deep reasoning, root-cause analysis, release review |
150
+
151
+ ### Contract-to-tier mapping
152
+
153
+ | Contract | Tier |
154
+ |----------|------|
155
+ | `tiny-fix` | lite |
156
+ | `local-fix`, `local-build`, `shared-edit` | code |
157
+ | `find-cause`, `map-impact`, `review-release` | smart |
158
+
159
+ ### Escalation rule
160
+
161
+ When the same file or symbol fails `debugLoopThreshold` (default: 2) times in one session, UKit routes the next attempt one tier higher (capped at `smart`). Config: `orchestration.escalation` in `.ukit/storage/config.json`.
162
+
163
+ This is internal orchestration — end users do not need to know about tiers, thresholds, or escalation. The AI handles routing transparently.
164
+
139
165
  ## Project Snapshot
140
166
 
141
167
  - Project: {{project.name}} | Root: {{project.root}}
@@ -8,7 +8,7 @@
8
8
  },
9
9
  "compact": {
10
10
  "enabled": true,
11
- "tokenThreshold": 100000,
11
+ "tokenThreshold": 50000,
12
12
  "contextRotDetection": true,
13
13
  "askBeforeDrop": true,
14
14
  "codexContext": {