@hifullmoon/aicommit 2.4.1 → 2.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.aicommit.config.example.json +7 -1
- package/CHANGELOG.md +13 -1
- package/README.md +27 -18
- package/README.zh-CN.md +27 -18
- package/docs/distribution.md +1 -1
- package/docs/large-change-implementation-plan.md +14 -14
- package/docs/privacy.md +2 -2
- package/docs/troubleshooting.md +7 -3
- package/package.json +1 -1
- package/src/analysis-budget.js +17 -7
- package/src/analysis-cache.js +246 -0
- package/src/change-analysis.js +206 -54
- package/src/cli.js +15 -0
- package/src/completion.js +4 -1
- package/src/config.js +30 -2
- package/src/local-analysis.js +46 -16
- package/src/main.js +27 -2
- package/src/split.js +114 -15
|
@@ -144,6 +144,12 @@
|
|
|
144
144
|
"chunkInputTokens": 12000,
|
|
145
145
|
"maxTotalTokens": 200000,
|
|
146
146
|
"concurrency": 2,
|
|
147
|
-
"timeoutMs": 180000
|
|
147
|
+
"timeoutMs": 180000,
|
|
148
|
+
"cache": {
|
|
149
|
+
"enabled": true,
|
|
150
|
+
"ttlMs": 86400000,
|
|
151
|
+
"maxBytes": 33554432,
|
|
152
|
+
"allowUnprotected": false
|
|
153
|
+
}
|
|
148
154
|
}
|
|
149
155
|
}
|
package/CHANGELOG.md
CHANGED
|
@@ -4,6 +4,17 @@ This file lists notable user-facing changes. Internal refactors, test-only chang
|
|
|
4
4
|
|
|
5
5
|
## [Unreleased]
|
|
6
6
|
|
|
7
|
+
## [2.5.0] - 2026-09-10
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
|
|
11
|
+
- Added short-lived, content-addressed recovery caching for validated `deep` analysis chunks so identical snapshots resume after failures without repeating completed provider requests.
|
|
12
|
+
|
|
13
|
+
### Fixed
|
|
14
|
+
|
|
15
|
+
- Packed `deep` analysis fragments using the configured token estimate instead of character counts, avoiding unnecessary requests and premature aggregate-budget failures for ASCII-heavy diffs.
|
|
16
|
+
- Large split plans now preflight deep-analysis and downstream planning cost, batch local candidates hierarchically, and offer one complete conservative plan when the bounded budget cannot finish. Non-interactive committing requires the explicit `--allow-single-fallback` opt-in.
|
|
17
|
+
|
|
7
18
|
## [2.4.1] - 2026-09-10
|
|
8
19
|
|
|
9
20
|
### Fixed
|
|
@@ -206,7 +217,8 @@ This file lists notable user-facing changes. Internal refactors, test-only chang
|
|
|
206
217
|
- Added file-level split planning and execution with Git-state concurrency checks.
|
|
207
218
|
- Added provider presets and user/project configuration boundaries.
|
|
208
219
|
|
|
209
|
-
[Unreleased]: https://github.com/hi-fullmoon/AICommit/compare/v2.
|
|
220
|
+
[Unreleased]: https://github.com/hi-fullmoon/AICommit/compare/v2.5.0...HEAD
|
|
221
|
+
[2.5.0]: https://github.com/hi-fullmoon/AICommit/releases/tag/v2.5.0
|
|
210
222
|
[2.4.1]: https://github.com/hi-fullmoon/AICommit/releases/tag/v2.4.1
|
|
211
223
|
[2.4.0]: https://github.com/hi-fullmoon/AICommit/releases/tag/v2.4.0
|
|
212
224
|
[2.3.0]: https://github.com/hi-fullmoon/AICommit/releases/tag/v2.3.0
|
package/README.md
CHANGED
|
@@ -197,6 +197,7 @@ This is the only supported user-config shape. Earlier flat or provider-level `mo
|
|
|
197
197
|
| `maxFileDiffChars` | Target per-file fragment size; remaining content is analyzed in subsequent chunks (default: `3000`) |
|
|
198
198
|
| `splitMaxDiffChars` | Context character budget for each split-planning request (default: `16000`) |
|
|
199
199
|
| `splitMaxPlanFiles` | Files or candidate groups per planning request; larger changes use hierarchical planning (default: `100`) |
|
|
200
|
+
| `largeChange` | Large-change strategy, budgets, and short-lived chunk recovery cache; personal config only, retaining protected summaries for 24 hours up to 32 MiB by default |
|
|
200
201
|
| `diffContextLines` | Context lines around each diff hunk (`git diff --unified=<n>`); lower values mean fewer tokens (default: `1`) |
|
|
201
202
|
| `stripFiles` | Extra files to stub out of the diff like lock files, matched by basename with `*`/`?` wildcards, e.g. `["*.min.js", "*.map", "*.snap"]` (default: `[]`; project-level entries are merged with user-level ones, not replaced) |
|
|
202
203
|
| `regenerateWithDiff` | `true` re-sends the full diff on every regenerate for more varied rewrites; `false` (default) only asks the model to reword its previous message, which is far cheaper |
|
|
@@ -311,6 +312,7 @@ aicommit split --dry-run # review a split plan without creating commits
|
|
|
311
312
|
aicommit --yes # non-interactively commit already staged changes
|
|
312
313
|
aicommit --yes --dry-run # non-interactively preview all changes; restores staging
|
|
313
314
|
aicommit split --scope=all --yes # non-interactively plan and commit all working-tree changes
|
|
315
|
+
aicommit split --scope=all --yes --allow-single-fallback # explicitly permit a conservative fallback commit
|
|
314
316
|
aicommit split plan --scope=staged --file=/tmp/split-plan.json --yes
|
|
315
317
|
aicommit split apply --file=/tmp/split-plan.json --yes
|
|
316
318
|
aicommit split resume --yes # resume an interrupted split transaction
|
|
@@ -324,20 +326,21 @@ aicommit --yes --output=json # emit one schema-validated JSON result on stdout
|
|
|
324
326
|
aicommit -h # help
|
|
325
327
|
```
|
|
326
328
|
|
|
327
|
-
| Option
|
|
328
|
-
|
|
|
329
|
-
| `-l`, `--lang`
|
|
330
|
-
| `-p`, `--provider`
|
|
331
|
-
| `-m`, `--model`
|
|
332
|
-
| `--scope`
|
|
333
|
-
| `--file`
|
|
334
|
-
| `--dry-run`
|
|
335
|
-
| `-y`, `--yes`
|
|
336
|
-
| `--
|
|
337
|
-
| `--
|
|
338
|
-
| `--
|
|
339
|
-
|
|
|
340
|
-
| `-
|
|
329
|
+
| Option | Description |
|
|
330
|
+
| ------------------------- | ------------------------------------------------------------------------------ |
|
|
331
|
+
| `-l`, `--lang` | Commit message language (`zh` or `en`) |
|
|
332
|
+
| `-p`, `--provider` | Use the named provider from `providers` |
|
|
333
|
+
| `-m`, `--model` | Use a named model profile from the selected provider |
|
|
334
|
+
| `--scope` | `staged` or `all` scope for `aicommit split` and `aicommit split plan` |
|
|
335
|
+
| `--file` | JSON plan path for `aicommit split plan` and `aicommit split apply` |
|
|
336
|
+
| `--dry-run` | Generate and review a message or split plan without creating commits |
|
|
337
|
+
| `-y`, `--yes` | Accept without prompts; normal mode requires explicitly staged changes |
|
|
338
|
+
| `--allow-single-fallback` | Explicitly permit non-interactive split to create one complete fallback commit |
|
|
339
|
+
| `--reasoning` | Enable reasoning with `low`, `medium`, `high`, `xhigh`, or `max` effort |
|
|
340
|
+
| `--no-reasoning` | Explicitly disable reasoning when the selected provider/model supports it |
|
|
341
|
+
| `--output` | `text` (default) or one JSON object; commit/split JSON flows require `--yes` |
|
|
342
|
+
| `-v`, `--version` | Show version |
|
|
343
|
+
| `-h`, `--help` | Show help |
|
|
341
344
|
|
|
342
345
|
### Configuration inspection
|
|
343
346
|
|
|
@@ -478,7 +481,7 @@ The default `largeChange.strategy: "auto"` inventories every file locally, group
|
|
|
478
481
|
|
|
479
482
|
A normal commit typically needs one model request, with no per-file AI calls or recursive model reduction. The inventory contains at most 16 representative groups under a UTF-8 byte budget, prioritizing coverage across code, configuration, tests, and other categories. It explicitly describes sampling limits. Both terminal and JSON output distinguish fully analyzed files, representative excerpts, and metadata-only files. Provider retries, response recovery, policy correction, and user-requested regeneration can still add requests.
|
|
480
483
|
|
|
481
|
-
Split mode builds local candidates and
|
|
484
|
+
Split mode builds local candidates and sends them in bounded batches of at most `splitMaxPlanFiles`, then merges the batch plans hierarchically. Every file remains represented even when the complete candidate inventory cannot fit one request. If `deep` analysis exhausts its aggregate budget or the hierarchy cannot converge, interactive and dry-run flows produce one conservative all-files plan with an explicit warning instead of using incomplete model output. Non-interactive committing stops unless `--allow-single-fallback` explicitly authorizes that degradation. Small changes keep the existing request path.
|
|
482
485
|
|
|
483
486
|
For exhaustive chunk-by-chunk model analysis, opt in through personal configuration:
|
|
484
487
|
|
|
@@ -489,11 +492,17 @@ For exhaustive chunk-by-chunk model analysis, opt in through personal configurat
|
|
|
489
492
|
"chunkInputTokens": 12000,
|
|
490
493
|
"maxTotalTokens": 200000,
|
|
491
494
|
"concurrency": 2,
|
|
492
|
-
"timeoutMs": 180000
|
|
495
|
+
"timeoutMs": 180000,
|
|
496
|
+
"cache": {
|
|
497
|
+
"enabled": true,
|
|
498
|
+
"ttlMs": 86400000,
|
|
499
|
+
"maxBytes": 33554432,
|
|
500
|
+
"allowUnprotected": false
|
|
501
|
+
}
|
|
493
502
|
}
|
|
494
503
|
}
|
|
495
504
|
```
|
|
496
505
|
|
|
497
|
-
`deep` spends more requests and tokens, with a maximum of 256 requests. Repository configuration cannot change this personal strategy or raise
|
|
506
|
+
`deep` spends more requests and tokens, with a maximum of 256 requests. Before dispatch, a preflight estimate covers initial chunks, required reductions, and the minimum hierarchical planning tree; an impossible deep run switches to the local inventory path, and validated cache hits are excluded from that estimate. Validated initial fact chunks are stored briefly under Git metadata, reused when the same snapshot is retried after failure or interruption, and removed after complete generation succeeds. The cache does not directly store captured diffs, reasoning, credentials, or complete provider responses; it stores model summaries that may contain code-derived details. Unprotected original input is not persisted unless personal configuration explicitly enables `allowUnprotected`. Repository configuration cannot change this personal strategy, enable unprotected caching, or raise spending/cache ceilings. Both strategies use conservative token estimates; cache hits consume no request or token budget. Incomplete model output is never committed: budget/capacity fallback is a new complete plan containing every reviewed file; interactive runs show it for review, while non-interactive committing requires `--allow-single-fallback`.
|
|
498
507
|
|
|
499
|
-
Complete patches and larger untracked text are captured in local temporary files, with descriptors opened only during reads and writes. Files are cleaned on normal exit or cancellation; crashes may leave them behind. Content reads are bounded. Lines exceeding 1 MiB
|
|
508
|
+
Complete patches and larger untracked text are captured in local temporary files, with descriptors opened only during reads and writes. Files are cleaned on normal exit or cancellation; crashes may leave them behind. Content reads are bounded. Lines exceeding 1 MiB and experimental hunk planning for large changes fail explicitly; file-level planning capacity uses the complete conservative fallback.
|
package/README.zh-CN.md
CHANGED
|
@@ -199,6 +199,7 @@ aicommit -p deepseek -m reasoner
|
|
|
199
199
|
| `maxFileDiffChars` | 单文件正文分块参考大小;剩余内容继续分析(默认:`3000`) |
|
|
200
200
|
| `splitMaxDiffChars` | 每次批次规划的上下文字符预算(默认:`16000`) |
|
|
201
201
|
| `splitMaxPlanFiles` | 每次规划的文件或候选组数量上限;超限分层规划(默认:`100`) |
|
|
202
|
+
| `largeChange` | 大变更策略、预算和短期分块恢复缓存;缓存只接受个人配置,默认保留受保护摘要 24 小时、最多 32 MiB |
|
|
202
203
|
| `diffContextLines` | 每个 diff hunk 周围的上下文行数(`git diff --unified=<n>`);越小越节省 token(默认:`1`) |
|
|
203
204
|
| `stripFiles` | 额外替换为占位的文件,按 basename 使用 `*` / `?` 通配,如 `["*.min.js", "*.map", "*.snap"]`(默认:`[]`;项目项与用户项合并而非覆盖) |
|
|
204
205
|
| `regenerateWithDiff` | `true` 表示每次重写都重发完整 diff,以获得更多变化;`false`(默认)只要求模型改写上一条消息,成本更低 |
|
|
@@ -313,6 +314,7 @@ aicommit split --dry-run # 审阅拆分计划,但不创建提交
|
|
|
313
314
|
aicommit --yes # 非交互提交已明确暂存的变更
|
|
314
315
|
aicommit --yes --dry-run # 非交互预览所有变更;退出时恢复暂存状态
|
|
315
316
|
aicommit split --scope=all --yes # 非交互规划并提交所有工作区变更
|
|
317
|
+
aicommit split --scope=all --yes --allow-single-fallback # 明确允许规划预算耗尽后的保守提交
|
|
316
318
|
aicommit split plan --scope=staged --file=/tmp/split-plan.json --yes
|
|
317
319
|
aicommit split apply --file=/tmp/split-plan.json --yes
|
|
318
320
|
aicommit split resume --yes # 恢复中断的拆分事务
|
|
@@ -326,20 +328,21 @@ aicommit --yes --output=json # 向 stdout 输出一个通过 schema 校验的 JS
|
|
|
326
328
|
aicommit -h # 帮助
|
|
327
329
|
```
|
|
328
330
|
|
|
329
|
-
| 选项
|
|
330
|
-
|
|
|
331
|
-
| `-l`, `--lang`
|
|
332
|
-
| `-p`, `--provider`
|
|
333
|
-
| `-m`, `--model`
|
|
334
|
-
| `--scope`
|
|
335
|
-
| `--file`
|
|
336
|
-
| `--dry-run`
|
|
337
|
-
| `-y`, `--yes`
|
|
338
|
-
| `--
|
|
339
|
-
| `--
|
|
340
|
-
| `--
|
|
341
|
-
|
|
|
342
|
-
| `-
|
|
331
|
+
| 选项 | 说明 |
|
|
332
|
+
| ------------------------- | ------------------------------------------------------------------- |
|
|
333
|
+
| `-l`, `--lang` | 提交信息语言:`zh` 或 `en` |
|
|
334
|
+
| `-p`, `--provider` | 使用 `providers` 中的命名 Provider |
|
|
335
|
+
| `-m`, `--model` | 使用所选 Provider 下的命名模型配置 |
|
|
336
|
+
| `--scope` | `aicommit split` 和 `aicommit split plan` 的范围:`staged` 或 `all` |
|
|
337
|
+
| `--file` | `aicommit split plan` 和 `aicommit split apply` 的 JSON 计划路径 |
|
|
338
|
+
| `--dry-run` | 生成并审阅消息或拆分计划,但不创建提交 |
|
|
339
|
+
| `-y`, `--yes` | 不提示直接接受;普通模式要求变更已明确暂存 |
|
|
340
|
+
| `--allow-single-fallback` | 明确允许非交互拆分创建一个覆盖完整变更的保守提交 |
|
|
341
|
+
| `--reasoning` | 启用推理,可选强度:`low`、`medium`、`high`、`xhigh` 或 `max` |
|
|
342
|
+
| `--no-reasoning` | 所选 Provider / 模型支持时显式关闭推理 |
|
|
343
|
+
| `--output` | `text`(默认)或单个 JSON 对象;提交 / 拆分的 JSON 流程要求 `--yes` |
|
|
344
|
+
| `-v`, `--version` | 显示版本 |
|
|
345
|
+
| `-h`, `--help` | 显示帮助 |
|
|
343
346
|
|
|
344
347
|
### 配置检查
|
|
345
348
|
|
|
@@ -480,7 +483,7 @@ exec zsh
|
|
|
480
483
|
|
|
481
484
|
普通提交通常只需要一次模型请求,不会为每个文件调用 AI,也不会递归调用模型汇总。摘要最多包含 16 个代表组,并受 UTF-8 字节预算约束;优先覆盖代码、配置和测试等不同类别。摘要明确说明抽样范围,界面和 JSON 分别报告全文分析、代表片段和仅元数据的文件数,不把抽样视为完整理解。提供方重试、响应恢复、格式修正和用户重新生成仍可能增加请求。
|
|
482
485
|
|
|
483
|
-
|
|
486
|
+
批次提交先在本地建立候选组,再按每批最多 `splitMaxPlanFiles` 个候选发送,并分层合并各批计划;完整候选清单放不进一次请求时,仍会保留每个文件。如果 `deep` 分析耗尽总预算,或分层规划无法收敛,交互和 dry-run 流程会明确警告并生成一个覆盖全部文件的保守计划,而不会采用不完整的模型结果;非交互提交默认停止,只有显式传入 `--allow-single-fallback` 才允许该降级。小变更保持原有请求路径。
|
|
484
487
|
|
|
485
488
|
确实需要逐块 AI 分析时,在个人配置中设置:
|
|
486
489
|
|
|
@@ -491,11 +494,17 @@ exec zsh
|
|
|
491
494
|
"chunkInputTokens": 12000,
|
|
492
495
|
"maxTotalTokens": 200000,
|
|
493
496
|
"concurrency": 2,
|
|
494
|
-
"timeoutMs": 180000
|
|
497
|
+
"timeoutMs": 180000,
|
|
498
|
+
"cache": {
|
|
499
|
+
"enabled": true,
|
|
500
|
+
"ttlMs": 86400000,
|
|
501
|
+
"maxBytes": 33554432,
|
|
502
|
+
"allowUnprotected": false
|
|
503
|
+
}
|
|
495
504
|
}
|
|
496
505
|
}
|
|
497
506
|
```
|
|
498
507
|
|
|
499
|
-
`deep` 会增加请求和 token 消耗,最多 256
|
|
508
|
+
`deep` 会增加请求和 token 消耗,最多 256 次请求。请求前会预估初始分块、必要归并和最小分层规划树的成本;确定无法装入总预算时直接切换到本地清单路径,已验证的缓存命中不计入这次预估。通过完整校验的初始事实分块会短期写入 Git 元数据目录;同一快照失败或中断后重试时直接复用,完整生成成功后清理。缓存不直接保存捕获的 diff、推理、凭据或完整 Provider 响应,只保存可能包含代码派生细节的模型摘要;选择发送未保护的原始内容时默认不落盘,只有个人配置显式设置 `allowUnprotected: true` 才允许。仓库配置不能切换策略、启用未保护缓存或提高费用与缓存上限。两种策略均使用保守 token 估算,未知 usage 按预留额度计入;缓存命中不计请求或 token。模型的不完整输出永远不会被提交:预算或容量降级会重新生成一个包含全部已审核文件的完整计划;交互模式先展示确认,非交互提交则要求 `--allow-single-fallback`。
|
|
500
509
|
|
|
501
|
-
完整补丁及较大未跟踪文本暂存到本地临时文件,只在读写时打开文件句柄,正常结束或取消时清理;异常崩溃可能遗留临时文件。正文读取有界。单行超过 1 MiB
|
|
510
|
+
完整补丁及较大未跟踪文本暂存到本地临时文件,只在读写时打开文件句柄,正常结束或取消时清理;异常崩溃可能遗留临时文件。正文读取有界。单行超过 1 MiB 和大变更的实验性 hunk 规划会明确报错;文件级规划容量不足时使用覆盖完整变更的保守回退。
|
package/docs/distribution.md
CHANGED
|
@@ -33,7 +33,7 @@ The release workflow uses npm Trusted Publishing without a long-lived `NPM_TOKEN
|
|
|
33
33
|
```bash
|
|
34
34
|
workdir=$(mktemp -d)
|
|
35
35
|
cd "$workdir"
|
|
36
|
-
npm install --package-lock-only @hifullmoon/aicommit@2.
|
|
36
|
+
npm install --package-lock-only @hifullmoon/aicommit@2.5.0
|
|
37
37
|
npm audit signatures
|
|
38
38
|
```
|
|
39
39
|
|
|
@@ -112,7 +112,7 @@
|
|
|
112
112
|
- [ ] 本地展开组 ID 到文件路径,校验完整覆盖、唯一归属及合法引用,替换超额文件兜底提交。
|
|
113
113
|
- [ ] 为最终分组生成消息,复用已有分析;大量分组的消息生成也纳入总预算。
|
|
114
114
|
- [ ] 未解决归属显示为待处理项,允许在现有审阅中调整;未解决前不执行、不导出可执行计划。
|
|
115
|
-
- [ ]
|
|
115
|
+
- [ ] 非交互模式绝不采用不完整分析或非法模型分组;确定性的预算/容量耗尽只有在显式允许单提交降级时才使用覆盖全部文件的保守计划,其余情况在自动暂存前失败。
|
|
116
116
|
- [ ] 继续生成现有 split plan 工件,由已有 apply / checkpoint / resume 流程执行;内部分析 ID 不改变外部路径语义。
|
|
117
117
|
- [ ] 验证实验性 hunk 归属和跨块片段映射;无法处理时给出明确错误。
|
|
118
118
|
|
|
@@ -136,18 +136,18 @@
|
|
|
136
136
|
|
|
137
137
|
使用临时仓库和模拟 Provider 构造测试,不把巨型 fixture 提交到仓库。确定性测试验证边界和完整性,语义评估单独记录,不用字符串快照冒充摘要质量验证。
|
|
138
138
|
|
|
139
|
-
| 场景 | 必须验证
|
|
140
|
-
| ------------------------------------- |
|
|
141
|
-
| 普通小提交 | 一次生成请求,已有消息策略和交互保持兼容
|
|
142
|
-
| 10,000 个小文件 |
|
|
143
|
-
| 超过 64 MiB 的源码补丁 | 流式读取、指纹校验与内存规模符合预期
|
|
144
|
-
| 巨型 hunk、超长单行 | 不导致超限请求或无界内存,覆盖状态诚实
|
|
145
|
-
| 二进制、lock、生成文件、批量重命名 | 正文策略正确,路径仍完整纳入提交
|
|
146
|
-
| 跨目录实现与测试 | 可形成同一逻辑组,分块不强制分组
|
|
147
|
-
| 敏感文件、未跟踪符号链接 | 保持现有保护行为,不因分块绕过
|
|
148
|
-
| 429、响应截断、错误 ID、usage 缺失 | 重试有界、校验拒绝非法响应、预算保守记账
|
|
149
|
-
| 超时、Ctrl+C、Git 失败、并发修改 | 清理资源、不误提交、不破坏真实 index
|
|
150
|
-
| split 导出、apply、hook 失败与 resume | 原有工件和事务恢复兼容
|
|
139
|
+
| 场景 | 必须验证 |
|
|
140
|
+
| ------------------------------------- | ------------------------------------------------------------ |
|
|
141
|
+
| 普通小提交 | 一次生成请求,已有消息策略和交互保持兼容 |
|
|
142
|
+
| 10,000 个小文件 | 全量清单完整,请求有界;预算不足时生成明确标记的完整保守计划 |
|
|
143
|
+
| 超过 64 MiB 的源码补丁 | 流式读取、指纹校验与内存规模符合预期 |
|
|
144
|
+
| 巨型 hunk、超长单行 | 不导致超限请求或无界内存,覆盖状态诚实 |
|
|
145
|
+
| 二进制、lock、生成文件、批量重命名 | 正文策略正确,路径仍完整纳入提交 |
|
|
146
|
+
| 跨目录实现与测试 | 可形成同一逻辑组,分块不强制分组 |
|
|
147
|
+
| 敏感文件、未跟踪符号链接 | 保持现有保护行为,不因分块绕过 |
|
|
148
|
+
| 429、响应截断、错误 ID、usage 缺失 | 重试有界、校验拒绝非法响应、预算保守记账 |
|
|
149
|
+
| 超时、Ctrl+C、Git 失败、并发修改 | 清理资源、不误提交、不破坏真实 index |
|
|
150
|
+
| split 导出、apply、hook 失败与 resume | 原有工件和事务恢复兼容 |
|
|
151
151
|
|
|
152
152
|
- [ ] 记录峰值 RSS、总耗时、请求数、输入 / 输出 token 及覆盖情况;至少比较两个补丁体量,确认正文内存有界。
|
|
153
153
|
- [ ] 记录小提交相对基线的额外耗时,测量后确定可接受阈值及最终预算默认值。
|
|
@@ -177,7 +177,7 @@
|
|
|
177
177
|
- Git 内容使用直接写临时文件的方式捕获,Node 按 64 KiB 读取,避免同步 stdout 缓冲上限;没有引入长期运行的异步 Git 子进程。临时文件空间取决于变更大小,Git 捕获有 120 秒命令超时。
|
|
178
178
|
- token 计数采用保守估算并结合 Provider 模型窗口。Provider 上下文超限明确失败,尚未实现 tokenizer 精确计数和 Provider 报错后的自动缩块。
|
|
179
179
|
- 分析不完整时统一停止,不生成可直接接受的部分草稿。完整分析结果的改写继续使用现有审阅流程。
|
|
180
|
-
-
|
|
180
|
+
- 批次分层规划允许合并候选组;深度分析会先预估成本,候选清单可分批并继续分层合并。预算耗尽或规划无法收敛时,不采用半份模型结果,而是明确警告并生成覆盖全部文件的单提交保守计划。大变更不支持实验性 hunk 规划,单行超过 1 MiB 明确拒绝。
|
|
181
181
|
- 未新增真实 Provider 的语义质量基准;本次验证使用模拟 Provider,不能据此保证任意模型的摘要和分组质量。
|
|
182
182
|
|
|
183
183
|
### 验证结果
|
package/docs/privacy.md
CHANGED
|
@@ -57,6 +57,6 @@ Use `aicommit config show` to inspect effective local state without revealing cr
|
|
|
57
57
|
|
|
58
58
|
## Large-change snapshots / 大变更快照
|
|
59
59
|
|
|
60
|
-
Default large-change analysis sends a bounded local inventory and selected protected excerpts to the configured provider, without per-file model requests. Explicit `deep` analysis sends protected fragments and intermediate factual summaries. Both local inventories and model summaries remain untrusted input. Full Git patches and larger untracked text are captured in private local temporary files, with bounded memory reads.
|
|
60
|
+
Default large-change analysis sends a bounded local inventory and selected protected excerpts to the configured provider, without per-file model requests. Explicit `deep` analysis sends protected fragments and intermediate factual summaries. Both local inventories and model summaries remain untrusted input. Full Git patches and larger untracked text are captured in private local temporary files, with bounded memory reads. Validated initial `deep` summaries may be stored under private Git metadata for up to 24 hours so an interrupted identical snapshot can resume; entries contain generic fragment IDs and model summaries, not the captured diff, reasoning, credentials, or complete provider responses. Because summaries are derived from repository content, they may still contain code details. Successful generation removes its cache. Unprotected original input is not persisted unless the user explicitly enables `largeChange.cache.allowUnprotected` in personal configuration. Cache entries are not added to JSON output or split checkpoints.
|
|
61
61
|
|
|
62
|
-
默认大变更分析向已配置 Provider 发送受限本地清单和选定的保护后片段,不逐文件调用模型。明确选择 `deep` 时才发送分块片段和中间事实摘要。本地清单和模型摘要均视为不可信输入。完整 Git
|
|
62
|
+
默认大变更分析向已配置 Provider 发送受限本地清单和选定的保护后片段,不逐文件调用模型。明确选择 `deep` 时才发送分块片段和中间事实摘要。本地清单和模型摘要均视为不可信输入。完整 Git 补丁和较大未跟踪文本保存在本地私有临时文件中,按块读取。通过校验的初始 `deep` 摘要可在私有 Git 元数据目录中保留最多 24 小时,让完全相同的快照在中断后续跑;缓存只含通用片段 ID 和模型摘要,不直接保存捕获的 diff、推理、凭据或完整 Provider 响应。摘要源自仓库内容,仍可能包含代码细节。完整生成成功后会删除本次缓存。未保护原始内容仅在用户通过个人配置显式启用 `largeChange.cache.allowUnprotected` 时持久化。缓存不会加入 JSON 输出或 split checkpoint。
|
package/docs/troubleshooting.md
CHANGED
|
@@ -36,10 +36,14 @@ If a failure remains, capture `aicommit doctor --output=json`, Node/Git versions
|
|
|
36
36
|
|
|
37
37
|
## Large-change limits / 大变更限制
|
|
38
38
|
|
|
39
|
-
Default `auto` analysis builds a local inventory and selects bounded excerpts
|
|
39
|
+
Default `auto` analysis builds a local inventory and selects bounded excerpts, then batches oversized split inventories and merges their plans hierarchically. Deep analysis still has fixed request (256) and depth (8) limits. If its aggregate token/time budget is exhausted or hierarchical planning cannot converge, split mode emits an explicit warning and can fall back to one complete all-files plan; it never uses a partial model plan. Interactive and dry-run flows show that plan, while non-interactive committing requires the explicit `--allow-single-fallback` option. Token/time limits remain configurable in personal settings. Unknown model token counts use conservative estimates; an oversized request is not dispatched, and provider-context errors are not blindly replayed.
|
|
40
40
|
|
|
41
|
-
默认 `auto`
|
|
41
|
+
默认 `auto` 在本地建立清单并选择受限片段;拆分候选过多时会分批请求,再分层合并计划。深度分析仍有固定的请求数(256)和汇总层级(8)上限。如果总 token/时间预算耗尽,或分层规划无法收敛,拆分模式会明确警告,并可降级为一个覆盖全部文件的计划,绝不会采用模型返回的半份计划。交互和 dry-run 流程会展示该计划;非交互提交必须显式传入 `--allow-single-fallback`。token 和时间预算仍可在个人配置中调整。未知模型采用保守 token 估算,单次输入超限时不会发送该请求;Provider 上下文超限不会盲目重试。
|
|
42
42
|
|
|
43
|
-
|
|
43
|
+
Validated initial `deep` chunks are cached under private Git metadata after a failed or interrupted run. A retry of the identical snapshot and model settings reports cached chunks and requests only the remainder. Any input/model/protection change causes a miss. Successful generation clears the active cache; stale entries expire after 24 hours. Unprotected input is not cached unless personal `largeChange.cache.allowUnprotected` is explicitly enabled.
|
|
44
|
+
|
|
45
|
+
失败或中断后,已经通过校验的初始 `deep` 分块会缓存在私有 Git 元数据中。使用相同快照和模型设置重试时会报告缓存命中,并只请求剩余分块;输入、模型或保护模式发生任何变化都会失效。完整生成成功后清理当前缓存,遗留项 24 小时后过期。未保护内容只有在个人配置显式启用 `largeChange.cache.allowUnprotected` 时才缓存。
|
|
46
|
+
|
|
47
|
+
Text lines over 1 MiB fail explicitly. Large-change hunk plans are unsupported; use file-level split. Metadata-only files remain in the complete plan. Invalid provider output remains a response-format error; only deterministic budget/capacity exhaustion activates the complete all-files fallback.
|
|
44
48
|
|
|
45
49
|
单行超过 1 MiB 会明确报错。大变更暂不支持 hunk 规划,请使用文件级拆分。仅统计的文件仍纳入完整计划。无效、重复或遗漏 ID 会报响应格式错误,不会自动归入兜底组。
|
package/package.json
CHANGED
package/src/analysis-budget.js
CHANGED
|
@@ -6,6 +6,12 @@ export const DEFAULT_LARGE_CHANGE = Object.freeze({
|
|
|
6
6
|
maxTotalTokens: 200000,
|
|
7
7
|
concurrency: 2,
|
|
8
8
|
timeoutMs: 180000,
|
|
9
|
+
cache: Object.freeze({
|
|
10
|
+
enabled: true,
|
|
11
|
+
ttlMs: 24 * 60 * 60 * 1000,
|
|
12
|
+
maxBytes: 32 * 1024 * 1024,
|
|
13
|
+
allowUnprotected: false,
|
|
14
|
+
}),
|
|
9
15
|
});
|
|
10
16
|
|
|
11
17
|
export function estimateTokens(text) {
|
|
@@ -40,21 +46,25 @@ export function createAnalysisBudget(settings = {}) {
|
|
|
40
46
|
};
|
|
41
47
|
},
|
|
42
48
|
reserve(input, output) {
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
49
|
+
const exhausted = !this.remainingMs()
|
|
50
|
+
? 'time'
|
|
51
|
+
: requests >= 256
|
|
52
|
+
? 'requests'
|
|
53
|
+
: charged + input + output + reserveFinal > limits.maxTotalTokens
|
|
54
|
+
? 'tokens'
|
|
55
|
+
: null;
|
|
56
|
+
if (exhausted) {
|
|
48
57
|
throw fail(
|
|
49
58
|
ERROR_CATEGORIES.PROVIDER,
|
|
50
|
-
|
|
51
|
-
{ data: { analysis: this.snapshot() } },
|
|
59
|
+
`Large-change analysis reached its ${exhausted} budget.`,
|
|
60
|
+
{ data: { analysis: { ...this.snapshot(), exhausted } } },
|
|
52
61
|
);
|
|
53
62
|
}
|
|
54
63
|
if (input > limits.chunkInputTokens) {
|
|
55
64
|
throw fail(
|
|
56
65
|
ERROR_CATEGORIES.PROVIDER,
|
|
57
66
|
'Analysis request exceeds largeChange.chunkInputTokens; shorten repository context or increase the personal input budget.',
|
|
67
|
+
{ data: { analysis: { ...this.snapshot(), exhausted: 'input' } } },
|
|
58
68
|
);
|
|
59
69
|
}
|
|
60
70
|
charged += input + output;
|
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
import { createHash } from 'node:crypto';
|
|
2
|
+
import { execFileSync } from 'node:child_process';
|
|
3
|
+
import {
|
|
4
|
+
existsSync,
|
|
5
|
+
lstatSync,
|
|
6
|
+
mkdirSync,
|
|
7
|
+
readFileSync,
|
|
8
|
+
readdirSync,
|
|
9
|
+
renameSync,
|
|
10
|
+
rmSync,
|
|
11
|
+
statSync,
|
|
12
|
+
utimesSync,
|
|
13
|
+
writeFileSync,
|
|
14
|
+
} from 'node:fs';
|
|
15
|
+
import { dirname, isAbsolute, join, resolve } from 'node:path';
|
|
16
|
+
|
|
17
|
+
export const ANALYSIS_CACHE_KIND = 'aicommit-analysis-cache-entry';
|
|
18
|
+
export const ANALYSIS_CACHE_VERSION = 1;
|
|
19
|
+
// Bump whenever the fragment-analysis prompt or summary contract changes.
|
|
20
|
+
export const ANALYSIS_CONTRACT_VERSION = 1;
|
|
21
|
+
const MAX_ENTRY_BYTES = 128 * 1024;
|
|
22
|
+
const SHA256_RE = /^[0-9a-f]{64}$/;
|
|
23
|
+
|
|
24
|
+
function object(value) {
|
|
25
|
+
return value && typeof value === 'object' && !Array.isArray(value);
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
function canonical(value) {
|
|
29
|
+
if (Array.isArray(value)) return value.map(canonical);
|
|
30
|
+
if (!object(value)) return value;
|
|
31
|
+
return Object.fromEntries(
|
|
32
|
+
Object.keys(value)
|
|
33
|
+
.sort()
|
|
34
|
+
.filter((key) => value[key] !== undefined)
|
|
35
|
+
.map((key) => [key, canonical(value[key])]),
|
|
36
|
+
);
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
function digest(value) {
|
|
40
|
+
return createHash('sha256')
|
|
41
|
+
.update(JSON.stringify(canonical(value)))
|
|
42
|
+
.digest('hex');
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
function cacheRoot(projectRoot) {
|
|
46
|
+
const raw = execFileSync(
|
|
47
|
+
'git',
|
|
48
|
+
['rev-parse', '--git-path', `aicommit/analysis-cache/v${ANALYSIS_CACHE_VERSION}`],
|
|
49
|
+
{
|
|
50
|
+
cwd: projectRoot,
|
|
51
|
+
encoding: 'utf8',
|
|
52
|
+
stdio: ['pipe', 'pipe', 'pipe'],
|
|
53
|
+
},
|
|
54
|
+
).trim();
|
|
55
|
+
return isAbsolute(raw) ? raw : resolve(projectRoot, raw);
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
function namespaceSignature(config, protect) {
|
|
59
|
+
return digest({
|
|
60
|
+
contractVersion: ANALYSIS_CONTRACT_VERSION,
|
|
61
|
+
providerType: config.providerType || '',
|
|
62
|
+
apiUrl: config.apiUrl,
|
|
63
|
+
modelId: config.modelId,
|
|
64
|
+
reasoning: config.reasoning,
|
|
65
|
+
extraBody: config.extraBody,
|
|
66
|
+
protect,
|
|
67
|
+
});
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
function normalizeGroups(groups) {
|
|
71
|
+
if (!Array.isArray(groups) || !groups.length || groups.length > 16) {
|
|
72
|
+
throw new Error('Analysis cache groups must be a non-empty bounded array.');
|
|
73
|
+
}
|
|
74
|
+
return groups.map((group) => {
|
|
75
|
+
if (
|
|
76
|
+
!object(group) ||
|
|
77
|
+
!Array.isArray(group.ids) ||
|
|
78
|
+
!group.ids.length ||
|
|
79
|
+
group.ids.some((id) => typeof id !== 'string' || !id || id.length > 128) ||
|
|
80
|
+
typeof group.summary !== 'string' ||
|
|
81
|
+
!group.summary.trim() ||
|
|
82
|
+
group.summary.length > 2000
|
|
83
|
+
) {
|
|
84
|
+
throw new Error('Analysis cache contains an invalid group.');
|
|
85
|
+
}
|
|
86
|
+
return { ids: [...group.ids], summary: group.summary };
|
|
87
|
+
});
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
function validateEntry(input, expectedIds) {
|
|
91
|
+
if (
|
|
92
|
+
!object(input) ||
|
|
93
|
+
Object.keys(input).some(
|
|
94
|
+
(key) => !['kind', 'version', 'createdAt', 'ids', 'groups'].includes(key),
|
|
95
|
+
) ||
|
|
96
|
+
input.kind !== ANALYSIS_CACHE_KIND ||
|
|
97
|
+
input.version !== ANALYSIS_CACHE_VERSION ||
|
|
98
|
+
typeof input.createdAt !== 'string' ||
|
|
99
|
+
!Number.isFinite(Date.parse(input.createdAt)) ||
|
|
100
|
+
!Array.isArray(input.ids) ||
|
|
101
|
+
input.ids.length !== expectedIds.length ||
|
|
102
|
+
input.ids.some((id, index) => id !== expectedIds[index])
|
|
103
|
+
) {
|
|
104
|
+
throw new Error('Analysis cache entry is invalid or belongs to another chunk.');
|
|
105
|
+
}
|
|
106
|
+
return normalizeGroups(input.groups);
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
function namespaceRecords(root) {
|
|
110
|
+
if (!existsSync(root)) return [];
|
|
111
|
+
const records = [];
|
|
112
|
+
for (const entry of readdirSync(root, { withFileTypes: true })) {
|
|
113
|
+
if (!entry.isDirectory() || entry.isSymbolicLink()) continue;
|
|
114
|
+
const path = join(root, entry.name);
|
|
115
|
+
let bytes = 0;
|
|
116
|
+
let modified = lstatSync(path).mtimeMs;
|
|
117
|
+
for (const child of readdirSync(path, { withFileTypes: true })) {
|
|
118
|
+
if (!child.isFile() || child.isSymbolicLink()) continue;
|
|
119
|
+
const stat = statSync(join(path, child.name));
|
|
120
|
+
bytes += stat.size;
|
|
121
|
+
modified = Math.max(modified, stat.mtimeMs);
|
|
122
|
+
}
|
|
123
|
+
records.push({ path, bytes, modified });
|
|
124
|
+
}
|
|
125
|
+
return records;
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
function prune(root, settings, keep = null, now = Date.now()) {
|
|
129
|
+
try {
|
|
130
|
+
let records = namespaceRecords(root);
|
|
131
|
+
for (const record of records) {
|
|
132
|
+
if (record.path !== keep && now - record.modified > settings.ttlMs) {
|
|
133
|
+
rmSync(record.path, { recursive: true, force: true });
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
records = namespaceRecords(root).sort((left, right) => left.modified - right.modified);
|
|
137
|
+
let total = records.reduce((sum, record) => sum + record.bytes, 0);
|
|
138
|
+
for (const record of records) {
|
|
139
|
+
if (total <= settings.maxBytes) break;
|
|
140
|
+
if (record.path === keep) continue;
|
|
141
|
+
rmSync(record.path, { recursive: true, force: true });
|
|
142
|
+
total -= record.bytes;
|
|
143
|
+
}
|
|
144
|
+
} catch {
|
|
145
|
+
// A cache maintenance failure must never block commit generation.
|
|
146
|
+
}
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
export function createAnalysisCache({ projectRoot, snapshotFingerprint, config, protect = true }) {
|
|
150
|
+
const settings = config.largeChange?.cache;
|
|
151
|
+
if (
|
|
152
|
+
!settings?.enabled ||
|
|
153
|
+
config.largeChange?.strategy !== 'deep' ||
|
|
154
|
+
!SHA256_RE.test(snapshotFingerprint || '') ||
|
|
155
|
+
(!protect && !settings.allowUnprotected)
|
|
156
|
+
) {
|
|
157
|
+
return null;
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
try {
|
|
161
|
+
const root = cacheRoot(projectRoot);
|
|
162
|
+
const signature = namespaceSignature(config, protect);
|
|
163
|
+
const namespace = join(root, `${snapshotFingerprint}.${signature}`);
|
|
164
|
+
prune(root, settings);
|
|
165
|
+
mkdirSync(namespace, { recursive: true, mode: 0o700 });
|
|
166
|
+
|
|
167
|
+
return {
|
|
168
|
+
keyFor(task, input) {
|
|
169
|
+
return digest({
|
|
170
|
+
contractVersion: ANALYSIS_CONTRACT_VERSION,
|
|
171
|
+
signature,
|
|
172
|
+
task,
|
|
173
|
+
input,
|
|
174
|
+
});
|
|
175
|
+
},
|
|
176
|
+
read(key, expectedIds, validate) {
|
|
177
|
+
if (!SHA256_RE.test(key)) return null;
|
|
178
|
+
const path = join(namespace, `${key}.json`);
|
|
179
|
+
try {
|
|
180
|
+
const stat = lstatSync(path);
|
|
181
|
+
if (!stat.isFile() || stat.isSymbolicLink() || stat.size > MAX_ENTRY_BYTES) return null;
|
|
182
|
+
const groups = validateEntry(JSON.parse(readFileSync(path, 'utf8')), expectedIds);
|
|
183
|
+
validate?.(groups);
|
|
184
|
+
const now = new Date();
|
|
185
|
+
utimesSync(namespace, now, now);
|
|
186
|
+
return groups;
|
|
187
|
+
} catch {
|
|
188
|
+
return null;
|
|
189
|
+
}
|
|
190
|
+
},
|
|
191
|
+
write(key, ids, groups) {
|
|
192
|
+
if (!SHA256_RE.test(key)) return false;
|
|
193
|
+
let entry;
|
|
194
|
+
try {
|
|
195
|
+
entry = {
|
|
196
|
+
kind: ANALYSIS_CACHE_KIND,
|
|
197
|
+
version: ANALYSIS_CACHE_VERSION,
|
|
198
|
+
createdAt: new Date().toISOString(),
|
|
199
|
+
ids: [...ids],
|
|
200
|
+
groups: normalizeGroups(groups),
|
|
201
|
+
};
|
|
202
|
+
} catch {
|
|
203
|
+
return false;
|
|
204
|
+
}
|
|
205
|
+
const path = join(namespace, `${key}.json`);
|
|
206
|
+
const temporary = `${path}.${process.pid}.${Date.now()}.tmp`;
|
|
207
|
+
try {
|
|
208
|
+
const contents = JSON.stringify(entry) + '\n';
|
|
209
|
+
const bytes = Buffer.byteLength(contents);
|
|
210
|
+
const current = namespaceRecords(root).find((record) => record.path === namespace);
|
|
211
|
+
let existingBytes = 0;
|
|
212
|
+
try {
|
|
213
|
+
existingBytes = lstatSync(path).size;
|
|
214
|
+
} catch (error) {
|
|
215
|
+
if (error.code !== 'ENOENT') return false;
|
|
216
|
+
}
|
|
217
|
+
if (
|
|
218
|
+
bytes > MAX_ENTRY_BYTES ||
|
|
219
|
+
(current?.bytes || 0) - existingBytes + bytes > settings.maxBytes
|
|
220
|
+
) {
|
|
221
|
+
return false;
|
|
222
|
+
}
|
|
223
|
+
writeFileSync(temporary, contents, { encoding: 'utf8', mode: 0o600 });
|
|
224
|
+
renameSync(temporary, path);
|
|
225
|
+
prune(root, settings, namespace);
|
|
226
|
+
return true;
|
|
227
|
+
} catch {
|
|
228
|
+
rmSync(temporary, { force: true });
|
|
229
|
+
return false;
|
|
230
|
+
}
|
|
231
|
+
},
|
|
232
|
+
clear() {
|
|
233
|
+
try {
|
|
234
|
+
rmSync(namespace, { recursive: true, force: true });
|
|
235
|
+
const parent = dirname(namespace);
|
|
236
|
+
if (existsSync(parent) && readdirSync(parent).length === 0) rmSync(parent);
|
|
237
|
+
} catch {
|
|
238
|
+
// Cache cleanup is best-effort and does not affect the completed result.
|
|
239
|
+
}
|
|
240
|
+
},
|
|
241
|
+
path: namespace,
|
|
242
|
+
};
|
|
243
|
+
} catch {
|
|
244
|
+
return null;
|
|
245
|
+
}
|
|
246
|
+
}
|