dennice 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dennice-0.1.0/LICENSE +3 -0
- dennice-0.1.0/PKG-INFO +384 -0
- dennice-0.1.0/README.md +360 -0
- dennice-0.1.0/pyproject.toml +47 -0
- dennice-0.1.0/setup.cfg +4 -0
- dennice-0.1.0/src/dennice/__init__.py +6 -0
- dennice-0.1.0/src/dennice/__main__.py +3 -0
- dennice-0.1.0/src/dennice/benchmark/__init__.py +1 -0
- dennice-0.1.0/src/dennice/benchmark/builtin_tasks/__init__.py +1 -0
- dennice-0.1.0/src/dennice/benchmark/builtin_tasks/fixtures/query_history.csv +7 -0
- dennice-0.1.0/src/dennice/benchmark/builtin_tasks/fixtures/run_results.json +13 -0
- dennice-0.1.0/src/dennice/benchmark/builtin_tasks/fixtures/warehouse_history.csv +5 -0
- dennice-0.1.0/src/dennice/benchmark/builtin_tasks/snowflake_cost_001.yaml +22 -0
- dennice-0.1.0/src/dennice/benchmark/dataset.py +85 -0
- dennice-0.1.0/src/dennice/benchmark/metrics.py +23 -0
- dennice-0.1.0/src/dennice/benchmark/outcomes.py +127 -0
- dennice-0.1.0/src/dennice/benchmark/runner.py +97 -0
- dennice-0.1.0/src/dennice/benchmark/schema.py +94 -0
- dennice-0.1.0/src/dennice/benchmark/store.py +21 -0
- dennice-0.1.0/src/dennice/cli/__init__.py +1 -0
- dennice-0.1.0/src/dennice/cli/app.py +166 -0
- dennice-0.1.0/src/dennice/cognition/__init__.py +1 -0
- dennice-0.1.0/src/dennice/cognition/policies/__init__.py +1 -0
- dennice-0.1.0/src/dennice/cognition/policies/abstraction.md +5 -0
- dennice-0.1.0/src/dennice/cognition/policies/causal_categorization.md +5 -0
- dennice-0.1.0/src/dennice/cognition/policies/constraint_reasoning.md +5 -0
- dennice-0.1.0/src/dennice/cognition/policies/contradiction_resolution.md +5 -0
- dennice-0.1.0/src/dennice/cognition/policies/critical_inquiry.md +6 -0
- dennice-0.1.0/src/dennice/cognition/policies/decomposition.md +5 -0
- dennice-0.1.0/src/dennice/cognition/policies/empirical_induction.md +5 -0
- dennice-0.1.0/src/dennice/cognition/registry.py +47 -0
- dennice-0.1.0/src/dennice/cognition/taxonomy.py +25 -0
- dennice-0.1.0/src/dennice/core/__init__.py +1 -0
- dennice-0.1.0/src/dennice/core/accounting.py +94 -0
- dennice-0.1.0/src/dennice/core/attachments.py +49 -0
- dennice-0.1.0/src/dennice/core/checks.py +46 -0
- dennice-0.1.0/src/dennice/core/codex_auth.py +23 -0
- dennice-0.1.0/src/dennice/core/config.py +218 -0
- dennice-0.1.0/src/dennice/core/events.py +22 -0
- dennice-0.1.0/src/dennice/core/goals.py +136 -0
- dennice-0.1.0/src/dennice/core/harness.py +313 -0
- dennice-0.1.0/src/dennice/core/hooks.py +109 -0
- dennice-0.1.0/src/dennice/core/jsonrpc.py +153 -0
- dennice-0.1.0/src/dennice/core/mcp.py +246 -0
- dennice-0.1.0/src/dennice/core/model_catalog.py +77 -0
- dennice-0.1.0/src/dennice/core/models.py +169 -0
- dennice-0.1.0/src/dennice/core/native_gate.py +161 -0
- dennice-0.1.0/src/dennice/core/native_hook_client.py +42 -0
- dennice-0.1.0/src/dennice/core/process.py +49 -0
- dennice-0.1.0/src/dennice/core/skills.py +107 -0
- dennice-0.1.0/src/dennice/core/tools.py +208 -0
- dennice-0.1.0/src/dennice/core/verification.py +88 -0
- dennice-0.1.0/src/dennice/executors/__init__.py +9 -0
- dennice-0.1.0/src/dennice/executors/api.py +425 -0
- dennice-0.1.0/src/dennice/executors/base.py +11 -0
- dennice-0.1.0/src/dennice/executors/claude.py +550 -0
- dennice-0.1.0/src/dennice/executors/codex.py +138 -0
- dennice-0.1.0/src/dennice/executors/codex_appserver.py +246 -0
- dennice-0.1.0/src/dennice/executors/copilot.py +254 -0
- dennice-0.1.0/src/dennice/executors/factory.py +41 -0
- dennice-0.1.0/src/dennice/executors/mock.py +19 -0
- dennice-0.1.0/src/dennice/prompting/__init__.py +1 -0
- dennice-0.1.0/src/dennice/prompting/composer.py +53 -0
- dennice-0.1.0/src/dennice/routing/__init__.py +1 -0
- dennice-0.1.0/src/dennice/routing/base.py +10 -0
- dennice-0.1.0/src/dennice/routing/codex.py +163 -0
- dennice-0.1.0/src/dennice/routing/factory.py +44 -0
- dennice-0.1.0/src/dennice/routing/openjev.py +249 -0
- dennice-0.1.0/src/dennice/routing/oracle.py +18 -0
- dennice-0.1.0/src/dennice/routing/policy.py +73 -0
- dennice-0.1.0/src/dennice/routing/routing_decision.schema.json +79 -0
- dennice-0.1.0/src/dennice/routing/rule.py +61 -0
- dennice-0.1.0/src/dennice/runs/__init__.py +1 -0
- dennice-0.1.0/src/dennice/runs/ownership.py +39 -0
- dennice-0.1.0/src/dennice/runs/provider_sessions.py +33 -0
- dennice-0.1.0/src/dennice/runs/sessions.py +88 -0
- dennice-0.1.0/src/dennice/runs/store.py +264 -0
- dennice-0.1.0/src/dennice/tui/__init__.py +1 -0
- dennice-0.1.0/src/dennice/tui/app.py +3000 -0
- dennice-0.1.0/src/dennice/tui/extensions.py +112 -0
- dennice-0.1.0/src/dennice/tui/files.py +1948 -0
- dennice-0.1.0/src/dennice/tui/transcript.py +63 -0
- dennice-0.1.0/src/dennice.egg-info/PKG-INFO +384 -0
- dennice-0.1.0/src/dennice.egg-info/SOURCES.txt +122 -0
- dennice-0.1.0/src/dennice.egg-info/dependency_links.txt +1 -0
- dennice-0.1.0/src/dennice.egg-info/entry_points.txt +2 -0
- dennice-0.1.0/src/dennice.egg-info/requires.txt +14 -0
- dennice-0.1.0/src/dennice.egg-info/top_level.txt +1 -0
- dennice-0.1.0/tests/test_api_and_skills.py +198 -0
- dennice-0.1.0/tests/test_benchmark.py +76 -0
- dennice-0.1.0/tests/test_chat_interaction.py +113 -0
- dennice-0.1.0/tests/test_claude_runtime.py +273 -0
- dennice-0.1.0/tests/test_claude_stdio_integration.py +181 -0
- dennice-0.1.0/tests/test_cli.py +49 -0
- dennice-0.1.0/tests/test_cli_diagnostics.py +55 -0
- dennice-0.1.0/tests/test_cli_status.py +68 -0
- dennice-0.1.0/tests/test_codex_appserver.py +242 -0
- dennice-0.1.0/tests/test_codex_router_process.py +86 -0
- dennice-0.1.0/tests/test_codex_stdio_integration.py +111 -0
- dennice-0.1.0/tests/test_copilot_runtime.py +210 -0
- dennice-0.1.0/tests/test_crash_recovery.py +207 -0
- dennice-0.1.0/tests/test_files.py +188 -0
- dennice-0.1.0/tests/test_harness.py +70 -0
- dennice-0.1.0/tests/test_jsonrpc.py +119 -0
- dennice-0.1.0/tests/test_mascot.py +16 -0
- dennice-0.1.0/tests/test_mcp_stdio.py +110 -0
- dennice-0.1.0/tests/test_mcp_stdio_faults.py +71 -0
- dennice-0.1.0/tests/test_models.py +177 -0
- dennice-0.1.0/tests/test_openjev.py +116 -0
- dennice-0.1.0/tests/test_outcome_evaluation.py +63 -0
- dennice-0.1.0/tests/test_platform_contract.py +46 -0
- dennice-0.1.0/tests/test_production_boundaries.py +373 -0
- dennice-0.1.0/tests/test_registry.py +48 -0
- dennice-0.1.0/tests/test_remaining_extensions.py +229 -0
- dennice-0.1.0/tests/test_routing.py +12 -0
- dennice-0.1.0/tests/test_run_reliability.py +212 -0
- dennice-0.1.0/tests/test_session_search.py +101 -0
- dennice-0.1.0/tests/test_session_usability.py +306 -0
- dennice-0.1.0/tests/test_session_workspace.py +203 -0
- dennice-0.1.0/tests/test_smoke.py +122 -0
- dennice-0.1.0/tests/test_taxonomy.py +7 -0
- dennice-0.1.0/tests/test_tui.py +888 -0
- dennice-0.1.0/tests/test_usage_accounting.py +125 -0
- dennice-0.1.0/tests/test_verification.py +108 -0
dennice-0.1.0/LICENSE
ADDED
dennice-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,384 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: dennice
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A cognitive routing and evaluation harness for data analytics agents.
|
|
5
|
+
Author: Dennice contributors
|
|
6
|
+
License: Apache-2.0
|
|
7
|
+
Requires-Python: >=3.11
|
|
8
|
+
Description-Content-Type: text/markdown
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Requires-Dist: pydantic<3,>=2.7
|
|
11
|
+
Requires-Dist: Pillow>=10
|
|
12
|
+
Requires-Dist: PyYAML>=6.0
|
|
13
|
+
Requires-Dist: textual>=0.70
|
|
14
|
+
Requires-Dist: typer<1,>=0.12
|
|
15
|
+
Requires-Dist: httpx<1,>=0.27
|
|
16
|
+
Requires-Dist: jsonschema<5,>=4.20
|
|
17
|
+
Requires-Dist: regex>=2024.11.6
|
|
18
|
+
Requires-Dist: mcp<2,>=1.28
|
|
19
|
+
Provides-Extra: dev
|
|
20
|
+
Requires-Dist: pytest>=8; extra == "dev"
|
|
21
|
+
Requires-Dist: pytest-asyncio>=0.23; extra == "dev"
|
|
22
|
+
Requires-Dist: ruff>=0.6; extra == "dev"
|
|
23
|
+
Dynamic: license-file
|
|
24
|
+
|
|
25
|
+
# Dennice
|
|
26
|
+
|
|
27
|
+
Dennice is a local-first AI harness for analytics and engineering work. It makes three decisions explicit before an agent acts: what kind of reasoning the task needs, which approved model should execute it, and what evidence is needed to call the work complete.
|
|
28
|
+
|
|
29
|
+
Use it when a single general-purpose chat prompt is not enough. Dennice classifies the task, attaches concise reasoning guidance, can choose a model and reasoning effort from a controlled pool, and preserves an inspectable local record of the run.
|
|
30
|
+
|
|
31
|
+
## What Dennice offers
|
|
32
|
+
|
|
33
|
+
- **Cognitive routing.** A System 1 router identifies a task family plus primary and supporting reasoning demands.
|
|
34
|
+
- **PA policies.** Versioned philosophical-assistant policies give the executor task-appropriate investigative guidance without requiring private reasoning traces.
|
|
35
|
+
- **Model routing.** Fixed, shadow, and automatic modes select only from an approved pool inside the active provider.
|
|
36
|
+
- **Provider choice.** Run work through Codex, Claude Code, GitHub Copilot, OpenAI API, Anthropic API, a local OpenAI-compatible server, or an offline mock executor.
|
|
37
|
+
- **Controlled execution.** Permission profiles, per-action approvals, tool controls, hooks, and MCP integration keep authority explicit.
|
|
38
|
+
- **Evidence and recovery.** Local run traces capture routing, policies, effective model, approvals, tool events, usage, verification, and interrupted work.
|
|
39
|
+
|
|
40
|
+
Dennice does not make an answer correct by itself. Its routing rules and provider capability declarations must be evaluated on representative work before they are used for consequential decisions.
|
|
41
|
+
|
|
42
|
+
## Install and run
|
|
43
|
+
|
|
44
|
+
Dennice requires Python 3.11 or later.
|
|
45
|
+
|
|
46
|
+
```bash
|
|
47
|
+
git clone https://github.com/keith-fajardo/dennice.git
|
|
48
|
+
cd dennice
|
|
49
|
+
python -m venv .venv
|
|
50
|
+
source .venv/bin/activate
|
|
51
|
+
python -m pip install -e ".[dev]"
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
On Windows PowerShell:
|
|
55
|
+
|
|
56
|
+
```powershell
|
|
57
|
+
py -m venv .venv
|
|
58
|
+
.\.venv\Scripts\Activate.ps1
|
|
59
|
+
py -m pip install -e ".[dev]"
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
Create an offline project configuration and launch the terminal interface:
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
dennice init
|
|
66
|
+
dennice
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
`dennice init` creates `dennice.yaml`, benchmark folders, and `.dennice/runs/`. Its initial configuration uses the offline Rule router and Mock executor, so it makes no provider request.
|
|
70
|
+
|
|
71
|
+
You can also use the CLI directly:
|
|
72
|
+
|
|
73
|
+
```bash
|
|
74
|
+
dennice classify "Why did Snowflake credits increase yesterday?" --json
|
|
75
|
+
dennice run "Investigate why Snowflake credits increased yesterday."
|
|
76
|
+
dennice benchmark list
|
|
77
|
+
dennice benchmark run --mode router --json
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
In the TUI, use `Ctrl+S` to configure the router and executor independently, save, then submit a task.
|
|
81
|
+
|
|
82
|
+
Setup's **Test executor (live)** starts one bounded provider turn and may consume provider quota or credits. It uses an empty temporary workspace and a 30-second ceiling (or a lower configured limit). API requests are capped at 128 output tokens and one model call; native CLIs may make internal calls that Dennice cannot count, and usage reports can cross the 4,096-token stop threshold before the turn is interrupted. Claude's built-in tools are disabled for this check. **Test router** also contacts the selected router unless it is the offline Rule router.
|
|
83
|
+
|
|
84
|
+
## How a run works
|
|
85
|
+
|
|
86
|
+
```text
|
|
87
|
+
Task and allowed context
|
|
88
|
+
│
|
|
89
|
+
▼
|
|
90
|
+
System 1: task family and cognitive assessment
|
|
91
|
+
│
|
|
92
|
+
├──► optional model and effort selection
|
|
93
|
+
│
|
|
94
|
+
▼
|
|
95
|
+
PA policy selection and prompt composition
|
|
96
|
+
│
|
|
97
|
+
▼
|
|
98
|
+
System 2: selected provider executor and permitted tools
|
|
99
|
+
│
|
|
100
|
+
▼
|
|
101
|
+
Verification, trace, and session state
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
The router identifies a task family, a primary cognitive demand, optional supporting demands, and an execution assessment. The primary demand selects the main PA policy; supporting demands can add a bounded number of supplementary policies.
|
|
105
|
+
|
|
106
|
+
| Cognitive demand | What it guides |
|
|
107
|
+
|---|---|
|
|
108
|
+
| `critical_inquiry` | Clarifying assumptions and discriminating hypotheses |
|
|
109
|
+
| `empirical_induction` | Evidence, measurements, and cautious inference |
|
|
110
|
+
| `decomposition` | Separating a task into independently testable parts |
|
|
111
|
+
| `constraint_reasoning` | Requirements, invariants, and business rules |
|
|
112
|
+
| `causal_categorization` | Entities, failure modes, and causal structure |
|
|
113
|
+
| `contradiction_resolution` | Conflicting definitions or claims |
|
|
114
|
+
| `abstraction` | Reusable concepts and architecture |
|
|
115
|
+
|
|
116
|
+
PA policies shape the execution prompt. They do not grant permissions, execute tools, select a provider, or prove a conclusion.
|
|
117
|
+
|
|
118
|
+
## Configure a provider
|
|
119
|
+
|
|
120
|
+
System 1 and System 2 are separate. For example, OpenJev can classify the task while Claude Code executes it.
|
|
121
|
+
|
|
122
|
+
| Provider | Role | Authentication |
|
|
123
|
+
|---|---|---|
|
|
124
|
+
| Rule | Offline System 1 router | None |
|
|
125
|
+
| Codex | Router or executor | Existing local Codex CLI login; ChatGPT mode by default |
|
|
126
|
+
| Jev | Hosted System 1 router | Environment-variable API key |
|
|
127
|
+
| OpenJev | Local System 1 router | None; loopback endpoint |
|
|
128
|
+
| Claude Code | Executor | Existing Claude Code login |
|
|
129
|
+
| GitHub Copilot CLI | Executor | Existing Copilot login |
|
|
130
|
+
| OpenAI API / Anthropic API | Executor | Provider API key in environment |
|
|
131
|
+
| Local | Executor | Loopback OpenAI-compatible server |
|
|
132
|
+
| Mock | Offline executor | None |
|
|
133
|
+
|
|
134
|
+
Dennice stores only environment-variable names, never provider secrets. The API adapters use separately billed API credentials. CLI adapters use the provider CLI's active authentication, which may itself be a subscription or API key.
|
|
135
|
+
|
|
136
|
+
### Codex, Claude Code, and Copilot
|
|
137
|
+
|
|
138
|
+
Install and authenticate the provider’s own CLI before choosing it in Setup. Dennice invokes that local tool and does not collect subscription credentials.
|
|
139
|
+
|
|
140
|
+
[Codex app-server's account status](https://learn.chatgpt.com/docs/app-server) distinguishes ChatGPT and API-key authentication. Dennice checks `account/read` before Codex executor turns and System 1 classification, defaulting each to ChatGPT authentication. If the CLI reports API-key or another provider mode, the model call stops before inference. Choose **API key** or **Provider default / other** in the respective Setup section only when you intend that mode. The YAML levers are `executor.codex_cli_auth` and `router.codex_cli_auth`, each set to `chatgpt | api_key | provider_default`. Dennice records only the executor's authentication mode in its run trace; this preflight does not prove the eventual billing record.
|
|
141
|
+
|
|
142
|
+
[Claude Code's authentication documentation](https://code.claude.com/docs/en/authentication) says environment variables can select cloud providers, gateways, and credentials ahead of the saved login. Dennice checks `claude auth status` before each Claude run and defaults to requiring subscription authentication (`claude.ai` login or its subscription OAuth token). API key/token variables, custom endpoints, cloud-provider selectors and credentials, and alternate Anthropic profiles/federation settings stop the subscription-mode run before model inference. Select **API key** or **Provider default / other** in Setup only when you intend that CLI authentication and its billing. The equivalent YAML lever is `executor.claude_cli_auth: subscription | api_key | provider_default`; `provider_default` explicitly accepts the CLI's current non-subscription method. Dennice records only the normalized method, not the CLI's status JSON or credentials. This preflight does not prove the provider's eventual billing record.
|
|
143
|
+
|
|
144
|
+
These adapters currently target local, user-operated installations. [OpenAI's app-server documentation](https://learn.chatgpt.com/docs/app-server) distinguishes local/open-source authentication from commercial or hosted use, and [Anthropic's Claude Code guidance](https://code.claude.com/docs/en/legal-and-compliance) sets conditions for products that run its binary. Review the [provider integration mapping](docs/provider-integration-review-2026-10-04.md) before distributing or hosting Dennice.
|
|
145
|
+
|
|
146
|
+
- Codex defaults to a read-only sandbox. Workspace-write must be explicitly selected.
|
|
147
|
+
- Claude Code uses a restricted plan profile by default. Its write profile exposes only the allowed file tools, and exact edits require approval.
|
|
148
|
+
- Copilot uses a narrow read or file-write allowlist. Dennice does not enable shell or MCP tools for the Copilot adapter.
|
|
149
|
+
- Copilot runs through the user's stored GitHub Copilot CLI login (the CLI may fall back to the authenticated `gh` account if no Copilot login is stored). Dennice removes `COPILOT_PROVIDER_*`, `COPILOT_MODEL`, and any saved BYOK registry from that subprocess, and supplies an empty temporary provider registry. It stops before startup if `COPILOT_GITHUB_TOKEN`, `GH_TOKEN`, or `GITHUB_TOKEN` has a non-empty value, because those variables can override the saved Copilot account. Unset them, sign in with `copilot login`, and verify the CLI's active account before testing. GitHub's individual-account terms allow AI inputs and outputs to be used for model improvement unless the account opts out in its settings; Dennice's local OTel content-capture setting does not change GitHub's policy. Check the current [provider integration mapping](docs/provider-integration-review-2026-10-04.md) and account privacy setting before sending workspace content.
|
|
150
|
+
- Copilot token totals are read from a private, temporary local OpenTelemetry file with message-content capture disabled. A budgeted run fails if the installed CLI does not produce valid usage counters; Copilot credits and provider cost multipliers are not currency and are not included.
|
|
151
|
+
|
|
152
|
+
Refresh the provider’s model catalog in Setup before choosing a model. Availability, quota, and supported effort levels remain account-specific. Claude labels include the CLI-reported resolved ID, such as `claude-sonnet-5-5`.
|
|
153
|
+
|
|
154
|
+
### API and local execution
|
|
155
|
+
|
|
156
|
+
Set API credentials in the environment before starting Dennice:
|
|
157
|
+
|
|
158
|
+
```bash
|
|
159
|
+
export OPENAI_API_KEY="your-openai-api-key"
|
|
160
|
+
export ANTHROPIC_API_KEY="your-anthropic-api-key"
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
Choose the API provider in Setup, refresh its catalog, and select a specific model. For a local provider, enter an OpenAI-compatible `/v1` endpoint, such as:
|
|
164
|
+
|
|
165
|
+
```text
|
|
166
|
+
Ollama: http://127.0.0.1:11434/v1
|
|
167
|
+
LM Studio: http://127.0.0.1:1234/v1
|
|
168
|
+
```
|
|
169
|
+
|
|
170
|
+
API/local execution can use Dennice’s native tool loop when `/tools on` is set. Local endpoints must be loopback. Hosted API keys are sent only to official provider HTTPS hosts, and redirects are refused.
|
|
171
|
+
|
|
172
|
+
### Local OpenJev router
|
|
173
|
+
|
|
174
|
+
OpenJev is a local System 1 decision service, not the chat model. It receives the task and typed classification questions, then returns task-family and cognitive-demand scores.
|
|
175
|
+
|
|
176
|
+
```yaml
|
|
177
|
+
router:
|
|
178
|
+
provider: openjev
|
|
179
|
+
model: openjev
|
|
180
|
+
openjev:
|
|
181
|
+
endpoint: http://127.0.0.1:8771/v1/systemone
|
|
182
|
+
model: openjev
|
|
183
|
+
```
|
|
184
|
+
|
|
185
|
+
The endpoint must be loopback HTTP(S). See the [OpenJev documentation](https://huggingface.co/openjev/openjev) for serving requirements.
|
|
186
|
+
|
|
187
|
+
## Automatic model and effort routing
|
|
188
|
+
|
|
189
|
+
Automatic routing chooses only from models explicitly approved for the active provider. It does not change provider, login, permission mode, or tool authority. The provider CLI still determines whether its own active authentication uses a subscription or API billing; check that separately as described above.
|
|
190
|
+
|
|
191
|
+
| Mode | Result |
|
|
192
|
+
|---|---|
|
|
193
|
+
| `fixed` | Use the configured executor model and effort. |
|
|
194
|
+
| `shadow` | Record the recommended route, but use the configured model and effort. |
|
|
195
|
+
| `auto` | Run the eligible model and effort from the approved pool. |
|
|
196
|
+
|
|
197
|
+
The TUI status line keeps the configured model and effort visible. When a run selects different values, `Run route` (or `Last route` afterward) shows the effective model and effort. The `Ctx` meter uses the reported context for that route.
|
|
198
|
+
|
|
199
|
+
To enable automatic routing, define candidates with their known capabilities. This Codex example must be adapted to models your account supports:
|
|
200
|
+
|
|
201
|
+
```yaml
|
|
202
|
+
routing:
|
|
203
|
+
mode: auto
|
|
204
|
+
model_pinned: false
|
|
205
|
+
effort_pinned: false
|
|
206
|
+
# PA policies attach only when cognitive scores meet these thresholds.
|
|
207
|
+
# Scores are routing signals, not calibrated success probabilities.
|
|
208
|
+
primary_threshold: 0.8
|
|
209
|
+
supporting_threshold: 0.55
|
|
210
|
+
model_pool:
|
|
211
|
+
codex:
|
|
212
|
+
- model: gpt-6-luna
|
|
213
|
+
tier: lightweight
|
|
214
|
+
context_tokens: 32000
|
|
215
|
+
efforts: [low, medium, high]
|
|
216
|
+
- model: gpt-5.6-terra
|
|
217
|
+
tier: strong
|
|
218
|
+
context_tokens: 32000
|
|
219
|
+
efforts: [low, medium, high]
|
|
220
|
+
```
|
|
221
|
+
|
|
222
|
+
The built-in route policy uses low effort for lightweight work, medium effort for balanced work, and high effort for strong work. A simple low-stakes task can use a lightweight candidate; a moderate task uses balanced when available; complex, high-stakes, or uncertain work uses a strong candidate. If no compatible candidate exists, Auto stops before execution rather than guessing a capability or switching providers.
|
|
223
|
+
|
|
224
|
+
Candidates declare context size and tool/vision support. Dennice does not infer those capabilities from a model name. `model_pinned: true` restricts Auto to the current model; `effort_pinned: true` preserves the current effort. The current policy does not automatically select `xhigh`; choose it manually until an explicit policy is added.
|
|
225
|
+
|
|
226
|
+
Manage the same settings in the TUI:
|
|
227
|
+
|
|
228
|
+
```text
|
|
229
|
+
/routing fixed|shadow|auto
|
|
230
|
+
/routing pin-model|unpin-model|pin-effort|unpin-effort
|
|
231
|
+
/pool add <exact-model-id> lightweight|balanced|strong [tools] [vision] [context=8192] [efforts=low,medium,high]
|
|
232
|
+
/pool remove <exact-model-id>
|
|
233
|
+
```
|
|
234
|
+
|
|
235
|
+
Classifier confidence is not a probability that execution will succeed. The primary score must meet `primary_threshold` before any PA is attached; each supporting policy must also meet `supporting_threshold`.
|
|
236
|
+
|
|
237
|
+
## Use the terminal interface
|
|
238
|
+
|
|
239
|
+
The TUI stores sessions, transcripts, workspace context, and traces in the current project directory.
|
|
240
|
+
|
|
241
|
+
| Action | Command or shortcut |
|
|
242
|
+
|---|---|
|
|
243
|
+
| New session | `Ctrl+N` or `/new` |
|
|
244
|
+
| Open Setup | `Ctrl+S` |
|
|
245
|
+
| Stop active work | `Ctrl+X` |
|
|
246
|
+
| Show active configuration | `/config` |
|
|
247
|
+
| Switch session | `/session <number>` |
|
|
248
|
+
| Search sessions | `/sessions` |
|
|
249
|
+
| Rename session | `/rename <title>` |
|
|
250
|
+
| Choose model | `/model <name>` or Setup |
|
|
251
|
+
| Choose effort | `/effort <low|medium|high|xhigh|default>` |
|
|
252
|
+
| Choose permissions | `/permissions <read-only|read-write|plan>` |
|
|
253
|
+
| Set workspace directory | `/cwd "path"` |
|
|
254
|
+
| Attach image | `/attach "/absolute/path/image.png"` or `Ctrl+V` |
|
|
255
|
+
| Remove pending images | `/clear-images` |
|
|
256
|
+
| Compact earlier context | `/compact` |
|
|
257
|
+
| Start a bounded goal | `/goal <objective>` |
|
|
258
|
+
| Inspect recovery | `/recovery list` |
|
|
259
|
+
|
|
260
|
+
The composer accepts multiple lines; press `Ctrl+Enter` to submit. Direct local terminal commands begin with `!`, for example `! git status --short`. Dennice captures them as terminal output and keeps them out of the model conversation history.
|
|
261
|
+
|
|
262
|
+
### Sessions, workspace files, and images
|
|
263
|
+
|
|
264
|
+
Sessions survive restart in the same workspace. Local session search supports case-insensitive regular expressions such as `snowflake|billing`. Each session keeps its own workspace directory.
|
|
265
|
+
|
|
266
|
+
The file explorer provides bounded search, preview, explicit save, undo/redo, and regex replacement. It excludes linked, credential-like, binary, and oversized files. Deletes, moves, and writes require visible user actions; Dennice never silently bulk-overwrites a workspace. Secure file browsing and editing require POSIX semantics, including WSL; native Windows file editing fails closed.
|
|
267
|
+
|
|
268
|
+
You can attach up to eight PNG, JPEG, GIF, or WebP images of up to 10 MB each. Clipboard snapshots are stored under `.dennice/attachments/` and are not uploaded until a task is submitted. Codex receives native image attachments; Claude Code is directed to read the image file; API/local providers receive image content where supported. The System 1 router currently receives text rather than image pixels.
|
|
269
|
+
|
|
270
|
+
### Skills
|
|
271
|
+
|
|
272
|
+
Use `/skills` to browse local Claude and Codex skills, then `/skill <qualified-key> <task>` to run one. Skill text remains user-level context: it cannot expand permissions, automatically run scripts, or bypass approval.
|
|
273
|
+
|
|
274
|
+
## Hooks and MCP
|
|
275
|
+
|
|
276
|
+
Dennice has its own lifecycle hooks and MCP connector. They complement rather than replace Claude Code’s provider-native extension system.
|
|
277
|
+
|
|
278
|
+
Supported hook events are `before_route`, `after_route`, `before_execution`, `before_tool`, `after_tool`, `before_verification`, and `turn_complete`. Hooks receive bounded metadata and can allow or deny the harness lifecycle operation. They cannot grant permissions, override routing pins, or intercept every provider-native tool action.
|
|
279
|
+
|
|
280
|
+
MCP is available to API/local tool loops, not injected into the native Codex, Claude Code, or Copilot loops. MCP tools, resources, and prompts must be allowlisted; every access requires approval. MCP content is treated as untrusted data, never as system instructions.
|
|
281
|
+
|
|
282
|
+
Tools, hooks, and MCP are disabled by default:
|
|
283
|
+
|
|
284
|
+
```yaml
|
|
285
|
+
tools:
|
|
286
|
+
enabled: false
|
|
287
|
+
root: .
|
|
288
|
+
hooks:
|
|
289
|
+
- name: validate
|
|
290
|
+
event: before_execution
|
|
291
|
+
command: [python, scripts/validate.py]
|
|
292
|
+
enabled: false
|
|
293
|
+
required: true
|
|
294
|
+
mcp:
|
|
295
|
+
- name: example
|
|
296
|
+
transport: stdio
|
|
297
|
+
command: [python, scripts/mcp_server.py]
|
|
298
|
+
enabled: false
|
|
299
|
+
approved_tools: [lookup]
|
|
300
|
+
```
|
|
301
|
+
|
|
302
|
+
Use `/tools on`, `/hooks list|trust|enable|disable <name>`, and `/mcp list|trust|enable|disable|test <name>`. Trust is tied to the current launch and workspace; changing server settings invalidates it. Opening configuration does not launch a process.
|
|
303
|
+
|
|
304
|
+
## Verification, traces, and recovery
|
|
305
|
+
|
|
306
|
+
Dennice records the task, configuration snapshot, routing decision, PA policies, effective model route, approvals, tool events, usage, output, and verification result in `.dennice/runs/runs.sqlite3` by default.
|
|
307
|
+
|
|
308
|
+
```bash
|
|
309
|
+
dennice inspect <run-id>
|
|
310
|
+
```
|
|
311
|
+
|
|
312
|
+
`dennice run` exits nonzero for failed, timed-out, cancelled, or interrupted runs. Both commands label partial output as unverified and show the stopping reason; `dennice inspect --json` includes the complete trace.
|
|
313
|
+
|
|
314
|
+
When a native CLI fails, Dennice records its exit status or a bounded error category. It withholds raw provider stderr and error messages from the trace because they may contain prompt text or credentials. Inspect the provider CLI locally for detailed diagnostics.
|
|
315
|
+
|
|
316
|
+
Run data contains plaintext prompts, transcripts, image paths, configuration references, and output. Treat the runs directory as sensitive; local permissions are not encryption.
|
|
317
|
+
|
|
318
|
+
Interrupted work is never replayed automatically because an external action may already have happened. Use `/recovery inspect <run-id>` and `/recovery reconcile <run-id> <operation-id> <completed|not_executed|compensated> <evidence>` to record verified outcomes.
|
|
319
|
+
|
|
320
|
+
Goals add an objective and bounded retries:
|
|
321
|
+
|
|
322
|
+
```text
|
|
323
|
+
/goal Verify the dashboard query against fixture data
|
|
324
|
+
```
|
|
325
|
+
|
|
326
|
+
Configure meaningful completion checks first:
|
|
327
|
+
|
|
328
|
+
```yaml
|
|
329
|
+
verification:
|
|
330
|
+
required_files: [report.md]
|
|
331
|
+
commands:
|
|
332
|
+
- [python, -m, pytest, tests/test_report.py]
|
|
333
|
+
```
|
|
334
|
+
|
|
335
|
+
A nonempty answer receives an `unverified` run status unless an independent file or command check is configured and passes. Verification commands require approval and are not sandboxed. Budgets limit elapsed time, model calls, tool calls, output tokens, and reported total tokens, but provider usage may remain unknown.
|
|
336
|
+
|
|
337
|
+
For Codex, `budgets.max_total_tokens` defaults to `100000` and counts cumulative reported input plus output tokens for a run. The status line shows a separate `Run` meter against this limit, then keeps it as `Last run` when execution stops. `≥` means some usage is unknown; `pending` and `unknown` mean no usable provider report has arrived. The `Ctx` indicator shows the latest model request against the model's context window, not the remaining run budget. Native usage reports arrive after model activity, so a run can cross the limit before Dennice interrupts it. Raise the limit in `dennice.yaml` only when the task warrants the additional usage:
|
|
338
|
+
|
|
339
|
+
```yaml
|
|
340
|
+
budgets:
|
|
341
|
+
max_total_tokens: 200000
|
|
342
|
+
```
|
|
343
|
+
|
|
344
|
+
API, Claude, Codex, and Copilot adapters fail a budgeted run when the token counts needed to enforce the cap are missing. Copilot emits usage only after its CLI invocation completes, so Dennice can report an over-limit result but cannot interrupt an internal Copilot model call at the exact token boundary. Copilot usage metering uses a temporary local OTel file and explicitly disables prompt/response content capture.
|
|
345
|
+
|
|
346
|
+
## Privacy and operational boundaries
|
|
347
|
+
|
|
348
|
+
- The router cannot choose a provider, grant permission, execute a tool, or create arbitrary routes.
|
|
349
|
+
- Codex classification runs its CLI in a disposable directory with a read-only sandbox; it receives the task text and opt-in routing history, not the project working directory.
|
|
350
|
+
- Auto-routing stays within the active provider’s approved pool.
|
|
351
|
+
- Set `privacy.router_history: true` before prior conversation turns are shared with the router.
|
|
352
|
+
- `privacy.local_only: true` rejects remote/CLI providers and unsandboxed hooks or MCP processes. It does not sandbox arbitrary local processes.
|
|
353
|
+
- Hosted routers reject redirects and implicit proxies; OpenJev accepts loopback endpoints only.
|
|
354
|
+
- Cancelling stops ongoing work but cannot undo an external side effect.
|
|
355
|
+
|
|
356
|
+
Review the selected provider, model, permission profile, tool settings, and workspace directory before authorizing work with side effects.
|
|
357
|
+
|
|
358
|
+
## Known limits
|
|
359
|
+
|
|
360
|
+
- The Rule router is a narrow deterministic development classifier. Use Codex, Jev, or OpenJev for broader task classification.
|
|
361
|
+
- The Rule router lacks a full assessment for most tasks, so automatic routing treats them conservatively.
|
|
362
|
+
- OpenJev’s current typed request does not ask whether tools or vision are needed. Attached images are still detected by the route policy.
|
|
363
|
+
- Cognitive scores are uncalibrated, so PA thresholds need held-out evaluation before they can be treated as quality controls.
|
|
364
|
+
- Model capability declarations, context sizes, and effort support must be verified with the provider and account.
|
|
365
|
+
- Current benchmarks measure routing labels, not general task quality, cost savings, or provider parity.
|
|
366
|
+
- Provider-native sessions and tools remain provider-specific. Dennice cannot intercept every action inside a native provider loop.
|
|
367
|
+
- Full platform qualification, held-out routing evaluations, and some provider integration evidence remain release gates.
|
|
368
|
+
|
|
369
|
+
See the [production-harness plan](docs/production-harness-plan.md), [recovery and accounting](docs/recovery-and-accounting.md), [platform and evaluation](docs/platform-and-evaluation.md), [extension qualification](docs/extension-qualification.md), and [provider integration review](docs/provider-integration-review-2026-10-04.md) for implementation boundaries and release criteria.
|
|
370
|
+
|
|
371
|
+
## Develop and test
|
|
372
|
+
|
|
373
|
+
```bash
|
|
374
|
+
pytest -q
|
|
375
|
+
ruff check src tests
|
|
376
|
+
```
|
|
377
|
+
|
|
378
|
+
The bundled Snowflake benchmark is a synthetic development fixture. `dennice init` copies its three evidence files into `benchmarks/fixtures/`; benchmark runs validate and pass their bounded contents as untrusted task data. Missing or escaping fixture paths fail before a model call. The gold answer and cognitive labels are never added to the task prompt. This example scores routing labels only; use controlled representative tasks and independent completion checks to evaluate routing, PA policies, model selection, cost, or latency.
|
|
379
|
+
|
|
380
|
+
## Windows notes
|
|
381
|
+
|
|
382
|
+
The TUI works in Windows terminals. For Codex, Claude Code, and Copilot, install and authenticate the provider CLI in Git Bash or WSL and ensure `bash` is on `PATH`; Dennice invokes those CLIs through Bash on Windows.
|
|
383
|
+
|
|
384
|
+
Claude Code supports Windows through WSL or Git for Windows. GitHub Copilot CLI is installed with `npm install -g @github/copilot` and authenticated with `copilot login`. Provider access remains account and organization dependent.
|