otto-cli-agent 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (129) hide show
  1. agent/README.md +22 -0
  2. agent/__init__.py +0 -0
  3. agent/cli/README.md +77 -0
  4. agent/cli/__init__.py +0 -0
  5. agent/cli/art.py +371 -0
  6. agent/cli/chat.py +309 -0
  7. agent/cli/clipboard.py +106 -0
  8. agent/cli/context.py +56 -0
  9. agent/cli/doctor.py +60 -0
  10. agent/cli/errors.py +58 -0
  11. agent/cli/eval.py +130 -0
  12. agent/cli/eval_claw.py +414 -0
  13. agent/cli/eval_compaction.py +106 -0
  14. agent/cli/eval_hle.py +92 -0
  15. agent/cli/eval_memory.py +253 -0
  16. agent/cli/eval_swe.py +172 -0
  17. agent/cli/lessons.py +97 -0
  18. agent/cli/main.py +67 -0
  19. agent/cli/modals.py +570 -0
  20. agent/cli/models.py +64 -0
  21. agent/cli/output.py +54 -0
  22. agent/cli/route.py +78 -0
  23. agent/cli/sessions.py +125 -0
  24. agent/cli/setup_screen.py +562 -0
  25. agent/cli/shell.py +548 -0
  26. agent/cli/tui.py +1807 -0
  27. agent/cli/ui.py +14 -0
  28. agent/cli/usage_panel.py +159 -0
  29. agent/config/README.md +7 -0
  30. agent/config/__init__.py +0 -0
  31. agent/config/envfile.py +76 -0
  32. agent/eval/README.md +76 -0
  33. agent/eval/__init__.py +0 -0
  34. agent/eval/claw_bench.py +1031 -0
  35. agent/eval/compaction_bench.py +229 -0
  36. agent/eval/data/README.md +10 -0
  37. agent/eval/data/claw/README.md +108 -0
  38. agent/eval/data/claw/llm_judge-gemini.patch +57 -0
  39. agent/eval/data/claw/otto.yaml +35 -0
  40. agent/eval/failures.py +276 -0
  41. agent/eval/golden/README.md +33 -0
  42. agent/eval/golden/code_01.json +6 -0
  43. agent/eval/golden/code_02.json +6 -0
  44. agent/eval/golden/code_03.json +6 -0
  45. agent/eval/golden/code_04.json +6 -0
  46. agent/eval/golden/code_05.json +6 -0
  47. agent/eval/golden/code_06.json +6 -0
  48. agent/eval/golden/math_01.json +6 -0
  49. agent/eval/golden/math_02.json +6 -0
  50. agent/eval/golden/math_03.json +6 -0
  51. agent/eval/golden/math_04.json +6 -0
  52. agent/eval/golden/math_05.json +6 -0
  53. agent/eval/golden/math_06.json +6 -0
  54. agent/eval/golden/nphard_gcp_01.json +6 -0
  55. agent/eval/golden/nphard_ksp_01.json +6 -0
  56. agent/eval/golden/nphard_math_binpacking_01.json +6 -0
  57. agent/eval/golden/nphard_math_clique_01.json +6 -0
  58. agent/eval/golden/nphard_math_setcover_01.json +6 -0
  59. agent/eval/golden/nphard_math_subsetsum_01.json +6 -0
  60. agent/eval/golden/nphard_tsp_01.json +6 -0
  61. agent/eval/golden/nphard_tsp_02.json +6 -0
  62. agent/eval/hle_bench.py +273 -0
  63. agent/eval/langfuse_sync.py +172 -0
  64. agent/eval/memory_bench.py +538 -0
  65. agent/eval/runner.py +174 -0
  66. agent/eval/single_agent.py +120 -0
  67. agent/eval/swe_bench.py +604 -0
  68. agent/eval/terminal_bench.py +345 -0
  69. agent/memory/README.md +102 -0
  70. agent/memory/__init__.py +42 -0
  71. agent/memory/embeddings.py +302 -0
  72. agent/memory/hashing.py +15 -0
  73. agent/memory/lessons.py +483 -0
  74. agent/memory/queue.py +531 -0
  75. agent/memory/retrieval.py +493 -0
  76. agent/memory/session.py +60 -0
  77. agent/memory/sessions.py +436 -0
  78. agent/memory/store.py +429 -0
  79. agent/memory/tokens.py +60 -0
  80. agent/memory/wiring.py +146 -0
  81. agent/pipeline/README.md +135 -0
  82. agent/pipeline/__init__.py +0 -0
  83. agent/pipeline/browsing.py +609 -0
  84. agent/pipeline/budget.py +403 -0
  85. agent/pipeline/codemap.py +254 -0
  86. agent/pipeline/evidence.py +325 -0
  87. agent/pipeline/execution.py +67 -0
  88. agent/pipeline/modes.py +137 -0
  89. agent/pipeline/native.py +1137 -0
  90. agent/pipeline/nodes.py +3644 -0
  91. agent/pipeline/pricing.py +209 -0
  92. agent/pipeline/progress.py +139 -0
  93. agent/pipeline/rag.py +139 -0
  94. agent/pipeline/research.py +1325 -0
  95. agent/pipeline/run.py +528 -0
  96. agent/pipeline/screen.py +77 -0
  97. agent/pipeline/state.py +220 -0
  98. agent/pipeline/toolkit.py +328 -0
  99. agent/pipeline/tools.py +1990 -0
  100. agent/pipeline/tracing.py +147 -0
  101. agent/pipeline/usage.py +251 -0
  102. agent/pipeline/vision.py +84 -0
  103. agent/pipeline/walkthrough.py +735 -0
  104. agent/pipeline/workspace.py +229 -0
  105. agent/router/README.md +60 -0
  106. agent/router/__init__.py +0 -0
  107. agent/router/automap.py +114 -0
  108. agent/router/health.py +229 -0
  109. agent/router/llm_provider/README.md +38 -0
  110. agent/router/llm_provider/__init__.py +202 -0
  111. agent/router/llm_provider/anthropic_provider.py +128 -0
  112. agent/router/llm_provider/base.py +507 -0
  113. agent/router/llm_provider/custom.py +152 -0
  114. agent/router/llm_provider/gemini_provider.py +122 -0
  115. agent/router/llm_provider/inception_provider.py +687 -0
  116. agent/router/llm_provider/openai_provider.py +151 -0
  117. agent/router/llm_provider/retired.py +145 -0
  118. agent/router/llm_provider/temperature.py +371 -0
  119. agent/router/mapping.py +579 -0
  120. agent/router/outcomes.py +363 -0
  121. agent/router/overrides.py +389 -0
  122. agent/router/reload.py +28 -0
  123. agent/router/router.py +413 -0
  124. agent/router/setup.py +123 -0
  125. otto_cli_agent-0.1.0.dist-info/METADATA +115 -0
  126. otto_cli_agent-0.1.0.dist-info/RECORD +129 -0
  127. otto_cli_agent-0.1.0.dist-info/WHEEL +4 -0
  128. otto_cli_agent-0.1.0.dist-info/entry_points.txt +2 -0
  129. otto_cli_agent-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,1325 @@
1
+ """The research workflow: a long document, built section by section.
2
+
3
+ A predefined code path, not a mode. The agent loop (nodes.py) is the right
4
+ shape for a task with a checkable result and the wrong shape for a document:
5
+ its prompt asks for "the numbers, the names, the decision", its replies stop
6
+ at the seat's max_tokens, `read_file` clips at 4000 characters, `_compact`
7
+ never shrinks the model's own prose, and the budget tells it to stop
8
+ gathering at a fifth of its calls. Asked for ten generations of a fictional
9
+ empire with a 500-word narrative each, it wrote a simulation, printed the
10
+ numbers, and answered in 300 words -- and the judge, whose criteria may not
11
+ mention length, approved. Every one of those pushes toward compression, and
12
+ none of them is the model being lazy.
13
+
14
+ So here the model never decides control flow. The workflow does, in this
15
+ order, and the model fills in content:
16
+
17
+ 1. OUTLINE (one call, the plan seat): a JSON plan -- sections, each with a
18
+ brief, a word minimum, the sub-headings it must contain, one-line checks,
19
+ and questions to look up first -- plus the seed of the continuity ledger.
20
+ 2. per section, in order:
21
+ GATHER (optional, a bounded agent loop in `find` mode) writes notes to a
22
+ file; WRITE (one call, the reason seat) returns the section as plain
23
+ Markdown ending in a fenced ledger block -- no ACTION protocol, so the
24
+ 8192 tokens go to prose and nothing is truncated at a line that happens
25
+ to start with `FINAL:`; CHECK in code (words, required parts, protocol
26
+ leak, truncation) and optionally one call against the outline's checks;
27
+ REVISE once if anything failed, keeping the better draft.
28
+ 3. ASSEMBLE in code: title, abstract, contents, the sections in order.
29
+ 4. CONVERT when the request named docx, pdf or xlsx.
30
+ 5. Hand the evaluator a REPORT computed from the files -- words against
31
+ minimums, parts found, checks passed -- and the weakest section in
32
+ full. It cannot read thirty thousand words through a 4000-character
33
+ clip, so it judges the report and may spend its one check on the file.
34
+
35
+ Continuity is the ledger: a small JSON of named entities, settled facts and
36
+ open threads that each writer receives, updates, and hands to the next, so
37
+ the tenth section can refer back to a law the first one passed. The writer
38
+ also sees the tail of the previous section. When a writer returns no usable
39
+ ledger the previous one is carried, so continuity is never worse than "the
40
+ last 1500 characters plus whatever ledger survived".
41
+
42
+ Everything the workflow spends goes through nodes.py's `_call`, so the
43
+ budget, the usage ledger and `model_calls` are right without any plumbing
44
+ of their own. What it does with a budget: skip the reconnaissance note
45
+ (workers must not be told to stop looking on their first iteration), drop
46
+ gathering and checks in the wrap-up stretch, and when the budget is spent,
47
+ assemble what exists, say which sections are missing, and end without a
48
+ judgment -- the same exit the loop takes, for the same reason.
49
+
50
+ Files, under `otto_research/<task-slug>/` in the workspace: outline.json,
51
+ ledger.json, sections/NN-slug.md, notes/NN-slug.md, document.md and the
52
+ converted file if one was asked for. A re-entry after an evaluator rejection
53
+ reads those back rather than carrying them in state, rewrites the weakest
54
+ sections with the judge's feedback, and reassembles.
55
+ """
56
+
57
+ from __future__ import annotations
58
+
59
+ import dataclasses
60
+ import json
61
+ import logging
62
+ import re
63
+ from dataclasses import dataclass, field
64
+
65
+ from langchain_core.messages import AIMessage, HumanMessage, SystemMessage
66
+ from langgraph.graph import END
67
+ from langgraph.types import Command
68
+
69
+ from agent.pipeline import nodes as pn
70
+ from agent.pipeline.budget import Budget, current_budget
71
+ from agent.pipeline.progress import report as report_progress
72
+ from agent.pipeline.state import AgentState
73
+ from agent.pipeline.tools import execute_python, reachable_tools, write_file
74
+ from agent.pipeline.toolkit import render_note
75
+ from agent.pipeline.workspace import OutsideWorkspace, resolve_in_workspace, workspace_note
76
+ from agent.router.llm_provider.base import ProviderError
77
+ from agent.router.mapping import Task
78
+
79
+ logger = logging.getLogger(__name__)
80
+
81
+ #: Where a document and its working files live, relative to the workspace.
82
+ RESEARCH_DIR = "otto_research"
83
+ #: An outline longer than this is a book, and a run that long is not what a
84
+ #: single turn's budget can pay for.
85
+ MAX_SECTIONS = 40
86
+ #: The floor and ceiling on a section's word minimum. The ceiling is the
87
+ #: writer's reply: the reason seat allows 8192 tokens, reasoning included,
88
+ #: and a section that needs more than 2000 words of prose plus a ledger is
89
+ #: one the outline must split.
90
+ MIN_SECTION_WORDS = 150
91
+ MAX_SECTION_WORDS = 2000
92
+ #: How much of the previous section the writer sees, so the join reads as
93
+ #: one document even when the ledger is thin.
94
+ PREVIOUS_TAIL_CHARS = 1500
95
+ #: How much gathered material the writer sees for one section.
96
+ NOTES_CHARS = 6000
97
+ #: How much request text the writer sees. The outline saw all of it.
98
+ REQUEST_CHARS = 4000
99
+ #: The ledger's size cap. Past it, facts are dropped from the MIDDLE: the
100
+ #: founding facts and the latest ones are the two ends a late section most
101
+ #: needs, and the first live run lost the Merchant Charter of Generation 1
102
+ #: by Generation 10 under an oldest-first rule at 3000.
103
+ LEDGER_MAX_CHARS = 6000
104
+ #: A gathering worker's iterations -- fewer than a delegate's, because it has
105
+ #: one question list and one file to write.
106
+ GATHER_ITERATIONS = 6
107
+ #: How many times the outline call may be asked for one JSON object.
108
+ MAX_OUTLINE_ATTEMPTS = 2
109
+ #: How many sections an evaluator rejection rewrites, weakest first.
110
+ MAX_REJECTION_REWRITES = 3
111
+ #: What a section costs at the full tier: gather, write, check, revise. The
112
+ #: tier drops when the budget cannot afford that for every section left.
113
+ CALLS_PER_SECTION_FULL = 4
114
+ #: Calls held back for the judgment at the end.
115
+ JUDGMENT_RESERVE = 2
116
+ #: Reply room for the outline and the writers, in tokens. The seats' own
117
+ #: settings are sized for the loop's replies -- a tool call, an answer. A
118
+ #: ten-section outline is 17,000 characters of JSON and stopped at the plan
119
+ #: seat's 4096 on the first live run (parse failed twice, the run fell back
120
+ #: to the loop); a 2000-word section plus a ledger on a reasoning model, whose
121
+ #: reasoning counts against the same cap, needs more than 8192.
122
+ OUTLINE_MAX_TOKENS = 16384
123
+ SECTION_MAX_TOKENS = 16384
124
+
125
+ FORMATS = ("md", "docx", "pdf", "xlsx")
126
+
127
+
128
+ @dataclass(frozen=True, slots=True)
129
+ class Section:
130
+ index: int
131
+ slug: str
132
+ title: str
133
+ brief: str
134
+ min_words: int
135
+ required_parts: tuple[str, ...]
136
+ checks: tuple[str, ...]
137
+ research_questions: tuple[str, ...]
138
+
139
+ @property
140
+ def path(self) -> str:
141
+ return f"sections/{self.index:02d}-{self.slug}.md"
142
+
143
+ @property
144
+ def notes_path(self) -> str:
145
+ return f"notes/{self.index:02d}-{self.slug}.md"
146
+
147
+
148
+ @dataclass(frozen=True, slots=True)
149
+ class Outline:
150
+ title: str
151
+ abstract: str
152
+ format: str
153
+ ledger: dict
154
+ sections: tuple[Section, ...]
155
+
156
+
157
+ @dataclass(slots=True)
158
+ class SectionCheck:
159
+ """What the code can say about one section without a model."""
160
+
161
+ words: int
162
+ min_words: int
163
+ missing_parts: list[str] = field(default_factory=list)
164
+ failed_checks: list[str] = field(default_factory=list)
165
+ leaked_protocol: bool = False
166
+ truncated: bool = False
167
+
168
+ @property
169
+ def ok(self) -> bool:
170
+ return not self.failures()
171
+
172
+ def failures(self) -> list[str]:
173
+ out = []
174
+ if self.leaked_protocol:
175
+ out.append("the reply was a tool call, not the section -- write the prose itself")
176
+ if self.words < self.min_words:
177
+ out.append(f"{self.words} words of prose; the minimum is {self.min_words}")
178
+ for part in self.missing_parts:
179
+ out.append(f"no `### {part}` heading -- that part is required, named exactly")
180
+ if self.truncated:
181
+ out.append("the section appears cut off: it does not end in a sentence "
182
+ "and has no ledger block")
183
+ out.extend(f"failed: {check}" for check in self.failed_checks)
184
+ return out
185
+
186
+
187
+ # --------------------------------------------------------------------------
188
+ # prompts
189
+ # --------------------------------------------------------------------------
190
+
191
+ OUTLINE_PROMPT = (
192
+ "Plan a long document that satisfies the request below. Reply with ONE "
193
+ "JSON object and nothing else -- no fence, no prose -- shaped like:\n"
194
+ '{"title": str, "abstract": str, "format": "md" | "docx" | "pdf" | "xlsx", '
195
+ '"ledger": {"entities": {"<name>": "<one-line current state>"}, '
196
+ '"facts": [str], "open_threads": [str]}, '
197
+ '"sections": [{"title": str, "brief": str, "min_words": int, '
198
+ '"required_parts": [str], "checks": [str], "research_questions": [str]}]}\n\n'
199
+ "The CRITERIA are binding: every count, length and required part in them "
200
+ "must be reachable from this outline. If the request names N parts, there "
201
+ "are N sections (or N groups of sections) -- never fewer. A section is "
202
+ "what one writer produces in one sitting: 150 to 2000 words; split "
203
+ "anything longer into more sections. `brief` says what the section covers "
204
+ "and how it connects to what came before. `required_parts` are the "
205
+ "sub-headings it must contain, in order, named exactly; when the request "
206
+ "names parts every section must have, list them for every section. "
207
+ "`min_words` is the least the whole section may run to, and it must "
208
+ "cover every minimum the request states for its parts. `checks` are "
209
+ "one-line statements someone could verify from the section alone; leave "
210
+ "it empty when the counts say it all. `research_questions` are things to "
211
+ "look up on the web or in the workspace before writing -- empty for "
212
+ "anything written from the request itself, such as fiction or analysis "
213
+ "of material already given. `ledger` seeds the continuity state: the "
214
+ "named things and settled facts the whole document must keep straight. "
215
+ "`format` is what the request asked the file to be; md when it did not say. "
216
+ "KEEP IT COMPACT: the abstract under 80 words, each brief under 60, "
217
+ "required part names short (the heading text only, such as \"Ruler "
218
+ "Profile\", never a sentence). The plan is read by machines; the writing "
219
+ "happens later."
220
+ )
221
+
222
+ WRITER_PROMPT = (
223
+ "You are writing ONE section of a longer document that is being assembled "
224
+ "section by section. You will be shown the request, the outline, this "
225
+ "section's brief, and a CONTINUITY LEDGER: the named things and settled "
226
+ "facts from the sections already written. Everything in the ledger is "
227
+ "true in this document. Build on it, refer back to it by name where the "
228
+ "brief calls for it, and never contradict or re-introduce what earlier "
229
+ "sections established.\n\n"
230
+ "Write the section in full, in Markdown, starting with the heading you "
231
+ "are given and using `### ` headings for each required part, named "
232
+ "exactly as listed. The minimum length is counted in words of prose -- "
233
+ "write the thing itself, not a plan or a summary of what it would "
234
+ "contain. No preamble and no closing remarks about the document.\n\n"
235
+ "After the prose, add a fenced block that starts with ```ledger and "
236
+ "holds the ledger as JSON, updated with what this section established: "
237
+ "entities whose state changed, facts added, threads opened or closed. "
238
+ "Keep it under 6000 characters -- drop the least important facts rather "
239
+ "than exceed it, but never the founding ones. Nothing after the closing "
240
+ "fence."
241
+ )
242
+
243
+ SECTION_CHECK_PROMPT = (
244
+ "Below is one section of a longer document and a list of statements it "
245
+ "must satisfy. Read the section, then reply with exactly:\nFAILED:\n"
246
+ "followed by one line per statement that is NOT satisfied, quoting the "
247
+ "statement, or the single word NONE. Nothing else."
248
+ )
249
+
250
+ RESEARCH_WORKER_CONTRACT = (
251
+ "You have one bounded job for a longer document another process is "
252
+ "assembling. Find out the following and nothing beyond it:\n\n{questions}\n\n"
253
+ "Use web_search for the open web and read_file / execute_bash for this "
254
+ "workspace. Write what you found -- specifics, numbers, names, and the "
255
+ "source of each -- to the file {notes_path} with write_file. Then reply "
256
+ "with exactly\nFINAL:\n<five lines: the findings someone who cannot see "
257
+ "your working can write from>"
258
+ )
259
+
260
+ RESEARCH_JUDGE_NOTE = (
261
+ "The table above is EXACT: the word counts, the parts found and the "
262
+ "check results were computed by code from the section files, not "
263
+ "reported by the model that wrote them, so there is nothing to recount. "
264
+ "Judge whether those numbers and the weakest section (below, in full) "
265
+ "satisfy the criteria. The document is at {path} in the workspace, one "
266
+ "file per section under {sections_dir}; if one criterion truly cannot "
267
+ "be settled from what is here, read a section file with read_file -- "
268
+ "one tool call per reply, and your LAST reply must be the verdict in "
269
+ "the FINAL format, whatever you have or have not managed to check."
270
+ )
271
+
272
+
273
+ # --------------------------------------------------------------------------
274
+ # small helpers
275
+ # --------------------------------------------------------------------------
276
+
277
+ _WORD = re.compile(r"\b\w+\b")
278
+ _HEADING = re.compile(r"^\s{0,3}#{1,6}\s+(.*?)\s*#*\s*$")
279
+ _LEDGER_BLOCK = re.compile(r"```ledger[^\n]*\n(.*?)\n[ \t]*```", re.S)
280
+ _SENTENCE_END = re.compile(r"[.!?:;\"'”’)\]*`_]\s*$")
281
+ _FORMAT_WORDS = re.compile(
282
+ r"\b(docx|word doc\w*|ms word|pdf|xlsx|excel|spreadsheet)\b", re.I,
283
+ )
284
+
285
+
286
+ def _slugify(text: str, *, limit: int = 40) -> str:
287
+ slug = re.sub(r"[^a-z0-9]+", "-", text.lower()).strip("-")
288
+ return slug[:limit].strip("-") or "section"
289
+
290
+
291
+ def _words(body: str) -> int:
292
+ """Words of prose: headings and fenced blocks do not count."""
293
+ total = 0
294
+ in_fence = False
295
+ for line in body.splitlines():
296
+ if line.strip().startswith("```"):
297
+ in_fence = not in_fence
298
+ continue
299
+ if in_fence or _HEADING.match(line):
300
+ continue
301
+ total += len(_WORD.findall(line))
302
+ return total
303
+
304
+
305
+ def _headings(body: str) -> list[str]:
306
+ found = []
307
+ in_fence = False
308
+ for line in body.splitlines():
309
+ if line.strip().startswith("```"):
310
+ in_fence = not in_fence
311
+ continue
312
+ if not in_fence and (m := _HEADING.match(line)):
313
+ found.append(m.group(1))
314
+ return found
315
+
316
+
317
+ def _anchor(title: str) -> str:
318
+ return re.sub(r"[^a-z0-9 -]", "", title.lower()).strip().replace(" ", "-")
319
+
320
+
321
+ def _write(rel: str, content: str) -> bool:
322
+ result = write_file(f"{rel}\n{content}")
323
+ if result.returncode != 0:
324
+ logger.warning("research: could not write %s: %s", rel, result.stderr)
325
+ return False
326
+ return True
327
+
328
+
329
+ def _read(rel: str) -> str | None:
330
+ try:
331
+ path = resolve_in_workspace(rel)
332
+ except OutsideWorkspace:
333
+ return None
334
+ try:
335
+ return path.read_text()
336
+ except OSError:
337
+ return None
338
+
339
+
340
+ def _exists(rel: str) -> bool:
341
+ try:
342
+ return resolve_in_workspace(rel).exists()
343
+ except OutsideWorkspace:
344
+ return False
345
+
346
+
347
+ def _task_slug(task_text: str) -> str:
348
+ base = _slugify(" ".join(task_text.split()[:6]), limit=48) or "document"
349
+ slug, n = base, 2
350
+ while _exists(f"{RESEARCH_DIR}/{slug}"):
351
+ slug = f"{base}-{n}"
352
+ n += 1
353
+ return slug
354
+
355
+
356
+ def _with_room(llm, max_tokens: int):
357
+ """The seat's model with at least `max_tokens` of reply allowed.
358
+
359
+ The routed model is a pydantic object whose cap is `max_tokens` on
360
+ three vendors and `max_output_tokens` on Gemini; a model_copy with a
361
+ bigger one is the same move nodes.py's _call makes for a diffusing
362
+ retry. A model with no cap set, or one already roomier, is returned as
363
+ it is."""
364
+ for attr in ("max_tokens", "max_output_tokens"):
365
+ current = getattr(llm, attr, None)
366
+ if isinstance(current, int) and current < max_tokens:
367
+ try:
368
+ return llm.model_copy(update={attr: max_tokens})
369
+ except Exception: # not a pydantic model; use it as it is
370
+ return llm
371
+ return llm
372
+
373
+
374
+ def _emit_board(*lines: str) -> None:
375
+ pn._emit({"research": {"board": list(lines)}})
376
+
377
+
378
+ def _spent(budget: Budget | None) -> bool:
379
+ return budget is not None and budget.spent()
380
+
381
+
382
+ def _tier(budget: Budget | None, remaining_sections: int) -> str:
383
+ """How much each remaining section may cost: "full", "lean" or "bare".
384
+
385
+ Full is gather + write + check + revise. Lean drops gathering and the
386
+ model check, keeping one revision for a section the code checks fail.
387
+ Bare is the write alone. The wrap-up stretch is always bare: the budget
388
+ has said so, and a workflow that argued would be the loop it replaced.
389
+ """
390
+ if budget is None:
391
+ return "full"
392
+ if budget.phase() == "wrap_up":
393
+ return "bare"
394
+ if budget.max_model_calls is None:
395
+ return "full"
396
+ remaining = budget.max_model_calls - budget.calls - JUDGMENT_RESERVE
397
+ per_section = remaining / max(remaining_sections, 1)
398
+ if per_section >= CALLS_PER_SECTION_FULL:
399
+ return "full"
400
+ if per_section >= 2:
401
+ return "lean"
402
+ return "bare"
403
+
404
+
405
+ # --------------------------------------------------------------------------
406
+ # outline
407
+ # --------------------------------------------------------------------------
408
+
409
+ def _parse_outline(text: str) -> Outline:
410
+ """One JSON object out of the outline reply, validated and clamped.
411
+ Raises ValueError with a reason the model can act on."""
412
+ stripped = pn._strip_code_fence(text).strip()
413
+ start, end = stripped.find("{"), stripped.rfind("}")
414
+ if start < 0 or end <= start:
415
+ raise ValueError("no JSON object in the reply")
416
+ try:
417
+ data = json.loads(stripped[start:end + 1])
418
+ except json.JSONDecodeError as exc:
419
+ raise ValueError(f"invalid JSON: {exc.msg} at character {exc.pos}") from None
420
+ if not isinstance(data, dict):
421
+ raise ValueError("the JSON must be an object")
422
+ raw_sections = data.get("sections")
423
+ if not isinstance(raw_sections, list) or not raw_sections:
424
+ raise ValueError("`sections` must be a non-empty list")
425
+ title = str(data.get("title") or "").strip() or "Document"
426
+ fmt = str(data.get("format") or "md").strip().lower()
427
+ if fmt not in FORMATS:
428
+ fmt = "md"
429
+
430
+ sections: list[Section] = []
431
+ used: set[str] = set()
432
+ for i, raw in enumerate(raw_sections[:MAX_SECTIONS], start=1):
433
+ if not isinstance(raw, dict) or not str(raw.get("title") or "").strip():
434
+ raise ValueError(f"section {i} has no title")
435
+ stitle = str(raw["title"]).strip()
436
+ slug, n = _slugify(stitle), 2
437
+ while slug in used:
438
+ slug = f"{_slugify(stitle)}-{n}"
439
+ n += 1
440
+ used.add(slug)
441
+ try:
442
+ min_words = int(raw.get("min_words") or MIN_SECTION_WORDS)
443
+ except (TypeError, ValueError):
444
+ min_words = MIN_SECTION_WORDS
445
+ min_words = max(MIN_SECTION_WORDS, min(MAX_SECTION_WORDS, min_words))
446
+ sections.append(Section(
447
+ index=i, slug=slug, title=stitle,
448
+ brief=str(raw.get("brief") or "").strip(),
449
+ min_words=min_words,
450
+ required_parts=_strings(raw.get("required_parts")),
451
+ checks=_strings(raw.get("checks")),
452
+ research_questions=_strings(raw.get("research_questions")),
453
+ ))
454
+ return Outline(
455
+ title=title,
456
+ abstract=str(data.get("abstract") or "").strip(),
457
+ format=fmt,
458
+ ledger=_normalise_ledger(data.get("ledger")) or _empty_ledger(),
459
+ sections=tuple(sections),
460
+ )
461
+
462
+
463
+ def _strings(value) -> tuple[str, ...]:
464
+ if not isinstance(value, list):
465
+ return ()
466
+ return tuple(str(v).strip() for v in value if str(v).strip())
467
+
468
+
469
+ def _outline(task_text: str, criteria: list[str]) -> Outline:
470
+ """Phase one: the plan, on the plan seat. Raises ValueError when two
471
+ attempts produced no usable JSON, ProviderError when the seat is down."""
472
+ llm = _with_room(pn.ROUTER.chat_model(Task.PLAN), OUTLINE_MAX_TOKENS)
473
+ body = f"TASK:\n{task_text}"
474
+ if criteria:
475
+ body += "\n\nCRITERIA (binding):\n" + "\n".join(f"- {c}" for c in criteria)
476
+ if note := workspace_note():
477
+ body += f"\n\nWORKSPACE:\n{note}"
478
+ messages: list = [SystemMessage(OUTLINE_PROMPT), HumanMessage(body)]
479
+ error: ValueError | None = None
480
+ for _ in range(MAX_OUTLINE_ATTEMPTS):
481
+ reply = pn._call(llm, messages)
482
+ try:
483
+ return _parse_outline(reply)
484
+ except ValueError as exc:
485
+ error = exc
486
+ messages.append(AIMessage(reply or "(empty)"))
487
+ messages.append(HumanMessage(
488
+ f"That was not one usable JSON object: {exc}. Reply again with "
489
+ "only the object, shaped exactly as described."
490
+ ))
491
+ raise error or ValueError("no outline")
492
+
493
+
494
+ def _outline_to_json(outline: Outline) -> str:
495
+ return json.dumps(dataclasses.asdict(outline), indent=1)
496
+
497
+
498
+ def _outline_from_json(text: str) -> Outline:
499
+ data = json.loads(text)
500
+ return Outline(
501
+ title=data["title"], abstract=data.get("abstract", ""),
502
+ format=data.get("format", "md"),
503
+ ledger=_normalise_ledger(data.get("ledger")) or _empty_ledger(),
504
+ sections=tuple(Section(
505
+ index=s["index"], slug=s["slug"], title=s["title"], brief=s.get("brief", ""),
506
+ min_words=s.get("min_words", MIN_SECTION_WORDS),
507
+ required_parts=tuple(s.get("required_parts", ())),
508
+ checks=tuple(s.get("checks", ())),
509
+ research_questions=tuple(s.get("research_questions", ())),
510
+ ) for s in data["sections"]),
511
+ )
512
+
513
+
514
+ # --------------------------------------------------------------------------
515
+ # the ledger
516
+ # --------------------------------------------------------------------------
517
+
518
+ def _empty_ledger() -> dict:
519
+ return {"entities": {}, "facts": [], "open_threads": []}
520
+
521
+
522
+ def _normalise_ledger(value) -> dict | None:
523
+ """The three keys, each the right shape, or None for anything that is
524
+ not a ledger at all. Missing keys are filled; wrong-typed ones dropped."""
525
+ if not isinstance(value, dict):
526
+ return None
527
+ entities = value.get("entities")
528
+ facts = value.get("facts")
529
+ threads = value.get("open_threads")
530
+ return {
531
+ "entities": ({str(k): str(v) for k, v in entities.items()}
532
+ if isinstance(entities, dict) else {}),
533
+ "facts": [str(f) for f in facts] if isinstance(facts, list) else [],
534
+ "open_threads": [str(t) for t in threads] if isinstance(threads, list) else [],
535
+ }
536
+
537
+
538
+ def _split_ledger(reply: str) -> tuple[str, dict | None]:
539
+ """The section's prose and its ledger, apart. The LAST ```ledger block
540
+ is the ledger; the prose is everything else, fence-stripped if the model
541
+ wrapped the whole reply. None when there is no block or it is not JSON."""
542
+ blocks = list(_LEDGER_BLOCK.finditer(reply))
543
+ if not blocks:
544
+ return pn._strip_code_fence(reply).strip(), None
545
+ last = blocks[-1]
546
+ body = (reply[:last.start()] + reply[last.end():]).strip()
547
+ body = pn._strip_code_fence(body).strip()
548
+ try:
549
+ ledger = _normalise_ledger(json.loads(last.group(1)))
550
+ except json.JSONDecodeError:
551
+ ledger = None
552
+ return body, ledger
553
+
554
+
555
+ def _merge_ledger(old: dict, new: dict | None) -> dict:
556
+ """The next section's ledger: the old one when the writer returned none.
557
+
558
+ Otherwise a UNION, not a replacement. A writer rewriting the ledger
559
+ drops what its own section did not touch -- on the first live run the
560
+ founding charter was gone from the ledger by the tenth section, and the
561
+ tenth section did not mention it. So an entity the new ledger omits is
562
+ kept with its old state, a fact it omits is kept, and the writer's
563
+ versions win where both exist. Open threads are the writer's: closing
564
+ one is the point of the field.
565
+
566
+ Trimmed to LEDGER_MAX_CHARS from the middle of the facts, so the
567
+ founding facts and the latest survive; then open threads; entities go
568
+ last, oldest first, because a name is the cheapest thing to keep and the
569
+ most expensive to lose.
570
+ """
571
+ if new is None:
572
+ return old
573
+ entities = {**old.get("entities", {}), **new.get("entities", {})}
574
+ facts = list(old.get("facts", []))
575
+ facts += [f for f in new.get("facts", []) if f not in facts]
576
+ merged = {"entities": entities, "facts": facts,
577
+ "open_threads": list(new.get("open_threads", []))}
578
+ while len(json.dumps(merged)) > LEDGER_MAX_CHARS:
579
+ if len(merged["facts"]) > 2:
580
+ middle = len(merged["facts"]) // 2
581
+ merged["facts"] = merged["facts"][:middle] + merged["facts"][middle + 1:]
582
+ elif merged["open_threads"]:
583
+ merged["open_threads"] = merged["open_threads"][1:]
584
+ elif merged["entities"]:
585
+ first = next(iter(merged["entities"]))
586
+ merged["entities"] = {k: v for k, v in merged["entities"].items() if k != first}
587
+ else:
588
+ break
589
+ return merged
590
+
591
+
592
+ # --------------------------------------------------------------------------
593
+ # gathering
594
+ # --------------------------------------------------------------------------
595
+
596
+ def _spawn_worker(state: AgentState, instruction: str, *, mode: str = "find",
597
+ max_iterations: int = GATHER_ITERATIONS,
598
+ actions: list[str]) -> str:
599
+ """One bounded agent loop with a contract, no parent conversation, and
600
+ a report back -- the delegate shape (nodes.py's _delegate), with the
601
+ caller rather than a model choosing the mode. The child's actions join
602
+ the record; its conversation is discarded."""
603
+ child: list = [SystemMessage(pn.compose_agent_prompt(reachable_tools(),
604
+ may_delegate=False))]
605
+ for extra in (workspace_note(), render_note()):
606
+ if extra:
607
+ child.append(SystemMessage(extra))
608
+ child.append(HumanMessage(instruction))
609
+ child.append(pn._mode_message(mode))
610
+ taken: list[str] = []
611
+ try:
612
+ output, why, _ = pn._agent_loop(
613
+ state, child, mode=mode, actions=taken, mode_log=[],
614
+ max_iterations=max_iterations, may_delegate=False,
615
+ )
616
+ except pn.NeedsUserInput as exc:
617
+ # A worker cannot pause the run; the workflow has no way to relay a
618
+ # question and the writer can work without the answer.
619
+ actions.append(f"research(worker): asked a question and was refused: "
620
+ f"{exc.question[:80]}")
621
+ return ""
622
+ except ProviderError as exc:
623
+ actions.append(f"research(worker): failed: {exc}")
624
+ return ""
625
+ actions.extend(f"{mode}(worker): {line}" for line in taken)
626
+ return output.strip()
627
+
628
+
629
+ def _gather(state: AgentState, section: Section, dir_rel: str,
630
+ actions: list[str]) -> str:
631
+ notes_path = f"{dir_rel}/{section.notes_path}"
632
+ instruction = RESEARCH_WORKER_CONTRACT.format(
633
+ questions="\n".join(f"- {q}" for q in section.research_questions),
634
+ notes_path=notes_path,
635
+ )
636
+ _emit_board(f"research: gathering for section {section.index}: "
637
+ f"{section.research_questions[0][:60]}")
638
+ report = _spawn_worker(state, instruction, actions=actions)
639
+ notes = _read(notes_path) or ""
640
+ combined = "\n\n".join(part for part in (notes.strip(), report) if part)
641
+ return combined[:NOTES_CHARS]
642
+
643
+
644
+ # --------------------------------------------------------------------------
645
+ # writing and checking
646
+ # --------------------------------------------------------------------------
647
+
648
+ def _writer_body(section: Section, *, task_text: str, outline: Outline, ledger: dict,
649
+ previous_tail: str, notes: str,
650
+ revise: tuple[str, list[str]] | None) -> str:
651
+ listing = "\n".join(
652
+ f"{s.index}. {s.title}" + (" <- this one" if s.index == section.index else "")
653
+ for s in outline.sections
654
+ )
655
+ parts = [
656
+ f"THE REQUEST:\n{task_text[:REQUEST_CHARS]}",
657
+ f"THE DOCUMENT: {outline.title}\n{listing}",
658
+ f"THIS SECTION -- start with this heading:\n## {section.title}\n\n{section.brief}",
659
+ ]
660
+ if section.required_parts:
661
+ parts.append("REQUIRED PARTS (as ### headings, in this order, named exactly):\n"
662
+ + "\n".join(f"- {p}" for p in section.required_parts))
663
+ parts.append(f"MINIMUM: {section.min_words} words of prose, not counting headings.")
664
+ if section.checks:
665
+ parts.append("CHECKS THIS SECTION MUST PASS:\n"
666
+ + "\n".join(f"- {c}" for c in section.checks))
667
+ parts.append(f"CONTINUITY LEDGER:\n{json.dumps(ledger, indent=1)}")
668
+ if previous_tail:
669
+ parts.append(f"THE PREVIOUS SECTION ENDS:\n...{previous_tail}")
670
+ if notes:
671
+ parts.append(f"NOTES GATHERED FOR THIS SECTION:\n{notes}")
672
+ if revise:
673
+ draft, failures = revise
674
+ parts.append(
675
+ f"YOUR DRAFT:\n{draft}\n\nIT FAILS:\n"
676
+ + "\n".join(f"- {f}" for f in failures)
677
+ + "\n\nReturn the whole section again, complete, fixed, with the ledger block."
678
+ )
679
+ return "\n\n".join(parts)
680
+
681
+
682
+ def _write_section(section: Section, *, task_text: str, outline: Outline, ledger: dict,
683
+ previous_tail: str, notes: str,
684
+ revise: tuple[str, list[str]] | None = None,
685
+ ) -> tuple[str | None, dict | None]:
686
+ """One call on the reason seat; the section's prose and its ledger.
687
+ (None, None) when the seat failed twice."""
688
+ llm = _with_room(pn.ROUTER.chat_model(Task.REASON), SECTION_MAX_TOKENS)
689
+ messages = [
690
+ SystemMessage(WRITER_PROMPT),
691
+ HumanMessage(_writer_body(
692
+ section, task_text=task_text, outline=outline, ledger=ledger,
693
+ previous_tail=previous_tail, notes=notes, revise=revise,
694
+ )),
695
+ ]
696
+ reply = ""
697
+ for attempt in range(2):
698
+ try:
699
+ reply = pn._call(llm, messages)
700
+ break
701
+ except ProviderError as exc:
702
+ logger.warning("research: writing section %d failed (%d/2): %s",
703
+ section.index, attempt + 1, exc)
704
+ if attempt == 1:
705
+ return None, None
706
+ body, new_ledger = _split_ledger(reply)
707
+ if body and not body.lstrip().startswith("## "):
708
+ body = f"## {section.title}\n\n{body}"
709
+ return body, new_ledger
710
+
711
+
712
+ def _check_section(section: Section, body: str, *, had_ledger: bool) -> SectionCheck:
713
+ check = SectionCheck(words=_words(body), min_words=section.min_words)
714
+ first = next((line for line in body.splitlines() if line.strip()), "")
715
+ check.leaked_protocol = bool(pn._NEXT_DIRECTIVE.match(first))
716
+ headings = [h.lower() for h in _headings(body)]
717
+ check.missing_parts = [
718
+ part for part in section.required_parts
719
+ if not any(part.lower() in h for h in headings)
720
+ ]
721
+ check.truncated = not had_ledger and not _SENTENCE_END.search(body.rstrip())
722
+ return check
723
+
724
+
725
+ def _judge_section(section: Section, body: str) -> list[str]:
726
+ """One call on the evaluate seat against the outline's own checks. The
727
+ statements it names as failed, or [] -- including on a provider error,
728
+ which is not the section's fault."""
729
+ llm = pn.ROUTER.chat_model(Task.EVALUATE)
730
+ try:
731
+ reply = pn._call(llm, [
732
+ SystemMessage(SECTION_CHECK_PROMPT),
733
+ HumanMessage("STATEMENTS:\n" + "\n".join(f"- {c}" for c in section.checks)
734
+ + f"\n\nSECTION:\n{body}"),
735
+ ])
736
+ except ProviderError as exc:
737
+ logger.warning("research: checking section %d failed: %s", section.index, exc)
738
+ return []
739
+ failed = []
740
+ after = reply.split("FAILED:", 1)[1] if "FAILED:" in reply else reply
741
+ for line in after.splitlines():
742
+ line = line.strip().lstrip("-*• ").strip()
743
+ if line and line.upper() != "NONE":
744
+ failed.append(line)
745
+ return failed
746
+
747
+
748
+ # --------------------------------------------------------------------------
749
+ # assembly, conversion, report
750
+ # --------------------------------------------------------------------------
751
+
752
+ def _assemble(outline: Outline, written: list[tuple[Section, str]],
753
+ incomplete: list[Section]) -> str:
754
+ total = sum(_words(body) for _, body in written)
755
+ lines = [f"# {outline.title}", ""]
756
+ if outline.abstract:
757
+ lines += [f"_{outline.abstract}_", ""]
758
+ status = f"{len(written)} sections, {total:,} words."
759
+ if incomplete:
760
+ status += (" INCOMPLETE -- not written: "
761
+ + ", ".join(f"{s.index} ({s.title})" for s in incomplete)
762
+ + " (budget ran out).")
763
+ lines += [status, "", "## Contents", ""]
764
+ lines += [f"{s.index}. [{s.title}](#{_anchor(s.title)})" for s, _ in written]
765
+ lines.append("")
766
+ for _, body in written:
767
+ lines += [body.strip(), ""]
768
+ return "\n".join(lines).rstrip() + "\n"
769
+
770
+
771
+ def _wanted_format(task_text: str, outline: Outline) -> str | None:
772
+ if outline.format != "md":
773
+ return outline.format
774
+ found = _FORMAT_WORDS.search(task_text)
775
+ if not found:
776
+ return None
777
+ word = found.group(1).lower()
778
+ if word.startswith("word") or word == "ms word" or word == "docx":
779
+ return "docx"
780
+ if word in ("xlsx", "excel", "spreadsheet"):
781
+ return "xlsx"
782
+ return "pdf"
783
+
784
+
785
+ def _convert(dir_rel: str, fmt: str) -> tuple[bool, str]:
786
+ """document.md -> document.<fmt> in the same directory, by a script on
787
+ the workspace's own interpreter. (ok, detail)."""
788
+ try:
789
+ source = resolve_in_workspace(f"{dir_rel}/document.md")
790
+ target = resolve_in_workspace(f"{dir_rel}/document.{fmt}")
791
+ except OutsideWorkspace as exc:
792
+ return False, str(exc)
793
+ script = (CONVERT_SCRIPTS[fmt]
794
+ .replace("__SOURCE__", str(source))
795
+ .replace("__TARGET__", str(target)))
796
+ result = execute_python(script)
797
+ if result.returncode == 0 and target.exists():
798
+ return True, f"document.{fmt}"
799
+ detail = (result.stderr or result.stdout or "").strip().splitlines()
800
+ return False, (detail[-1] if detail else f"exit {result.returncode}")
801
+
802
+
803
+ def _report(outline: Outline, document_path: str, rows: list[dict],
804
+ incomplete: list[Section], converted: tuple[str, bool, str] | None) -> str:
805
+ total = sum(r["words"] for r in rows)
806
+ head = (f"Document written to `{document_path}` -- {len(rows)} sections, "
807
+ f"{total:,} words.")
808
+ if converted:
809
+ fmt, ok, detail = converted
810
+ head += (f" Also `{detail}`." if ok
811
+ else f" Conversion to {fmt} failed ({detail}); the Markdown stands.")
812
+ lines = [head]
813
+ if incomplete:
814
+ lines.append("INCOMPLETE -- the budget ran out before: "
815
+ + ", ".join(f"{s.index}. {s.title}" for s in incomplete))
816
+ if outline.abstract:
817
+ lines += ["", outline.abstract]
818
+ lines += ["", "| # | section | words | min | parts | checks |",
819
+ "|---|---|---|---|---|---|"]
820
+ for r in rows:
821
+ lines.append(f"| {r['index']} | {r['title']} | {r['words']:,} | {r['min']:,} | "
822
+ f"{r['parts']} | {r['checks']} |")
823
+ return "\n".join(lines)
824
+
825
+
826
+ def _row(section: Section, check: SectionCheck, revised: bool) -> dict:
827
+ parts = (f"{len(section.required_parts) - len(check.missing_parts)}/"
828
+ f"{len(section.required_parts)}" if section.required_parts else "-")
829
+ if check.ok:
830
+ status = "ok"
831
+ else:
832
+ status = "; ".join(check.failures())[:120]
833
+ if revised:
834
+ status = f"revised once -- {status}"
835
+ return {"index": section.index, "title": section.title, "words": check.words,
836
+ "min": section.min_words, "parts": parts, "checks": status,
837
+ "ratio": check.words / max(section.min_words, 1), "ok": check.ok}
838
+
839
+
840
+ # --------------------------------------------------------------------------
841
+ # the node
842
+ # --------------------------------------------------------------------------
843
+
844
+ def run_research(state: AgentState) -> Command:
845
+ task_text = pn._requested(state)
846
+ criteria = [c["text"] for c in (state.get("checklist") or [])]
847
+ budget = current_budget()
848
+ if budget is not None:
849
+ budget.skip_recon()
850
+ actions: list[str] = []
851
+
852
+ document_path = state.get("document_path")
853
+ if state.get("feedback") and document_path and _exists(
854
+ f"{document_path.rsplit('/', 1)[0]}/outline.json"):
855
+ return _revise(state, task_text, document_path, budget, actions)
856
+
857
+ report_progress("phase", "outlining the document")
858
+ try:
859
+ outline = _outline(task_text, criteria)
860
+ except (ValueError, ProviderError) as exc:
861
+ _emit_board(f"research: could not outline the document ({exc}) -- "
862
+ "running it as an agent task")
863
+ return Command(
864
+ update={
865
+ "node": "research", "route": "agent",
866
+ "board": [f"research: could not outline the document ({exc}) -- "
867
+ "running it as an agent task"],
868
+ **_calls(budget),
869
+ },
870
+ goto="agent",
871
+ )
872
+ dir_rel = f"{RESEARCH_DIR}/{_task_slug(task_text)}"
873
+ _write(f"{dir_rel}/outline.json", _outline_to_json(outline))
874
+ _emit_board(f"research: outlined {len(outline.sections)} sections -- {outline.title}")
875
+ actions.append(f"research: outlined {len(outline.sections)} sections into {dir_rel}/")
876
+
877
+ ledger = outline.ledger
878
+ _write(f"{dir_rel}/ledger.json", json.dumps(ledger, indent=1))
879
+ written: list[tuple[Section, str]] = []
880
+ rows: list[dict] = []
881
+ incomplete: list[Section] = []
882
+ previous_tail = ""
883
+ n = len(outline.sections)
884
+ for section in outline.sections:
885
+ if _spent(budget):
886
+ incomplete.append(section)
887
+ continue
888
+ tier = _tier(budget, n - section.index + 1)
889
+ report_progress("phase", f"writing section {section.index} of {n}: {section.title}")
890
+
891
+ notes = ""
892
+ if tier == "full" and section.research_questions:
893
+ notes = _gather(state, section, dir_rel, actions)
894
+ if _spent(budget):
895
+ incomplete.append(section)
896
+ continue
897
+
898
+ body, new_ledger = _write_section(
899
+ section, task_text=task_text, outline=outline, ledger=ledger,
900
+ previous_tail=previous_tail, notes=notes,
901
+ )
902
+ if body is None:
903
+ incomplete.append(section)
904
+ actions.append(f"research: section {section.index} could not be written "
905
+ "(provider failure)")
906
+ continue
907
+ check = _check_section(section, body, had_ledger=new_ledger is not None)
908
+ if tier == "full" and section.checks and not _spent(budget):
909
+ check.failed_checks = _judge_section(section, body)
910
+
911
+ revised = False
912
+ if not check.ok and tier != "bare" and not _spent(budget):
913
+ _emit_board(f"research: section {section.index} needs a revision -- "
914
+ + "; ".join(check.failures())[:100])
915
+ body2, ledger2 = _write_section(
916
+ section, task_text=task_text, outline=outline, ledger=ledger,
917
+ previous_tail=previous_tail, notes=notes,
918
+ revise=(body, check.failures()),
919
+ )
920
+ if body2 is not None:
921
+ check2 = _check_section(section, body2, had_ledger=ledger2 is not None)
922
+ if check.failed_checks and tier == "full" and not _spent(budget):
923
+ check2.failed_checks = _judge_section(section, body2)
924
+ if len(check2.failures()) <= len(check.failures()):
925
+ body, new_ledger, check = body2, ledger2, check2
926
+ revised = True
927
+
928
+ _write(f"{dir_rel}/{section.path}", body)
929
+ ledger = _merge_ledger(ledger, new_ledger)
930
+ if new_ledger is None:
931
+ _emit_board(f"research: section {section.index} returned no ledger -- "
932
+ "carrying the previous one")
933
+ _write(f"{dir_rel}/ledger.json", json.dumps(ledger, indent=1))
934
+ previous_tail = body[-PREVIOUS_TAIL_CHARS:]
935
+ written.append((section, body))
936
+ rows.append(_row(section, check, revised))
937
+ actions.append(f"research: wrote {section.path} ({check.words} words"
938
+ + (", revised once" if revised else "") + ")")
939
+ _emit_board(f"research: section {section.index}/{n} -- {check.words} words"
940
+ + ("" if check.ok else " (" + "; ".join(check.failures())[:80] + ")"))
941
+
942
+ return _finish(state, task_text, outline, dir_rel, written, rows, incomplete,
943
+ budget, actions)
944
+
945
+
946
+ def _revise(state: AgentState, task_text: str, document_path: str,
947
+ budget: Budget | None, actions: list[str]) -> Command:
948
+ """Re-entry after an evaluator rejection: rewrite the weakest sections
949
+ with the judge's feedback, reassemble, and go back for judgment."""
950
+ dir_rel = document_path.rsplit("/", 1)[0]
951
+ feedback = state.get("feedback") or ""
952
+ try:
953
+ outline = _outline_from_json(_read(f"{dir_rel}/outline.json") or "")
954
+ except (ValueError, KeyError, TypeError) as exc:
955
+ return Command(
956
+ update={"node": "research", "route": "agent",
957
+ "board": [f"research: could not reload the outline ({exc}) -- "
958
+ "handing the rejection to the agent"],
959
+ **_calls(budget)},
960
+ goto="agent",
961
+ )
962
+ # The ledger as it stands at the END -- the one before each rewritten
963
+ # section is not kept. A rewrite of section 3 may therefore see facts
964
+ # from 4 onward; one round of that is a smaller wrong than reconstructing
965
+ # ten ledgers would cost.
966
+ ledger = _normalise_ledger(json.loads(_read(f"{dir_rel}/ledger.json") or "{}")) \
967
+ or outline.ledger
968
+ report_progress("phase", "revising the document")
969
+ resubmit = _no_verdict(feedback)
970
+ _emit_board("research: the judge reached no verdict -- resubmitting the document unchanged"
971
+ if resubmit else f"research: revising after rejection -- {feedback[:80]}")
972
+
973
+ bodies: dict[int, str] = {}
974
+ checks: dict[int, SectionCheck] = {}
975
+ incomplete: list[Section] = []
976
+ for section in outline.sections:
977
+ body = _read(f"{dir_rel}/{section.path}")
978
+ if body is None:
979
+ incomplete.append(section)
980
+ continue
981
+ bodies[section.index] = body
982
+ checks[section.index] = _check_section(section, body, had_ledger=True)
983
+
984
+ candidates = sorted(
985
+ (s for s in outline.sections if s.index in bodies),
986
+ key=lambda s: (checks[s.index].ok, checks[s.index].words / max(s.min_words, 1)),
987
+ )
988
+ rewritten: set[int] = set()
989
+ for section in ([] if resubmit else candidates[:MAX_REJECTION_REWRITES]):
990
+ if _spent(budget):
991
+ break
992
+ previous = bodies.get(section.index - 1, "")[-PREVIOUS_TAIL_CHARS:]
993
+ old = bodies[section.index]
994
+ body, new_ledger = _write_section(
995
+ section, task_text=task_text, outline=outline, ledger=ledger,
996
+ previous_tail=previous, notes="",
997
+ revise=(old, [f"the evaluator rejected the document: {feedback}",
998
+ *checks[section.index].failures()]),
999
+ )
1000
+ if body is None:
1001
+ continue
1002
+ check = _check_section(section, body, had_ledger=new_ledger is not None)
1003
+ if len(check.failures()) <= len(checks[section.index].failures()):
1004
+ bodies[section.index] = body
1005
+ checks[section.index] = check
1006
+ ledger = _merge_ledger(ledger, new_ledger)
1007
+ _write(f"{dir_rel}/{section.path}", body)
1008
+ rewritten.add(section.index)
1009
+ actions.append(f"research: rewrote {section.path} after rejection "
1010
+ f"({check.words} words)")
1011
+ _write(f"{dir_rel}/ledger.json", json.dumps(ledger, indent=1))
1012
+
1013
+ written = [(s, bodies[s.index]) for s in outline.sections if s.index in bodies]
1014
+ rows = [_row(s, checks[s.index], s.index in rewritten) for s, _ in written]
1015
+ return _finish(state, task_text, outline, dir_rel, written, rows, incomplete,
1016
+ budget, actions,
1017
+ board=["research: the judge reached no verdict -- resubmitted unchanged"
1018
+ if resubmit else
1019
+ f"research: revising after rejection -- {feedback[:80]}"])
1020
+
1021
+
1022
+ def _finish(state: AgentState, task_text: str, outline: Outline, dir_rel: str,
1023
+ written: list[tuple[Section, str]], rows: list[dict],
1024
+ incomplete: list[Section], budget: Budget | None,
1025
+ actions: list[str], board: list[str] | None = None) -> Command:
1026
+ board = list(board or [])
1027
+ report_progress("phase", "assembling the document")
1028
+ document_path = f"{dir_rel}/document.md"
1029
+ _write(document_path, _assemble(outline, written, incomplete))
1030
+ actions.append(f"research: assembled {document_path} ({len(written)} sections)")
1031
+
1032
+ converted = None
1033
+ if (fmt := _wanted_format(task_text, outline)) and written:
1034
+ report_progress("phase", f"converting to {fmt}")
1035
+ ok, detail = _convert(dir_rel, fmt)
1036
+ converted = (fmt, ok, detail)
1037
+ actions.append(f"research: converted to {fmt}" if ok
1038
+ else f"research: conversion to {fmt} failed: {detail}")
1039
+
1040
+ report = _report(outline, document_path, rows, incomplete, converted)
1041
+ context = _judge_context(document_path, dir_rel, written, rows)
1042
+
1043
+ update = {
1044
+ "node": "research", "route": "research",
1045
+ "output": report, "context": context,
1046
+ "document_path": document_path,
1047
+ "actions": actions, "feedback": "",
1048
+ **_calls(budget),
1049
+ }
1050
+ if _spent(budget) or not written:
1051
+ why = ("ran out of budget" if _spent(budget)
1052
+ else "wrote no sections")
1053
+ _emit_board(f"research: {why} -- answering with what exists, unverified")
1054
+ return Command(
1055
+ update={**update, "final_output": report,
1056
+ "board": board + [f"research: {why} after {len(written)} of "
1057
+ f"{len(outline.sections)} sections -- answering "
1058
+ "with what exists, unverified"]},
1059
+ goto=END,
1060
+ )
1061
+ return Command(
1062
+ update={**update, "board": board + [f"research: document assembled -- "
1063
+ f"{len(written)} sections, ready for judgment"]},
1064
+ goto="evaluator",
1065
+ )
1066
+
1067
+
1068
+ def _judge_context(document_path: str, dir_rel: str,
1069
+ written: list[tuple[Section, str]], rows: list[dict]) -> str:
1070
+ """What the evaluator sees under CONTEXT GATHERED: that the table is
1071
+ exact, where the files are, and the weakest section in full (the
1072
+ evaluator clips it to its own limit)."""
1073
+ context = RESEARCH_JUDGE_NOTE.format(path=document_path,
1074
+ sections_dir=f"{dir_rel}/sections/")
1075
+ weakest = min(rows, key=lambda r: r["ratio"], default=None)
1076
+ if weakest is not None:
1077
+ body = next(b for s, b in written if s.index == weakest["index"])
1078
+ context += (f"\n\nWEAKEST SECTION IN FULL ({weakest['title']}, "
1079
+ f"{weakest['words']} words, minimum {weakest['min']}):\n{body}")
1080
+ return context
1081
+
1082
+
1083
+ def _no_verdict(feedback: str) -> bool:
1084
+ """Whether a rejection is the judge failing to judge rather than a
1085
+ finding: it ran out of its own replies, or answered in a tool-call
1086
+ format the loop does not read. Seen live on the first document run --
1087
+ three rejections, none about the document, each one paying for a
1088
+ revision pass. Not something to rewrite sections over."""
1089
+ head = pn.NO_VERDICT_NOTE.split(",", 1)[0]
1090
+ return feedback.startswith(head) or "<function_calls>" in feedback or "<invoke " in feedback
1091
+
1092
+
1093
+ def _calls(budget: Budget | None) -> dict:
1094
+ return {"model_calls": budget.calls} if budget is not None else {}
1095
+
1096
+
1097
+ # --------------------------------------------------------------------------
1098
+ # conversion scripts. Run by execute_python on the workspace's interpreter;
1099
+ # __SOURCE__ / __TARGET__ are replaced with absolute paths (str.replace, not
1100
+ # .format, so the braces below are safe). Each prints "wrote <target>".
1101
+ # --------------------------------------------------------------------------
1102
+
1103
+ _DOCX_SCRIPT = r'''
1104
+ import re
1105
+ from docx import Document
1106
+
1107
+ SOURCE = r"""__SOURCE__"""
1108
+ TARGET = r"""__TARGET__"""
1109
+ text = open(SOURCE, encoding="utf-8").read()
1110
+ doc = Document()
1111
+
1112
+
1113
+ def clean(s):
1114
+ s = re.sub(r"\*\*(.+?)\*\*", r"\1", s)
1115
+ s = re.sub(r"(?<!\w)_(.+?)_(?!\w)", r"\1", s)
1116
+ s = re.sub(r"\[([^\]]+)\]\([^)]*\)", r"\1", s)
1117
+ s = re.sub(r"`([^`]*)`", r"\1", s)
1118
+ return s
1119
+
1120
+
1121
+ para = []
1122
+
1123
+
1124
+ def flush():
1125
+ global para
1126
+ if para:
1127
+ doc.add_paragraph(clean(" ".join(para)))
1128
+ para = []
1129
+
1130
+
1131
+ in_fence = False
1132
+ for line in text.splitlines():
1133
+ if line.strip().startswith("```"):
1134
+ flush()
1135
+ in_fence = not in_fence
1136
+ continue
1137
+ if in_fence:
1138
+ doc.add_paragraph(line, style="No Spacing")
1139
+ continue
1140
+ s = line.rstrip()
1141
+ if not s.strip():
1142
+ flush()
1143
+ continue
1144
+ m = re.match(r"^(#{1,6})\s+(.*)$", s)
1145
+ if m:
1146
+ flush()
1147
+ doc.add_heading(clean(m.group(2)), min(len(m.group(1)) - 1, 4))
1148
+ continue
1149
+ m = re.match(r"^\s*(?:[-*]|\d+\.)\s+(.*)$", s)
1150
+ if m:
1151
+ flush()
1152
+ style = "List Number" if s.lstrip()[0].isdigit() else "List Bullet"
1153
+ doc.add_paragraph(clean(m.group(1)), style=style)
1154
+ continue
1155
+ if s.lstrip().startswith("|"):
1156
+ flush()
1157
+ continue
1158
+ para.append(s.strip())
1159
+ flush()
1160
+ doc.save(TARGET)
1161
+ print("wrote", TARGET)
1162
+ '''
1163
+
1164
+ _PDF_SCRIPT = r'''
1165
+ import re
1166
+ from xml.sax.saxutils import escape
1167
+ from reportlab.lib.pagesizes import A4
1168
+ from reportlab.lib.styles import getSampleStyleSheet
1169
+ from reportlab.lib.units import cm
1170
+ from reportlab.platypus import Paragraph, Preformatted, SimpleDocTemplate, Spacer
1171
+
1172
+ SOURCE = r"""__SOURCE__"""
1173
+ TARGET = r"""__TARGET__"""
1174
+ text = open(SOURCE, encoding="utf-8").read()
1175
+ styles = getSampleStyleSheet()
1176
+ heading = {1: styles["Title"], 2: styles["Heading1"], 3: styles["Heading2"]}
1177
+
1178
+
1179
+ def clean(s):
1180
+ s = escape(s)
1181
+ s = re.sub(r"\*\*(.+?)\*\*", r"<b>\1</b>", s)
1182
+ s = re.sub(r"(?<!\w)_(.+?)_(?!\w)", r"<i>\1</i>", s)
1183
+ s = re.sub(r"\[([^\]]+)\]\([^)]*\)", r"\1", s)
1184
+ s = re.sub(r"`([^`]*)`", r"\1", s)
1185
+ return s
1186
+
1187
+
1188
+ story = []
1189
+ para = []
1190
+ fence = []
1191
+ in_fence = False
1192
+
1193
+
1194
+ def flush():
1195
+ global para
1196
+ if para:
1197
+ story.append(Paragraph(clean(" ".join(para)), styles["BodyText"]))
1198
+ story.append(Spacer(1, 0.2 * cm))
1199
+ para = []
1200
+
1201
+
1202
+ for line in text.splitlines():
1203
+ if line.strip().startswith("```"):
1204
+ if in_fence:
1205
+ story.append(Preformatted("\n".join(fence), styles["Code"]))
1206
+ fence = []
1207
+ else:
1208
+ flush()
1209
+ in_fence = not in_fence
1210
+ continue
1211
+ if in_fence:
1212
+ fence.append(line)
1213
+ continue
1214
+ s = line.rstrip()
1215
+ if not s.strip():
1216
+ flush()
1217
+ continue
1218
+ m = re.match(r"^(#{1,6})\s+(.*)$", s)
1219
+ if m:
1220
+ flush()
1221
+ story.append(Paragraph(clean(m.group(2)),
1222
+ heading.get(len(m.group(1)), styles["Heading3"])))
1223
+ continue
1224
+ m = re.match(r"^\s*(?:[-*]|\d+\.)\s+(.*)$", s)
1225
+ if m:
1226
+ flush()
1227
+ story.append(Paragraph("• " + clean(m.group(1)), styles["BodyText"]))
1228
+ continue
1229
+ if s.lstrip().startswith("|"):
1230
+ flush()
1231
+ continue
1232
+ para.append(s.strip())
1233
+ flush()
1234
+ SimpleDocTemplate(TARGET, pagesize=A4, leftMargin=2 * cm, rightMargin=2 * cm,
1235
+ topMargin=2 * cm, bottomMargin=2 * cm).build(story)
1236
+ print("wrote", TARGET)
1237
+ '''
1238
+
1239
+ _XLSX_SCRIPT = r'''
1240
+ import re
1241
+ from openpyxl import Workbook
1242
+ from openpyxl.styles import Font
1243
+
1244
+ SOURCE = r"""__SOURCE__"""
1245
+ TARGET = r"""__TARGET__"""
1246
+ text = open(SOURCE, encoding="utf-8").read()
1247
+ wb = Workbook()
1248
+ contents = wb.active
1249
+ contents.title = "Contents"
1250
+ contents.append(["#", "Section", "Words"])
1251
+ for cell in ("A1", "B1", "C1"):
1252
+ contents[cell].font = Font(bold=True)
1253
+ contents.column_dimensions["B"].width = 60
1254
+
1255
+
1256
+ def sheet_name(title, used):
1257
+ base = re.sub(r"[\[\]:*?/\\]", " ", title).strip()[:28] or "Section"
1258
+ name, n = base, 2
1259
+ while name in used or name == "Contents":
1260
+ name = f"{base[:25]}-{n}"
1261
+ n += 1
1262
+ used.add(name)
1263
+ return name
1264
+
1265
+
1266
+ used = set()
1267
+ sheet = None
1268
+ para = []
1269
+ index = 0
1270
+ words = 0
1271
+
1272
+
1273
+ def flush():
1274
+ global para
1275
+ if para and sheet is not None:
1276
+ sheet.append([" ".join(para)])
1277
+ para = []
1278
+
1279
+
1280
+ in_fence = False
1281
+ for line in text.splitlines():
1282
+ if line.strip().startswith("```"):
1283
+ in_fence = not in_fence
1284
+ continue
1285
+ s = line.rstrip()
1286
+ m = re.match(r"^## (.*)$", s)
1287
+ if m and not in_fence:
1288
+ flush()
1289
+ if sheet is not None:
1290
+ contents.append([index, sheet.title, words])
1291
+ if m.group(1).strip() == "Contents":
1292
+ sheet = None
1293
+ continue
1294
+ index += 1
1295
+ words = 0
1296
+ sheet = wb.create_sheet(sheet_name(m.group(1), used))
1297
+ sheet.append([m.group(1)])
1298
+ sheet["A1"].font = Font(bold=True)
1299
+ sheet.column_dimensions["A"].width = 120
1300
+ continue
1301
+ if sheet is None:
1302
+ continue
1303
+ if not s.strip():
1304
+ flush()
1305
+ continue
1306
+ m = re.match(r"^(#{3,6})\s+(.*)$", s)
1307
+ if m:
1308
+ flush()
1309
+ sheet.append([m.group(2)])
1310
+ sheet.cell(row=sheet.max_row, column=1).font = Font(bold=True)
1311
+ continue
1312
+ words += len(re.findall(r"\b\w+\b", s))
1313
+ para.append(s.strip())
1314
+ flush()
1315
+ if sheet is not None:
1316
+ contents.append([index, sheet.title, words])
1317
+ wb.save(TARGET)
1318
+ print("wrote", TARGET)
1319
+ '''
1320
+
1321
+ CONVERT_SCRIPTS: dict[str, str] = {
1322
+ "docx": _DOCX_SCRIPT,
1323
+ "pdf": _PDF_SCRIPT,
1324
+ "xlsx": _XLSX_SCRIPT,
1325
+ }