calkit-python 0.47.5__py3-none-any.whl → 0.47.7__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- calkit/agent_skills/build-paper-pipeline/SKILL.md +5 -2
- calkit/agent_skills/check-questions/SKILL.md +45 -5
- calkit/cli/latex.py +37 -0
- calkit/cli/list.py +18 -7
- calkit/cli/main/core.py +13 -2
- calkit/cli/sync.py +5 -1
- calkit/models/core.py +93 -6
- calkit/models/pipeline.py +52 -0
- calkit/notebooks.py +5 -0
- calkit/pipeline.py +3 -0
- calkit/questions.py +537 -53
- calkit/tests/cli/main/test_core.py +14 -0
- calkit/tests/cli/test_check.py +34 -17
- calkit/tests/cli/test_latex.py +78 -0
- calkit/tests/cli/test_list.py +12 -0
- calkit/tests/cli/test_notebooks.py +34 -0
- calkit/tests/cli/test_sync.py +11 -1
- calkit/tests/test_questions.py +415 -5
- {calkit_python-0.47.5.dist-info → calkit_python-0.47.7.dist-info}/METADATA +1 -1
- {calkit_python-0.47.5.dist-info → calkit_python-0.47.7.dist-info}/RECORD +36 -36
- {calkit_python-0.47.5.dist-info → calkit_python-0.47.7.dist-info}/WHEEL +1 -1
- {calkit_python-0.47.5.data → calkit_python-0.47.7.data}/data/etc/jupyter/jupyter_server_config.d/calkit.json +0 -0
- {calkit_python-0.47.5.data → calkit_python-0.47.7.data}/data/share/jupyter/labextensions/calkit/package.json +0 -0
- {calkit_python-0.47.5.data → calkit_python-0.47.7.data}/data/share/jupyter/labextensions/calkit/schemas/calkit/package.json.orig +0 -0
- {calkit_python-0.47.5.data → calkit_python-0.47.7.data}/data/share/jupyter/labextensions/calkit/schemas/calkit/plugin.json +0 -0
- {calkit_python-0.47.5.data → calkit_python-0.47.7.data}/data/share/jupyter/labextensions/calkit/static/506.c34b070098184e33.js +0 -0
- {calkit_python-0.47.5.data → calkit_python-0.47.7.data}/data/share/jupyter/labextensions/calkit/static/57d13a7399cec3c5.png +0 -0
- {calkit_python-0.47.5.data → calkit_python-0.47.7.data}/data/share/jupyter/labextensions/calkit/static/616.528427eb54a4a0c0.js +0 -0
- {calkit_python-0.47.5.data → calkit_python-0.47.7.data}/data/share/jupyter/labextensions/calkit/static/740.6bf87276e9e5788a.js +0 -0
- {calkit_python-0.47.5.data → calkit_python-0.47.7.data}/data/share/jupyter/labextensions/calkit/static/899.f9b9f9bf705a6493.js +0 -0
- {calkit_python-0.47.5.data → calkit_python-0.47.7.data}/data/share/jupyter/labextensions/calkit/static/935.46ecc6bf99aa593a.js +0 -0
- {calkit_python-0.47.5.data → calkit_python-0.47.7.data}/data/share/jupyter/labextensions/calkit/static/remoteEntry.109a3a8379365e59.js +0 -0
- {calkit_python-0.47.5.data → calkit_python-0.47.7.data}/data/share/jupyter/labextensions/calkit/static/style.js +0 -0
- {calkit_python-0.47.5.data → calkit_python-0.47.7.data}/data/share/jupyter/labextensions/calkit/static/third-party-licenses.json +0 -0
- {calkit_python-0.47.5.dist-info → calkit_python-0.47.7.dist-info}/entry_points.txt +0 -0
- {calkit_python-0.47.5.dist-info → calkit_python-0.47.7.dist-info}/licenses/LICENSE +0 -0
|
@@ -84,8 +84,11 @@ building a pipeline from scratch.
|
|
|
84
84
|
`calkit.yaml`: the `question` as the paper poses it, the `answer` as the
|
|
85
85
|
paper states it with `{name}` placeholders where numbers go, and
|
|
86
86
|
`evidence` naming the results file and key, the figure, and the
|
|
87
|
-
publication section that carries the argument.
|
|
88
|
-
|
|
87
|
+
publication section that carries the argument. The results file must be
|
|
88
|
+
one a stage from step 3 writes, never one written by hand. Where the
|
|
89
|
+
claim hinges on a threshold, write a conditional answer; the
|
|
90
|
+
`check-questions` skill covers both. Run `calkit check questions` and
|
|
91
|
+
fix what it reports.
|
|
89
92
|
|
|
90
93
|
6. **Run the pipeline, then check your own work.** Run `calkit run`, then
|
|
91
94
|
`calkit check repro`, which reads the manuscript back and reports any
|
|
@@ -21,12 +21,17 @@ Deterministic, done by `calkit check questions` — never re-derive by hand:
|
|
|
21
21
|
- every evidence path exists;
|
|
22
22
|
- every `value` key resolves in its file, and every `{name}` placeholder in
|
|
23
23
|
the prose resolves and formats;
|
|
24
|
+
- every clause of a conditional answer parses, names only evidence, and
|
|
25
|
+
renders, including the clauses the current values don't select;
|
|
24
26
|
- every publication `label` still exists in the LaTeX source;
|
|
25
27
|
- no evidence has changed (Git history for Git-tracked outputs, `dvc.lock`
|
|
26
28
|
for DVC-tracked ones) since the commit that last edited the question;
|
|
27
|
-
- each
|
|
28
|
-
|
|
29
|
-
|
|
29
|
+
- each `value` entry reads a file a pipeline stage produces, or one
|
|
30
|
+
declared with `imported_from`; anything else is an error, since a number
|
|
31
|
+
nothing computes is a magic number;
|
|
32
|
+
- each other evidence path is produced by a pipeline stage, or declared
|
|
33
|
+
with `imported_from` or `created_by`. This one is advisory: it is
|
|
34
|
+
reported as `unattributed` and does not fail the check.
|
|
30
35
|
|
|
31
36
|
Judgment, done here — the check reads paths and hashes, and cannot read a
|
|
32
37
|
sentence:
|
|
@@ -35,6 +40,8 @@ sentence:
|
|
|
35
40
|
this skill exists;
|
|
36
41
|
- does the answer still follow from the evidence, given what changed;
|
|
37
42
|
- are numbers retyped into the prose that should be `{name}` placeholders;
|
|
43
|
+
- does a claim resting on a threshold use a conditional answer, with a
|
|
44
|
+
threshold chosen before the value was known;
|
|
38
45
|
- is the answer concise, and does it point at the publication section that
|
|
39
46
|
carries the argument rather than repeating it.
|
|
40
47
|
|
|
@@ -104,7 +111,8 @@ evidence without showing them both.
|
|
|
104
111
|
from. If a stage should produce it, that is a pipeline gap worth
|
|
105
112
|
reporting. If it was imported or made by hand, declare it under
|
|
106
113
|
`figures`, `datasets`, or `publications` with `imported_from` or
|
|
107
|
-
`created_by` so the project says so.
|
|
114
|
+
`created_by` so the project says so. A `value` entry with no stage is
|
|
115
|
+
reported as an error rather than `unattributed`: give it a stage.
|
|
108
116
|
4. For each **error**, fix the reference: a missing path means the pipeline
|
|
109
117
|
has not been run or pulled; a bad key or placeholder means a results
|
|
110
118
|
file was restructured; a missing label means the publication was
|
|
@@ -127,7 +135,39 @@ project, not a tidy-up.
|
|
|
127
135
|
- One claim per question, two to four sentences. Say what was found and
|
|
128
136
|
what it means; leave the reasoning to the publication.
|
|
129
137
|
- Numbers come from `value` evidence via placeholders, formatted to the
|
|
130
|
-
precision the claim needs: `{ratio:.1f}x`, `{error:.0%}`.
|
|
138
|
+
precision the claim needs: `{ratio:.1f}x`, `{error:.0%}`. When several
|
|
139
|
+
come from one file, e.g., the outputs of one calculation, name them
|
|
140
|
+
together in one `result` entry's `values`, a map of name to key.
|
|
141
|
+
- A `value` entry must read a file a pipeline stage writes. A placeholder
|
|
142
|
+
over a results file written by hand, including one you write, is a
|
|
143
|
+
retyped number with extra steps, and the check fails it. If no stage
|
|
144
|
+
produces the number, add one (`/calkit:add-pipeline-stage`) instead of
|
|
145
|
+
writing the file. Never declare a file `imported_from` or `created_by`
|
|
146
|
+
to clear that error unless it really came from there, and never edit a
|
|
147
|
+
results file to change what an answer says.
|
|
148
|
+
- When the claim itself depends on a value, not just the number in it,
|
|
149
|
+
e.g., significant or not, which method wins, write a conditional answer
|
|
150
|
+
so the wording follows the evidence on a rerun:
|
|
151
|
+
|
|
152
|
+
```yaml
|
|
153
|
+
answer:
|
|
154
|
+
if p < 0.05: "The closure cuts error by {improvement:.1f}x."
|
|
155
|
+
else: "The closure does not measurably reduce error."
|
|
156
|
+
evidence:
|
|
157
|
+
- kind: result
|
|
158
|
+
path: results/closure.json
|
|
159
|
+
values:
|
|
160
|
+
p: p-value
|
|
161
|
+
improvement: improvement
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
Conditions use the names of values, so those names must be valid Python
|
|
165
|
+
identifiers. Every branch must be a
|
|
166
|
+
claim the evidence would support if it held. The threshold is part of
|
|
167
|
+
the claim: take it from the field's convention or the user, never pick
|
|
168
|
+
it to make the current branch hold, and don't use branches to hedge a
|
|
169
|
+
claim that should simply be weakened.
|
|
170
|
+
|
|
131
171
|
- Point at the publication with a `publication` evidence entry carrying
|
|
132
172
|
`section` (for the reader) and `label` (for the check), instead of an
|
|
133
173
|
`explanation` that restates the argument.
|
calkit/cli/latex.py
CHANGED
|
@@ -137,6 +137,43 @@ def from_json(
|
|
|
137
137
|
json2latex.dump(cmd_name, formatted, f)
|
|
138
138
|
|
|
139
139
|
|
|
140
|
+
@latex_app.command(name="from-questions")
|
|
141
|
+
def from_questions(
|
|
142
|
+
output_fpaths: Annotated[
|
|
143
|
+
list[str],
|
|
144
|
+
typer.Option("--output", "-o", help="Output LaTeX file path(s)."),
|
|
145
|
+
],
|
|
146
|
+
command_name: Annotated[
|
|
147
|
+
str,
|
|
148
|
+
typer.Option("--command", help="Command name to use in LaTeX output."),
|
|
149
|
+
] = "questions",
|
|
150
|
+
) -> None:
|
|
151
|
+
"""Write the project's questions and answers as a LaTeX command.
|
|
152
|
+
|
|
153
|
+
Each question's text, hypothesis, answer, and notes are rendered from
|
|
154
|
+
their evidence and exposed as, e.g., ``\\questions[staging.answer]``,
|
|
155
|
+
keyed by the question's ``name`` or its 1-based position.
|
|
156
|
+
"""
|
|
157
|
+
import json2latex
|
|
158
|
+
|
|
159
|
+
import calkit.questions
|
|
160
|
+
|
|
161
|
+
for out_path in output_fpaths:
|
|
162
|
+
if not out_path.endswith(".tex"):
|
|
163
|
+
raise_error("Output file must be a .tex file")
|
|
164
|
+
ck_info = calkit.load_calkit_info()
|
|
165
|
+
try:
|
|
166
|
+
values = calkit.questions.latex_values(ck_info)
|
|
167
|
+
except ValueError as e:
|
|
168
|
+
raise_error(str(e))
|
|
169
|
+
for out_path in output_fpaths:
|
|
170
|
+
outdir = os.path.dirname(out_path)
|
|
171
|
+
if outdir:
|
|
172
|
+
os.makedirs(outdir, exist_ok=True)
|
|
173
|
+
with open(out_path, "w") as f:
|
|
174
|
+
json2latex.dump(command_name, values, f)
|
|
175
|
+
|
|
176
|
+
|
|
140
177
|
def _tex_cmd(
|
|
141
178
|
tex_cmd: list[str],
|
|
142
179
|
environment: str | None,
|
calkit/cli/list.py
CHANGED
|
@@ -197,7 +197,10 @@ def _echo_question(n: int, question: str | dict) -> None:
|
|
|
197
197
|
question = dict(question)
|
|
198
198
|
text = question.pop("question", "")
|
|
199
199
|
typer.echo(f"{n}. question: {text}")
|
|
200
|
+
# Rendering fills every field, so leave out the ones that aren't set
|
|
200
201
|
for k, v in question.items():
|
|
202
|
+
if v is None:
|
|
203
|
+
continue
|
|
201
204
|
if isinstance(v, dict):
|
|
202
205
|
typer.echo(f" {k}:")
|
|
203
206
|
for k1, v1 in v.items():
|
|
@@ -206,11 +209,17 @@ def _echo_question(n: int, question: str | dict) -> None:
|
|
|
206
209
|
typer.echo(f" {k}:")
|
|
207
210
|
for item in v:
|
|
208
211
|
if isinstance(item, dict):
|
|
212
|
+
item = {
|
|
213
|
+
k1: v1 for k1, v1 in item.items() if v1 is not None
|
|
214
|
+
}
|
|
209
215
|
for n1, (k1, v1) in enumerate(item.items()):
|
|
210
|
-
if n1 == 0
|
|
211
|
-
|
|
216
|
+
prefix = " - " if n1 == 0 else " "
|
|
217
|
+
if isinstance(v1, dict):
|
|
218
|
+
typer.echo(f"{prefix}{k1}:")
|
|
219
|
+
for k2, v2 in v1.items():
|
|
220
|
+
typer.echo(f" {k2}: {v2}")
|
|
212
221
|
else:
|
|
213
|
-
typer.echo(f"
|
|
222
|
+
typer.echo(f"{prefix}{k1}: {v1}")
|
|
214
223
|
else:
|
|
215
224
|
typer.echo(f" - {item}")
|
|
216
225
|
else:
|
|
@@ -243,7 +252,7 @@ def list_questions(
|
|
|
243
252
|
render_question,
|
|
244
253
|
)
|
|
245
254
|
|
|
246
|
-
def _texts(question: dict) -> list[str]:
|
|
255
|
+
def _texts(question: dict) -> list[str | dict]:
|
|
247
256
|
evidence = question.get("evidence") or []
|
|
248
257
|
return [question.get(f) or "" for f in TEMPLATED_FIELDS] + [
|
|
249
258
|
ev.get("explanation") or ""
|
|
@@ -260,16 +269,18 @@ def list_questions(
|
|
|
260
269
|
# let a fresh clone read as a project that types its braces. Any
|
|
261
270
|
# placeholder left standing counts, whether it names evidence that
|
|
262
271
|
# could not be read or names nothing at all: a brace meant to stay
|
|
263
|
-
# in the text is written '{{' and never reaches here.
|
|
272
|
+
# in the text is written '{{' and never reaches here. A conditional
|
|
273
|
+
# answer still in clauses is one whose conditions could not be read.
|
|
264
274
|
unfilled = any(
|
|
265
|
-
placeholders(text)
|
|
275
|
+
isinstance(text, dict) or placeholders(text)
|
|
266
276
|
for q in rendered
|
|
267
277
|
if isinstance(q, dict)
|
|
268
278
|
for text in _texts(q)
|
|
269
279
|
)
|
|
270
280
|
if unfilled:
|
|
271
281
|
warn(
|
|
272
|
-
"Some placeholders could not be filled from
|
|
282
|
+
"Some placeholders or conditions could not be filled from "
|
|
283
|
+
"the evidence. "
|
|
273
284
|
"Run 'calkit check questions' to see why; 'calkit pull' if "
|
|
274
285
|
"the results files are not here yet.",
|
|
275
286
|
err=json_output,
|
calkit/cli/main/core.py
CHANGED
|
@@ -684,10 +684,16 @@ def get_status(
|
|
|
684
684
|
typer.echo()
|
|
685
685
|
if "dvc" in categories:
|
|
686
686
|
print_sep("DVC")
|
|
687
|
+
from dvc.exceptions import NotDvcRepoError
|
|
688
|
+
|
|
687
689
|
try:
|
|
688
690
|
calkit.dvc.get_dvc_repo()
|
|
689
|
-
except
|
|
691
|
+
except NotDvcRepoError:
|
|
690
692
|
typer.echo("This is not a DVC repository.\n")
|
|
693
|
+
except Exception as e:
|
|
694
|
+
typer.echo(
|
|
695
|
+
f"Failed to open DVC repo: {e.__class__.__name__}: {e}\n"
|
|
696
|
+
)
|
|
691
697
|
else:
|
|
692
698
|
zip_path_map = calkit.dvc.zip.get_zip_path_map()
|
|
693
699
|
dvc_repo = calkit.dvc.get_dvc_repo()
|
|
@@ -2699,14 +2705,19 @@ def run(
|
|
|
2699
2705
|
os.environ.pop("CALKIT_PIPELINE_RUNNING", None)
|
|
2700
2706
|
raise_error(f"Pipeline compilation failed: {e}")
|
|
2701
2707
|
# Initialize DVC repo if necessary
|
|
2708
|
+
from dvc.exceptions import NotDvcRepoError
|
|
2709
|
+
|
|
2702
2710
|
try:
|
|
2703
2711
|
calkit.dvc.get_dvc_repo()
|
|
2704
|
-
except
|
|
2712
|
+
except NotDvcRepoError:
|
|
2705
2713
|
if not quiet:
|
|
2706
2714
|
typer.echo("Initializing DVC repo")
|
|
2707
2715
|
result = calkit.dvc.init()
|
|
2708
2716
|
if result != 0:
|
|
2709
2717
|
raise_error("Failed to initialize DVC repo")
|
|
2718
|
+
except Exception as e:
|
|
2719
|
+
# E.g., DVC's site cache dir isn't writable, which 'dvc init' can't fix
|
|
2720
|
+
raise_error(f"Failed to open DVC repo: {e.__class__.__name__}: {e}")
|
|
2710
2721
|
# Convert deps into target stage names
|
|
2711
2722
|
# TODO: This could probably be merged back upstream into DVC
|
|
2712
2723
|
if dvc_stages is None:
|
calkit/cli/sync.py
CHANGED
|
@@ -39,12 +39,16 @@ def sync_dvc(
|
|
|
39
39
|
no_check_auth: Annotated[bool, typer.Option("--no-check-auth")] = False,
|
|
40
40
|
) -> None:
|
|
41
41
|
"""Sync the DVC repository by pulling and then pushing."""
|
|
42
|
+
from dvc.exceptions import NotDvcRepoError
|
|
43
|
+
|
|
42
44
|
from calkit.cli.main.core import pull, push
|
|
43
45
|
|
|
44
46
|
try:
|
|
45
47
|
calkit.dvc.get_dvc_repo()
|
|
46
|
-
except
|
|
48
|
+
except NotDvcRepoError:
|
|
47
49
|
raise_error("No DVC repository found. Run 'calkit init' first.")
|
|
50
|
+
except Exception as e:
|
|
51
|
+
raise_error(f"Failed to open DVC repo: {e.__class__.__name__}: {e}")
|
|
48
52
|
if not calkit.dvc.get_remotes():
|
|
49
53
|
raise_error(
|
|
50
54
|
"No DVC remotes configured. Add a remote with "
|
calkit/models/core.py
CHANGED
|
@@ -1703,8 +1703,12 @@ class FigureEvidence(BaseModel):
|
|
|
1703
1703
|
|
|
1704
1704
|
class ResultsEvidence(BaseModel):
|
|
1705
1705
|
"""Evidence in the form of a results file: a set of values, a table, a
|
|
1706
|
-
map, whatever the pipeline wrote.
|
|
1707
|
-
|
|
1706
|
+
map, whatever the pipeline wrote.
|
|
1707
|
+
|
|
1708
|
+
``values`` names related values within it, like the fields of a struct
|
|
1709
|
+
or object,
|
|
1710
|
+
so each can be templated into the answer as ``value`` evidence would be
|
|
1711
|
+
without an entry per value.
|
|
1708
1712
|
"""
|
|
1709
1713
|
|
|
1710
1714
|
kind: Literal["result"] = "result"
|
|
@@ -1716,6 +1720,14 @@ class ResultsEvidence(BaseModel):
|
|
|
1716
1720
|
"results file."
|
|
1717
1721
|
),
|
|
1718
1722
|
)
|
|
1723
|
+
values: dict[str, str] | None = Field(
|
|
1724
|
+
default=None,
|
|
1725
|
+
description=(
|
|
1726
|
+
"Values within the results file, mapping each name, under which "
|
|
1727
|
+
"it can be templated into the question's text, to its key. Names "
|
|
1728
|
+
"must be unique within the question."
|
|
1729
|
+
),
|
|
1730
|
+
)
|
|
1719
1731
|
explanation: str | None = None
|
|
1720
1732
|
git_ref: str | None = Field(
|
|
1721
1733
|
default=None,
|
|
@@ -1725,11 +1737,17 @@ class ResultsEvidence(BaseModel):
|
|
|
1725
1737
|
),
|
|
1726
1738
|
)
|
|
1727
1739
|
|
|
1740
|
+
@model_validator(mode="after")
|
|
1741
|
+
def _key_or_values(self) -> ResultsEvidence:
|
|
1742
|
+
if self.key is not None and self.values is not None:
|
|
1743
|
+
raise ValueError("a result takes 'values' or 'key', not both")
|
|
1744
|
+
return self
|
|
1745
|
+
|
|
1728
1746
|
|
|
1729
1747
|
_KEY_DESCRIPTION = (
|
|
1730
|
-
"Key of the value within the results file.
|
|
1731
|
-
"
|
|
1732
|
-
"
|
|
1748
|
+
"Key of the value within the results file. It is split on dots and "
|
|
1749
|
+
"walked into nested objects, taking the longest run of parts that names "
|
|
1750
|
+
"a key at each level, with integers indexing lists."
|
|
1733
1751
|
)
|
|
1734
1752
|
|
|
1735
1753
|
|
|
@@ -1817,6 +1835,34 @@ class PublicationEvidence(BaseModel):
|
|
|
1817
1835
|
)
|
|
1818
1836
|
|
|
1819
1837
|
|
|
1838
|
+
class DocumentEvidence(BaseModel):
|
|
1839
|
+
"""Evidence in the form of a document cited by path, e.g., a Markdown
|
|
1840
|
+
write-up, without declaring it as a publication.
|
|
1841
|
+
|
|
1842
|
+
The document must be built by a pipeline stage, usually a Markdown
|
|
1843
|
+
stage, so its numbers are injected from the results and go stale with
|
|
1844
|
+
them; one written by hand is an error, like a typed-in value.
|
|
1845
|
+
"""
|
|
1846
|
+
|
|
1847
|
+
kind: Literal["document"] = "document"
|
|
1848
|
+
path: str
|
|
1849
|
+
section: str | None = Field(
|
|
1850
|
+
default=None,
|
|
1851
|
+
description=(
|
|
1852
|
+
"Section of the document where the evidence is presented, as a "
|
|
1853
|
+
"reader would find it, e.g., '4.2' or 'Results'."
|
|
1854
|
+
),
|
|
1855
|
+
)
|
|
1856
|
+
explanation: str | None = None
|
|
1857
|
+
git_ref: str | None = Field(
|
|
1858
|
+
default=None,
|
|
1859
|
+
description=(
|
|
1860
|
+
"Git reference (branch, tag, or commit hash) pointing to the "
|
|
1861
|
+
"version of the repository where the document can be found."
|
|
1862
|
+
),
|
|
1863
|
+
)
|
|
1864
|
+
|
|
1865
|
+
|
|
1820
1866
|
class Question(BaseModel):
|
|
1821
1867
|
"""A question the project hopes to answer.
|
|
1822
1868
|
|
|
@@ -1839,9 +1885,26 @@ class Question(BaseModel):
|
|
|
1839
1885
|
new evidence is how to say it still holds.
|
|
1840
1886
|
"""
|
|
1841
1887
|
|
|
1888
|
+
name: str | None = Field(
|
|
1889
|
+
default=None,
|
|
1890
|
+
description=(
|
|
1891
|
+
"Name for the question, e.g., for quoting its answer in a "
|
|
1892
|
+
"document through 'calkit latex from-questions'. Unlike its "
|
|
1893
|
+
"position in the list, it survives questions being added or "
|
|
1894
|
+
"reordered. Must be unique among the project's questions."
|
|
1895
|
+
),
|
|
1896
|
+
)
|
|
1842
1897
|
question: str
|
|
1843
1898
|
hypothesis: str | None = None
|
|
1844
|
-
answer: str | None =
|
|
1899
|
+
answer: str | dict[str, str] | None = Field(
|
|
1900
|
+
default=None,
|
|
1901
|
+
description=(
|
|
1902
|
+
"The claim the evidence supports. A mapping keyed by 'if "
|
|
1903
|
+
"<condition>', 'elif <condition>' and 'else' picks its wording "
|
|
1904
|
+
"from the evidence, so an answer resting on a threshold states "
|
|
1905
|
+
"the other outcome instead of going stale when a value moves."
|
|
1906
|
+
),
|
|
1907
|
+
)
|
|
1845
1908
|
notes: str | None = Field(
|
|
1846
1909
|
default=None,
|
|
1847
1910
|
description=(
|
|
@@ -1858,10 +1921,23 @@ class Question(BaseModel):
|
|
|
1858
1921
|
| ValueEvidence
|
|
1859
1922
|
| TableEvidence
|
|
1860
1923
|
| PublicationEvidence
|
|
1924
|
+
| DocumentEvidence
|
|
1861
1925
|
]
|
|
1862
1926
|
| None
|
|
1863
1927
|
) = None
|
|
1864
1928
|
|
|
1929
|
+
@field_validator("name")
|
|
1930
|
+
@classmethod
|
|
1931
|
+
def check_name_not_a_position(cls, v: str | None) -> str | None:
|
|
1932
|
+
# A question is also addressable by its position, so an all-digit
|
|
1933
|
+
# name would be ambiguous with some other question's number
|
|
1934
|
+
if v is not None and v.isdigit():
|
|
1935
|
+
raise ValueError(
|
|
1936
|
+
f"Question name {v!r} can't be a number, since questions "
|
|
1937
|
+
"are also addressed by position"
|
|
1938
|
+
)
|
|
1939
|
+
return v
|
|
1940
|
+
|
|
1865
1941
|
|
|
1866
1942
|
class ProjectInfo(BaseModel):
|
|
1867
1943
|
"""All of the project's information or metadata, written to the
|
|
@@ -2066,3 +2142,14 @@ class ProjectInfo(BaseModel):
|
|
|
2066
2142
|
description="Overleaf sync configuration, keyed by the path of the "
|
|
2067
2143
|
"synced directory.",
|
|
2068
2144
|
)
|
|
2145
|
+
|
|
2146
|
+
@field_validator("questions")
|
|
2147
|
+
@classmethod
|
|
2148
|
+
def check_question_names_unique(
|
|
2149
|
+
cls, v: list[str | Question]
|
|
2150
|
+
) -> list[str | Question]:
|
|
2151
|
+
names = [q.name for q in v if isinstance(q, Question) and q.name]
|
|
2152
|
+
dupes = sorted({n for n in names if names.count(n) > 1})
|
|
2153
|
+
if dupes:
|
|
2154
|
+
raise ValueError(f"Question names must be unique: {dupes}")
|
|
2155
|
+
return v
|
calkit/models/pipeline.py
CHANGED
|
@@ -254,6 +254,7 @@ class Stage(BaseModel):
|
|
|
254
254
|
"marimo-html-wasm",
|
|
255
255
|
"markdown",
|
|
256
256
|
"procedure",
|
|
257
|
+
"questions-to-latex",
|
|
257
258
|
] = Field(description="What kind of stage this is.")
|
|
258
259
|
environment: str = Field(
|
|
259
260
|
description="Name of the environment in which to run this stage."
|
|
@@ -1343,6 +1344,56 @@ class JsonToLatexStage(Stage):
|
|
|
1343
1344
|
return outs
|
|
1344
1345
|
|
|
1345
1346
|
|
|
1347
|
+
class QuestionsToLatexStage(Stage):
|
|
1348
|
+
"""The project's questions and answers, rendered for a LaTeX document.
|
|
1349
|
+
|
|
1350
|
+
Its inputs are ``calkit.yaml`` and every file the questions cite as
|
|
1351
|
+
evidence, added when the pipeline is compiled, so the output reruns
|
|
1352
|
+
when an answer or a value it reads changes.
|
|
1353
|
+
"""
|
|
1354
|
+
|
|
1355
|
+
kind: Literal["questions-to-latex"] = "questions-to-latex"
|
|
1356
|
+
environment: str = "_system"
|
|
1357
|
+
command_name: str = Field(
|
|
1358
|
+
default="questions",
|
|
1359
|
+
description=(
|
|
1360
|
+
"Name of the LaTeX command the document quotes questions "
|
|
1361
|
+
"through, e.g., 'questions' for \\questions[staging.answer]."
|
|
1362
|
+
),
|
|
1363
|
+
)
|
|
1364
|
+
wdir: None = Field(
|
|
1365
|
+
default=None,
|
|
1366
|
+
description="Not supported; the stage reads the project's "
|
|
1367
|
+
"calkit.yaml and evidence from the project root.",
|
|
1368
|
+
)
|
|
1369
|
+
|
|
1370
|
+
@property
|
|
1371
|
+
def dvc_cmd(self) -> str:
|
|
1372
|
+
cmd = "calkit latex from-questions"
|
|
1373
|
+
for out in self.outputs:
|
|
1374
|
+
out_path = out if isinstance(out, str) else out.path
|
|
1375
|
+
cmd += f" --output {shlex.quote(out_path)}"
|
|
1376
|
+
return cmd + f" --command {shlex.quote(self.command_name)}"
|
|
1377
|
+
|
|
1378
|
+
@property
|
|
1379
|
+
def dvc_outs(self) -> list[str | dict]:
|
|
1380
|
+
"""Stored with Git by default, like other generated LaTeX."""
|
|
1381
|
+
outs: list[str | dict] = []
|
|
1382
|
+
for out in self.outputs:
|
|
1383
|
+
if isinstance(out, str):
|
|
1384
|
+
outs.append({out: dict(cache=False, persist=False)})
|
|
1385
|
+
elif isinstance(out, PathOutput):
|
|
1386
|
+
outs.append(
|
|
1387
|
+
{
|
|
1388
|
+
out.path: dict(
|
|
1389
|
+
cache=out.storage == "dvc",
|
|
1390
|
+
persist=not out.delete_before_run,
|
|
1391
|
+
)
|
|
1392
|
+
}
|
|
1393
|
+
)
|
|
1394
|
+
return outs
|
|
1395
|
+
|
|
1396
|
+
|
|
1346
1397
|
class MatlabScriptStage(Stage):
|
|
1347
1398
|
kind: Literal["matlab-script"]
|
|
1348
1399
|
script_path: RelativeChildPathString = Field(
|
|
@@ -2096,6 +2147,7 @@ class Pipeline(BaseModel):
|
|
|
2096
2147
|
| LatexStage
|
|
2097
2148
|
| QuartoStage
|
|
2098
2149
|
| JsonToLatexStage
|
|
2150
|
+
| QuestionsToLatexStage
|
|
2099
2151
|
| MatlabScriptStage
|
|
2100
2152
|
| MatlabCommandStage
|
|
2101
2153
|
| ShellCommandStage
|
calkit/notebooks.py
CHANGED
|
@@ -96,6 +96,11 @@ def clean_notebook(nb: dict) -> dict:
|
|
|
96
96
|
if cell.get("cell_type") == "code":
|
|
97
97
|
cell["outputs"] = []
|
|
98
98
|
cell["execution_count"] = None
|
|
99
|
+
# Cell IDs are editor residue, and nbstripout rewrites them to
|
|
100
|
+
# ordinals. Keeping them would make the cleaned copy depend on whether
|
|
101
|
+
# the source had been through a clean filter, which differs between a
|
|
102
|
+
# working tree and a fresh clone of the same commit.
|
|
103
|
+
cell.pop("id", None)
|
|
99
104
|
# Clean metadata but keep tags
|
|
100
105
|
if "tags" in cell.get("metadata", {}):
|
|
101
106
|
cell["metadata"] = {"tags": cell["metadata"]["tags"]}
|
calkit/pipeline.py
CHANGED
|
@@ -1634,6 +1634,7 @@ def to_dvc(
|
|
|
1634
1634
|
"""
|
|
1635
1635
|
import calkit.dvc.zip
|
|
1636
1636
|
import calkit.markdown
|
|
1637
|
+
import calkit.questions
|
|
1637
1638
|
from calkit.environments import get_env_input_paths, get_env_lock_fpath
|
|
1638
1639
|
|
|
1639
1640
|
if ck_info is None:
|
|
@@ -1645,6 +1646,8 @@ def to_dvc(
|
|
|
1645
1646
|
# options and iteration need no knowledge of Markdown.
|
|
1646
1647
|
markdown = calkit.markdown.expand_ck_info(ck_info, wdir=wdir)
|
|
1647
1648
|
ck_info = markdown.ck_info
|
|
1649
|
+
# Likewise give questions-to-latex stages the evidence they read
|
|
1650
|
+
ck_info = calkit.questions.expand_questions_stages(ck_info)
|
|
1648
1651
|
if write and markdown.environments:
|
|
1649
1652
|
_write_markdown_environments(markdown, wdir=wdir)
|
|
1650
1653
|
# Everything Markdown derives is rewritten on every compile, so it
|