calkit-python 0.47.6__py3-none-any.whl → 0.47.7__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. calkit/agent_skills/check-questions/SKILL.md +9 -9
  2. calkit/cli/latex.py +37 -0
  3. calkit/cli/list.py +12 -3
  4. calkit/cli/main/core.py +13 -2
  5. calkit/cli/sync.py +5 -1
  6. calkit/models/core.py +84 -5
  7. calkit/models/pipeline.py +52 -0
  8. calkit/notebooks.py +5 -0
  9. calkit/pipeline.py +3 -0
  10. calkit/questions.py +309 -37
  11. calkit/tests/cli/main/test_core.py +14 -0
  12. calkit/tests/cli/test_latex.py +78 -0
  13. calkit/tests/cli/test_list.py +12 -0
  14. calkit/tests/cli/test_notebooks.py +34 -0
  15. calkit/tests/cli/test_sync.py +11 -1
  16. calkit/tests/test_questions.py +260 -4
  17. {calkit_python-0.47.6.dist-info → calkit_python-0.47.7.dist-info}/METADATA +1 -1
  18. {calkit_python-0.47.6.dist-info → calkit_python-0.47.7.dist-info}/RECORD +34 -34
  19. {calkit_python-0.47.6.dist-info → calkit_python-0.47.7.dist-info}/WHEEL +1 -1
  20. {calkit_python-0.47.6.data → calkit_python-0.47.7.data}/data/etc/jupyter/jupyter_server_config.d/calkit.json +0 -0
  21. {calkit_python-0.47.6.data → calkit_python-0.47.7.data}/data/share/jupyter/labextensions/calkit/package.json +0 -0
  22. {calkit_python-0.47.6.data → calkit_python-0.47.7.data}/data/share/jupyter/labextensions/calkit/schemas/calkit/package.json.orig +0 -0
  23. {calkit_python-0.47.6.data → calkit_python-0.47.7.data}/data/share/jupyter/labextensions/calkit/schemas/calkit/plugin.json +0 -0
  24. {calkit_python-0.47.6.data → calkit_python-0.47.7.data}/data/share/jupyter/labextensions/calkit/static/506.c34b070098184e33.js +0 -0
  25. {calkit_python-0.47.6.data → calkit_python-0.47.7.data}/data/share/jupyter/labextensions/calkit/static/57d13a7399cec3c5.png +0 -0
  26. {calkit_python-0.47.6.data → calkit_python-0.47.7.data}/data/share/jupyter/labextensions/calkit/static/616.528427eb54a4a0c0.js +0 -0
  27. {calkit_python-0.47.6.data → calkit_python-0.47.7.data}/data/share/jupyter/labextensions/calkit/static/740.6bf87276e9e5788a.js +0 -0
  28. {calkit_python-0.47.6.data → calkit_python-0.47.7.data}/data/share/jupyter/labextensions/calkit/static/899.f9b9f9bf705a6493.js +0 -0
  29. {calkit_python-0.47.6.data → calkit_python-0.47.7.data}/data/share/jupyter/labextensions/calkit/static/935.46ecc6bf99aa593a.js +0 -0
  30. {calkit_python-0.47.6.data → calkit_python-0.47.7.data}/data/share/jupyter/labextensions/calkit/static/remoteEntry.109a3a8379365e59.js +0 -0
  31. {calkit_python-0.47.6.data → calkit_python-0.47.7.data}/data/share/jupyter/labextensions/calkit/static/style.js +0 -0
  32. {calkit_python-0.47.6.data → calkit_python-0.47.7.data}/data/share/jupyter/labextensions/calkit/static/third-party-licenses.json +0 -0
  33. {calkit_python-0.47.6.dist-info → calkit_python-0.47.7.dist-info}/entry_points.txt +0 -0
  34. {calkit_python-0.47.6.dist-info → calkit_python-0.47.7.dist-info}/licenses/LICENSE +0 -0
@@ -135,7 +135,9 @@ project, not a tidy-up.
135
135
  - One claim per question, two to four sentences. Say what was found and
136
136
  what it means; leave the reasoning to the publication.
137
137
  - Numbers come from `value` evidence via placeholders, formatted to the
138
- precision the claim needs: `{ratio:.1f}x`, `{error:.0%}`.
138
+ precision the claim needs: `{ratio:.1f}x`, `{error:.0%}`. When several
139
+ come from one file, e.g., the outputs of one calculation, name them
140
+ together in one `result` entry's `values`, a map of name to key.
139
141
  - A `value` entry must read a file a pipeline stage writes. A placeholder
140
142
  over a results file written by hand, including one you write, is a
141
143
  retyped number with extra steps, and the check fails it. If no stage
@@ -152,17 +154,15 @@ project, not a tidy-up.
152
154
  if p < 0.05: "The closure cuts error by {improvement:.1f}x."
153
155
  else: "The closure does not measurably reduce error."
154
156
  evidence:
155
- - kind: value
157
+ - kind: result
156
158
  path: results/closure.json
157
- key: p-value
158
- name: p
159
- - kind: value
160
- path: results/closure.json
161
- key: improvement
159
+ values:
160
+ p: p-value
161
+ improvement: improvement
162
162
  ```
163
163
 
164
- Conditions name `value` evidence, so those names must be valid Python
165
- identifiers; give the entry a `name` otherwise. Every branch must be a
164
+ Conditions use the names of values, so those names must be valid Python
165
+ identifiers. Every branch must be a
166
166
  claim the evidence would support if it held. The threshold is part of
167
167
  the claim: take it from the field's convention or the user, never pick
168
168
  it to make the current branch hold, and don't use branches to hedge a
calkit/cli/latex.py CHANGED
@@ -137,6 +137,43 @@ def from_json(
137
137
  json2latex.dump(cmd_name, formatted, f)
138
138
 
139
139
 
140
+ @latex_app.command(name="from-questions")
141
+ def from_questions(
142
+ output_fpaths: Annotated[
143
+ list[str],
144
+ typer.Option("--output", "-o", help="Output LaTeX file path(s)."),
145
+ ],
146
+ command_name: Annotated[
147
+ str,
148
+ typer.Option("--command", help="Command name to use in LaTeX output."),
149
+ ] = "questions",
150
+ ) -> None:
151
+ """Write the project's questions and answers as a LaTeX command.
152
+
153
+ Each question's text, hypothesis, answer, and notes are rendered from
154
+ their evidence and exposed as, e.g., ``\\questions[staging.answer]``,
155
+ keyed by the question's ``name`` or its 1-based position.
156
+ """
157
+ import json2latex
158
+
159
+ import calkit.questions
160
+
161
+ for out_path in output_fpaths:
162
+ if not out_path.endswith(".tex"):
163
+ raise_error("Output file must be a .tex file")
164
+ ck_info = calkit.load_calkit_info()
165
+ try:
166
+ values = calkit.questions.latex_values(ck_info)
167
+ except ValueError as e:
168
+ raise_error(str(e))
169
+ for out_path in output_fpaths:
170
+ outdir = os.path.dirname(out_path)
171
+ if outdir:
172
+ os.makedirs(outdir, exist_ok=True)
173
+ with open(out_path, "w") as f:
174
+ json2latex.dump(command_name, values, f)
175
+
176
+
140
177
  def _tex_cmd(
141
178
  tex_cmd: list[str],
142
179
  environment: str | None,
calkit/cli/list.py CHANGED
@@ -197,7 +197,10 @@ def _echo_question(n: int, question: str | dict) -> None:
197
197
  question = dict(question)
198
198
  text = question.pop("question", "")
199
199
  typer.echo(f"{n}. question: {text}")
200
+ # Rendering fills every field, so leave out the ones that aren't set
200
201
  for k, v in question.items():
202
+ if v is None:
203
+ continue
201
204
  if isinstance(v, dict):
202
205
  typer.echo(f" {k}:")
203
206
  for k1, v1 in v.items():
@@ -206,11 +209,17 @@ def _echo_question(n: int, question: str | dict) -> None:
206
209
  typer.echo(f" {k}:")
207
210
  for item in v:
208
211
  if isinstance(item, dict):
212
+ item = {
213
+ k1: v1 for k1, v1 in item.items() if v1 is not None
214
+ }
209
215
  for n1, (k1, v1) in enumerate(item.items()):
210
- if n1 == 0:
211
- typer.echo(f" - {k1}: {v1}")
216
+ prefix = " - " if n1 == 0 else " "
217
+ if isinstance(v1, dict):
218
+ typer.echo(f"{prefix}{k1}:")
219
+ for k2, v2 in v1.items():
220
+ typer.echo(f" {k2}: {v2}")
212
221
  else:
213
- typer.echo(f" {k1}: {v1}")
222
+ typer.echo(f"{prefix}{k1}: {v1}")
214
223
  else:
215
224
  typer.echo(f" - {item}")
216
225
  else:
calkit/cli/main/core.py CHANGED
@@ -684,10 +684,16 @@ def get_status(
684
684
  typer.echo()
685
685
  if "dvc" in categories:
686
686
  print_sep("DVC")
687
+ from dvc.exceptions import NotDvcRepoError
688
+
687
689
  try:
688
690
  calkit.dvc.get_dvc_repo()
689
- except Exception:
691
+ except NotDvcRepoError:
690
692
  typer.echo("This is not a DVC repository.\n")
693
+ except Exception as e:
694
+ typer.echo(
695
+ f"Failed to open DVC repo: {e.__class__.__name__}: {e}\n"
696
+ )
691
697
  else:
692
698
  zip_path_map = calkit.dvc.zip.get_zip_path_map()
693
699
  dvc_repo = calkit.dvc.get_dvc_repo()
@@ -2699,14 +2705,19 @@ def run(
2699
2705
  os.environ.pop("CALKIT_PIPELINE_RUNNING", None)
2700
2706
  raise_error(f"Pipeline compilation failed: {e}")
2701
2707
  # Initialize DVC repo if necessary
2708
+ from dvc.exceptions import NotDvcRepoError
2709
+
2702
2710
  try:
2703
2711
  calkit.dvc.get_dvc_repo()
2704
- except Exception:
2712
+ except NotDvcRepoError:
2705
2713
  if not quiet:
2706
2714
  typer.echo("Initializing DVC repo")
2707
2715
  result = calkit.dvc.init()
2708
2716
  if result != 0:
2709
2717
  raise_error("Failed to initialize DVC repo")
2718
+ except Exception as e:
2719
+ # E.g., DVC's site cache dir isn't writable, which 'dvc init' can't fix
2720
+ raise_error(f"Failed to open DVC repo: {e.__class__.__name__}: {e}")
2710
2721
  # Convert deps into target stage names
2711
2722
  # TODO: This could probably be merged back upstream into DVC
2712
2723
  if dvc_stages is None:
calkit/cli/sync.py CHANGED
@@ -39,12 +39,16 @@ def sync_dvc(
39
39
  no_check_auth: Annotated[bool, typer.Option("--no-check-auth")] = False,
40
40
  ) -> None:
41
41
  """Sync the DVC repository by pulling and then pushing."""
42
+ from dvc.exceptions import NotDvcRepoError
43
+
42
44
  from calkit.cli.main.core import pull, push
43
45
 
44
46
  try:
45
47
  calkit.dvc.get_dvc_repo()
46
- except Exception:
48
+ except NotDvcRepoError:
47
49
  raise_error("No DVC repository found. Run 'calkit init' first.")
50
+ except Exception as e:
51
+ raise_error(f"Failed to open DVC repo: {e.__class__.__name__}: {e}")
48
52
  if not calkit.dvc.get_remotes():
49
53
  raise_error(
50
54
  "No DVC remotes configured. Add a remote with "
calkit/models/core.py CHANGED
@@ -1703,8 +1703,12 @@ class FigureEvidence(BaseModel):
1703
1703
 
1704
1704
  class ResultsEvidence(BaseModel):
1705
1705
  """Evidence in the form of a results file: a set of values, a table, a
1706
- map, whatever the pipeline wrote. For one value inside such a file, use
1707
- ``value`` evidence, which can be templated into the answer.
1706
+ map, whatever the pipeline wrote.
1707
+
1708
+ ``values`` names related values within it, like the fields of a struct
1709
+ or object,
1710
+ so each can be templated into the answer as ``value`` evidence would be
1711
+ without an entry per value.
1708
1712
  """
1709
1713
 
1710
1714
  kind: Literal["result"] = "result"
@@ -1716,6 +1720,14 @@ class ResultsEvidence(BaseModel):
1716
1720
  "results file."
1717
1721
  ),
1718
1722
  )
1723
+ values: dict[str, str] | None = Field(
1724
+ default=None,
1725
+ description=(
1726
+ "Values within the results file, mapping each name, under which "
1727
+ "it can be templated into the question's text, to its key. Names "
1728
+ "must be unique within the question."
1729
+ ),
1730
+ )
1719
1731
  explanation: str | None = None
1720
1732
  git_ref: str | None = Field(
1721
1733
  default=None,
@@ -1725,11 +1737,17 @@ class ResultsEvidence(BaseModel):
1725
1737
  ),
1726
1738
  )
1727
1739
 
1740
+ @model_validator(mode="after")
1741
+ def _key_or_values(self) -> ResultsEvidence:
1742
+ if self.key is not None and self.values is not None:
1743
+ raise ValueError("a result takes 'values' or 'key', not both")
1744
+ return self
1745
+
1728
1746
 
1729
1747
  _KEY_DESCRIPTION = (
1730
- "Key of the value within the results file. A key present at the top "
1731
- "level is used as-is; otherwise it is split on dots and walked into "
1732
- "nested objects, with integers indexing lists."
1748
+ "Key of the value within the results file. It is split on dots and "
1749
+ "walked into nested objects, taking the longest run of parts that names "
1750
+ "a key at each level, with integers indexing lists."
1733
1751
  )
1734
1752
 
1735
1753
 
@@ -1817,6 +1835,34 @@ class PublicationEvidence(BaseModel):
1817
1835
  )
1818
1836
 
1819
1837
 
1838
+ class DocumentEvidence(BaseModel):
1839
+ """Evidence in the form of a document cited by path, e.g., a Markdown
1840
+ write-up, without declaring it as a publication.
1841
+
1842
+ The document must be built by a pipeline stage, usually a Markdown
1843
+ stage, so its numbers are injected from the results and go stale with
1844
+ them; one written by hand is an error, like a typed-in value.
1845
+ """
1846
+
1847
+ kind: Literal["document"] = "document"
1848
+ path: str
1849
+ section: str | None = Field(
1850
+ default=None,
1851
+ description=(
1852
+ "Section of the document where the evidence is presented, as a "
1853
+ "reader would find it, e.g., '4.2' or 'Results'."
1854
+ ),
1855
+ )
1856
+ explanation: str | None = None
1857
+ git_ref: str | None = Field(
1858
+ default=None,
1859
+ description=(
1860
+ "Git reference (branch, tag, or commit hash) pointing to the "
1861
+ "version of the repository where the document can be found."
1862
+ ),
1863
+ )
1864
+
1865
+
1820
1866
  class Question(BaseModel):
1821
1867
  """A question the project hopes to answer.
1822
1868
 
@@ -1839,6 +1885,15 @@ class Question(BaseModel):
1839
1885
  new evidence is how to say it still holds.
1840
1886
  """
1841
1887
 
1888
+ name: str | None = Field(
1889
+ default=None,
1890
+ description=(
1891
+ "Name for the question, e.g., for quoting its answer in a "
1892
+ "document through 'calkit latex from-questions'. Unlike its "
1893
+ "position in the list, it survives questions being added or "
1894
+ "reordered. Must be unique among the project's questions."
1895
+ ),
1896
+ )
1842
1897
  question: str
1843
1898
  hypothesis: str | None = None
1844
1899
  answer: str | dict[str, str] | None = Field(
@@ -1866,10 +1921,23 @@ class Question(BaseModel):
1866
1921
  | ValueEvidence
1867
1922
  | TableEvidence
1868
1923
  | PublicationEvidence
1924
+ | DocumentEvidence
1869
1925
  ]
1870
1926
  | None
1871
1927
  ) = None
1872
1928
 
1929
+ @field_validator("name")
1930
+ @classmethod
1931
+ def check_name_not_a_position(cls, v: str | None) -> str | None:
1932
+ # A question is also addressable by its position, so an all-digit
1933
+ # name would be ambiguous with some other question's number
1934
+ if v is not None and v.isdigit():
1935
+ raise ValueError(
1936
+ f"Question name {v!r} can't be a number, since questions "
1937
+ "are also addressed by position"
1938
+ )
1939
+ return v
1940
+
1873
1941
 
1874
1942
  class ProjectInfo(BaseModel):
1875
1943
  """All of the project's information or metadata, written to the
@@ -2074,3 +2142,14 @@ class ProjectInfo(BaseModel):
2074
2142
  description="Overleaf sync configuration, keyed by the path of the "
2075
2143
  "synced directory.",
2076
2144
  )
2145
+
2146
+ @field_validator("questions")
2147
+ @classmethod
2148
+ def check_question_names_unique(
2149
+ cls, v: list[str | Question]
2150
+ ) -> list[str | Question]:
2151
+ names = [q.name for q in v if isinstance(q, Question) and q.name]
2152
+ dupes = sorted({n for n in names if names.count(n) > 1})
2153
+ if dupes:
2154
+ raise ValueError(f"Question names must be unique: {dupes}")
2155
+ return v
calkit/models/pipeline.py CHANGED
@@ -254,6 +254,7 @@ class Stage(BaseModel):
254
254
  "marimo-html-wasm",
255
255
  "markdown",
256
256
  "procedure",
257
+ "questions-to-latex",
257
258
  ] = Field(description="What kind of stage this is.")
258
259
  environment: str = Field(
259
260
  description="Name of the environment in which to run this stage."
@@ -1343,6 +1344,56 @@ class JsonToLatexStage(Stage):
1343
1344
  return outs
1344
1345
 
1345
1346
 
1347
+ class QuestionsToLatexStage(Stage):
1348
+ """The project's questions and answers, rendered for a LaTeX document.
1349
+
1350
+ Its inputs are ``calkit.yaml`` and every file the questions cite as
1351
+ evidence, added when the pipeline is compiled, so the output reruns
1352
+ when an answer or a value it reads changes.
1353
+ """
1354
+
1355
+ kind: Literal["questions-to-latex"] = "questions-to-latex"
1356
+ environment: str = "_system"
1357
+ command_name: str = Field(
1358
+ default="questions",
1359
+ description=(
1360
+ "Name of the LaTeX command the document quotes questions "
1361
+ "through, e.g., 'questions' for \\questions[staging.answer]."
1362
+ ),
1363
+ )
1364
+ wdir: None = Field(
1365
+ default=None,
1366
+ description="Not supported; the stage reads the project's "
1367
+ "calkit.yaml and evidence from the project root.",
1368
+ )
1369
+
1370
+ @property
1371
+ def dvc_cmd(self) -> str:
1372
+ cmd = "calkit latex from-questions"
1373
+ for out in self.outputs:
1374
+ out_path = out if isinstance(out, str) else out.path
1375
+ cmd += f" --output {shlex.quote(out_path)}"
1376
+ return cmd + f" --command {shlex.quote(self.command_name)}"
1377
+
1378
+ @property
1379
+ def dvc_outs(self) -> list[str | dict]:
1380
+ """Stored with Git by default, like other generated LaTeX."""
1381
+ outs: list[str | dict] = []
1382
+ for out in self.outputs:
1383
+ if isinstance(out, str):
1384
+ outs.append({out: dict(cache=False, persist=False)})
1385
+ elif isinstance(out, PathOutput):
1386
+ outs.append(
1387
+ {
1388
+ out.path: dict(
1389
+ cache=out.storage == "dvc",
1390
+ persist=not out.delete_before_run,
1391
+ )
1392
+ }
1393
+ )
1394
+ return outs
1395
+
1396
+
1346
1397
  class MatlabScriptStage(Stage):
1347
1398
  kind: Literal["matlab-script"]
1348
1399
  script_path: RelativeChildPathString = Field(
@@ -2096,6 +2147,7 @@ class Pipeline(BaseModel):
2096
2147
  | LatexStage
2097
2148
  | QuartoStage
2098
2149
  | JsonToLatexStage
2150
+ | QuestionsToLatexStage
2099
2151
  | MatlabScriptStage
2100
2152
  | MatlabCommandStage
2101
2153
  | ShellCommandStage
calkit/notebooks.py CHANGED
@@ -96,6 +96,11 @@ def clean_notebook(nb: dict) -> dict:
96
96
  if cell.get("cell_type") == "code":
97
97
  cell["outputs"] = []
98
98
  cell["execution_count"] = None
99
+ # Cell IDs are editor residue, and nbstripout rewrites them to
100
+ # ordinals. Keeping them would make the cleaned copy depend on whether
101
+ # the source had been through a clean filter, which differs between a
102
+ # working tree and a fresh clone of the same commit.
103
+ cell.pop("id", None)
99
104
  # Clean metadata but keep tags
100
105
  if "tags" in cell.get("metadata", {}):
101
106
  cell["metadata"] = {"tags": cell["metadata"]["tags"]}
calkit/pipeline.py CHANGED
@@ -1634,6 +1634,7 @@ def to_dvc(
1634
1634
  """
1635
1635
  import calkit.dvc.zip
1636
1636
  import calkit.markdown
1637
+ import calkit.questions
1637
1638
  from calkit.environments import get_env_input_paths, get_env_lock_fpath
1638
1639
 
1639
1640
  if ck_info is None:
@@ -1645,6 +1646,8 @@ def to_dvc(
1645
1646
  # options and iteration need no knowledge of Markdown.
1646
1647
  markdown = calkit.markdown.expand_ck_info(ck_info, wdir=wdir)
1647
1648
  ck_info = markdown.ck_info
1649
+ # Likewise give questions-to-latex stages the evidence they read
1650
+ ck_info = calkit.questions.expand_questions_stages(ck_info)
1648
1651
  if write and markdown.environments:
1649
1652
  _write_markdown_environments(markdown, wdir=wdir)
1650
1653
  # Everything Markdown derives is rewritten on every compile, so it