refkit 0.2.1__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. {refkit-0.2.1 → refkit-0.3.0}/.agent-plugin/skills/refkit/references/inspect.md +19 -0
  2. {refkit-0.2.1 → refkit-0.3.0}/Cargo.lock +2 -2
  3. {refkit-0.2.1 → refkit-0.3.0}/Cargo.toml +2 -2
  4. {refkit-0.2.1 → refkit-0.3.0}/PKG-INFO +1 -1
  5. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/lib.rs +1 -0
  6. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/library/guard.rs +92 -10
  7. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/library/mod.rs +3 -1
  8. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/raw/parse.rs +11 -2
  9. refkit-0.3.0/crates/refkit-core/src/raw/resolve.rs +384 -0
  10. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/raw.rs +13 -0
  11. {refkit-0.2.1 → refkit-0.3.0}/packages/refkit/rust/src/raw.rs +17 -1
  12. {refkit-0.2.1 → refkit-0.3.0}/pyproject.toml +1 -1
  13. {refkit-0.2.1 → refkit-0.3.0}/src/refkit/_native.pyi +2 -0
  14. {refkit-0.2.1 → refkit-0.3.0}/src/refkit/types.py +6 -0
  15. {refkit-0.2.1 → refkit-0.3.0}/.agent-plugin/plugin.json +0 -0
  16. {refkit-0.2.1 → refkit-0.3.0}/.agent-plugin/skills/refkit/SKILL.md +0 -0
  17. {refkit-0.2.1 → refkit-0.3.0}/.agent-plugin/skills/refkit/agents/openai.yaml +0 -0
  18. {refkit-0.2.1 → refkit-0.3.0}/.agent-plugin/skills/refkit/references/contracts.md +0 -0
  19. {refkit-0.2.1 → refkit-0.3.0}/.agent-plugin/skills/refkit/references/edit.md +0 -0
  20. {refkit-0.2.1 → refkit-0.3.0}/.agent-plugin/skills/refkit/references/render.md +0 -0
  21. {refkit-0.2.1 → refkit-0.3.0}/.agent-plugin/skills/refkit/references/tidy.md +0 -0
  22. {refkit-0.2.1 → refkit-0.3.0}/LICENSE +0 -0
  23. {refkit-0.2.1 → refkit-0.3.0}/README.md +0 -0
  24. {refkit-0.2.1 → refkit-0.3.0}/build_backend.py +0 -0
  25. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/Cargo.toml +0 -0
  26. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/document.rs +0 -0
  27. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/library/diagnostic.rs +0 -0
  28. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/library/parse.rs +0 -0
  29. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/library/recovery.rs +0 -0
  30. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/library/source.rs +0 -0
  31. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/raw/edit.rs +0 -0
  32. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/raw/sanitize.rs +0 -0
  33. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/raw/tests.rs +0 -0
  34. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/render/bibliography.rs +0 -0
  35. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/render/citation.rs +0 -0
  36. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/render/html.rs +0 -0
  37. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/render/mod.rs +0 -0
  38. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/render/text.rs +0 -0
  39. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/render_tree.rs +0 -0
  40. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/source.rs +0 -0
  41. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/strings.rs +0 -0
  42. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/style/validate.rs +0 -0
  43. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/style.rs +0 -0
  44. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/tidy/duplicates.rs +0 -0
  45. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/tidy/keys.rs +0 -0
  46. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/tidy/latex.rs +0 -0
  47. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/tidy/mod.rs +0 -0
  48. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/tidy/options.rs +0 -0
  49. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/tidy/references.rs +0 -0
  50. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/tidy/render/sort.rs +0 -0
  51. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/tidy/render/value.rs +0 -0
  52. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/tidy/render.rs +0 -0
  53. {refkit-0.2.1 → refkit-0.3.0}/crates/refkit-core/src/tidy/unicode.rs +0 -0
  54. {refkit-0.2.1 → refkit-0.3.0}/packages/refkit/rust/Cargo.toml +0 -0
  55. {refkit-0.2.1 → refkit-0.3.0}/packages/refkit/rust/src/citation.rs +0 -0
  56. {refkit-0.2.1 → refkit-0.3.0}/packages/refkit/rust/src/conversion.rs +0 -0
  57. {refkit-0.2.1 → refkit-0.3.0}/packages/refkit/rust/src/document.rs +0 -0
  58. {refkit-0.2.1 → refkit-0.3.0}/packages/refkit/rust/src/entry.rs +0 -0
  59. {refkit-0.2.1 → refkit-0.3.0}/packages/refkit/rust/src/errors.rs +0 -0
  60. {refkit-0.2.1 → refkit-0.3.0}/packages/refkit/rust/src/filesystem.rs +0 -0
  61. {refkit-0.2.1 → refkit-0.3.0}/packages/refkit/rust/src/lib.rs +0 -0
  62. {refkit-0.2.1 → refkit-0.3.0}/packages/refkit/rust/src/library.rs +0 -0
  63. {refkit-0.2.1 → refkit-0.3.0}/packages/refkit/rust/src/module.rs +0 -0
  64. {refkit-0.2.1 → refkit-0.3.0}/packages/refkit/rust/src/rendered.rs +0 -0
  65. {refkit-0.2.1 → refkit-0.3.0}/packages/refkit/rust/src/repr.rs +0 -0
  66. {refkit-0.2.1 → refkit-0.3.0}/packages/refkit/rust/src/style.rs +0 -0
  67. {refkit-0.2.1 → refkit-0.3.0}/packages/refkit/rust/src/tidy.rs +0 -0
  68. {refkit-0.2.1 → refkit-0.3.0}/src/refkit/__init__.py +0 -0
  69. {refkit-0.2.1 → refkit-0.3.0}/src/refkit/__init__.pyi +0 -0
  70. {refkit-0.2.1 → refkit-0.3.0}/src/refkit/agent.py +0 -0
  71. {refkit-0.2.1 → refkit-0.3.0}/src/refkit/py.typed +0 -0
@@ -31,6 +31,25 @@ Return the entry preview, diagnostic preview, and their total counts together. D
31
31
 
32
32
  Use `Library.read(path)` for files, `Library.parse_yaml(source)` for [Hayagriva bibliography YAML](https://github.com/typst/hayagriva), and `Library.select(selector)` for [Hayagriva selectors](https://github.com/typst/hayagriva#selectors). Use `project(..., keys=selected_keys)` to constrain subsequent output.
33
33
 
34
+ Use `BibDocument.resolve()` for every source field, including custom fields,
35
+ with string macros and concatenations expanded. It preserves TeX grouping and
36
+ escapes. Results are detached records with `key`, `entry_type`, and `fields`.
37
+ The call uses current edits and raises `ParseError` for ambiguous or invalid
38
+ source. It resolves `crossref` and `xdata` as field text. Choose `Library` for
39
+ inherited citation metadata.
40
+
41
+ ```python
42
+ import refkit as rk
43
+
44
+ document = rk.BibDocument.parse("""
45
+ @string{host = {https://example.org/}}
46
+ @misc{guide, custom_link = host # {guide}, title = {A {Guide}}}
47
+ """)
48
+ entry = document.resolve()[0]
49
+ assert entry["fields"]["custom_link"] == "https://example.org/guide"
50
+ assert entry["fields"]["title"] == "A {Guide}"
51
+ ```
52
+
34
53
  When parsing cannot retain an entry, handle the typed failure and report its diagnostics:
35
54
 
36
55
  ```python
@@ -455,7 +455,7 @@ dependencies = [
455
455
 
456
456
  [[package]]
457
457
  name = "refkit-core"
458
- version = "0.2.1"
458
+ version = "0.3.0"
459
459
  dependencies = [
460
460
  "biblatex",
461
461
  "hayagriva",
@@ -468,7 +468,7 @@ dependencies = [
468
468
 
469
469
  [[package]]
470
470
  name = "refkit-native"
471
- version = "0.2.1"
471
+ version = "0.3.0"
472
472
  dependencies = [
473
473
  "pyo3",
474
474
  "refkit-core",
@@ -3,7 +3,7 @@ members = ["crates/refkit-core", "packages/refkit/rust"]
3
3
  resolver = "3"
4
4
 
5
5
  [workspace.package]
6
- version = "0.2.1"
6
+ version = "0.3.0"
7
7
  edition = "2024"
8
8
  rust-version = "1.88"
9
9
  license = "Apache-2.0"
@@ -11,7 +11,7 @@ repository = "https://github.com/peter-gy/refkit"
11
11
 
12
12
  [workspace.dependencies]
13
13
  pyo3 = "0.29.0"
14
- refkit-core = { version = "0.2.1", path = "crates/refkit-core" }
14
+ refkit-core = { version = "0.3.0", path = "crates/refkit-core" }
15
15
  serde = { version = "1.0.228", features = ["derive"] }
16
16
  serde_json = "1.0.150"
17
17
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: refkit
3
- Version: 0.2.1
3
+ Version: 0.3.0
4
4
  Classifier: Development Status :: 3 - Alpha
5
5
  Classifier: Intended Audience :: Developers
6
6
  Classifier: License :: OSI Approved :: Apache Software License
@@ -15,6 +15,7 @@ pub use library::{
15
15
  };
16
16
  pub use raw::{
17
17
  RawBlockInfo, RawDocument, RawEditError, RawEntryId, RawEntryInfo, RawFieldId, RawFieldInfo,
18
+ ResolvedBibEntry,
18
19
  };
19
20
  pub use render::{
20
21
  RenderedOutput, is_bundled_locale, render_library_bibliography, render_library_citation,
@@ -11,7 +11,11 @@ const MAX_EXPANDED_BYTES: usize = 16 * 1024 * 1024;
11
11
  const MAX_STEPS: usize = 100_000;
12
12
 
13
13
  pub(crate) fn validate_source(source: &str) -> Result<(), Diagnostic> {
14
- if source.len() > MAX_SOURCE_BYTES {
14
+ validate_source_size(source.len())
15
+ }
16
+
17
+ pub(crate) fn validate_source_size(bytes: usize) -> Result<(), Diagnostic> {
18
+ if bytes > MAX_SOURCE_BYTES {
15
19
  return Err(limit(None, "bibliography source exceeds 16 MiB"));
16
20
  }
17
21
  Ok(())
@@ -61,7 +65,7 @@ pub(crate) fn normalize_reference<'a>(
61
65
  ) -> Result<Vec<String>, Diagnostic> {
62
66
  let lookup = abbreviations
63
67
  .iter()
64
- .map(|pair| (pair.key.v, &pair.value.v))
68
+ .map(|pair| (pair.key.v.to_string(), &pair.value.v))
65
69
  .collect();
66
70
  resolve_field(field, &lookup)?;
67
71
  let span = field.first().map_or(0, |chunk| chunk.span.start)
@@ -96,7 +100,7 @@ pub(crate) fn normalize_reference<'a>(
96
100
 
97
101
  pub(crate) fn resolve_field(
98
102
  field: &Field<'_>,
99
- abbreviations: &HashMap<&str, &Field<'_>>,
103
+ abbreviations: &HashMap<String, &Field<'_>>,
100
104
  ) -> Result<String, Diagnostic> {
101
105
  let mut output = String::new();
102
106
  let mut ancestors = Vec::new();
@@ -107,18 +111,25 @@ pub(crate) fn resolve_field(
107
111
  &mut ancestors,
108
112
  &mut steps,
109
113
  &mut output,
114
+ false,
110
115
  )?;
111
116
  Ok(output)
112
117
  }
113
118
 
114
119
  fn expand<'a>(
115
120
  field: &Field<'a>,
116
- abbreviations: &HashMap<&str, &Field<'a>>,
121
+ abbreviations: &HashMap<String, &Field<'a>>,
117
122
  ancestors: &mut Vec<String>,
118
123
  steps: &mut usize,
119
124
  output: &mut String,
125
+ strict: bool,
120
126
  ) -> Result<(), Diagnostic> {
121
- validate_value(field)?;
127
+ validate_value(field).map_err(|mut diagnostic| {
128
+ if strict {
129
+ diagnostic.span = field.first().map(|chunk| chunk.span.clone());
130
+ }
131
+ diagnostic
132
+ })?;
122
133
  for chunk in field {
123
134
  *steps += 1;
124
135
  if *steps > MAX_STEPS || ancestors.len() >= MAX_DEPTH {
@@ -130,17 +141,33 @@ fn expand<'a>(
130
141
  match chunk.v {
131
142
  RawChunk::Normal(text) => output.push_str(text),
132
143
  RawChunk::Abbreviation(name) => {
133
- if ancestors.iter().any(|ancestor| ancestor == name) {
144
+ let key = if strict {
145
+ name.to_ascii_lowercase()
146
+ } else {
147
+ name.to_string()
148
+ };
149
+ if ancestors.contains(&key) {
134
150
  return Err(Diagnostic::error(
135
151
  "cyclic_abbreviation",
136
152
  Some(chunk.span.clone()),
137
153
  format!("cyclic BibTeX abbreviation {name:?}"),
138
154
  ));
139
155
  }
140
- if let Some(value) = abbreviations.get(name) {
141
- ancestors.push(name.to_string());
142
- expand(value, abbreviations, ancestors, steps, output)?;
156
+ if let Some(value) = abbreviations.get(&key) {
157
+ ancestors.push(key);
158
+ expand(value, abbreviations, ancestors, steps, output, strict)?;
143
159
  ancestors.pop();
160
+ } else if strict {
161
+ match month(&key) {
162
+ Some(month) => output.push_str(month),
163
+ None => {
164
+ return Err(Diagnostic::error(
165
+ "unknown_abbreviation",
166
+ Some(chunk.span.clone()),
167
+ format!("unknown BibTeX abbreviation {name:?}"),
168
+ ));
169
+ }
170
+ }
144
171
  } else {
145
172
  output.push_str(name);
146
173
  }
@@ -156,6 +183,60 @@ fn expand<'a>(
156
183
  Ok(())
157
184
  }
158
185
 
186
+ pub(crate) struct FieldResolver<'a> {
187
+ abbreviations: HashMap<String, &'a Field<'a>>,
188
+ steps: usize,
189
+ bytes: usize,
190
+ }
191
+
192
+ impl<'a> FieldResolver<'a> {
193
+ pub(crate) fn new(abbreviations: HashMap<String, &'a Field<'a>>) -> Self {
194
+ Self {
195
+ abbreviations,
196
+ steps: 0,
197
+ bytes: 0,
198
+ }
199
+ }
200
+
201
+ pub(crate) fn resolve(&mut self, field: &Field<'a>) -> Result<String, Diagnostic> {
202
+ let mut output = String::new();
203
+ expand(
204
+ field,
205
+ &self.abbreviations,
206
+ &mut Vec::new(),
207
+ &mut self.steps,
208
+ &mut output,
209
+ true,
210
+ )?;
211
+ self.bytes = self.bytes.saturating_add(output.len());
212
+ if self.bytes > MAX_EXPANDED_BYTES {
213
+ return Err(limit(
214
+ field.first().map(|chunk| chunk.span.clone()),
215
+ "bibliography expansion exceeds 16 MiB",
216
+ ));
217
+ }
218
+ Ok(output)
219
+ }
220
+ }
221
+
222
+ fn month(name: &str) -> Option<&'static str> {
223
+ Some(match name {
224
+ "jan" => "January",
225
+ "feb" => "February",
226
+ "mar" => "March",
227
+ "apr" => "April",
228
+ "may" => "May",
229
+ "jun" => "June",
230
+ "jul" => "July",
231
+ "aug" => "August",
232
+ "sep" => "September",
233
+ "oct" => "October",
234
+ "nov" => "November",
235
+ "dec" => "December",
236
+ _ => return None,
237
+ })
238
+ }
239
+
159
240
  pub(crate) fn validate_raw(raw: &RawBibliography<'_>) -> Result<(), Diagnostic> {
160
241
  for abbreviation in &raw.abbreviations {
161
242
  validate_value(&abbreviation.value.v)?;
@@ -163,7 +244,7 @@ pub(crate) fn validate_raw(raw: &RawBibliography<'_>) -> Result<(), Diagnostic>
163
244
  let abbreviations: HashMap<_, _> = raw
164
245
  .abbreviations
165
246
  .iter()
166
- .map(|pair| (pair.key.v, &pair.value.v))
247
+ .map(|pair| (pair.key.v.to_string(), &pair.value.v))
167
248
  .collect();
168
249
  let has_references = raw
169
250
  .entries
@@ -183,6 +264,7 @@ pub(crate) fn validate_raw(raw: &RawBibliography<'_>) -> Result<(), Diagnostic>
183
264
  &mut ancestors,
184
265
  &mut expansion_steps,
185
266
  &mut value,
267
+ false,
186
268
  )
187
269
  .map_err(|mut diagnostic| {
188
270
  diagnostic.entry = Some(entry.v.key.v.to_string());
@@ -15,7 +15,9 @@ use crate::quoted;
15
15
  use crate::strings::entry_type_name;
16
16
 
17
17
  pub use self::diagnostic::{Diagnostic, DiagnosticAction, DiagnosticSeverity, ParseFailure};
18
- pub(crate) use self::guard::{normalize_reference, validate_literal, validate_source};
18
+ pub(crate) use self::guard::{
19
+ FieldResolver, normalize_reference, validate_literal, validate_source, validate_source_size,
20
+ };
19
21
  pub use self::parse::parse_bibtex_report;
20
22
  use self::parse::{parse_biblatex_library, parse_hayagriva_yaml};
21
23
 
@@ -775,6 +775,11 @@ fn skip_field_trivia(body: &str, cursor: &mut usize) {
775
775
  }
776
776
 
777
777
  fn parse_assignment(body: &str) -> Option<(String, String)> {
778
+ let (key, value, _) = parse_assignment_atoms(body)?;
779
+ Some((key, value))
780
+ }
781
+
782
+ pub(super) fn parse_assignment_atoms(body: &str) -> Option<(String, String, Vec<RawValueAtom>)> {
778
783
  let equals = body.find('=')?;
779
784
  let key = body[..equals].trim().to_ascii_lowercase();
780
785
  if !is_valid_identifier(&key) {
@@ -782,13 +787,17 @@ fn parse_assignment(body: &str) -> Option<(String, String)> {
782
787
  }
783
788
  let mut cursor = equals + 1;
784
789
  skip_field_space(body, &mut cursor);
785
- let (value, end, _, _, _) = parse_value(body, cursor, 0).ok()?;
790
+ let (value, end, _, _, atoms) = parse_value(body, cursor, 0).ok()?;
786
791
  cursor = end;
787
792
  skip_field_trivia(body, &mut cursor);
793
+ if body[cursor..].starts_with(',') {
794
+ cursor += 1;
795
+ skip_field_trivia(body, &mut cursor);
796
+ }
788
797
  if cursor != body.len() {
789
798
  return None;
790
799
  }
791
- Some((key, value))
800
+ Some((key, value, atoms))
792
801
  }
793
802
 
794
803
  fn parse_preamble_value(body: &str) -> String {
@@ -0,0 +1,384 @@
1
+ use std::collections::{BTreeMap, HashMap, HashSet};
2
+ use std::ops::Range;
3
+
4
+ use biblatex::{Field, RawChunk, Spanned};
5
+
6
+ use super::parse::{parse_assignment_atoms, parse_raw_document};
7
+ use super::{RawBlock, RawDocument, RawValueAtom, RawValueMode, ResolvedBibEntry};
8
+ use crate::library::{
9
+ Diagnostic, FieldResolver, ParseFailure, validate_source, validate_source_size,
10
+ };
11
+
12
+ pub(super) fn resolve_document(
13
+ document: &RawDocument,
14
+ ) -> Result<Vec<ResolvedBibEntry>, ParseFailure> {
15
+ let (changed, edited_bytes) = document
16
+ .data
17
+ .entry_blocks
18
+ .iter()
19
+ .flat_map(|entry| &entry.field_blocks)
20
+ .filter(|field| field.changed)
21
+ .fold((false, 0usize), |(_, bytes), field| {
22
+ (true, bytes.saturating_add(field.value.len()))
23
+ });
24
+ validate_source_size(edited_bytes)?;
25
+ let current;
26
+ let data = if changed {
27
+ let source = document
28
+ .render()
29
+ .map_err(|message| Diagnostic::error("syntax_error", None, message))?;
30
+ validate_source(&source)?;
31
+ current = parse_raw_document(&source);
32
+ &current
33
+ } else {
34
+ validate_source_size(
35
+ document
36
+ .data
37
+ .blocks
38
+ .last()
39
+ .map_or(0, |block| block.span().end),
40
+ )?;
41
+ &document.data
42
+ };
43
+ let mut definitions = Vec::new();
44
+ for block in &data.blocks {
45
+ match block {
46
+ RawBlock::StringDef { raw, span, .. } => {
47
+ let body_start = raw.find(['{', '(']).expect("parsed block has an opener") + 1;
48
+ let (key, _, atoms) = parse_assignment_atoms(&raw[body_start..raw.len() - 1])
49
+ .ok_or_else(|| {
50
+ Diagnostic::error(
51
+ "syntax_error",
52
+ Some(span.clone()),
53
+ "invalid BibTeX string definition".to_string(),
54
+ )
55
+ })?;
56
+ definitions.push((key, atoms, span.clone()));
57
+ }
58
+ RawBlock::Failed { error, span, .. } => {
59
+ return Err(
60
+ Diagnostic::error("syntax_error", Some(span.clone()), error.clone()).into(),
61
+ );
62
+ }
63
+ _ => {}
64
+ }
65
+ }
66
+ let definitions: Vec<_> = definitions
67
+ .iter()
68
+ .map(|(key, atoms, span)| (key, as_field(atoms, span)))
69
+ .collect();
70
+ let abbreviations: HashMap<_, _> = definitions
71
+ .iter()
72
+ .map(|(key, value)| ((*key).clone(), value))
73
+ .collect();
74
+ let mut resolver = FieldResolver::new(abbreviations);
75
+ let mut keys = HashSet::new();
76
+ let mut entries = Vec::with_capacity(data.entry_blocks.len());
77
+ for entry in &data.entry_blocks {
78
+ if entry.key.is_empty() || !keys.insert(&entry.key) {
79
+ let code = if entry.key.is_empty() {
80
+ "syntax_error"
81
+ } else {
82
+ "duplicate_key"
83
+ };
84
+ let mut diagnostic = Diagnostic::error(
85
+ code,
86
+ Some(entry.span.clone()),
87
+ format!(
88
+ "BibTeX entry key {:?} must be nonempty and unique",
89
+ entry.key
90
+ ),
91
+ );
92
+ diagnostic.entry = Some(entry.key.clone());
93
+ return Err(diagnostic.into());
94
+ }
95
+ let mut fields = BTreeMap::new();
96
+ for field in &entry.field_blocks {
97
+ let name = field.name.to_ascii_lowercase();
98
+ let result = if fields.contains_key(&name) {
99
+ Err(Diagnostic::error(
100
+ "duplicate_field",
101
+ Some(field.span.clone()),
102
+ format!("duplicate BibTeX field {name:?}"),
103
+ ))
104
+ } else if field.value_mode == RawValueMode::Missing {
105
+ Err(Diagnostic::error(
106
+ "syntax_error",
107
+ Some(field.span.clone()),
108
+ format!("BibTeX field {name:?} requires an assignment"),
109
+ ))
110
+ } else {
111
+ resolver.resolve(&as_field(&field.value_atoms, &field.span))
112
+ };
113
+ let value = result.map_err(|mut diagnostic| {
114
+ diagnostic.entry = Some(entry.key.clone());
115
+ diagnostic.field = Some(name.clone());
116
+ diagnostic
117
+ })?;
118
+ fields.insert(name, value);
119
+ }
120
+ entries.push(ResolvedBibEntry {
121
+ key: entry.key.clone(),
122
+ entry_type: entry.kind.to_ascii_lowercase(),
123
+ fields,
124
+ });
125
+ }
126
+ Ok(entries)
127
+ }
128
+
129
+ fn as_field<'a>(atoms: &'a [RawValueAtom], span: &Range<usize>) -> Field<'a> {
130
+ atoms
131
+ .iter()
132
+ .map(|atom| {
133
+ let value = if atom.value_mode == RawValueMode::Bare
134
+ && !atom.value.bytes().all(|byte| byte.is_ascii_digit())
135
+ {
136
+ RawChunk::Abbreviation(&atom.value)
137
+ } else {
138
+ RawChunk::Normal(&atom.value)
139
+ };
140
+ Spanned::new(value, span.clone())
141
+ })
142
+ .collect()
143
+ }
144
+
145
+ #[cfg(test)]
146
+ mod tests {
147
+ use super::*;
148
+
149
+ #[test]
150
+ fn resolves_source_fields_with_macros_and_preserves_tex() {
151
+ let source = r#"@string{publisher = "Example Press"}
152
+ @string{URL = "https://example.test/"}
153
+ @CustomType(Work, TITLE = {A {Protected} \LaTeX{} title},
154
+ HOWPUBLISHED = url # {paper}, publisher = PUBLISHER, year = 2026,
155
+ custom = { keep spaces }, month = SEP, crossref = {Parent})
156
+ @book{Parent, title = {Parent Title}}"#;
157
+ let document = RawDocument::parse(source);
158
+ let entries = document.resolve().unwrap();
159
+ assert_eq!(
160
+ entries
161
+ .iter()
162
+ .map(|entry| entry.key.as_str())
163
+ .collect::<Vec<_>>(),
164
+ ["Work", "Parent"]
165
+ );
166
+ assert_eq!(entries[0].entry_type, "customtype");
167
+ assert_eq!(
168
+ entries[0].fields,
169
+ BTreeMap::from([
170
+ (
171
+ "title".to_string(),
172
+ r"A {Protected} \LaTeX{} title".to_string()
173
+ ),
174
+ (
175
+ "howpublished".to_string(),
176
+ "https://example.test/paper".to_string()
177
+ ),
178
+ ("publisher".to_string(), "Example Press".to_string()),
179
+ ("year".to_string(), "2026".to_string()),
180
+ ("custom".to_string(), " keep spaces ".to_string()),
181
+ ("month".to_string(), "September".to_string()),
182
+ ("crossref".to_string(), "Parent".to_string()),
183
+ ])
184
+ );
185
+ assert_eq!(document.render().unwrap(), source);
186
+ }
187
+
188
+ #[test]
189
+ fn resolution_observes_edits_and_matches_rendered_readback() {
190
+ let mut document = RawDocument::parse(
191
+ r#"@string{prefix="old"}@misc{work,title=prefix # { title},year=2025}"#,
192
+ );
193
+ let entry = document.unique_entry("work").unwrap().unwrap();
194
+ let field = document.unique_field(entry, "title").unwrap().unwrap();
195
+ document
196
+ .set_field_value(entry, field, r#"new # "literal""#.to_string())
197
+ .unwrap();
198
+ let before = document.render().unwrap();
199
+ let entries = document.resolve().unwrap();
200
+ assert_eq!(entries[0].fields["title"], r#"new # "literal""#);
201
+ assert_eq!(entries, RawDocument::parse(&before).resolve().unwrap());
202
+ assert_eq!(document.render().unwrap(), before);
203
+ }
204
+
205
+ #[test]
206
+ fn reference_fields_resolve_values_and_preserve_authored_fields() {
207
+ let entries = RawDocument::parse(
208
+ r#"@string{parent="P"}@book{P,title={Parent}}
209
+ @misc{child,crossref=parent,xdata=parent # {, X},note={own}}"#,
210
+ )
211
+ .resolve()
212
+ .unwrap();
213
+ assert_eq!(
214
+ entries[1].fields,
215
+ BTreeMap::from([
216
+ ("crossref".to_string(), "P".to_string()),
217
+ ("xdata".to_string(), "P, X".to_string()),
218
+ ("note".to_string(), "own".to_string()),
219
+ ])
220
+ );
221
+ }
222
+
223
+ #[test]
224
+ fn macro_lookup_uses_final_definitions_and_preserves_literal_escapes() {
225
+ let entries = RawDocument::parse(
226
+ r#"@misc{Work,title=pub,month=JAN,note={A \} B}}
227
+ @misc{work,title={lowercase key}}
228
+ @string{PUB={old}}@string{pub=next,}@string{next={new}}@string{jan={Custom Month}}"#,
229
+ )
230
+ .resolve()
231
+ .unwrap();
232
+ assert_eq!(entries[0].fields["title"], "new");
233
+ assert_eq!(entries[0].fields["month"], "Custom Month");
234
+ assert_eq!(entries[0].fields["note"], r"A \} B");
235
+ assert_eq!(entries[1].key, "work");
236
+ }
237
+
238
+ #[test]
239
+ fn bounds_source_and_edited_input_size() {
240
+ let source = " ".repeat(16 * 1024 * 1024 + 1);
241
+ assert_eq!(
242
+ RawDocument::parse(&source)
243
+ .resolve()
244
+ .unwrap_err()
245
+ .diagnostics[0]
246
+ .code,
247
+ "resource_limit"
248
+ );
249
+ let mut document = RawDocument::parse("@misc{work,title={old}}");
250
+ let entry = document.unique_entry("work").unwrap().unwrap();
251
+ let field = document.unique_field(entry, "title").unwrap().unwrap();
252
+ document.set_field_value(entry, field, source).unwrap();
253
+ assert_eq!(
254
+ document.resolve().unwrap_err().diagnostics[0].code,
255
+ "resource_limit"
256
+ );
257
+ }
258
+
259
+ #[test]
260
+ fn resolves_edits_that_bring_source_within_the_size_limit() {
261
+ let source = format!("@misc{{work,title={{{}}}}}", "x".repeat(16 * 1024 * 1024));
262
+ let mut document = RawDocument::parse(&source);
263
+ let entry = document.unique_entry("work").unwrap().unwrap();
264
+ let field = document.unique_field(entry, "title").unwrap().unwrap();
265
+ document
266
+ .set_field_value(entry, field, "small".to_string())
267
+ .unwrap();
268
+
269
+ let entries = document.resolve().unwrap();
270
+ assert_eq!(entries[0].fields["title"], "small");
271
+ assert_eq!(
272
+ entries,
273
+ RawDocument::parse(&document.render().unwrap())
274
+ .resolve()
275
+ .unwrap()
276
+ );
277
+ }
278
+
279
+ #[test]
280
+ fn nesting_failures_have_valid_source_spans_for_unicode_expressions() {
281
+ let source = format!(
282
+ "@string{{accent=\"é\"}}@misc{{work,title=accent # {{{}é{}}}}}",
283
+ "{".repeat(64),
284
+ "}".repeat(64),
285
+ );
286
+ let failure = RawDocument::parse(&source).resolve().unwrap_err();
287
+ let diagnostic = &failure.diagnostics[0];
288
+ assert_eq!(diagnostic.code, "resource_limit");
289
+ assert_eq!(diagnostic.entry.as_deref(), Some("work"));
290
+ assert_eq!(diagnostic.field.as_deref(), Some("title"));
291
+ assert!(
292
+ diagnostic
293
+ .span
294
+ .as_ref()
295
+ .is_some_and(|span| source.get(span.clone()).is_some())
296
+ );
297
+ }
298
+
299
+ #[test]
300
+ fn rejects_ambiguous_fields_and_invalid_dependencies() {
301
+ for (source, code) in [
302
+ ("@misc{work,title=missing}", "unknown_abbreviation"),
303
+ (
304
+ "@string{a=B}@string{b=A}@misc{work,title=a}",
305
+ "cyclic_abbreviation",
306
+ ),
307
+ ("@misc{work,title={a},TITLE={b}}", "duplicate_field"),
308
+ ] {
309
+ let failure = RawDocument::parse(source).resolve().unwrap_err();
310
+ let diagnostic = &failure.diagnostics[0];
311
+ assert_eq!(diagnostic.code, code);
312
+ assert_eq!(diagnostic.entry.as_deref(), Some("work"));
313
+ assert_eq!(diagnostic.field.as_deref(), Some("title"));
314
+ assert!(
315
+ diagnostic
316
+ .span
317
+ .as_ref()
318
+ .is_some_and(|span| source.get(span.clone()).is_some())
319
+ );
320
+ }
321
+ for (source, code) in [
322
+ ("@misc{work}@misc{work}", "duplicate_key"),
323
+ ("@misc{work,title={unfinished", "syntax_error"),
324
+ ("@string{a=}", "syntax_error"),
325
+ ] {
326
+ assert_eq!(
327
+ RawDocument::parse(source)
328
+ .resolve()
329
+ .unwrap_err()
330
+ .diagnostics[0]
331
+ .code,
332
+ code
333
+ );
334
+ }
335
+ }
336
+
337
+ #[test]
338
+ fn bounds_aggregate_expansion_and_dependency_depth() {
339
+ let mut source = "@string{x0={}}".to_string();
340
+ for index in 1..20 {
341
+ source.push_str(&format!(
342
+ "@string{{x{index}=x{} # x{}}}",
343
+ index - 1,
344
+ index - 1
345
+ ));
346
+ }
347
+ source.push_str("@misc{work,title=x19}");
348
+ assert_eq!(
349
+ RawDocument::parse(&source)
350
+ .resolve()
351
+ .unwrap_err()
352
+ .diagnostics[0]
353
+ .code,
354
+ "resource_limit"
355
+ );
356
+
357
+ let mut source = "@string{x0={ok}}".to_string();
358
+ for index in 1..65 {
359
+ source.push_str(&format!("@string{{x{index}=x{}}}", index - 1));
360
+ }
361
+ source.push_str("@misc{work,title=x64}");
362
+ assert_eq!(
363
+ RawDocument::parse(&source)
364
+ .resolve()
365
+ .unwrap_err()
366
+ .diagnostics[0]
367
+ .code,
368
+ "resource_limit"
369
+ );
370
+
371
+ let source = format!(
372
+ "@string{{a={{{}}}}}@misc{{work,a=a,b=a,c=a}}",
373
+ "x".repeat(6 * 1024 * 1024)
374
+ );
375
+ assert_eq!(
376
+ RawDocument::parse(&source)
377
+ .resolve()
378
+ .unwrap_err()
379
+ .diagnostics[0]
380
+ .code,
381
+ "resource_limit"
382
+ );
383
+ }
384
+ }
@@ -1,3 +1,4 @@
1
+ use std::collections::BTreeMap;
1
2
  use std::fmt;
2
3
  use std::ops::Range;
3
4
 
@@ -5,6 +6,7 @@ use indexmap::IndexMap;
5
6
 
6
7
  mod edit;
7
8
  mod parse;
9
+ mod resolve;
8
10
  mod sanitize;
9
11
  #[cfg(test)]
10
12
  mod tests;
@@ -113,6 +115,13 @@ pub struct RawDocument {
113
115
  data: RawDocumentData,
114
116
  }
115
117
 
118
+ #[derive(Debug, Clone, PartialEq, Eq)]
119
+ pub struct ResolvedBibEntry {
120
+ pub key: String,
121
+ pub entry_type: String,
122
+ pub fields: BTreeMap<String, String>,
123
+ }
124
+
116
125
  #[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
117
126
  pub struct RawEntryId(usize);
118
127
 
@@ -259,6 +268,10 @@ impl RawDocument {
259
268
  }
260
269
  }
261
270
 
271
+ pub fn resolve(&self) -> Result<Vec<ResolvedBibEntry>, crate::ParseFailure> {
272
+ resolve::resolve_document(self)
273
+ }
274
+
262
275
  pub fn entry_count(&self) -> usize {
263
276
  self.data.entry_blocks.len()
264
277
  }
@@ -8,7 +8,7 @@ use pyo3::prelude::*;
8
8
  use pyo3::types::{PyDict, PyList, PyModule};
9
9
 
10
10
  use crate::conversion::diagnostics_to_py;
11
- use crate::errors::RefkitError;
11
+ use crate::errors::{RefkitError, library_error_to_py};
12
12
  use crate::filesystem::{read_bibtex, write_bibtex};
13
13
  use crate::repr::quoted;
14
14
  use crate::tidy::{TidyOptions, TidyResult, tidy_error_to_py};
@@ -103,6 +103,22 @@ impl BibDocument {
103
103
  py.detach(move || render_document(&data))
104
104
  }
105
105
 
106
+ fn resolve(&self, py: Python<'_>) -> PyResult<Py<PyAny>> {
107
+ let data = self.doc.borrow().clone();
108
+ let entries = py
109
+ .detach(move || data.resolve())
110
+ .map_err(|error| library_error_to_py(py, refkit_core::LibraryError::Biblatex(error)))?;
111
+ let values = PyList::empty(py);
112
+ for entry in entries {
113
+ let value = PyDict::new(py);
114
+ value.set_item("key", entry.key)?;
115
+ value.set_item("entry_type", entry.entry_type)?;
116
+ value.set_item("fields", entry.fields)?;
117
+ values.append(value)?;
118
+ }
119
+ Ok(values.into_any().unbind())
120
+ }
121
+
106
122
  #[pyo3(signature = (*, options = None))]
107
123
  fn tidy(
108
124
  &self,
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "refkit"
3
- version = "0.2.1"
3
+ version = "0.3.0"
4
4
  description = "Fast Python citation parsing, rendering, and BibTeX editing backed by Rust"
5
5
  readme = "README.md"
6
6
  license = "Apache-2.0"
@@ -9,6 +9,7 @@ from .types import (
9
9
  RawBlock,
10
10
  RawFailedBlock,
11
11
  RenderedTree,
12
+ ResolvedBibEntry,
12
13
  TidyRename,
13
14
  )
14
15
 
@@ -268,5 +269,6 @@ class BibDocument:
268
269
  @property
269
270
  def blocks(self) -> list[RawBlock]: ...
270
271
  def to_bibtex(self) -> str: ...
272
+ def resolve(self) -> list[ResolvedBibEntry]: ...
271
273
  def tidy(self, *, options: TidyOptions | None = None) -> TidyResult: ...
272
274
  def write(self, path: str | PathLike[str]) -> None: ...
@@ -62,6 +62,12 @@ class ProjectionRow(TypedDict, total=False):
62
62
  volume: str | None
63
63
 
64
64
 
65
+ class ResolvedBibEntry(TypedDict):
66
+ key: str
67
+ entry_type: str
68
+ fields: dict[str, str]
69
+
70
+
65
71
  class RenderedFormatting(TypedDict):
66
72
  font_style: Literal["Normal", "Italic"]
67
73
  font_variant: Literal["Normal", "SmallCaps"]
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes