opencomb 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
opencomb/combiner.py ADDED
@@ -0,0 +1,232 @@
1
+ """Core combining logic for code and configuration files."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import ast
6
+ import json
7
+ from pathlib import Path
8
+ from typing import Any
9
+
10
+ import yaml
11
+
12
+ try:
13
+ import tomllib # Python 3.11+
14
+ except ImportError:
15
+ import tomli as tomllib # type: ignore
16
+
17
+ import tomli_w
18
+
19
+
20
+ class CodeCombiner:
21
+ """Combine multiple Python source files / snippets into one coherent module."""
22
+
23
+ def __init__(self, separator: str = "\n\n# --- OpenComb separator ---\n\n"):
24
+ self.separator = separator
25
+
26
+ def combine_files(
27
+ self,
28
+ files: list[str | Path],
29
+ *,
30
+ add_headers: bool = True,
31
+ deduplicate_imports: bool = True,
32
+ ) -> str:
33
+ """
34
+ Combine multiple Python files into a single source string.
35
+
36
+ Args:
37
+ files: List of file paths to combine.
38
+ add_headers: Whether to add a comment header for each file.
39
+ deduplicate_imports: Try to keep only unique import statements at the top.
40
+
41
+ Returns:
42
+ Combined Python source code as a string.
43
+ """
44
+ contents: list[str] = []
45
+ all_imports: list[str] = []
46
+ seen_imports: set[str] = set()
47
+
48
+ for file_path in files:
49
+ path = Path(file_path)
50
+ if not path.exists():
51
+ raise FileNotFoundError(f"File not found: {path}")
52
+
53
+ source = path.read_text(encoding="utf-8")
54
+
55
+ if deduplicate_imports:
56
+ imports, body = self._split_imports(source)
57
+ for imp in imports:
58
+ normalized = imp.strip()
59
+ if normalized and normalized not in seen_imports:
60
+ seen_imports.add(normalized)
61
+ all_imports.append(normalized)
62
+ source_body = body
63
+ else:
64
+ source_body = source
65
+
66
+ if add_headers:
67
+ header = f"# === Source: {path.name} ===\n"
68
+ contents.append(header + source_body.strip())
69
+ else:
70
+ contents.append(source_body.strip())
71
+
72
+ result_parts: list[str] = []
73
+ if all_imports:
74
+ result_parts.append("\n".join(all_imports))
75
+ result_parts.append("")
76
+
77
+ result_parts.append(self.separator.join(contents))
78
+ return "\n".join(result_parts).strip() + "\n"
79
+
80
+ def combine_snippets(
81
+ self,
82
+ snippets: list[str],
83
+ *,
84
+ names: list[str] | None = None,
85
+ ) -> str:
86
+ """Combine in-memory code snippets."""
87
+ if names is None:
88
+ names = [f"snippet_{i}" for i in range(len(snippets))]
89
+
90
+ if len(names) != len(snippets):
91
+ raise ValueError("names and snippets must have the same length")
92
+
93
+ parts = []
94
+ for name, snippet in zip(names, snippets):
95
+ parts.append(f"# === {name} ===\n{snippet.strip()}")
96
+
97
+ return self.separator.join(parts) + "\n"
98
+
99
+ @staticmethod
100
+ def _split_imports(source: str) -> tuple[list[str], str]:
101
+ """Naive but useful import splitter using AST when possible."""
102
+ try:
103
+ tree = ast.parse(source)
104
+ except SyntaxError:
105
+ # Fallback: treat everything as body
106
+ return [], source
107
+
108
+ imports: list[str] = []
109
+ lines = source.splitlines(keepends=True)
110
+
111
+ import_end = 0
112
+ for node in tree.body:
113
+ if isinstance(node, (ast.Import, ast.ImportFrom)):
114
+ # Collect the original source lines for this import
115
+ start = node.lineno - 1
116
+ end = getattr(node, "end_lineno", node.lineno)
117
+ import_text = "".join(lines[start:end]).rstrip()
118
+ imports.append(import_text)
119
+ import_end = max(import_end, end)
120
+ else:
121
+ break
122
+
123
+ body = "".join(lines[import_end:]).lstrip("\n")
124
+ return imports, body
125
+
126
+
127
+ class ConfigMerger:
128
+ """Intelligently merge configuration files (YAML, JSON, TOML)."""
129
+
130
+ SUPPORTED = {".yaml", ".yml", ".json", ".toml"}
131
+
132
+ def merge_files(
133
+ self,
134
+ files: list[str | Path],
135
+ *,
136
+ strategy: str = "deep",
137
+ ) -> dict[str, Any]:
138
+ """
139
+ Merge multiple config files.
140
+
141
+ Args:
142
+ files: Config files in order (later files override earlier ones).
143
+ strategy: "deep" (recursive merge) or "shallow" (top-level only).
144
+
145
+ Returns:
146
+ Merged dictionary.
147
+ """
148
+ result: dict[str, Any] = {}
149
+
150
+ for file_path in files:
151
+ path = Path(file_path)
152
+ if not path.exists():
153
+ raise FileNotFoundError(f"Config file not found: {path}")
154
+
155
+ data = self._load(path)
156
+ if not isinstance(data, dict):
157
+ raise ValueError(f"Config must be a mapping/object: {path}")
158
+
159
+ if strategy == "deep":
160
+ result = self._deep_merge(result, data)
161
+ else:
162
+ result.update(data)
163
+
164
+ return result
165
+
166
+ def merge_dicts(
167
+ self,
168
+ *dicts: dict[str, Any],
169
+ strategy: str = "deep",
170
+ ) -> dict[str, Any]:
171
+ """Merge multiple dictionaries."""
172
+ result: dict[str, Any] = {}
173
+ for d in dicts:
174
+ if strategy == "deep":
175
+ result = self._deep_merge(result, d)
176
+ else:
177
+ result.update(d)
178
+ return result
179
+
180
+ def save(
181
+ self,
182
+ data: dict[str, Any],
183
+ path: str | Path,
184
+ *,
185
+ format: str | None = None,
186
+ ) -> None:
187
+ """Save merged config to a file."""
188
+ path = Path(path)
189
+ fmt = format or path.suffix.lower().lstrip(".")
190
+
191
+ if fmt in ("yaml", "yml"):
192
+ path.write_text(
193
+ yaml.dump(data, default_flow_style=False, allow_unicode=True, sort_keys=False),
194
+ encoding="utf-8",
195
+ )
196
+ elif fmt == "json":
197
+ path.write_text(
198
+ json.dumps(data, indent=2, ensure_ascii=False) + "\n",
199
+ encoding="utf-8",
200
+ )
201
+ elif fmt == "toml":
202
+ path.write_bytes(tomli_w.dumps(data).encode("utf-8"))
203
+ else:
204
+ raise ValueError(f"Unsupported format: {fmt}. Use yaml, json or toml.")
205
+
206
+ def _load(self, path: Path) -> Any:
207
+ suffix = path.suffix.lower()
208
+ content = path.read_text(encoding="utf-8")
209
+
210
+ if suffix in (".yaml", ".yml"):
211
+ return yaml.safe_load(content) or {}
212
+ if suffix == ".json":
213
+ return json.loads(content)
214
+ if suffix == ".toml":
215
+ return tomllib.loads(content)
216
+
217
+ raise ValueError(f"Unsupported config format: {suffix}")
218
+
219
+ @staticmethod
220
+ def _deep_merge(base: dict[str, Any], override: dict[str, Any]) -> dict[str, Any]:
221
+ """Recursively merge two dictionaries. Lists are replaced, not merged."""
222
+ result = base.copy()
223
+ for key, value in override.items():
224
+ if (
225
+ key in result
226
+ and isinstance(result[key], dict)
227
+ and isinstance(value, dict)
228
+ ):
229
+ result[key] = ConfigMerger._deep_merge(result[key], value)
230
+ else:
231
+ result[key] = value
232
+ return result
opencomb/prompt.py ADDED
@@ -0,0 +1,149 @@
1
+ """Prompt combining and management for LLM workflows."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pathlib import Path
6
+ from typing import Any
7
+
8
+
9
+ class PromptCombiner:
10
+ """
11
+ Combine system prompts, user prompts, few-shot examples and context
12
+ into a single well-structured prompt.
13
+ """
14
+
15
+ def __init__(
16
+ self,
17
+ system_separator: str = "\n\n",
18
+ example_header: str = "### Example {n}\n",
19
+ section_header: str = "## {title}\n\n",
20
+ ):
21
+ self.system_separator = system_separator
22
+ self.example_header = example_header
23
+ self.section_header = section_header
24
+
25
+ def combine(
26
+ self,
27
+ *,
28
+ system: str | list[str] | None = None,
29
+ instruction: str | None = None,
30
+ context: str | list[str] | None = None,
31
+ examples: list[dict[str, str]] | None = None,
32
+ user: str | None = None,
33
+ extra_sections: dict[str, str] | None = None,
34
+ ) -> str:
35
+ """
36
+ Build a complete prompt from components.
37
+
38
+ Args:
39
+ system: One or more system messages.
40
+ instruction: Main task instruction.
41
+ context: Background information / retrieved documents.
42
+ examples: List of {"input": ..., "output": ...} few-shot examples.
43
+ user: The actual user query.
44
+ extra_sections: Arbitrary named sections.
45
+
46
+ Returns:
47
+ Fully combined prompt string.
48
+ """
49
+ parts: list[str] = []
50
+
51
+ # System
52
+ if system:
53
+ if isinstance(system, str):
54
+ systems = [system]
55
+ else:
56
+ systems = system
57
+ system_text = self.system_separator.join(s.strip() for s in systems if s.strip())
58
+ if system_text:
59
+ parts.append(self.section_header.format(title="System") + system_text)
60
+
61
+ # Instruction
62
+ if instruction and instruction.strip():
63
+ parts.append(self.section_header.format(title="Instruction") + instruction.strip())
64
+
65
+ # Context
66
+ if context:
67
+ if isinstance(context, str):
68
+ contexts = [context]
69
+ else:
70
+ contexts = context
71
+ ctx_text = "\n\n".join(c.strip() for c in contexts if c.strip())
72
+ if ctx_text:
73
+ parts.append(self.section_header.format(title="Context") + ctx_text)
74
+
75
+ # Few-shot examples
76
+ if examples:
77
+ example_blocks = []
78
+ for i, ex in enumerate(examples, 1):
79
+ header = self.example_header.format(n=i)
80
+ block = header
81
+ if "input" in ex:
82
+ block += f"**Input:**\n{ex['input'].strip()}\n\n"
83
+ if "output" in ex:
84
+ block += f"**Output:**\n{ex['output'].strip()}\n"
85
+ example_blocks.append(block.strip())
86
+ if example_blocks:
87
+ parts.append(
88
+ self.section_header.format(title="Examples")
89
+ + "\n\n".join(example_blocks)
90
+ )
91
+
92
+ # Extra sections
93
+ if extra_sections:
94
+ for title, content in extra_sections.items():
95
+ if content and content.strip():
96
+ parts.append(
97
+ self.section_header.format(title=title) + content.strip()
98
+ )
99
+
100
+ # User query
101
+ if user and user.strip():
102
+ parts.append(self.section_header.format(title="User") + user.strip())
103
+
104
+ return "\n\n".join(parts).strip() + "\n"
105
+
106
+ def from_files(
107
+ self,
108
+ *,
109
+ system_files: list[str | Path] | None = None,
110
+ instruction_file: str | Path | None = None,
111
+ context_files: list[str | Path] | None = None,
112
+ examples_file: str | Path | None = None,
113
+ user_file: str | Path | None = None,
114
+ ) -> str:
115
+ """Load components from files and combine them."""
116
+ system = None
117
+ if system_files:
118
+ system = [Path(f).read_text(encoding="utf-8") for f in system_files]
119
+
120
+ instruction = None
121
+ if instruction_file:
122
+ instruction = Path(instruction_file).read_text(encoding="utf-8")
123
+
124
+ context = None
125
+ if context_files:
126
+ context = [Path(f).read_text(encoding="utf-8") for f in context_files]
127
+
128
+ examples = None
129
+ if examples_file:
130
+ import yaml
131
+ data = yaml.safe_load(Path(examples_file).read_text(encoding="utf-8"))
132
+ if isinstance(data, list):
133
+ examples = data
134
+
135
+ user = None
136
+ if user_file:
137
+ user = Path(user_file).read_text(encoding="utf-8")
138
+
139
+ return self.combine(
140
+ system=system,
141
+ instruction=instruction,
142
+ context=context,
143
+ examples=examples,
144
+ user=user,
145
+ )
146
+
147
+ def save(self, prompt: str, path: str | Path) -> None:
148
+ """Save the combined prompt to a file."""
149
+ Path(path).write_text(prompt, encoding="utf-8")
opencomb/recipe.py ADDED
@@ -0,0 +1,165 @@
1
+ """Recipe system – declarative combining of code, configs, prompts and templates."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pathlib import Path
6
+ from typing import Any
7
+
8
+ import yaml
9
+
10
+ from opencomb.combiner import CodeCombiner, ConfigMerger
11
+ from opencomb.combinatorial import CombinatorialGenerator
12
+ from opencomb.prompt import PromptCombiner
13
+ from opencomb.template import TemplateRenderer
14
+
15
+
16
+ class RecipeRunner:
17
+ """
18
+ Execute OpenComb recipes defined in YAML.
19
+
20
+ A recipe can combine code, merge configs, render templates,
21
+ build prompts and generate combinatorial sets in one go.
22
+ """
23
+
24
+ def __init__(self, base_dir: str | Path | None = None):
25
+ self.base_dir = Path(base_dir) if base_dir else Path.cwd()
26
+
27
+ def load(self, recipe_path: str | Path) -> dict[str, Any]:
28
+ """Load a recipe YAML file."""
29
+ path = Path(recipe_path)
30
+ if not path.is_absolute():
31
+ path = self.base_dir / path
32
+ data = yaml.safe_load(path.read_text(encoding="utf-8"))
33
+ if not isinstance(data, dict):
34
+ raise ValueError("Recipe must be a YAML mapping")
35
+ return data
36
+
37
+ def run(self, recipe: dict[str, Any] | str | Path, *, dry_run: bool = False) -> dict[str, Any]:
38
+ """
39
+ Execute a recipe.
40
+
41
+ Returns a dict with results of each step.
42
+ """
43
+ if not isinstance(recipe, dict):
44
+ recipe = self.load(recipe)
45
+
46
+ results: dict[str, Any] = {}
47
+ name = recipe.get("name", "unnamed")
48
+ results["_recipe"] = name
49
+
50
+ # 1. Code combining
51
+ if "combine" in recipe:
52
+ step = recipe["combine"]
53
+ files = [self._resolve(f) for f in step.get("files", [])]
54
+ output = step.get("output")
55
+ combiner = CodeCombiner()
56
+ combined = combiner.combine_files(
57
+ files,
58
+ add_headers=step.get("headers", True),
59
+ deduplicate_imports=step.get("dedupe_imports", True),
60
+ )
61
+ results["combine"] = combined
62
+ if output and not dry_run:
63
+ out_path = self._resolve(output)
64
+ out_path.parent.mkdir(parents=True, exist_ok=True)
65
+ out_path.write_text(combined, encoding="utf-8")
66
+ results["combine_output"] = str(out_path)
67
+
68
+ # 2. Config merging
69
+ if "merge" in recipe:
70
+ step = recipe["merge"]
71
+ files = [self._resolve(f) for f in step.get("files", [])]
72
+ output = step.get("output")
73
+ strategy = step.get("strategy", "deep")
74
+ merger = ConfigMerger()
75
+ merged = merger.merge_files(files, strategy=strategy)
76
+ results["merge"] = merged
77
+ if output and not dry_run:
78
+ out_path = self._resolve(output)
79
+ out_path.parent.mkdir(parents=True, exist_ok=True)
80
+ fmt = step.get("format")
81
+ merger.save(merged, out_path, format=fmt)
82
+ results["merge_output"] = str(out_path)
83
+
84
+ # 3. Template rendering
85
+ if "template" in recipe:
86
+ step = recipe["template"]
87
+ template_dirs = step.get("dirs", ["."])
88
+ renderer = TemplateRenderer(
89
+ [self._resolve(d) for d in template_dirs],
90
+ strict=step.get("strict", True),
91
+ )
92
+ data = step.get("data", {})
93
+ if "file" in step:
94
+ rendered = renderer.render_file(step["file"], **data)
95
+ elif "files" in step:
96
+ rendered = renderer.render_files(
97
+ step["files"],
98
+ separator=step.get("separator", "\n\n"),
99
+ **data,
100
+ )
101
+ elif "string" in step:
102
+ rendered = renderer.render_string(step["string"], **data)
103
+ else:
104
+ raise ValueError("template step needs 'file', 'files' or 'string'")
105
+ results["template"] = rendered
106
+ if "output" in step and not dry_run:
107
+ out_path = self._resolve(step["output"])
108
+ out_path.parent.mkdir(parents=True, exist_ok=True)
109
+ out_path.write_text(rendered, encoding="utf-8")
110
+ results["template_output"] = str(out_path)
111
+
112
+ # 4. Prompt building
113
+ if "prompt" in recipe:
114
+ step = recipe["prompt"]
115
+ combiner = PromptCombiner()
116
+ prompt = combiner.combine(
117
+ system=step.get("system"),
118
+ instruction=step.get("instruction"),
119
+ context=step.get("context"),
120
+ examples=step.get("examples"),
121
+ user=step.get("user"),
122
+ extra_sections=step.get("extra_sections"),
123
+ )
124
+ results["prompt"] = prompt
125
+ if "output" in step and not dry_run:
126
+ out_path = self._resolve(step["output"])
127
+ out_path.parent.mkdir(parents=True, exist_ok=True)
128
+ out_path.write_text(prompt, encoding="utf-8")
129
+ results["prompt_output"] = str(out_path)
130
+
131
+ # 5. Combinatorial generation
132
+ if "generate" in recipe:
133
+ step = recipe["generate"]
134
+ params = step.get("params", {})
135
+ method = step.get("method", "cartesian")
136
+ limit = step.get("limit")
137
+ seed = step.get("seed")
138
+ gen = CombinatorialGenerator(seed=seed)
139
+
140
+ if method == "cartesian":
141
+ combos = gen.cartesian(params, limit=limit)
142
+ elif method == "pairwise":
143
+ combos = gen.pairwise(params, limit=limit)
144
+ elif method == "sample":
145
+ combos = gen.sample(params, n=limit or 10)
146
+ else:
147
+ raise ValueError(f"Unknown generate method: {method}")
148
+
149
+ results["generate"] = combos
150
+ if "output" in step and not dry_run:
151
+ import json
152
+ out_path = self._resolve(step["output"])
153
+ out_path.parent.mkdir(parents=True, exist_ok=True)
154
+ with out_path.open("w", encoding="utf-8") as f:
155
+ for c in combos:
156
+ f.write(json.dumps(c, ensure_ascii=False) + "\n")
157
+ results["generate_output"] = str(out_path)
158
+
159
+ return results
160
+
161
+ def _resolve(self, path: str | Path) -> Path:
162
+ p = Path(path)
163
+ if p.is_absolute():
164
+ return p
165
+ return self.base_dir / p
opencomb/template.py ADDED
@@ -0,0 +1,68 @@
1
+ """Jinja2-powered template rendering and combining."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pathlib import Path
6
+ from typing import Any
7
+
8
+ from jinja2 import Environment, FileSystemLoader, StrictUndefined, Template, select_autoescape
9
+
10
+
11
+ class TemplateRenderer:
12
+ """Render and combine Jinja2 templates with data."""
13
+
14
+ def __init__(
15
+ self,
16
+ template_dirs: list[str | Path] | None = None,
17
+ *,
18
+ strict: bool = True,
19
+ ):
20
+ search_paths = [str(p) for p in (template_dirs or ["."])]
21
+ self.env = Environment(
22
+ loader=FileSystemLoader(search_paths),
23
+ autoescape=select_autoescape(enabled_extensions=("html", "xml")),
24
+ undefined=StrictUndefined if strict else None,
25
+ trim_blocks=True,
26
+ lstrip_blocks=True,
27
+ keep_trailing_newline=True,
28
+ )
29
+
30
+ def render_string(self, template_str: str, **context: Any) -> str:
31
+ """Render a template from a string."""
32
+ tmpl = self.env.from_string(template_str)
33
+ return tmpl.render(**context)
34
+
35
+ def render_file(self, template_name: str, **context: Any) -> str:
36
+ """Render a template file (relative to template_dirs)."""
37
+ tmpl = self.env.get_template(template_name)
38
+ return tmpl.render(**context)
39
+
40
+ def render_files(
41
+ self,
42
+ template_names: list[str],
43
+ *,
44
+ separator: str = "\n\n",
45
+ **context: Any,
46
+ ) -> str:
47
+ """Render multiple templates and join them."""
48
+ parts = [self.render_file(name, **context) for name in template_names]
49
+ return separator.join(parts)
50
+
51
+ def combine_and_render(
52
+ self,
53
+ templates: list[str | Path],
54
+ *,
55
+ data: dict[str, Any] | None = None,
56
+ separator: str = "\n\n# --- template separator ---\n\n",
57
+ ) -> str:
58
+ """
59
+ Load multiple template files, concatenate them, then render once.
60
+ Useful when you want shared macros / blocks across files.
61
+ """
62
+ contents = []
63
+ for t in templates:
64
+ path = Path(t)
65
+ contents.append(path.read_text(encoding="utf-8"))
66
+
67
+ combined = separator.join(contents)
68
+ return self.render_string(combined, **(data or {}))