bhtool 0.3.1__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,13 +1,13 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: bhtool
3
- Version: 0.3.1
3
+ Version: 0.4.0
4
4
  Summary: Assorted CLI utilities for Bosco’s workflows (dev, media, macOS, LLM helpers).
5
5
  Author-email: Bosco Ho <apposite@gmail.com>
6
6
  Requires-Python: >=3.13
7
7
  Requires-Dist: cyclopts>=4.10.2
8
- Requires-Dist: microeval>=0.5.0
9
8
  Requires-Dist: numpy>=1.26.4
10
9
  Requires-Dist: path>=16.14.0
10
+ Requires-Dist: pydantic-ai>=2.55.0
11
11
  Requires-Dist: pydash>=8.0.3
12
12
  Requires-Dist: python-dotenv>=1.2.2
13
13
  Requires-Dist: rich>=15.0.0
@@ -48,7 +48,7 @@ Assorted CLI utilities that are handy in Bosco’s day-to-day work—version bum
48
48
 
49
49
  ### Movies + LLM
50
50
 
51
- `b movies --service <name>` selects the backend for one run (`openai`, `groq`, `ollama`, `bedrock` — see `b movies --help`). `LLM_SERVICE` sets the default. Set `LLM_SERVICE` in the process environment or in **`bhtool/.env`** (loaded from the installed package directory). Model defaults come from [microeval](https://pypi.org/project/microeval/)’s config for that service.
51
+ `b movies` asks an LLM for normalized names. `--service <name>` selects a provider for one run (`groq`, `google`, `openai`, `anthropic`, `mistral`, `deepseek`, `xai`, `bedrock`, `ollama` — see `b movies --help`). Without it, the provider comes from `LLM_SERVICE`, else from the first API key in the environment, else `ollama`. Set keys in the process environment or in **`bhtool/.env`** (loaded from the installed package directory). Each provider uses one cheap model from `bhtool/llm_services.py` through [pydantic-ai](https://pypi.org/project/pydantic-ai/).
52
52
 
53
53
  ---
54
54
 
@@ -32,7 +32,7 @@ Assorted CLI utilities that are handy in Bosco’s day-to-day work—version bum
32
32
 
33
33
  ### Movies + LLM
34
34
 
35
- `b movies --service <name>` selects the backend for one run (`openai`, `groq`, `ollama`, `bedrock` — see `b movies --help`). `LLM_SERVICE` sets the default. Set `LLM_SERVICE` in the process environment or in **`bhtool/.env`** (loaded from the installed package directory). Model defaults come from [microeval](https://pypi.org/project/microeval/)’s config for that service.
35
+ `b movies` asks an LLM for normalized names. `--service <name>` selects a provider for one run (`groq`, `google`, `openai`, `anthropic`, `mistral`, `deepseek`, `xai`, `bedrock`, `ollama` — see `b movies --help`). Without it, the provider comes from `LLM_SERVICE`, else from the first API key in the environment, else `ollama`. Set keys in the process environment or in **`bhtool/.env`** (loaded from the installed package directory). Each provider uses one cheap model from `bhtool/llm_services.py` through [pydantic-ai](https://pypi.org/project/pydantic-ai/).
36
36
 
37
37
  ---
38
38
 
@@ -116,7 +116,7 @@ def movies_cmd(
116
116
 
117
117
  :param root_dir: Root directory containing movies (positional); default is current working directory.
118
118
  :param execute: If true, perform renames; otherwise dry run (table output).
119
- :param service: LLM service for normalization; default from LLM_SERVICE, else openai.
119
+ :param service: LLM service for normalization; default is detected from the environment.
120
120
  """
121
121
  import bhtool.list_movies as list_movies_mod
122
122
 
@@ -3,7 +3,6 @@
3
3
  import asyncio
4
4
  import json
5
5
  import os
6
- import re
7
6
  import sys
8
7
  import textwrap
9
8
 
@@ -16,12 +15,19 @@ from rich.panel import Panel
16
15
  from rich.table import Table
17
16
  from rich.text import Text
18
17
 
19
- from microeval.llm import get_llm_client
18
+ from pydantic import BaseModel, Field
19
+ from pydantic_ai import Agent
20
+ from pydantic_ai.models import infer_model
21
+
22
+ from bhtool.llm_services import OLLAMA_BASE_URL, detect_service, model_for
20
23
 
21
24
  _pkg_root = Path(__file__).parent
22
25
 
23
26
  load_dotenv(_pkg_root / ".env")
24
27
 
28
+ # Suppress the pydantic-ai first-run banner in this CLI.
29
+ os.environ.setdefault("PYDANTIC_AI_NO_BANNER", "1")
30
+
25
31
  _MOVIES_BORDER = "cyan"
26
32
  _MOVIES_TITLE = "movies"
27
33
 
@@ -90,16 +96,10 @@ NORMALIZE_SYSTEM_PROMPT = textwrap.dedent("""
90
96
  extension stays unchanged in the output.
91
97
  - Preserve the original spelling of the title; only fix casing and
92
98
  punctuation.
93
- - Return valid JSON only, no markdown or extra text.
94
99
  """).strip()
95
100
 
96
101
  NORMALIZE_USER_PROMPT_TEMPLATE = textwrap.dedent("""
97
- Convert these names to normalized form. Return a single JSON object with
98
- two keys:
99
- - "directories": list of {{"original": "<dir name>",
100
- "normalized": "<normalized dir name>"}}
101
- - "files": list of {{"original": "<filename with extension>",
102
- "normalized": "<normalized base name without extension>"}}
102
+ Convert these names to normalized form.
103
103
 
104
104
  Directory names:
105
105
  {dir_list}
@@ -109,81 +109,79 @@ NORMALIZE_USER_PROMPT_TEMPLATE = textwrap.dedent("""
109
109
  """).strip()
110
110
 
111
111
 
112
- def _parse_llm_json_object(result: dict) -> dict | None:
113
- text = (result.get("text") or "").strip()
114
- if not text:
115
- return None
116
- fence = re.search(r"```(?:json)?\s*([\s\S]*?)```", text)
117
- if fence:
118
- text = fence.group(1).strip()
119
- try:
120
- data = json.loads(text)
121
- return data if isinstance(data, dict) else None
122
- except json.JSONDecodeError:
123
- pass
124
- start, end = text.find("{"), text.rfind("}")
125
- if start != -1 and end > start:
126
- try:
127
- data = json.loads(text[start : end + 1])
128
- return data if isinstance(data, dict) else None
129
- except json.JSONDecodeError:
130
- return None
131
- return None
112
+ class NameMapping(BaseModel):
113
+ """One original name and its normalized form."""
114
+
115
+ original: str = Field(description="The original directory or file name.")
116
+ normalized: str = Field(description="The normalized name.")
117
+
118
+
119
+ class NormalizedNameMapping(BaseModel):
120
+ """Normalized names for the directories and files in one root directory."""
121
+
122
+ directories: list[NameMapping] = Field(
123
+ default_factory=list, description="Normalized directory names."
124
+ )
125
+ files: list[NameMapping] = Field(
126
+ default_factory=list, description="Normalized file names."
127
+ )
132
128
 
133
129
 
134
130
  async def normalize_names_with_llm(
135
- dir_names, file_names, service="openai", *, console: Console | None = None
131
+ dir_names, file_names, service=None, *, console: Console | None = None
136
132
  ):
137
133
  if not dir_names and not file_names:
138
134
  return {"directories": [], "files": []}
139
135
 
140
136
  console = console or Console()
137
+ service = service or detect_service()
141
138
  dir_list = "\n".join(f"- {n}" for n in dir_names) or "(none)"
142
139
  file_list = "\n".join(f"- {n}" for n in file_names) or "(none)"
143
140
  user_content = NORMALIZE_USER_PROMPT_TEMPLATE.format(
144
141
  dir_list=dir_list,
145
142
  file_list=file_list,
146
143
  )
147
- messages = [
148
- {"role": "system", "content": NORMALIZE_SYSTEM_PROMPT},
149
- {"role": "user", "content": user_content},
150
- ]
151
144
 
152
145
  try:
153
- async with get_llm_client(service) as client:
154
- llm_body = Text.assemble(
155
- ("service ", ""),
156
- (client.service, "bold"),
157
- (" model ", ""),
158
- (client.model, "bold"),
159
- ("\n", ""),
160
- ("Requesting normalized names from the LLM…", "dim"),
161
- )
162
- console.print(_panel_lines(llm_body, title="LLM", border="blue"))
163
- result = await client.get_completion(messages)
146
+ if service == "ollama":
147
+ os.environ.setdefault("OLLAMA_BASE_URL", OLLAMA_BASE_URL)
148
+ model = infer_model(model_for(service))
149
+ agent = Agent(
150
+ model,
151
+ instructions=NORMALIZE_SYSTEM_PROMPT,
152
+ output_type=NormalizedNameMapping,
153
+ )
154
+ llm_body = Text.assemble(
155
+ ("service ", ""),
156
+ (service, "bold"),
157
+ (" model ", ""),
158
+ (model.model_name, "bold"),
159
+ ("\n", ""),
160
+ ("Requesting normalized names from the LLM…", "dim"),
161
+ )
162
+ console.print(_panel_lines(llm_body, title="LLM", border="blue"))
163
+ result = await agent.run(user_content)
164
+ mapping = result.output
164
165
  except Exception as e:
165
166
  err = Console(stderr=True)
166
167
  err.print(CycloptsPanel(str(e), title="Error", style="red"))
167
168
  err.print(
168
169
  CycloptsPanel(
169
- "Set LLM_SERVICE (openai, groq, ollama, bedrock) and the matching API key or "
170
- "AWS/Ollama; use bhtool/.env or your environment — see README.",
170
+ "Set LLM_SERVICE or --service and the matching API key. Providers: "
171
+ "groq, google, openai, anthropic, mistral, deepseek, xai, bedrock, "
172
+ "ollama. Use bhtool/.env or your environment — see README.",
171
173
  title="Hint",
172
174
  style="yellow",
173
175
  )
174
176
  )
175
177
  sys.exit(1)
176
- if result.get("text", "").strip().startswith("Error:"):
177
- raise RuntimeError(result["text"])
178
178
 
179
- data = _parse_llm_json_object(result)
180
- if not data:
181
- raise RuntimeError("LLM response could not be parsed as JSON")
182
- return {"directories": data.get("directories", []), "files": data.get("files", [])}
179
+ return mapping.model_dump()
183
180
 
184
181
 
185
- def run_normalize_with_llm(root_dir, service="openai", *, console: Console | None = None):
182
+ def run_normalize_with_llm(root_dir, service=None, *, console: Console | None = None):
186
183
  console = console or Console()
184
+ service = service or detect_service()
187
185
  root = Path(root_dir)
188
186
  video_suffixes = {".avi", ".mkv", ".mp4", ".m4v", ".mov", ".wmv", ".webm"}
189
187
  skip_names = {
@@ -219,7 +217,7 @@ def run_normalize_with_llm(root_dir, service="openai", *, console: Console | Non
219
217
  f" {'video file' if len(file_names) == 1 else 'video files'} · ",
220
218
  "dim",
221
219
  ),
222
- ("LLM_SERVICE=", "dim"),
220
+ ("service ", "dim"),
223
221
  (service, "bold"),
224
222
  )
225
223
  console.print(_panel_lines(scan, title="Scan", border=_MOVIES_BORDER))
@@ -364,7 +362,7 @@ def rename(
364
362
  )
365
363
  console.print(_panel_lines(intro, title=_MOVIES_TITLE, border=_MOVIES_BORDER))
366
364
  root = _root_dir(root_dir)
367
- svc = service or os.environ.get("LLM_SERVICE", "openai")
365
+ svc = service or detect_service()
368
366
  mapping = run_normalize_with_llm(root, svc, console=console)
369
367
  mapping_path = Path.cwd() / "movie_mapping.json"
370
368
  mapping_path.write_text(json.dumps(mapping, indent=2))
@@ -0,0 +1,77 @@
1
+ #!/usr/bin/env python3
2
+
3
+ """Provider defaults for the movies LLM command.
4
+
5
+ One cheap model per provider, picked for the short name-normalization prompt.
6
+ Prices change, so check them before you add a provider. This module has no
7
+ heavy imports, so the CLI can show the list without loading pydantic-ai.
8
+ """
9
+
10
+ import os
11
+ from typing import Literal, get_args
12
+
13
+ LLMService = Literal[
14
+ "groq",
15
+ "google",
16
+ "openai",
17
+ "anthropic",
18
+ "mistral",
19
+ "deepseek",
20
+ "xai",
21
+ "bedrock",
22
+ "ollama",
23
+ ]
24
+
25
+ PROVIDERS = get_args(LLMService)
26
+
27
+ # pydantic-ai model id is "<provider>:<model>".
28
+ # The value is the cheapest model that can follow the normalize instructions.
29
+ PROVIDER_MODELS = {
30
+ "groq": "openai/gpt-oss-20b", # $0.075 / $0.30 per 1M
31
+ "google": "gemini-2.5-flash-lite", # $0.10 / $0.40
32
+ "openai": "gpt-6-luna", # $0.10 / $0.50
33
+ "anthropic": "claude-haiku-5-5", # $0.10 / $0.50
34
+ "mistral": "ministral-3-8b-25-12", # $0.15 / $0.15
35
+ "deepseek": "deepseek-flash", # $0.15 / $0.60 off-peak
36
+ "xai": "grok-4.3", # $1.25 / $2.50
37
+ "bedrock": "amazon.nova-lite-v1:0", # region-dependent
38
+ "ollama": "llama3.2", # free, local
39
+ }
40
+
41
+ # Environment variable that enables a provider. Auto-detect reads these in order.
42
+ PROVIDER_ENV_VARS = {
43
+ "groq": ("GROQ_API_KEY",),
44
+ "google": ("GOOGLE_API_KEY", "GEMINI_API_KEY"),
45
+ "openai": ("OPENAI_API_KEY",),
46
+ "anthropic": ("ANTHROPIC_API_KEY",),
47
+ "mistral": ("MISTRAL_API_KEY",),
48
+ "deepseek": ("DEEPSEEK_API_KEY",),
49
+ "xai": ("XAI_API_KEY",),
50
+ "bedrock": ("AWS_ACCESS_KEY_ID", "AWS_PROFILE", "AWS_BEARER_TOKEN_BEDROCK"),
51
+ "ollama": ("OLLAMA_BASE_URL",),
52
+ }
53
+
54
+ # pydantic-ai needs this value for the Ollama provider.
55
+ OLLAMA_BASE_URL = "http://localhost:11434"
56
+
57
+
58
+ def model_for(service: str) -> str:
59
+ """Return the pydantic-ai model id for a provider, for example 'openai:gpt-6-luna'."""
60
+ try:
61
+ model = PROVIDER_MODELS[service]
62
+ except KeyError:
63
+ raise ValueError(
64
+ f"Unknown service {service!r}; choose one of {', '.join(PROVIDERS)}"
65
+ ) from None
66
+ return f"{service}:{model}"
67
+
68
+
69
+ def detect_service() -> str:
70
+ """Pick a provider from LLM_SERVICE, else from the environment, else ollama."""
71
+ env_service = os.environ.get("LLM_SERVICE")
72
+ if env_service:
73
+ return env_service.lower()
74
+ for provider in PROVIDERS:
75
+ if any(os.environ.get(var) for var in PROVIDER_ENV_VARS[provider]):
76
+ return provider
77
+ return "ollama"
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "bhtool"
3
- version = "0.3.1"
3
+ version = "0.4.0"
4
4
  description = "Assorted CLI utilities for Bosco’s workflows (dev, media, macOS, LLM helpers)."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.13"
@@ -15,7 +15,7 @@ dependencies = [
15
15
  "cyclopts>=4.10.2",
16
16
  "python-dotenv>=1.2.2",
17
17
  "rich>=15.0.0",
18
- "microeval>=0.5.0",
18
+ "pydantic-ai>=2.55.0",
19
19
  ]
20
20
 
21
21
  [project.scripts]
@@ -1,14 +0,0 @@
1
- #!/usr/bin/env python3
2
-
3
- """LLM service names shared by the CLI and the movies command.
4
-
5
- Keep ``LLMService`` in sync with ``microeval.llm.LLM_CLIENTS``. This module has
6
- no heavy imports, so the CLI can show the choices without importing microeval
7
- at startup.
8
- """
9
-
10
- from typing import Literal, get_args
11
-
12
- LLMService = Literal["openai", "groq", "ollama", "bedrock"]
13
-
14
- LLM_SERVICES = get_args(LLMService)
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes