kinako-llama-cpp 0.3.24.post1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,148 @@
1
+ Metadata-Version: 2.4
2
+ Name: kinako-llama-cpp
3
+ Version: 0.3.24.post1
4
+ Summary: [Beta] A custom llama-cpp-python wheel tailored for Intel 10th Gen CPUs on Windows 11.
5
+ Home-page: https://github.com/sak301537/gemma4-portable-for-Windows11
6
+ Author: sora_sakurai/toshiaki_sakurai
7
+ Author-email: t301537@mbr.nifty.com
8
+ Classifier: Programming Language :: Python :: 3
9
+ Classifier: License :: OSI Approved :: MIT License
10
+ Classifier: Operating System :: Microsoft :: Windows :: Windows 11
11
+ Requires-Python: >=3.10
12
+ Description-Content-Type: text/markdown
13
+ Dynamic: author
14
+ Dynamic: author-email
15
+ Dynamic: classifier
16
+ Dynamic: description
17
+ Dynamic: description-content-type
18
+ Dynamic: home-page
19
+ Dynamic: requires-python
20
+ Dynamic: summary
21
+
22
+ # kinako-llama-cpp-python
23
+ **⚠️ This package is an early experimental/testing release.
24
+ This is a custom-built, wheel-embedded distribution of llama-cpp-python optimized to run LLMs like Gemma 4 smoothly on everyday, mid-to-low spec PCs running Windows 11 (specifically tested on Intel 7th, 8th, 9th, 10th, and 13th Gen CPUs).
25
+
26
+ **⚠️ 本パッケージは検証用の早期(取り急ぎの)リリースです。**
27
+ 知人・友人環境での動作検証を通じ、今後記述や構成を随時修正・アップデートしていく前提のプロジェクトです。
28
+ 本パッケージは、Windowsの身近な低スペックPC環境において、軽量LLMを動作させることを目的とした `llama-cpp-python` のカスタムビルド(Wheel同梱版)です。
29
+
30
+ > **開発者より:**
31
+ > 現時点ではWindows+CPU AVX2環境での動作を想定していますが、構成の厳密な検証はこれからです。(Intel 7,8,9,10,13世代/Win11で確認済)
32
+ > 実際に `pip install` してみて動いた・動かなかったなどのフィードバック(動作報告)をもとに、セットアップ設定やドキュメントを順次修正していきます。
33
+
34
+ ### インストール方法
35
+ ```bash
36
+ pip install kinako-llama-cpp
37
+
38
+ Important (English):
39
+ If you already have the official llama-cpp-python installed in your environment, installing this package directly may cause module conflicts inside the llama_cpp folder.
40
+ Please make sure to uninstall the official version before installing this package to prevent any broken dependencies.
41
+
42
+ Bash
43
+ ### ⚠️ すでに公式の llama-cpp-python をインストールしている方へ / For Existing Users
44
+
45
+ **重要 (Japanese):**
46
+ もし、すでに本家(公式)の `llama-cpp-python` をインストール済みの環境に本パッケージを導入する場合、中身のモジュール(`llama_cpp`)が衝突して正常に動作しなくなる恐れがあります。
47
+ 本パッケージを試す前に、必ず一度既存のパッケージをアンインストールしてください。
48
+
49
+ ```bash
50
+ # 既存の公式版を一度削除する
51
+ pip uninstall llama-cpp-python -y
52
+
53
+ # その後、本パッケージをインストールする
54
+ pip install kinako-llama-cpp
55
+
56
+ > Python program sample
57
+ >
58
+ import os
59
+ import sys
60
+ from llama_cpp import Llama
61
+
62
+ # ── 設定項目 / Configuration ──
63
+ MODEL_PATH = r"C:\pythonfiles\llm\gemma-4-E2B-it-RotorQuant-Q8_0.gguf"
64
+
65
+ print("--- LLMモデルを読み込んでいます(数秒〜十数秒かかります)... ---")
66
+ print("--- Loading LLM model (This may take a few seconds)... ---")
67
+
68
+ try:
69
+ llm = Llama(
70
+ model_path=MODEL_PATH,
71
+ n_ctx=2048,
72
+ n_batch=128, # 過去の履歴を読み直す速度をCPU向けに最適化 / Optimized for CPU
73
+ n_gpu_layers=0 # 完全CPUモード / CPU Only Mode
74
+ )
75
+ except Exception as e:
76
+ print(f"[ERROR] Model file not found or could not be loaded.")
77
+ print(f"\n【エラー】モデルファイルが見つからないか、読み込めませんでした。")
78
+ print(f"Please check the path: {MODEL_PATH}")
79
+ input("\nPress Enter to exit...")
80
+ sys.exit(1)
81
+
82
+ print("\n=========================================")
83
+ print(" Windows CPU-driven PC Chat")
84
+ print(" Type 'exit' to quit the chat. / 終了するには「exit」と入力してください。")
85
+ print("=========================================\n")
86
+
87
+ # 過去の会話履歴を保存するリスト / Chat history list
88
+ chat_history = []
89
+
90
+ while True:
91
+ try:
92
+ user_input = input("You / あなた: ")
93
+
94
+ if not user_input.strip():
95
+ continue
96
+
97
+ if user_input.strip().lower() == "exit":
98
+ print("Closing chat. Thank you! / チャットを終了します。お疲れ様でした!")
99
+ break
100
+
101
+ # -------------------------------------------------
102
+ # プロンプトの組み立て / Prompt Construction
103
+ # -------------------------------------------------
104
+ # 多言語に対応できるよう、指示を英語に統一し「ユーザーの言語に合わせる」ルールを追加
105
+ system_prompt = "System: Act as a helpful AI assistant. Reply kindly and politely within 400 characters. (Please respond in the same language the user speaks.)\n"
106
+ #日本語のみの場合
107
+ #system_prompt = "System: 親切なAIとして、400文字以内で、優しく丁寧に回答してください。\n"
108
+
109
+ # 履歴を直近の2回分(発言数に直すと最大4つ)に絞る
110
+ recent_history = chat_history[-4:]
111
+ history_text = "".join(recent_history)
112
+
113
+ # 今回の質問をドッキング
114
+ current_prompt = f"User: {user_input}\nAI: "
115
+ full_prompt = system_prompt + history_text + current_prompt
116
+
117
+ print("\nAI: ", end="", flush=True)
118
+
119
+ # ストリーミング実行 / Streaming Execution
120
+ response_stream = llm(
121
+ full_prompt,
122
+ max_tokens=500, # 余裕を持たせた上限設定
123
+ stop=["User:", "\nUser:", "System:", "\nSystem:"],
124
+ echo=False,
125
+ stream=True
126
+ )
127
+
128
+ # 1文字ずつ出力しながら、今回の回答テキストを記録
129
+ ai_response = ""
130
+ for chunk in response_stream:
131
+ text = chunk["choices"][0]["text"]
132
+ sys.stdout.write(text)
133
+ sys.stdout.flush()
134
+ ai_response += text
135
+
136
+ print("\n-----------------------------------------")
137
+
138
+ # ── 今回の会話を履歴リストに追加 ──
139
+ chat_history.append(f"User: {user_input}\n")
140
+ chat_history.append(f"AI: {ai_response.strip()}\n")
141
+
142
+ except KeyboardInterrupt:
143
+ print("\n Exit requested. Closing chat./ チャットを終了します。")
144
+ print("\n")
145
+ break
146
+ except Exception as e:
147
+ print(f"\n An error occurred / エラーが発生しました {e}")
148
+ break
@@ -0,0 +1,127 @@
1
+ # kinako-llama-cpp-python
2
+ **⚠️ This package is an early experimental/testing release.
3
+ This is a custom-built, wheel-embedded distribution of llama-cpp-python optimized to run LLMs like Gemma 4 smoothly on everyday, mid-to-low spec PCs running Windows 11 (specifically tested on Intel 7th, 8th, 9th, 10th, and 13th Gen CPUs).
4
+
5
+ **⚠️ 本パッケージは検証用の早期(取り急ぎの)リリースです。**
6
+ 知人・友人環境での動作検証を通じ、今後記述や構成を随時修正・アップデートしていく前提のプロジェクトです。
7
+ 本パッケージは、Windowsの身近な低スペックPC環境において、軽量LLMを動作させることを目的とした `llama-cpp-python` のカスタムビルド(Wheel同梱版)です。
8
+
9
+ > **開発者より:**
10
+ > 現時点ではWindows+CPU AVX2環境での動作を想定していますが、構成の厳密な検証はこれからです。(Intel 7,8,9,10,13世代/Win11で確認済)
11
+ > 実際に `pip install` してみて動いた・動かなかったなどのフィードバック(動作報告)をもとに、セットアップ設定やドキュメントを順次修正していきます。
12
+
13
+ ### インストール方法
14
+ ```bash
15
+ pip install kinako-llama-cpp
16
+
17
+ Important (English):
18
+ If you already have the official llama-cpp-python installed in your environment, installing this package directly may cause module conflicts inside the llama_cpp folder.
19
+ Please make sure to uninstall the official version before installing this package to prevent any broken dependencies.
20
+
21
+ Bash
22
+ ### ⚠️ すでに公式の llama-cpp-python をインストールしている方へ / For Existing Users
23
+
24
+ **重要 (Japanese):**
25
+ もし、すでに本家(公式)の `llama-cpp-python` をインストール済みの環境に本パッケージを導入する場合、中身のモジュール(`llama_cpp`)が衝突して正常に動作しなくなる恐れがあります。
26
+ 本パッケージを試す前に、必ず一度既存のパッケージをアンインストールしてください。
27
+
28
+ ```bash
29
+ # 既存の公式版を一度削除する
30
+ pip uninstall llama-cpp-python -y
31
+
32
+ # その後、本パッケージをインストールする
33
+ pip install kinako-llama-cpp
34
+
35
+ > Python program sample
36
+ >
37
+ import os
38
+ import sys
39
+ from llama_cpp import Llama
40
+
41
+ # ── 設定項目 / Configuration ──
42
+ MODEL_PATH = r"C:\pythonfiles\llm\gemma-4-E2B-it-RotorQuant-Q8_0.gguf"
43
+
44
+ print("--- LLMモデルを読み込んでいます(数秒〜十数秒かかります)... ---")
45
+ print("--- Loading LLM model (This may take a few seconds)... ---")
46
+
47
+ try:
48
+ llm = Llama(
49
+ model_path=MODEL_PATH,
50
+ n_ctx=2048,
51
+ n_batch=128, # 過去の履歴を読み直す速度をCPU向けに最適化 / Optimized for CPU
52
+ n_gpu_layers=0 # 完全CPUモード / CPU Only Mode
53
+ )
54
+ except Exception as e:
55
+ print(f"[ERROR] Model file not found or could not be loaded.")
56
+ print(f"\n【エラー】モデルファイルが見つからないか、読み込めませんでした。")
57
+ print(f"Please check the path: {MODEL_PATH}")
58
+ input("\nPress Enter to exit...")
59
+ sys.exit(1)
60
+
61
+ print("\n=========================================")
62
+ print(" Windows CPU-driven PC Chat")
63
+ print(" Type 'exit' to quit the chat. / 終了するには「exit」と入力してください。")
64
+ print("=========================================\n")
65
+
66
+ # 過去の会話履歴を保存するリスト / Chat history list
67
+ chat_history = []
68
+
69
+ while True:
70
+ try:
71
+ user_input = input("You / あなた: ")
72
+
73
+ if not user_input.strip():
74
+ continue
75
+
76
+ if user_input.strip().lower() == "exit":
77
+ print("Closing chat. Thank you! / チャットを終了します。お疲れ様でした!")
78
+ break
79
+
80
+ # -------------------------------------------------
81
+ # プロンプトの組み立て / Prompt Construction
82
+ # -------------------------------------------------
83
+ # 多言語に対応できるよう、指示を英語に統一し「ユーザーの言語に合わせる」ルールを追加
84
+ system_prompt = "System: Act as a helpful AI assistant. Reply kindly and politely within 400 characters. (Please respond in the same language the user speaks.)\n"
85
+ #日本語のみの場合
86
+ #system_prompt = "System: 親切なAIとして、400文字以内で、優しく丁寧に回答してください。\n"
87
+
88
+ # 履歴を直近の2回分(発言数に直すと最大4つ)に絞る
89
+ recent_history = chat_history[-4:]
90
+ history_text = "".join(recent_history)
91
+
92
+ # 今回の質問をドッキング
93
+ current_prompt = f"User: {user_input}\nAI: "
94
+ full_prompt = system_prompt + history_text + current_prompt
95
+
96
+ print("\nAI: ", end="", flush=True)
97
+
98
+ # ストリーミング実行 / Streaming Execution
99
+ response_stream = llm(
100
+ full_prompt,
101
+ max_tokens=500, # 余裕を持たせた上限設定
102
+ stop=["User:", "\nUser:", "System:", "\nSystem:"],
103
+ echo=False,
104
+ stream=True
105
+ )
106
+
107
+ # 1文字ずつ出力しながら、今回の回答テキストを記録
108
+ ai_response = ""
109
+ for chunk in response_stream:
110
+ text = chunk["choices"][0]["text"]
111
+ sys.stdout.write(text)
112
+ sys.stdout.flush()
113
+ ai_response += text
114
+
115
+ print("\n-----------------------------------------")
116
+
117
+ # ── 今回の会話を履歴リストに追加 ──
118
+ chat_history.append(f"User: {user_input}\n")
119
+ chat_history.append(f"AI: {ai_response.strip()}\n")
120
+
121
+ except KeyboardInterrupt:
122
+ print("\n Exit requested. Closing chat./ チャットを終了します。")
123
+ print("\n")
124
+ break
125
+ except Exception as e:
126
+ print(f"\n An error occurred / エラーが発生しました {e}")
127
+ break
@@ -0,0 +1,148 @@
1
+ Metadata-Version: 2.4
2
+ Name: kinako-llama-cpp
3
+ Version: 0.3.24.post1
4
+ Summary: [Beta] A custom llama-cpp-python wheel tailored for Intel 10th Gen CPUs on Windows 11.
5
+ Home-page: https://github.com/sak301537/gemma4-portable-for-Windows11
6
+ Author: sora_sakurai/toshiaki_sakurai
7
+ Author-email: t301537@mbr.nifty.com
8
+ Classifier: Programming Language :: Python :: 3
9
+ Classifier: License :: OSI Approved :: MIT License
10
+ Classifier: Operating System :: Microsoft :: Windows :: Windows 11
11
+ Requires-Python: >=3.10
12
+ Description-Content-Type: text/markdown
13
+ Dynamic: author
14
+ Dynamic: author-email
15
+ Dynamic: classifier
16
+ Dynamic: description
17
+ Dynamic: description-content-type
18
+ Dynamic: home-page
19
+ Dynamic: requires-python
20
+ Dynamic: summary
21
+
22
+ # kinako-llama-cpp-python
23
+ **⚠️ This package is an early experimental/testing release.
24
+ This is a custom-built, wheel-embedded distribution of llama-cpp-python optimized to run LLMs like Gemma 4 smoothly on everyday, mid-to-low spec PCs running Windows 11 (specifically tested on Intel 7th, 8th, 9th, 10th, and 13th Gen CPUs).
25
+
26
+ **⚠️ 本パッケージは検証用の早期(取り急ぎの)リリースです。**
27
+ 知人・友人環境での動作検証を通じ、今後記述や構成を随時修正・アップデートしていく前提のプロジェクトです。
28
+ 本パッケージは、Windowsの身近な低スペックPC環境において、軽量LLMを動作させることを目的とした `llama-cpp-python` のカスタムビルド(Wheel同梱版)です。
29
+
30
+ > **開発者より:**
31
+ > 現時点ではWindows+CPU AVX2環境での動作を想定していますが、構成の厳密な検証はこれからです。(Intel 7,8,9,10,13世代/Win11で確認済)
32
+ > 実際に `pip install` してみて動いた・動かなかったなどのフィードバック(動作報告)をもとに、セットアップ設定やドキュメントを順次修正していきます。
33
+
34
+ ### インストール方法
35
+ ```bash
36
+ pip install kinako-llama-cpp
37
+
38
+ Important (English):
39
+ If you already have the official llama-cpp-python installed in your environment, installing this package directly may cause module conflicts inside the llama_cpp folder.
40
+ Please make sure to uninstall the official version before installing this package to prevent any broken dependencies.
41
+
42
+ Bash
43
+ ### ⚠️ すでに公式の llama-cpp-python をインストールしている方へ / For Existing Users
44
+
45
+ **重要 (Japanese):**
46
+ もし、すでに本家(公式)の `llama-cpp-python` をインストール済みの環境に本パッケージを導入する場合、中身のモジュール(`llama_cpp`)が衝突して正常に動作しなくなる恐れがあります。
47
+ 本パッケージを試す前に、必ず一度既存のパッケージをアンインストールしてください。
48
+
49
+ ```bash
50
+ # 既存の公式版を一度削除する
51
+ pip uninstall llama-cpp-python -y
52
+
53
+ # その後、本パッケージをインストールする
54
+ pip install kinako-llama-cpp
55
+
56
+ > Python program sample
57
+ >
58
+ import os
59
+ import sys
60
+ from llama_cpp import Llama
61
+
62
+ # ── 設定項目 / Configuration ──
63
+ MODEL_PATH = r"C:\pythonfiles\llm\gemma-4-E2B-it-RotorQuant-Q8_0.gguf"
64
+
65
+ print("--- LLMモデルを読み込んでいます(数秒〜十数秒かかります)... ---")
66
+ print("--- Loading LLM model (This may take a few seconds)... ---")
67
+
68
+ try:
69
+ llm = Llama(
70
+ model_path=MODEL_PATH,
71
+ n_ctx=2048,
72
+ n_batch=128, # 過去の履歴を読み直す速度をCPU向けに最適化 / Optimized for CPU
73
+ n_gpu_layers=0 # 完全CPUモード / CPU Only Mode
74
+ )
75
+ except Exception as e:
76
+ print(f"[ERROR] Model file not found or could not be loaded.")
77
+ print(f"\n【エラー】モデルファイルが見つからないか、読み込めませんでした。")
78
+ print(f"Please check the path: {MODEL_PATH}")
79
+ input("\nPress Enter to exit...")
80
+ sys.exit(1)
81
+
82
+ print("\n=========================================")
83
+ print(" Windows CPU-driven PC Chat")
84
+ print(" Type 'exit' to quit the chat. / 終了するには「exit」と入力してください。")
85
+ print("=========================================\n")
86
+
87
+ # 過去の会話履歴を保存するリスト / Chat history list
88
+ chat_history = []
89
+
90
+ while True:
91
+ try:
92
+ user_input = input("You / あなた: ")
93
+
94
+ if not user_input.strip():
95
+ continue
96
+
97
+ if user_input.strip().lower() == "exit":
98
+ print("Closing chat. Thank you! / チャットを終了します。お疲れ様でした!")
99
+ break
100
+
101
+ # -------------------------------------------------
102
+ # プロンプトの組み立て / Prompt Construction
103
+ # -------------------------------------------------
104
+ # 多言語に対応できるよう、指示を英語に統一し「ユーザーの言語に合わせる」ルールを追加
105
+ system_prompt = "System: Act as a helpful AI assistant. Reply kindly and politely within 400 characters. (Please respond in the same language the user speaks.)\n"
106
+ #日本語のみの場合
107
+ #system_prompt = "System: 親切なAIとして、400文字以内で、優しく丁寧に回答してください。\n"
108
+
109
+ # 履歴を直近の2回分(発言数に直すと最大4つ)に絞る
110
+ recent_history = chat_history[-4:]
111
+ history_text = "".join(recent_history)
112
+
113
+ # 今回の質問をドッキング
114
+ current_prompt = f"User: {user_input}\nAI: "
115
+ full_prompt = system_prompt + history_text + current_prompt
116
+
117
+ print("\nAI: ", end="", flush=True)
118
+
119
+ # ストリーミング実行 / Streaming Execution
120
+ response_stream = llm(
121
+ full_prompt,
122
+ max_tokens=500, # 余裕を持たせた上限設定
123
+ stop=["User:", "\nUser:", "System:", "\nSystem:"],
124
+ echo=False,
125
+ stream=True
126
+ )
127
+
128
+ # 1文字ずつ出力しながら、今回の回答テキストを記録
129
+ ai_response = ""
130
+ for chunk in response_stream:
131
+ text = chunk["choices"][0]["text"]
132
+ sys.stdout.write(text)
133
+ sys.stdout.flush()
134
+ ai_response += text
135
+
136
+ print("\n-----------------------------------------")
137
+
138
+ # ── 今回の会話を履歴リストに追加 ──
139
+ chat_history.append(f"User: {user_input}\n")
140
+ chat_history.append(f"AI: {ai_response.strip()}\n")
141
+
142
+ except KeyboardInterrupt:
143
+ print("\n Exit requested. Closing chat./ チャットを終了します。")
144
+ print("\n")
145
+ break
146
+ except Exception as e:
147
+ print(f"\n An error occurred / エラーが発生しました {e}")
148
+ break
@@ -0,0 +1,6 @@
1
+ README.md
2
+ setup.py
3
+ kinako_llama_cpp.egg-info/PKG-INFO
4
+ kinako_llama_cpp.egg-info/SOURCES.txt
5
+ kinako_llama_cpp.egg-info/dependency_links.txt
6
+ kinako_llama_cpp.egg-info/top_level.txt
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,65 @@
1
+ import os
2
+ import zipfile
3
+ from setuptools import setup, find_packages
4
+ from setuptools.command.install import install
5
+
6
+ def check_conflict_and_abort():
7
+ try:
8
+ import llama_cpp
9
+ # すでに llama_cpp が存在する場合、詳細なエラーメッセージを出して終了
10
+ print("\n" + "="*70)
11
+ print("\nConflict detected: 'llama-cpp-python' or a similar module is already installed.")
12
+ print(" To prevent broken dependencies, please uninstall it first and try again.")
13
+ print("【⚠️ インストールを中断しました / Installation Aborted】")
14
+ print("\n[JA] すでに公式の 'llama-cpp-python' 等のモジュールが環境に存在します。")
15
+ print(" 衝突を防ぐため、一度既存のパッケージを削除してから再試行してください。")
16
+
17
+ print("\n >>> pip uninstall llama-cpp-python -y <<<")
18
+ print("="*70 + "\n")
19
+ sys.exit(1)
20
+ except ImportError:
21
+ pass
22
+
23
+ # ── インストール時に、同梱した本物のWheelを解凍して中身を展開する ──
24
+ class CustomInstallCommand(install):
25
+ def run(self):
26
+ # 1. まず本家との衝突がないかチェック
27
+ check_conflict_and_abort()
28
+
29
+ # 2. 衝突がなければ通常のインストール処理を実行
30
+ super().run()
31
+
32
+ # 3. 同梱されている本物のwheelファイルを展開
33
+ target_whl = "llama_cpp_python-0.3.24-py3-none-win_amd64.whl"
34
+ if os.path.exists(target_whl):
35
+ install_lib = self.install_lib
36
+ with zipfile.ZipFile(target_whl, 'r') as zip_ref:
37
+ zip_ref.extractall(install_lib)
38
+
39
+ with open("README.md", "r", encoding="utf-8") as fh:
40
+ long_description = fh.read()
41
+
42
+ setup(
43
+ name="kinako-llama-cpp",
44
+ version="0.3.24.post1", # 修正版としてバージョンを少し上げます
45
+ author="sora_sakurai/toshiaki_sakurai",
46
+ author_email="t301537@mbr.nifty.com",
47
+ description="[Beta] A custom llama-cpp-python wheel tailored for Intel 10th Gen CPUs on Windows 11.",
48
+ long_description=long_description,
49
+ long_description_content_type="text/markdown",
50
+ url="https://github.com/sak301537/gemma4-portable-for-Windows11",
51
+ classifiers=[
52
+ "Programming Language :: Python :: 3",
53
+ "License :: OSI Approved :: MIT License",
54
+ "Operating System :: Microsoft :: Windows :: Windows 11",
55
+ ],
56
+ python_requires=">=3.10",
57
+ packages=find_packages(),
58
+ package_data={
59
+ "": ["*.whl"], # ビルド時に元のwheelファイルを確実に含める
60
+ },
61
+ include_package_data=True,
62
+ cmdclass={
63
+ 'install': CustomInstallCommand, # 上記の解凍展開処理を登録
64
+ },
65
+ )