kinako-llama-cpp 0.3.24.post1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- kinako_llama_cpp-0.3.24.post1/PKG-INFO +148 -0
- kinako_llama_cpp-0.3.24.post1/README.md +127 -0
- kinako_llama_cpp-0.3.24.post1/kinako_llama_cpp.egg-info/PKG-INFO +148 -0
- kinako_llama_cpp-0.3.24.post1/kinako_llama_cpp.egg-info/SOURCES.txt +6 -0
- kinako_llama_cpp-0.3.24.post1/kinako_llama_cpp.egg-info/dependency_links.txt +1 -0
- kinako_llama_cpp-0.3.24.post1/kinako_llama_cpp.egg-info/top_level.txt +1 -0
- kinako_llama_cpp-0.3.24.post1/setup.cfg +4 -0
- kinako_llama_cpp-0.3.24.post1/setup.py +65 -0
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: kinako-llama-cpp
|
|
3
|
+
Version: 0.3.24.post1
|
|
4
|
+
Summary: [Beta] A custom llama-cpp-python wheel tailored for Intel 10th Gen CPUs on Windows 11.
|
|
5
|
+
Home-page: https://github.com/sak301537/gemma4-portable-for-Windows11
|
|
6
|
+
Author: sora_sakurai/toshiaki_sakurai
|
|
7
|
+
Author-email: t301537@mbr.nifty.com
|
|
8
|
+
Classifier: Programming Language :: Python :: 3
|
|
9
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
10
|
+
Classifier: Operating System :: Microsoft :: Windows :: Windows 11
|
|
11
|
+
Requires-Python: >=3.10
|
|
12
|
+
Description-Content-Type: text/markdown
|
|
13
|
+
Dynamic: author
|
|
14
|
+
Dynamic: author-email
|
|
15
|
+
Dynamic: classifier
|
|
16
|
+
Dynamic: description
|
|
17
|
+
Dynamic: description-content-type
|
|
18
|
+
Dynamic: home-page
|
|
19
|
+
Dynamic: requires-python
|
|
20
|
+
Dynamic: summary
|
|
21
|
+
|
|
22
|
+
# kinako-llama-cpp-python
|
|
23
|
+
**⚠️ This package is an early experimental/testing release.
|
|
24
|
+
This is a custom-built, wheel-embedded distribution of llama-cpp-python optimized to run LLMs like Gemma 4 smoothly on everyday, mid-to-low spec PCs running Windows 11 (specifically tested on Intel 7th, 8th, 9th, 10th, and 13th Gen CPUs).
|
|
25
|
+
|
|
26
|
+
**⚠️ 本パッケージは検証用の早期(取り急ぎの)リリースです。**
|
|
27
|
+
知人・友人環境での動作検証を通じ、今後記述や構成を随時修正・アップデートしていく前提のプロジェクトです。
|
|
28
|
+
本パッケージは、Windowsの身近な低スペックPC環境において、軽量LLMを動作させることを目的とした `llama-cpp-python` のカスタムビルド(Wheel同梱版)です。
|
|
29
|
+
|
|
30
|
+
> **開発者より:**
|
|
31
|
+
> 現時点ではWindows+CPU AVX2環境での動作を想定していますが、構成の厳密な検証はこれからです。(Intel 7,8,9,10,13世代/Win11で確認済)
|
|
32
|
+
> 実際に `pip install` してみて動いた・動かなかったなどのフィードバック(動作報告)をもとに、セットアップ設定やドキュメントを順次修正していきます。
|
|
33
|
+
|
|
34
|
+
### インストール方法
|
|
35
|
+
```bash
|
|
36
|
+
pip install kinako-llama-cpp
|
|
37
|
+
|
|
38
|
+
Important (English):
|
|
39
|
+
If you already have the official llama-cpp-python installed in your environment, installing this package directly may cause module conflicts inside the llama_cpp folder.
|
|
40
|
+
Please make sure to uninstall the official version before installing this package to prevent any broken dependencies.
|
|
41
|
+
|
|
42
|
+
Bash
|
|
43
|
+
### ⚠️ すでに公式の llama-cpp-python をインストールしている方へ / For Existing Users
|
|
44
|
+
|
|
45
|
+
**重要 (Japanese):**
|
|
46
|
+
もし、すでに本家(公式)の `llama-cpp-python` をインストール済みの環境に本パッケージを導入する場合、中身のモジュール(`llama_cpp`)が衝突して正常に動作しなくなる恐れがあります。
|
|
47
|
+
本パッケージを試す前に、必ず一度既存のパッケージをアンインストールしてください。
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
# 既存の公式版を一度削除する
|
|
51
|
+
pip uninstall llama-cpp-python -y
|
|
52
|
+
|
|
53
|
+
# その後、本パッケージをインストールする
|
|
54
|
+
pip install kinako-llama-cpp
|
|
55
|
+
|
|
56
|
+
> Python program sample
|
|
57
|
+
>
|
|
58
|
+
import os
|
|
59
|
+
import sys
|
|
60
|
+
from llama_cpp import Llama
|
|
61
|
+
|
|
62
|
+
# ── 設定項目 / Configuration ──
|
|
63
|
+
MODEL_PATH = r"C:\pythonfiles\llm\gemma-4-E2B-it-RotorQuant-Q8_0.gguf"
|
|
64
|
+
|
|
65
|
+
print("--- LLMモデルを読み込んでいます(数秒〜十数秒かかります)... ---")
|
|
66
|
+
print("--- Loading LLM model (This may take a few seconds)... ---")
|
|
67
|
+
|
|
68
|
+
try:
|
|
69
|
+
llm = Llama(
|
|
70
|
+
model_path=MODEL_PATH,
|
|
71
|
+
n_ctx=2048,
|
|
72
|
+
n_batch=128, # 過去の履歴を読み直す速度をCPU向けに最適化 / Optimized for CPU
|
|
73
|
+
n_gpu_layers=0 # 完全CPUモード / CPU Only Mode
|
|
74
|
+
)
|
|
75
|
+
except Exception as e:
|
|
76
|
+
print(f"[ERROR] Model file not found or could not be loaded.")
|
|
77
|
+
print(f"\n【エラー】モデルファイルが見つからないか、読み込めませんでした。")
|
|
78
|
+
print(f"Please check the path: {MODEL_PATH}")
|
|
79
|
+
input("\nPress Enter to exit...")
|
|
80
|
+
sys.exit(1)
|
|
81
|
+
|
|
82
|
+
print("\n=========================================")
|
|
83
|
+
print(" Windows CPU-driven PC Chat")
|
|
84
|
+
print(" Type 'exit' to quit the chat. / 終了するには「exit」と入力してください。")
|
|
85
|
+
print("=========================================\n")
|
|
86
|
+
|
|
87
|
+
# 過去の会話履歴を保存するリスト / Chat history list
|
|
88
|
+
chat_history = []
|
|
89
|
+
|
|
90
|
+
while True:
|
|
91
|
+
try:
|
|
92
|
+
user_input = input("You / あなた: ")
|
|
93
|
+
|
|
94
|
+
if not user_input.strip():
|
|
95
|
+
continue
|
|
96
|
+
|
|
97
|
+
if user_input.strip().lower() == "exit":
|
|
98
|
+
print("Closing chat. Thank you! / チャットを終了します。お疲れ様でした!")
|
|
99
|
+
break
|
|
100
|
+
|
|
101
|
+
# -------------------------------------------------
|
|
102
|
+
# プロンプトの組み立て / Prompt Construction
|
|
103
|
+
# -------------------------------------------------
|
|
104
|
+
# 多言語に対応できるよう、指示を英語に統一し「ユーザーの言語に合わせる」ルールを追加
|
|
105
|
+
system_prompt = "System: Act as a helpful AI assistant. Reply kindly and politely within 400 characters. (Please respond in the same language the user speaks.)\n"
|
|
106
|
+
#日本語のみの場合
|
|
107
|
+
#system_prompt = "System: 親切なAIとして、400文字以内で、優しく丁寧に回答してください。\n"
|
|
108
|
+
|
|
109
|
+
# 履歴を直近の2回分(発言数に直すと最大4つ)に絞る
|
|
110
|
+
recent_history = chat_history[-4:]
|
|
111
|
+
history_text = "".join(recent_history)
|
|
112
|
+
|
|
113
|
+
# 今回の質問をドッキング
|
|
114
|
+
current_prompt = f"User: {user_input}\nAI: "
|
|
115
|
+
full_prompt = system_prompt + history_text + current_prompt
|
|
116
|
+
|
|
117
|
+
print("\nAI: ", end="", flush=True)
|
|
118
|
+
|
|
119
|
+
# ストリーミング実行 / Streaming Execution
|
|
120
|
+
response_stream = llm(
|
|
121
|
+
full_prompt,
|
|
122
|
+
max_tokens=500, # 余裕を持たせた上限設定
|
|
123
|
+
stop=["User:", "\nUser:", "System:", "\nSystem:"],
|
|
124
|
+
echo=False,
|
|
125
|
+
stream=True
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
# 1文字ずつ出力しながら、今回の回答テキストを記録
|
|
129
|
+
ai_response = ""
|
|
130
|
+
for chunk in response_stream:
|
|
131
|
+
text = chunk["choices"][0]["text"]
|
|
132
|
+
sys.stdout.write(text)
|
|
133
|
+
sys.stdout.flush()
|
|
134
|
+
ai_response += text
|
|
135
|
+
|
|
136
|
+
print("\n-----------------------------------------")
|
|
137
|
+
|
|
138
|
+
# ── 今回の会話を履歴リストに追加 ──
|
|
139
|
+
chat_history.append(f"User: {user_input}\n")
|
|
140
|
+
chat_history.append(f"AI: {ai_response.strip()}\n")
|
|
141
|
+
|
|
142
|
+
except KeyboardInterrupt:
|
|
143
|
+
print("\n Exit requested. Closing chat./ チャットを終了します。")
|
|
144
|
+
print("\n")
|
|
145
|
+
break
|
|
146
|
+
except Exception as e:
|
|
147
|
+
print(f"\n An error occurred / エラーが発生しました {e}")
|
|
148
|
+
break
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
# kinako-llama-cpp-python
|
|
2
|
+
**⚠️ This package is an early experimental/testing release.
|
|
3
|
+
This is a custom-built, wheel-embedded distribution of llama-cpp-python optimized to run LLMs like Gemma 4 smoothly on everyday, mid-to-low spec PCs running Windows 11 (specifically tested on Intel 7th, 8th, 9th, 10th, and 13th Gen CPUs).
|
|
4
|
+
|
|
5
|
+
**⚠️ 本パッケージは検証用の早期(取り急ぎの)リリースです。**
|
|
6
|
+
知人・友人環境での動作検証を通じ、今後記述や構成を随時修正・アップデートしていく前提のプロジェクトです。
|
|
7
|
+
本パッケージは、Windowsの身近な低スペックPC環境において、軽量LLMを動作させることを目的とした `llama-cpp-python` のカスタムビルド(Wheel同梱版)です。
|
|
8
|
+
|
|
9
|
+
> **開発者より:**
|
|
10
|
+
> 現時点ではWindows+CPU AVX2環境での動作を想定していますが、構成の厳密な検証はこれからです。(Intel 7,8,9,10,13世代/Win11で確認済)
|
|
11
|
+
> 実際に `pip install` してみて動いた・動かなかったなどのフィードバック(動作報告)をもとに、セットアップ設定やドキュメントを順次修正していきます。
|
|
12
|
+
|
|
13
|
+
### インストール方法
|
|
14
|
+
```bash
|
|
15
|
+
pip install kinako-llama-cpp
|
|
16
|
+
|
|
17
|
+
Important (English):
|
|
18
|
+
If you already have the official llama-cpp-python installed in your environment, installing this package directly may cause module conflicts inside the llama_cpp folder.
|
|
19
|
+
Please make sure to uninstall the official version before installing this package to prevent any broken dependencies.
|
|
20
|
+
|
|
21
|
+
Bash
|
|
22
|
+
### ⚠️ すでに公式の llama-cpp-python をインストールしている方へ / For Existing Users
|
|
23
|
+
|
|
24
|
+
**重要 (Japanese):**
|
|
25
|
+
もし、すでに本家(公式)の `llama-cpp-python` をインストール済みの環境に本パッケージを導入する場合、中身のモジュール(`llama_cpp`)が衝突して正常に動作しなくなる恐れがあります。
|
|
26
|
+
本パッケージを試す前に、必ず一度既存のパッケージをアンインストールしてください。
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
# 既存の公式版を一度削除する
|
|
30
|
+
pip uninstall llama-cpp-python -y
|
|
31
|
+
|
|
32
|
+
# その後、本パッケージをインストールする
|
|
33
|
+
pip install kinako-llama-cpp
|
|
34
|
+
|
|
35
|
+
> Python program sample
|
|
36
|
+
>
|
|
37
|
+
import os
|
|
38
|
+
import sys
|
|
39
|
+
from llama_cpp import Llama
|
|
40
|
+
|
|
41
|
+
# ── 設定項目 / Configuration ──
|
|
42
|
+
MODEL_PATH = r"C:\pythonfiles\llm\gemma-4-E2B-it-RotorQuant-Q8_0.gguf"
|
|
43
|
+
|
|
44
|
+
print("--- LLMモデルを読み込んでいます(数秒〜十数秒かかります)... ---")
|
|
45
|
+
print("--- Loading LLM model (This may take a few seconds)... ---")
|
|
46
|
+
|
|
47
|
+
try:
|
|
48
|
+
llm = Llama(
|
|
49
|
+
model_path=MODEL_PATH,
|
|
50
|
+
n_ctx=2048,
|
|
51
|
+
n_batch=128, # 過去の履歴を読み直す速度をCPU向けに最適化 / Optimized for CPU
|
|
52
|
+
n_gpu_layers=0 # 完全CPUモード / CPU Only Mode
|
|
53
|
+
)
|
|
54
|
+
except Exception as e:
|
|
55
|
+
print(f"[ERROR] Model file not found or could not be loaded.")
|
|
56
|
+
print(f"\n【エラー】モデルファイルが見つからないか、読み込めませんでした。")
|
|
57
|
+
print(f"Please check the path: {MODEL_PATH}")
|
|
58
|
+
input("\nPress Enter to exit...")
|
|
59
|
+
sys.exit(1)
|
|
60
|
+
|
|
61
|
+
print("\n=========================================")
|
|
62
|
+
print(" Windows CPU-driven PC Chat")
|
|
63
|
+
print(" Type 'exit' to quit the chat. / 終了するには「exit」と入力してください。")
|
|
64
|
+
print("=========================================\n")
|
|
65
|
+
|
|
66
|
+
# 過去の会話履歴を保存するリスト / Chat history list
|
|
67
|
+
chat_history = []
|
|
68
|
+
|
|
69
|
+
while True:
|
|
70
|
+
try:
|
|
71
|
+
user_input = input("You / あなた: ")
|
|
72
|
+
|
|
73
|
+
if not user_input.strip():
|
|
74
|
+
continue
|
|
75
|
+
|
|
76
|
+
if user_input.strip().lower() == "exit":
|
|
77
|
+
print("Closing chat. Thank you! / チャットを終了します。お疲れ様でした!")
|
|
78
|
+
break
|
|
79
|
+
|
|
80
|
+
# -------------------------------------------------
|
|
81
|
+
# プロンプトの組み立て / Prompt Construction
|
|
82
|
+
# -------------------------------------------------
|
|
83
|
+
# 多言語に対応できるよう、指示を英語に統一し「ユーザーの言語に合わせる」ルールを追加
|
|
84
|
+
system_prompt = "System: Act as a helpful AI assistant. Reply kindly and politely within 400 characters. (Please respond in the same language the user speaks.)\n"
|
|
85
|
+
#日本語のみの場合
|
|
86
|
+
#system_prompt = "System: 親切なAIとして、400文字以内で、優しく丁寧に回答してください。\n"
|
|
87
|
+
|
|
88
|
+
# 履歴を直近の2回分(発言数に直すと最大4つ)に絞る
|
|
89
|
+
recent_history = chat_history[-4:]
|
|
90
|
+
history_text = "".join(recent_history)
|
|
91
|
+
|
|
92
|
+
# 今回の質問をドッキング
|
|
93
|
+
current_prompt = f"User: {user_input}\nAI: "
|
|
94
|
+
full_prompt = system_prompt + history_text + current_prompt
|
|
95
|
+
|
|
96
|
+
print("\nAI: ", end="", flush=True)
|
|
97
|
+
|
|
98
|
+
# ストリーミング実行 / Streaming Execution
|
|
99
|
+
response_stream = llm(
|
|
100
|
+
full_prompt,
|
|
101
|
+
max_tokens=500, # 余裕を持たせた上限設定
|
|
102
|
+
stop=["User:", "\nUser:", "System:", "\nSystem:"],
|
|
103
|
+
echo=False,
|
|
104
|
+
stream=True
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
# 1文字ずつ出力しながら、今回の回答テキストを記録
|
|
108
|
+
ai_response = ""
|
|
109
|
+
for chunk in response_stream:
|
|
110
|
+
text = chunk["choices"][0]["text"]
|
|
111
|
+
sys.stdout.write(text)
|
|
112
|
+
sys.stdout.flush()
|
|
113
|
+
ai_response += text
|
|
114
|
+
|
|
115
|
+
print("\n-----------------------------------------")
|
|
116
|
+
|
|
117
|
+
# ── 今回の会話を履歴リストに追加 ──
|
|
118
|
+
chat_history.append(f"User: {user_input}\n")
|
|
119
|
+
chat_history.append(f"AI: {ai_response.strip()}\n")
|
|
120
|
+
|
|
121
|
+
except KeyboardInterrupt:
|
|
122
|
+
print("\n Exit requested. Closing chat./ チャットを終了します。")
|
|
123
|
+
print("\n")
|
|
124
|
+
break
|
|
125
|
+
except Exception as e:
|
|
126
|
+
print(f"\n An error occurred / エラーが発生しました {e}")
|
|
127
|
+
break
|
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: kinako-llama-cpp
|
|
3
|
+
Version: 0.3.24.post1
|
|
4
|
+
Summary: [Beta] A custom llama-cpp-python wheel tailored for Intel 10th Gen CPUs on Windows 11.
|
|
5
|
+
Home-page: https://github.com/sak301537/gemma4-portable-for-Windows11
|
|
6
|
+
Author: sora_sakurai/toshiaki_sakurai
|
|
7
|
+
Author-email: t301537@mbr.nifty.com
|
|
8
|
+
Classifier: Programming Language :: Python :: 3
|
|
9
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
10
|
+
Classifier: Operating System :: Microsoft :: Windows :: Windows 11
|
|
11
|
+
Requires-Python: >=3.10
|
|
12
|
+
Description-Content-Type: text/markdown
|
|
13
|
+
Dynamic: author
|
|
14
|
+
Dynamic: author-email
|
|
15
|
+
Dynamic: classifier
|
|
16
|
+
Dynamic: description
|
|
17
|
+
Dynamic: description-content-type
|
|
18
|
+
Dynamic: home-page
|
|
19
|
+
Dynamic: requires-python
|
|
20
|
+
Dynamic: summary
|
|
21
|
+
|
|
22
|
+
# kinako-llama-cpp-python
|
|
23
|
+
**⚠️ This package is an early experimental/testing release.
|
|
24
|
+
This is a custom-built, wheel-embedded distribution of llama-cpp-python optimized to run LLMs like Gemma 4 smoothly on everyday, mid-to-low spec PCs running Windows 11 (specifically tested on Intel 7th, 8th, 9th, 10th, and 13th Gen CPUs).
|
|
25
|
+
|
|
26
|
+
**⚠️ 本パッケージは検証用の早期(取り急ぎの)リリースです。**
|
|
27
|
+
知人・友人環境での動作検証を通じ、今後記述や構成を随時修正・アップデートしていく前提のプロジェクトです。
|
|
28
|
+
本パッケージは、Windowsの身近な低スペックPC環境において、軽量LLMを動作させることを目的とした `llama-cpp-python` のカスタムビルド(Wheel同梱版)です。
|
|
29
|
+
|
|
30
|
+
> **開発者より:**
|
|
31
|
+
> 現時点ではWindows+CPU AVX2環境での動作を想定していますが、構成の厳密な検証はこれからです。(Intel 7,8,9,10,13世代/Win11で確認済)
|
|
32
|
+
> 実際に `pip install` してみて動いた・動かなかったなどのフィードバック(動作報告)をもとに、セットアップ設定やドキュメントを順次修正していきます。
|
|
33
|
+
|
|
34
|
+
### インストール方法
|
|
35
|
+
```bash
|
|
36
|
+
pip install kinako-llama-cpp
|
|
37
|
+
|
|
38
|
+
Important (English):
|
|
39
|
+
If you already have the official llama-cpp-python installed in your environment, installing this package directly may cause module conflicts inside the llama_cpp folder.
|
|
40
|
+
Please make sure to uninstall the official version before installing this package to prevent any broken dependencies.
|
|
41
|
+
|
|
42
|
+
Bash
|
|
43
|
+
### ⚠️ すでに公式の llama-cpp-python をインストールしている方へ / For Existing Users
|
|
44
|
+
|
|
45
|
+
**重要 (Japanese):**
|
|
46
|
+
もし、すでに本家(公式)の `llama-cpp-python` をインストール済みの環境に本パッケージを導入する場合、中身のモジュール(`llama_cpp`)が衝突して正常に動作しなくなる恐れがあります。
|
|
47
|
+
本パッケージを試す前に、必ず一度既存のパッケージをアンインストールしてください。
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
# 既存の公式版を一度削除する
|
|
51
|
+
pip uninstall llama-cpp-python -y
|
|
52
|
+
|
|
53
|
+
# その後、本パッケージをインストールする
|
|
54
|
+
pip install kinako-llama-cpp
|
|
55
|
+
|
|
56
|
+
> Python program sample
|
|
57
|
+
>
|
|
58
|
+
import os
|
|
59
|
+
import sys
|
|
60
|
+
from llama_cpp import Llama
|
|
61
|
+
|
|
62
|
+
# ── 設定項目 / Configuration ──
|
|
63
|
+
MODEL_PATH = r"C:\pythonfiles\llm\gemma-4-E2B-it-RotorQuant-Q8_0.gguf"
|
|
64
|
+
|
|
65
|
+
print("--- LLMモデルを読み込んでいます(数秒〜十数秒かかります)... ---")
|
|
66
|
+
print("--- Loading LLM model (This may take a few seconds)... ---")
|
|
67
|
+
|
|
68
|
+
try:
|
|
69
|
+
llm = Llama(
|
|
70
|
+
model_path=MODEL_PATH,
|
|
71
|
+
n_ctx=2048,
|
|
72
|
+
n_batch=128, # 過去の履歴を読み直す速度をCPU向けに最適化 / Optimized for CPU
|
|
73
|
+
n_gpu_layers=0 # 完全CPUモード / CPU Only Mode
|
|
74
|
+
)
|
|
75
|
+
except Exception as e:
|
|
76
|
+
print(f"[ERROR] Model file not found or could not be loaded.")
|
|
77
|
+
print(f"\n【エラー】モデルファイルが見つからないか、読み込めませんでした。")
|
|
78
|
+
print(f"Please check the path: {MODEL_PATH}")
|
|
79
|
+
input("\nPress Enter to exit...")
|
|
80
|
+
sys.exit(1)
|
|
81
|
+
|
|
82
|
+
print("\n=========================================")
|
|
83
|
+
print(" Windows CPU-driven PC Chat")
|
|
84
|
+
print(" Type 'exit' to quit the chat. / 終了するには「exit」と入力してください。")
|
|
85
|
+
print("=========================================\n")
|
|
86
|
+
|
|
87
|
+
# 過去の会話履歴を保存するリスト / Chat history list
|
|
88
|
+
chat_history = []
|
|
89
|
+
|
|
90
|
+
while True:
|
|
91
|
+
try:
|
|
92
|
+
user_input = input("You / あなた: ")
|
|
93
|
+
|
|
94
|
+
if not user_input.strip():
|
|
95
|
+
continue
|
|
96
|
+
|
|
97
|
+
if user_input.strip().lower() == "exit":
|
|
98
|
+
print("Closing chat. Thank you! / チャットを終了します。お疲れ様でした!")
|
|
99
|
+
break
|
|
100
|
+
|
|
101
|
+
# -------------------------------------------------
|
|
102
|
+
# プロンプトの組み立て / Prompt Construction
|
|
103
|
+
# -------------------------------------------------
|
|
104
|
+
# 多言語に対応できるよう、指示を英語に統一し「ユーザーの言語に合わせる」ルールを追加
|
|
105
|
+
system_prompt = "System: Act as a helpful AI assistant. Reply kindly and politely within 400 characters. (Please respond in the same language the user speaks.)\n"
|
|
106
|
+
#日本語のみの場合
|
|
107
|
+
#system_prompt = "System: 親切なAIとして、400文字以内で、優しく丁寧に回答してください。\n"
|
|
108
|
+
|
|
109
|
+
# 履歴を直近の2回分(発言数に直すと最大4つ)に絞る
|
|
110
|
+
recent_history = chat_history[-4:]
|
|
111
|
+
history_text = "".join(recent_history)
|
|
112
|
+
|
|
113
|
+
# 今回の質問をドッキング
|
|
114
|
+
current_prompt = f"User: {user_input}\nAI: "
|
|
115
|
+
full_prompt = system_prompt + history_text + current_prompt
|
|
116
|
+
|
|
117
|
+
print("\nAI: ", end="", flush=True)
|
|
118
|
+
|
|
119
|
+
# ストリーミング実行 / Streaming Execution
|
|
120
|
+
response_stream = llm(
|
|
121
|
+
full_prompt,
|
|
122
|
+
max_tokens=500, # 余裕を持たせた上限設定
|
|
123
|
+
stop=["User:", "\nUser:", "System:", "\nSystem:"],
|
|
124
|
+
echo=False,
|
|
125
|
+
stream=True
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
# 1文字ずつ出力しながら、今回の回答テキストを記録
|
|
129
|
+
ai_response = ""
|
|
130
|
+
for chunk in response_stream:
|
|
131
|
+
text = chunk["choices"][0]["text"]
|
|
132
|
+
sys.stdout.write(text)
|
|
133
|
+
sys.stdout.flush()
|
|
134
|
+
ai_response += text
|
|
135
|
+
|
|
136
|
+
print("\n-----------------------------------------")
|
|
137
|
+
|
|
138
|
+
# ── 今回の会話を履歴リストに追加 ──
|
|
139
|
+
chat_history.append(f"User: {user_input}\n")
|
|
140
|
+
chat_history.append(f"AI: {ai_response.strip()}\n")
|
|
141
|
+
|
|
142
|
+
except KeyboardInterrupt:
|
|
143
|
+
print("\n Exit requested. Closing chat./ チャットを終了します。")
|
|
144
|
+
print("\n")
|
|
145
|
+
break
|
|
146
|
+
except Exception as e:
|
|
147
|
+
print(f"\n An error occurred / エラーが発生しました {e}")
|
|
148
|
+
break
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import zipfile
|
|
3
|
+
from setuptools import setup, find_packages
|
|
4
|
+
from setuptools.command.install import install
|
|
5
|
+
|
|
6
|
+
def check_conflict_and_abort():
|
|
7
|
+
try:
|
|
8
|
+
import llama_cpp
|
|
9
|
+
# すでに llama_cpp が存在する場合、詳細なエラーメッセージを出して終了
|
|
10
|
+
print("\n" + "="*70)
|
|
11
|
+
print("\nConflict detected: 'llama-cpp-python' or a similar module is already installed.")
|
|
12
|
+
print(" To prevent broken dependencies, please uninstall it first and try again.")
|
|
13
|
+
print("【⚠️ インストールを中断しました / Installation Aborted】")
|
|
14
|
+
print("\n[JA] すでに公式の 'llama-cpp-python' 等のモジュールが環境に存在します。")
|
|
15
|
+
print(" 衝突を防ぐため、一度既存のパッケージを削除してから再試行してください。")
|
|
16
|
+
|
|
17
|
+
print("\n >>> pip uninstall llama-cpp-python -y <<<")
|
|
18
|
+
print("="*70 + "\n")
|
|
19
|
+
sys.exit(1)
|
|
20
|
+
except ImportError:
|
|
21
|
+
pass
|
|
22
|
+
|
|
23
|
+
# ── インストール時に、同梱した本物のWheelを解凍して中身を展開する ──
|
|
24
|
+
class CustomInstallCommand(install):
|
|
25
|
+
def run(self):
|
|
26
|
+
# 1. まず本家との衝突がないかチェック
|
|
27
|
+
check_conflict_and_abort()
|
|
28
|
+
|
|
29
|
+
# 2. 衝突がなければ通常のインストール処理を実行
|
|
30
|
+
super().run()
|
|
31
|
+
|
|
32
|
+
# 3. 同梱されている本物のwheelファイルを展開
|
|
33
|
+
target_whl = "llama_cpp_python-0.3.24-py3-none-win_amd64.whl"
|
|
34
|
+
if os.path.exists(target_whl):
|
|
35
|
+
install_lib = self.install_lib
|
|
36
|
+
with zipfile.ZipFile(target_whl, 'r') as zip_ref:
|
|
37
|
+
zip_ref.extractall(install_lib)
|
|
38
|
+
|
|
39
|
+
with open("README.md", "r", encoding="utf-8") as fh:
|
|
40
|
+
long_description = fh.read()
|
|
41
|
+
|
|
42
|
+
setup(
|
|
43
|
+
name="kinako-llama-cpp",
|
|
44
|
+
version="0.3.24.post1", # 修正版としてバージョンを少し上げます
|
|
45
|
+
author="sora_sakurai/toshiaki_sakurai",
|
|
46
|
+
author_email="t301537@mbr.nifty.com",
|
|
47
|
+
description="[Beta] A custom llama-cpp-python wheel tailored for Intel 10th Gen CPUs on Windows 11.",
|
|
48
|
+
long_description=long_description,
|
|
49
|
+
long_description_content_type="text/markdown",
|
|
50
|
+
url="https://github.com/sak301537/gemma4-portable-for-Windows11",
|
|
51
|
+
classifiers=[
|
|
52
|
+
"Programming Language :: Python :: 3",
|
|
53
|
+
"License :: OSI Approved :: MIT License",
|
|
54
|
+
"Operating System :: Microsoft :: Windows :: Windows 11",
|
|
55
|
+
],
|
|
56
|
+
python_requires=">=3.10",
|
|
57
|
+
packages=find_packages(),
|
|
58
|
+
package_data={
|
|
59
|
+
"": ["*.whl"], # ビルド時に元のwheelファイルを確実に含める
|
|
60
|
+
},
|
|
61
|
+
include_package_data=True,
|
|
62
|
+
cmdclass={
|
|
63
|
+
'install': CustomInstallCommand, # 上記の解凍展開処理を登録
|
|
64
|
+
},
|
|
65
|
+
)
|