lab-copy 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
lab_copy/__init__.py
ADDED
|
@@ -0,0 +1,459 @@
|
|
|
1
|
+
codes = {
|
|
2
|
+
1: """import numpy as np
|
|
3
|
+
import matplotlib.pyplot as plt
|
|
4
|
+
|
|
5
|
+
# Gridworld Environment Setup
|
|
6
|
+
GRID_SIZE = 5
|
|
7
|
+
ACTIONS = ['up', 'down', 'left', 'right']
|
|
8
|
+
ACTION_DICT = {'up': (-1, 0), 'down': (1, 0), 'left': (0, -1), 'right': (0, 1)}
|
|
9
|
+
GOAL_STATE = (4, 4)
|
|
10
|
+
MAX_STEPS = 50
|
|
11
|
+
|
|
12
|
+
def is_valid_state(state):
|
|
13
|
+
return 0 <= state[0] < GRID_SIZE and 0 <= state[1] < GRID_SIZE
|
|
14
|
+
|
|
15
|
+
def step(state, action):
|
|
16
|
+
move = ACTION_DICT[action]
|
|
17
|
+
new_state = (state[0] + move[0], state[1] + move[1])
|
|
18
|
+
if not is_valid_state(new_state):
|
|
19
|
+
new_state = state # stay if move goes out of bounds
|
|
20
|
+
reward = 10 if new_state == GOAL_STATE else 0
|
|
21
|
+
done = (new_state == GOAL_STATE)
|
|
22
|
+
return new_state, reward, done
|
|
23
|
+
|
|
24
|
+
# Agent Loop
|
|
25
|
+
def run_episode():
|
|
26
|
+
state = (0, 0)
|
|
27
|
+
total_reward = 0
|
|
28
|
+
trajectory = [state]
|
|
29
|
+
for step_num in range(MAX_STEPS):
|
|
30
|
+
action = np.random.choice(ACTIONS) # Random policy
|
|
31
|
+
next_state, reward, done = step(state, action)
|
|
32
|
+
trajectory.append(next_state)
|
|
33
|
+
total_reward += reward
|
|
34
|
+
state = next_state
|
|
35
|
+
if done:
|
|
36
|
+
break
|
|
37
|
+
return trajectory, total_reward
|
|
38
|
+
|
|
39
|
+
# Run 10 episodes and visualize one
|
|
40
|
+
for ep in range(10):
|
|
41
|
+
traj, reward = run_episode()
|
|
42
|
+
print(f"Episode {ep+1}: Total Reward = {reward}, Steps = {len(traj)}")
|
|
43
|
+
|
|
44
|
+
# Visualize trajectory of the last episode
|
|
45
|
+
def plot_trajectory(trajectory):
|
|
46
|
+
grid = np.zeros((GRID_SIZE, GRID_SIZE))
|
|
47
|
+
for (x, y) in trajectory:
|
|
48
|
+
grid[x, y] += 1
|
|
49
|
+
plt.imshow(grid, cmap='Blues', origin='upper')
|
|
50
|
+
plt.title("Agent Trajectory Heatmap")
|
|
51
|
+
plt.colorbar(label="Visits")
|
|
52
|
+
plt.scatter(0, 0, c='green', s=100, label='Start')
|
|
53
|
+
plt.scatter(4, 4, c='red', s=100, label='Goal')
|
|
54
|
+
plt.legend()
|
|
55
|
+
plt.grid(True)
|
|
56
|
+
plt.show()
|
|
57
|
+
|
|
58
|
+
plot_trajectory(traj)""",
|
|
59
|
+
2: """import numpy as np
|
|
60
|
+
import matplotlib.pyplot as plt
|
|
61
|
+
|
|
62
|
+
class EpsilonGreedyAgent:
|
|
63
|
+
def __init__(self, n_arms, epsilon):
|
|
64
|
+
self.n_arms = n_arms
|
|
65
|
+
self.epsilon = epsilon
|
|
66
|
+
self.counts = np.zeros(n_arms) # Number of times each arm was pulled
|
|
67
|
+
self.values = np.zeros(n_arms) # Estimated value (CTR) for each arm
|
|
68
|
+
self.total_reward = 0
|
|
69
|
+
self.actions = []
|
|
70
|
+
self.rewards = []
|
|
71
|
+
|
|
72
|
+
def select_action(self):
|
|
73
|
+
if np.random.rand() < self.epsilon:
|
|
74
|
+
return np.random.randint(self.n_arms) # Explore
|
|
75
|
+
else:
|
|
76
|
+
return np.argmax(self.values) # Exploit
|
|
77
|
+
|
|
78
|
+
def update(self, action, reward):
|
|
79
|
+
self.counts[action] += 1
|
|
80
|
+
self.values[action] += (reward - self.values[action]) / self.counts[action]
|
|
81
|
+
self.total_reward += reward
|
|
82
|
+
self.actions.append(action)
|
|
83
|
+
self.rewards.append(reward)
|
|
84
|
+
|
|
85
|
+
def simulate_bandit(true_ctrs, epsilon, n_rounds=1000):
|
|
86
|
+
n_arms = len(true_ctrs)
|
|
87
|
+
agent = EpsilonGreedyAgent(n_arms, epsilon)
|
|
88
|
+
optimal_arm = np.argmax(true_ctrs)
|
|
89
|
+
regrets = []
|
|
90
|
+
for t in range(n_rounds):
|
|
91
|
+
action = agent.select_action()
|
|
92
|
+
reward = np.random.rand() < true_ctrs[action]
|
|
93
|
+
agent.update(action, reward)
|
|
94
|
+
regret = true_ctrs[optimal_arm] - true_ctrs[action]
|
|
95
|
+
regrets.append(regret)
|
|
96
|
+
return agent, np.cumsum(regrets)
|
|
97
|
+
|
|
98
|
+
# ---------- Main Experiment ----------
|
|
99
|
+
np.random.seed(42)
|
|
100
|
+
n_arms = 10
|
|
101
|
+
true_ctrs = np.random.uniform(0.05, 0.5, n_arms)
|
|
102
|
+
print("True Click-Through Rates (CTR) per Ad:", np.round(true_ctrs, 2))
|
|
103
|
+
n_rounds = 1000
|
|
104
|
+
epsilons = [0.01, 0.1, 0.3]
|
|
105
|
+
agents = {}
|
|
106
|
+
regret_curves = {}
|
|
107
|
+
for epsilon in epsilons:
|
|
108
|
+
agent, regrets = simulate_bandit(true_ctrs, epsilon, n_rounds)
|
|
109
|
+
agents[epsilon] = agent
|
|
110
|
+
regret_curves[epsilon] = regrets
|
|
111
|
+
|
|
112
|
+
# ---------- Plotting Results ----------
|
|
113
|
+
plt.figure(figsize=(12, 5))
|
|
114
|
+
# Plot cumulative regret
|
|
115
|
+
plt.subplot(1, 2, 1)
|
|
116
|
+
for epsilon in epsilons:
|
|
117
|
+
plt.plot(regret_curves[epsilon], label=f'ε={epsilon}')
|
|
118
|
+
plt.title("Cumulative Regret")
|
|
119
|
+
plt.xlabel("Rounds")
|
|
120
|
+
plt.ylabel("Cumulative Regret")
|
|
121
|
+
plt.legend()
|
|
122
|
+
plt.grid(True)
|
|
123
|
+
|
|
124
|
+
plt.subplot(1, 2, 2)
|
|
125
|
+
bar_width = 0.25
|
|
126
|
+
x = np.arange(n_arms)
|
|
127
|
+
for i, epsilon in enumerate(epsilons):
|
|
128
|
+
plt.bar(x + i * bar_width,
|
|
129
|
+
agents[epsilon].values,
|
|
130
|
+
width=bar_width,
|
|
131
|
+
label=f'ε={epsilon}')
|
|
132
|
+
plt.axhline(np.max(true_ctrs), color='r', linestyle='--', label='Optimal CTR')
|
|
133
|
+
plt.xticks(x + bar_width, [f'Ad {i}' for i in range(n_arms)])
|
|
134
|
+
plt.ylabel("Estimated CTR")
|
|
135
|
+
plt.title("Estimated CTRs vs True CTR")
|
|
136
|
+
plt.legend()
|
|
137
|
+
plt.grid(True)
|
|
138
|
+
plt.tight_layout()
|
|
139
|
+
plt.show()""",
|
|
140
|
+
3: """import numpy as np
|
|
141
|
+
import mdptoolbox
|
|
142
|
+
import matplotlib.pyplot as plt
|
|
143
|
+
import random
|
|
144
|
+
|
|
145
|
+
# Grid Parameters
|
|
146
|
+
rows, cols = 5, 5
|
|
147
|
+
num_states = rows * cols
|
|
148
|
+
shelves = [(1, 1), (2, 2), (3, 3)] # Obstacle positions
|
|
149
|
+
actions = ['up', 'down', 'left', 'right']
|
|
150
|
+
num_actions = len(actions)
|
|
151
|
+
movement = {'up': (-1, 0), 'down': (1, 0), 'left': (0, -1), 'right': (0, 1)}
|
|
152
|
+
|
|
153
|
+
# Function to map (x, y) to index
|
|
154
|
+
def to_index(x, y):
|
|
155
|
+
return x * cols + y
|
|
156
|
+
|
|
157
|
+
# Function to randomly set a goal not on a shelf
|
|
158
|
+
def set_dynamic_goal():
|
|
159
|
+
possible = [(i, j) for i in range(rows) for j in range(cols) if (i, j) not in shelves]
|
|
160
|
+
return random.choice(possible)
|
|
161
|
+
|
|
162
|
+
# Randomly choose a goal position
|
|
163
|
+
goal_state = set_dynamic_goal()
|
|
164
|
+
print(" Current Goal Position:", goal_state)
|
|
165
|
+
|
|
166
|
+
# Transition and Reward Matrices
|
|
167
|
+
P = [np.zeros((num_states, num_states)) for _ in range(num_actions)]
|
|
168
|
+
R = np.zeros((num_states, num_actions))
|
|
169
|
+
|
|
170
|
+
for action_idx, action in enumerate(actions):
|
|
171
|
+
dx, dy = movement[action]
|
|
172
|
+
for x in range(rows):
|
|
173
|
+
for y in range(cols):
|
|
174
|
+
current_state = to_index(x, y)
|
|
175
|
+
if (x, y) == goal_state:
|
|
176
|
+
P[action_idx][current_state, current_state] = 1
|
|
177
|
+
R[current_state, action_idx] = 10
|
|
178
|
+
continue
|
|
179
|
+
|
|
180
|
+
outcomes = []
|
|
181
|
+
# Intended move (90%)
|
|
182
|
+
new_x, new_y = x + dx, y + dy
|
|
183
|
+
if (new_x, new_y) in shelves or not (0 <= new_x < rows and 0 <= new_y < cols):
|
|
184
|
+
new_state = current_state
|
|
185
|
+
reward = -5 if (new_x, new_y) in shelves else -1
|
|
186
|
+
else:
|
|
187
|
+
new_state = to_index(new_x, new_y)
|
|
188
|
+
reward = -1
|
|
189
|
+
outcomes.append((new_state, 0.9, reward))
|
|
190
|
+
|
|
191
|
+
# 10% misstep (wrong move in any of the other 3 directions)
|
|
192
|
+
other_actions = [a for i, a in enumerate(actions) if i != action_idx]
|
|
193
|
+
for mis_action in other_actions:
|
|
194
|
+
mx, my = movement[mis_action]
|
|
195
|
+
new_x, new_y = x + mx, y + my
|
|
196
|
+
if (new_x, new_y) in shelves or not (0 <= new_x < rows and 0 <= new_y < cols):
|
|
197
|
+
mis_state = current_state
|
|
198
|
+
mis_reward = -5 if (new_x, new_y) in shelves else -1
|
|
199
|
+
else:
|
|
200
|
+
mis_state = to_index(new_x, new_y)
|
|
201
|
+
mis_reward = -1
|
|
202
|
+
outcomes.append((mis_state, 0.1 / 3, mis_reward))
|
|
203
|
+
|
|
204
|
+
for s_next, prob, rew in outcomes:
|
|
205
|
+
P[action_idx][current_state, s_next] += prob
|
|
206
|
+
R[current_state, action_idx] += prob * rew # Expected reward
|
|
207
|
+
|
|
208
|
+
# Run Value Iteration
|
|
209
|
+
vi = mdptoolbox.mdp.ValueIteration(P, R, 0.9)
|
|
210
|
+
vi.run()
|
|
211
|
+
|
|
212
|
+
# Reshape policy to grid
|
|
213
|
+
policy_grid = np.array(vi.policy).reshape((rows, cols))
|
|
214
|
+
action_symbols = ['ā', 'ā', 'ā', 'ā']
|
|
215
|
+
policy_symbols = np.array([[action_symbols[a] for a in row] for row in policy_grid])
|
|
216
|
+
print("\\nš Optimal Policy Grid:")
|
|
217
|
+
print(policy_symbols)
|
|
218
|
+
|
|
219
|
+
# Plotting
|
|
220
|
+
plt.figure(figsize=(6, 6))
|
|
221
|
+
for x in range(rows):
|
|
222
|
+
for y in range(cols):
|
|
223
|
+
idx = to_index(x, y)
|
|
224
|
+
if (x, y) == goal_state:
|
|
225
|
+
plt.text(y, rows - x - 1, 'G', ha='center', va='center', fontsize=14, color='green')
|
|
226
|
+
elif (x, y) in shelves:
|
|
227
|
+
plt.text(y, rows - x - 1, 'S', ha='center', va='center', fontsize=14, color='red')
|
|
228
|
+
else:
|
|
229
|
+
plt.text(y, rows - x - 1, action_symbols[vi.policy[idx]], ha='center', va='center', fontsize=14)
|
|
230
|
+
plt.xticks(range(cols))
|
|
231
|
+
plt.yticks(range(rows))
|
|
232
|
+
plt.grid(True)
|
|
233
|
+
plt.title("Stochastic Optimal Policy with Dynamic Goal")
|
|
234
|
+
plt.show()""",
|
|
235
|
+
4: """import numpy as np
|
|
236
|
+
import matplotlib.pyplot as plt
|
|
237
|
+
from collections import defaultdict
|
|
238
|
+
import random
|
|
239
|
+
|
|
240
|
+
# ---------- Environment Setup ----------
|
|
241
|
+
GRID_SIZE = 6
|
|
242
|
+
ACTIONS = ['up', 'down', 'left', 'right']
|
|
243
|
+
ACTION_MAP = {'up': (-1, 0), 'down': (1, 0), 'left': (0, -1), 'right': (0, 1)}
|
|
244
|
+
MAX_STEPS = 50
|
|
245
|
+
DISCOUNT = 0.95
|
|
246
|
+
EPSILON = 0.1
|
|
247
|
+
EPISODES = 10000
|
|
248
|
+
|
|
249
|
+
# Static obstacles (permanent roadblocks)
|
|
250
|
+
static_obstacles = [(1, 3), (3, 2)]
|
|
251
|
+
hospital = (0, 0) # Ambulance dispatch center
|
|
252
|
+
|
|
253
|
+
# Rush hour control
|
|
254
|
+
def is_rush_hour(ep):
|
|
255
|
+
return ep % 1000 < 300 or ep % 1000 > 800 # Congested traffic windows
|
|
256
|
+
|
|
257
|
+
# Emergency severity and urgency
|
|
258
|
+
emergency_types = {
|
|
259
|
+
'minor': 20,
|
|
260
|
+
'moderate': 35,
|
|
261
|
+
'critical': 50
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
def is_valid(state):
|
|
265
|
+
x, y = state
|
|
266
|
+
return 0 <= x < GRID_SIZE and 0 <= y < GRID_SIZE
|
|
267
|
+
|
|
268
|
+
def get_dynamic_obstacles():
|
|
269
|
+
return [(2, 4), (4, 1), (3, 3)] if random.random() < 0.3 else []
|
|
270
|
+
|
|
271
|
+
def epsilon_greedy(state, Q):
|
|
272
|
+
if np.random.rand() < EPSILON or state not in Q:
|
|
273
|
+
return random.randint(0, len(ACTIONS) - 1)
|
|
274
|
+
else:
|
|
275
|
+
return np.argmax(Q[state])
|
|
276
|
+
|
|
277
|
+
def generate_emergency():
|
|
278
|
+
location = random.choice([
|
|
279
|
+
(i, j) for i in range(GRID_SIZE) for j in range(GRID_SIZE)
|
|
280
|
+
if (i, j) != hospital and (i, j) not in static_obstacles
|
|
281
|
+
])
|
|
282
|
+
severity = random.choice(list(emergency_types.keys()))
|
|
283
|
+
reward = emergency_types[severity]
|
|
284
|
+
return location, reward
|
|
285
|
+
|
|
286
|
+
# ---------- Monte Carlo Training ----------
|
|
287
|
+
Q = defaultdict(lambda: np.zeros(len(ACTIONS)))
|
|
288
|
+
Returns = defaultdict(list)
|
|
289
|
+
|
|
290
|
+
def run_episode(episode_num):
|
|
291
|
+
rush = is_rush_hour(episode_num)
|
|
292
|
+
prob_blocks = get_dynamic_obstacles()
|
|
293
|
+
all_obstacles = static_obstacles + prob_blocks
|
|
294
|
+
goal, goal_reward = generate_emergency()
|
|
295
|
+
state = hospital
|
|
296
|
+
episode = []
|
|
297
|
+
steps = 0
|
|
298
|
+
while steps < MAX_STEPS:
|
|
299
|
+
action_idx = epsilon_greedy(state, Q)
|
|
300
|
+
dx, dy = ACTION_MAP[ACTIONS[action_idx]]
|
|
301
|
+
next_state = (state[0] + dx, state[1] + dy)
|
|
302
|
+
|
|
303
|
+
if not is_valid(next_state) or next_state in all_obstacles:
|
|
304
|
+
reward = -10 if rush else -5
|
|
305
|
+
next_state = state
|
|
306
|
+
elif next_state == goal:
|
|
307
|
+
reward = goal_reward - steps
|
|
308
|
+
else:
|
|
309
|
+
reward = -2 if rush else -1
|
|
310
|
+
|
|
311
|
+
episode.append((state, action_idx, reward))
|
|
312
|
+
if next_state == goal:
|
|
313
|
+
break
|
|
314
|
+
state = next_state
|
|
315
|
+
steps += 1
|
|
316
|
+
return episode
|
|
317
|
+
|
|
318
|
+
for ep in range(EPISODES):
|
|
319
|
+
episode = run_episode(ep)
|
|
320
|
+
G = 0
|
|
321
|
+
visited = set()
|
|
322
|
+
for t in reversed(range(len(episode))):
|
|
323
|
+
s, a, r = episode[t]
|
|
324
|
+
G = DISCOUNT * G + r
|
|
325
|
+
if (s, a) not in visited:
|
|
326
|
+
Returns[(s, a)].append(G)
|
|
327
|
+
Q[s][a] = np.mean(Returns[(s, a)])
|
|
328
|
+
visited.add((s, a)) # ā
FIXED: used tuple, not list
|
|
329
|
+
print("š Training Complete: Smart Ambulance Dispatch Policy Learned.")
|
|
330
|
+
|
|
331
|
+
# ---------- Policy Visualization ----------
|
|
332
|
+
policy = np.full((GRID_SIZE, GRID_SIZE), '.', dtype=str)
|
|
333
|
+
for i in range(GRID_SIZE):
|
|
334
|
+
for j in range(GRID_SIZE):
|
|
335
|
+
state = (i, j)
|
|
336
|
+
if state in static_obstacles:
|
|
337
|
+
policy[i][j] = 'S'
|
|
338
|
+
elif state in Q:
|
|
339
|
+
best_action = np.argmax(Q[state])
|
|
340
|
+
policy[i][j] = ['ā', 'ā', 'ā', 'ā'][best_action]
|
|
341
|
+
else:
|
|
342
|
+
policy[i][j] = ' '
|
|
343
|
+
|
|
344
|
+
print("\\nš Learned Ambulance Dispatch Policy Grid:")
|
|
345
|
+
for row in policy:
|
|
346
|
+
print(' '.join(row))""",
|
|
347
|
+
5: """import numpy as np
|
|
348
|
+
import random
|
|
349
|
+
import matplotlib.pyplot as plt
|
|
350
|
+
|
|
351
|
+
# Configuration
|
|
352
|
+
FLOORS = 5
|
|
353
|
+
ACTIONS = ['stay', 'up', 'down']
|
|
354
|
+
ACTION_SPACE = {0: 'stay', 1: 'up', 2: 'down'}
|
|
355
|
+
N_ACTIONS = len(ACTIONS)
|
|
356
|
+
GAMMA = 0.9 # Discount factor
|
|
357
|
+
ALPHA = 0.1 # Learning rate
|
|
358
|
+
EPSILON = 0.1 # Exploration rate
|
|
359
|
+
EPISODES = 10000
|
|
360
|
+
MAX_STEPS = 50
|
|
361
|
+
USE_SARSA = False # š Set to True to use SARSA; False for Q-Learning
|
|
362
|
+
|
|
363
|
+
# Initialize Q-table: Q[state][action]
|
|
364
|
+
Q = np.zeros((FLOORS, FLOORS, N_ACTIONS)) # [elevator_floor][request_floor][action]
|
|
365
|
+
episode_rewards = []
|
|
366
|
+
|
|
367
|
+
# Helper functions
|
|
368
|
+
def select_action(state):
|
|
369
|
+
ef, rf = state
|
|
370
|
+
if random.random() < EPSILON:
|
|
371
|
+
return random.randint(0, N_ACTIONS - 1)
|
|
372
|
+
return np.argmax(Q[ef][rf])
|
|
373
|
+
|
|
374
|
+
def take_action(ef, action):
|
|
375
|
+
if action == 0: # stay
|
|
376
|
+
return ef
|
|
377
|
+
elif action == 1: # up
|
|
378
|
+
return min(ef + 1, FLOORS - 1)
|
|
379
|
+
elif action == 2: # down
|
|
380
|
+
return max(ef - 1, 0)
|
|
381
|
+
|
|
382
|
+
def get_reward(ef, rf, next_ef):
|
|
383
|
+
if ef == rf and next_ef == rf:
|
|
384
|
+
return 10
|
|
385
|
+
elif abs(next_ef - rf) < abs(ef - rf):
|
|
386
|
+
return 1
|
|
387
|
+
elif abs(next_ef - rf) > abs(ef - rf):
|
|
388
|
+
return -2
|
|
389
|
+
elif next_ef == ef:
|
|
390
|
+
return -1
|
|
391
|
+
return -5
|
|
392
|
+
|
|
393
|
+
# Training loop
|
|
394
|
+
for ep in range(EPISODES):
|
|
395
|
+
ef = random.randint(0, FLOORS - 1) # elevator floor
|
|
396
|
+
rf = random.randint(0, FLOORS - 1) # request floor
|
|
397
|
+
state = (ef, rf)
|
|
398
|
+
total_reward = 0
|
|
399
|
+
action = select_action(state)
|
|
400
|
+
|
|
401
|
+
for step in range(MAX_STEPS):
|
|
402
|
+
next_ef = take_action(ef, action)
|
|
403
|
+
reward = get_reward(ef, rf, next_ef)
|
|
404
|
+
total_reward += reward
|
|
405
|
+
next_state = (next_ef, rf)
|
|
406
|
+
next_action = select_action(next_state)
|
|
407
|
+
|
|
408
|
+
if USE_SARSA:
|
|
409
|
+
Q[ef][rf][action] += ALPHA * (reward + GAMMA * Q[next_ef][rf][next_action] - Q[ef][rf][action])
|
|
410
|
+
else: # Q-Learning
|
|
411
|
+
Q[ef][rf][action] += ALPHA * (reward + GAMMA * np.max(Q[next_ef][rf]) - Q[ef][rf][action])
|
|
412
|
+
|
|
413
|
+
ef = next_ef
|
|
414
|
+
rf = rf # request remains until served
|
|
415
|
+
state = next_state
|
|
416
|
+
action = next_action if USE_SARSA else select_action(state)
|
|
417
|
+
|
|
418
|
+
if ef == rf:
|
|
419
|
+
break # request served
|
|
420
|
+
|
|
421
|
+
episode_rewards.append(total_reward)
|
|
422
|
+
|
|
423
|
+
print(f"š¢ Training complete using {'SARSA' if USE_SARSA else 'Q-Learning'}.")
|
|
424
|
+
print("\\nš Learned Elevator Policy:")
|
|
425
|
+
for ef in range(FLOORS):
|
|
426
|
+
for rf in range(FLOORS):
|
|
427
|
+
best_action = np.argmax(Q[ef][rf])
|
|
428
|
+
print(f"Elevator at {ef}, Request at {rf} ā Action: {ACTION_SPACE[best_action]}")
|
|
429
|
+
|
|
430
|
+
# Plot learning curve
|
|
431
|
+
plt.plot(episode_rewards)
|
|
432
|
+
plt.title(f"Episode Rewards ({'SARSA' if USE_SARSA else 'Q-Learning'})")
|
|
433
|
+
plt.xlabel("Episode")
|
|
434
|
+
plt.ylabel("Total Reward")
|
|
435
|
+
plt.grid()
|
|
436
|
+
plt.show()""",
|
|
437
|
+
}
|
|
438
|
+
|
|
439
|
+
descriptions = {
|
|
440
|
+
1: "core structure of Reinforcement Learning by simulating an agent-environment interaction loop in a simple Gridworld",
|
|
441
|
+
2: "ε-Greedy strategy for solving the multi-armed bandit problem click-through rate (CTR)",
|
|
442
|
+
3: "Markov Decision Process (MDP) and solve it using Value Iteration",
|
|
443
|
+
4: "Monte Carlo Control for Emergency Ambulance Dispatch in Smart Cities",
|
|
444
|
+
5: "Q-Learning for Intelligent Elevator Control in a Smart Building",
|
|
445
|
+
}
|
|
446
|
+
|
|
447
|
+
def info():
|
|
448
|
+
"""Returns what the 5 codes are for."""
|
|
449
|
+
info_str = "This package contains 5 codes for the following:\\n"
|
|
450
|
+
for k, v in descriptions.items():
|
|
451
|
+
info_str += f"{k}. {v}\\n"
|
|
452
|
+
return info_str
|
|
453
|
+
|
|
454
|
+
def question(n):
|
|
455
|
+
"""Returns the code for the given question number."""
|
|
456
|
+
if n in codes:
|
|
457
|
+
return codes[n]
|
|
458
|
+
else:
|
|
459
|
+
return "Invalid question number. Please choose between 1 and 5."
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: lab_copy
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A package containing 5 specific codes.
|
|
5
|
+
Author: Your Name
|
|
6
|
+
Author-email: your.email@example.com
|
|
7
|
+
Classifier: Programming Language :: Python :: 3
|
|
8
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
9
|
+
Classifier: Operating System :: OS Independent
|
|
10
|
+
Requires-Python: >=3.6
|
|
11
|
+
Dynamic: author
|
|
12
|
+
Dynamic: author-email
|
|
13
|
+
Dynamic: classifier
|
|
14
|
+
Dynamic: requires-python
|
|
15
|
+
Dynamic: summary
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
lab_copy/__init__.py,sha256=njwz-l6qoxdY-SB13mTF-H_5Ty4E0Z_gDN2W-kXaAOc,15013
|
|
2
|
+
lab_copy-0.1.0.dist-info/METADATA,sha256=pN1EFBNbYpJutNPgwK-tEH4WZUmyUncKfM4s6FjmSFk,439
|
|
3
|
+
lab_copy-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
4
|
+
lab_copy-0.1.0.dist-info/top_level.txt,sha256=DbsIFHxtoewjU2SlS4x128s4_37Ou1PeQGXsB9GCXoU,9
|
|
5
|
+
lab_copy-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
lab_copy
|