lab-copy 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
lab_copy/__init__.py ADDED
@@ -0,0 +1,459 @@
1
+ codes = {
2
+ 1: """import numpy as np
3
+ import matplotlib.pyplot as plt
4
+
5
+ # Gridworld Environment Setup
6
+ GRID_SIZE = 5
7
+ ACTIONS = ['up', 'down', 'left', 'right']
8
+ ACTION_DICT = {'up': (-1, 0), 'down': (1, 0), 'left': (0, -1), 'right': (0, 1)}
9
+ GOAL_STATE = (4, 4)
10
+ MAX_STEPS = 50
11
+
12
+ def is_valid_state(state):
13
+ return 0 <= state[0] < GRID_SIZE and 0 <= state[1] < GRID_SIZE
14
+
15
+ def step(state, action):
16
+ move = ACTION_DICT[action]
17
+ new_state = (state[0] + move[0], state[1] + move[1])
18
+ if not is_valid_state(new_state):
19
+ new_state = state # stay if move goes out of bounds
20
+ reward = 10 if new_state == GOAL_STATE else 0
21
+ done = (new_state == GOAL_STATE)
22
+ return new_state, reward, done
23
+
24
+ # Agent Loop
25
+ def run_episode():
26
+ state = (0, 0)
27
+ total_reward = 0
28
+ trajectory = [state]
29
+ for step_num in range(MAX_STEPS):
30
+ action = np.random.choice(ACTIONS) # Random policy
31
+ next_state, reward, done = step(state, action)
32
+ trajectory.append(next_state)
33
+ total_reward += reward
34
+ state = next_state
35
+ if done:
36
+ break
37
+ return trajectory, total_reward
38
+
39
+ # Run 10 episodes and visualize one
40
+ for ep in range(10):
41
+ traj, reward = run_episode()
42
+ print(f"Episode {ep+1}: Total Reward = {reward}, Steps = {len(traj)}")
43
+
44
+ # Visualize trajectory of the last episode
45
+ def plot_trajectory(trajectory):
46
+ grid = np.zeros((GRID_SIZE, GRID_SIZE))
47
+ for (x, y) in trajectory:
48
+ grid[x, y] += 1
49
+ plt.imshow(grid, cmap='Blues', origin='upper')
50
+ plt.title("Agent Trajectory Heatmap")
51
+ plt.colorbar(label="Visits")
52
+ plt.scatter(0, 0, c='green', s=100, label='Start')
53
+ plt.scatter(4, 4, c='red', s=100, label='Goal')
54
+ plt.legend()
55
+ plt.grid(True)
56
+ plt.show()
57
+
58
+ plot_trajectory(traj)""",
59
+ 2: """import numpy as np
60
+ import matplotlib.pyplot as plt
61
+
62
+ class EpsilonGreedyAgent:
63
+ def __init__(self, n_arms, epsilon):
64
+ self.n_arms = n_arms
65
+ self.epsilon = epsilon
66
+ self.counts = np.zeros(n_arms) # Number of times each arm was pulled
67
+ self.values = np.zeros(n_arms) # Estimated value (CTR) for each arm
68
+ self.total_reward = 0
69
+ self.actions = []
70
+ self.rewards = []
71
+
72
+ def select_action(self):
73
+ if np.random.rand() < self.epsilon:
74
+ return np.random.randint(self.n_arms) # Explore
75
+ else:
76
+ return np.argmax(self.values) # Exploit
77
+
78
+ def update(self, action, reward):
79
+ self.counts[action] += 1
80
+ self.values[action] += (reward - self.values[action]) / self.counts[action]
81
+ self.total_reward += reward
82
+ self.actions.append(action)
83
+ self.rewards.append(reward)
84
+
85
+ def simulate_bandit(true_ctrs, epsilon, n_rounds=1000):
86
+ n_arms = len(true_ctrs)
87
+ agent = EpsilonGreedyAgent(n_arms, epsilon)
88
+ optimal_arm = np.argmax(true_ctrs)
89
+ regrets = []
90
+ for t in range(n_rounds):
91
+ action = agent.select_action()
92
+ reward = np.random.rand() < true_ctrs[action]
93
+ agent.update(action, reward)
94
+ regret = true_ctrs[optimal_arm] - true_ctrs[action]
95
+ regrets.append(regret)
96
+ return agent, np.cumsum(regrets)
97
+
98
+ # ---------- Main Experiment ----------
99
+ np.random.seed(42)
100
+ n_arms = 10
101
+ true_ctrs = np.random.uniform(0.05, 0.5, n_arms)
102
+ print("True Click-Through Rates (CTR) per Ad:", np.round(true_ctrs, 2))
103
+ n_rounds = 1000
104
+ epsilons = [0.01, 0.1, 0.3]
105
+ agents = {}
106
+ regret_curves = {}
107
+ for epsilon in epsilons:
108
+ agent, regrets = simulate_bandit(true_ctrs, epsilon, n_rounds)
109
+ agents[epsilon] = agent
110
+ regret_curves[epsilon] = regrets
111
+
112
+ # ---------- Plotting Results ----------
113
+ plt.figure(figsize=(12, 5))
114
+ # Plot cumulative regret
115
+ plt.subplot(1, 2, 1)
116
+ for epsilon in epsilons:
117
+ plt.plot(regret_curves[epsilon], label=f'ε={epsilon}')
118
+ plt.title("Cumulative Regret")
119
+ plt.xlabel("Rounds")
120
+ plt.ylabel("Cumulative Regret")
121
+ plt.legend()
122
+ plt.grid(True)
123
+
124
+ plt.subplot(1, 2, 2)
125
+ bar_width = 0.25
126
+ x = np.arange(n_arms)
127
+ for i, epsilon in enumerate(epsilons):
128
+ plt.bar(x + i * bar_width,
129
+ agents[epsilon].values,
130
+ width=bar_width,
131
+ label=f'ε={epsilon}')
132
+ plt.axhline(np.max(true_ctrs), color='r', linestyle='--', label='Optimal CTR')
133
+ plt.xticks(x + bar_width, [f'Ad {i}' for i in range(n_arms)])
134
+ plt.ylabel("Estimated CTR")
135
+ plt.title("Estimated CTRs vs True CTR")
136
+ plt.legend()
137
+ plt.grid(True)
138
+ plt.tight_layout()
139
+ plt.show()""",
140
+ 3: """import numpy as np
141
+ import mdptoolbox
142
+ import matplotlib.pyplot as plt
143
+ import random
144
+
145
+ # Grid Parameters
146
+ rows, cols = 5, 5
147
+ num_states = rows * cols
148
+ shelves = [(1, 1), (2, 2), (3, 3)] # Obstacle positions
149
+ actions = ['up', 'down', 'left', 'right']
150
+ num_actions = len(actions)
151
+ movement = {'up': (-1, 0), 'down': (1, 0), 'left': (0, -1), 'right': (0, 1)}
152
+
153
+ # Function to map (x, y) to index
154
+ def to_index(x, y):
155
+ return x * cols + y
156
+
157
+ # Function to randomly set a goal not on a shelf
158
+ def set_dynamic_goal():
159
+ possible = [(i, j) for i in range(rows) for j in range(cols) if (i, j) not in shelves]
160
+ return random.choice(possible)
161
+
162
+ # Randomly choose a goal position
163
+ goal_state = set_dynamic_goal()
164
+ print(" Current Goal Position:", goal_state)
165
+
166
+ # Transition and Reward Matrices
167
+ P = [np.zeros((num_states, num_states)) for _ in range(num_actions)]
168
+ R = np.zeros((num_states, num_actions))
169
+
170
+ for action_idx, action in enumerate(actions):
171
+ dx, dy = movement[action]
172
+ for x in range(rows):
173
+ for y in range(cols):
174
+ current_state = to_index(x, y)
175
+ if (x, y) == goal_state:
176
+ P[action_idx][current_state, current_state] = 1
177
+ R[current_state, action_idx] = 10
178
+ continue
179
+
180
+ outcomes = []
181
+ # Intended move (90%)
182
+ new_x, new_y = x + dx, y + dy
183
+ if (new_x, new_y) in shelves or not (0 <= new_x < rows and 0 <= new_y < cols):
184
+ new_state = current_state
185
+ reward = -5 if (new_x, new_y) in shelves else -1
186
+ else:
187
+ new_state = to_index(new_x, new_y)
188
+ reward = -1
189
+ outcomes.append((new_state, 0.9, reward))
190
+
191
+ # 10% misstep (wrong move in any of the other 3 directions)
192
+ other_actions = [a for i, a in enumerate(actions) if i != action_idx]
193
+ for mis_action in other_actions:
194
+ mx, my = movement[mis_action]
195
+ new_x, new_y = x + mx, y + my
196
+ if (new_x, new_y) in shelves or not (0 <= new_x < rows and 0 <= new_y < cols):
197
+ mis_state = current_state
198
+ mis_reward = -5 if (new_x, new_y) in shelves else -1
199
+ else:
200
+ mis_state = to_index(new_x, new_y)
201
+ mis_reward = -1
202
+ outcomes.append((mis_state, 0.1 / 3, mis_reward))
203
+
204
+ for s_next, prob, rew in outcomes:
205
+ P[action_idx][current_state, s_next] += prob
206
+ R[current_state, action_idx] += prob * rew # Expected reward
207
+
208
+ # Run Value Iteration
209
+ vi = mdptoolbox.mdp.ValueIteration(P, R, 0.9)
210
+ vi.run()
211
+
212
+ # Reshape policy to grid
213
+ policy_grid = np.array(vi.policy).reshape((rows, cols))
214
+ action_symbols = ['↑', '↓', '←', '→']
215
+ policy_symbols = np.array([[action_symbols[a] for a in row] for row in policy_grid])
216
+ print("\\nšŸ“ Optimal Policy Grid:")
217
+ print(policy_symbols)
218
+
219
+ # Plotting
220
+ plt.figure(figsize=(6, 6))
221
+ for x in range(rows):
222
+ for y in range(cols):
223
+ idx = to_index(x, y)
224
+ if (x, y) == goal_state:
225
+ plt.text(y, rows - x - 1, 'G', ha='center', va='center', fontsize=14, color='green')
226
+ elif (x, y) in shelves:
227
+ plt.text(y, rows - x - 1, 'S', ha='center', va='center', fontsize=14, color='red')
228
+ else:
229
+ plt.text(y, rows - x - 1, action_symbols[vi.policy[idx]], ha='center', va='center', fontsize=14)
230
+ plt.xticks(range(cols))
231
+ plt.yticks(range(rows))
232
+ plt.grid(True)
233
+ plt.title("Stochastic Optimal Policy with Dynamic Goal")
234
+ plt.show()""",
235
+ 4: """import numpy as np
236
+ import matplotlib.pyplot as plt
237
+ from collections import defaultdict
238
+ import random
239
+
240
+ # ---------- Environment Setup ----------
241
+ GRID_SIZE = 6
242
+ ACTIONS = ['up', 'down', 'left', 'right']
243
+ ACTION_MAP = {'up': (-1, 0), 'down': (1, 0), 'left': (0, -1), 'right': (0, 1)}
244
+ MAX_STEPS = 50
245
+ DISCOUNT = 0.95
246
+ EPSILON = 0.1
247
+ EPISODES = 10000
248
+
249
+ # Static obstacles (permanent roadblocks)
250
+ static_obstacles = [(1, 3), (3, 2)]
251
+ hospital = (0, 0) # Ambulance dispatch center
252
+
253
+ # Rush hour control
254
+ def is_rush_hour(ep):
255
+ return ep % 1000 < 300 or ep % 1000 > 800 # Congested traffic windows
256
+
257
+ # Emergency severity and urgency
258
+ emergency_types = {
259
+ 'minor': 20,
260
+ 'moderate': 35,
261
+ 'critical': 50
262
+ }
263
+
264
+ def is_valid(state):
265
+ x, y = state
266
+ return 0 <= x < GRID_SIZE and 0 <= y < GRID_SIZE
267
+
268
+ def get_dynamic_obstacles():
269
+ return [(2, 4), (4, 1), (3, 3)] if random.random() < 0.3 else []
270
+
271
+ def epsilon_greedy(state, Q):
272
+ if np.random.rand() < EPSILON or state not in Q:
273
+ return random.randint(0, len(ACTIONS) - 1)
274
+ else:
275
+ return np.argmax(Q[state])
276
+
277
+ def generate_emergency():
278
+ location = random.choice([
279
+ (i, j) for i in range(GRID_SIZE) for j in range(GRID_SIZE)
280
+ if (i, j) != hospital and (i, j) not in static_obstacles
281
+ ])
282
+ severity = random.choice(list(emergency_types.keys()))
283
+ reward = emergency_types[severity]
284
+ return location, reward
285
+
286
+ # ---------- Monte Carlo Training ----------
287
+ Q = defaultdict(lambda: np.zeros(len(ACTIONS)))
288
+ Returns = defaultdict(list)
289
+
290
+ def run_episode(episode_num):
291
+ rush = is_rush_hour(episode_num)
292
+ prob_blocks = get_dynamic_obstacles()
293
+ all_obstacles = static_obstacles + prob_blocks
294
+ goal, goal_reward = generate_emergency()
295
+ state = hospital
296
+ episode = []
297
+ steps = 0
298
+ while steps < MAX_STEPS:
299
+ action_idx = epsilon_greedy(state, Q)
300
+ dx, dy = ACTION_MAP[ACTIONS[action_idx]]
301
+ next_state = (state[0] + dx, state[1] + dy)
302
+
303
+ if not is_valid(next_state) or next_state in all_obstacles:
304
+ reward = -10 if rush else -5
305
+ next_state = state
306
+ elif next_state == goal:
307
+ reward = goal_reward - steps
308
+ else:
309
+ reward = -2 if rush else -1
310
+
311
+ episode.append((state, action_idx, reward))
312
+ if next_state == goal:
313
+ break
314
+ state = next_state
315
+ steps += 1
316
+ return episode
317
+
318
+ for ep in range(EPISODES):
319
+ episode = run_episode(ep)
320
+ G = 0
321
+ visited = set()
322
+ for t in reversed(range(len(episode))):
323
+ s, a, r = episode[t]
324
+ G = DISCOUNT * G + r
325
+ if (s, a) not in visited:
326
+ Returns[(s, a)].append(G)
327
+ Q[s][a] = np.mean(Returns[(s, a)])
328
+ visited.add((s, a)) # āœ… FIXED: used tuple, not list
329
+ print("šŸš‘ Training Complete: Smart Ambulance Dispatch Policy Learned.")
330
+
331
+ # ---------- Policy Visualization ----------
332
+ policy = np.full((GRID_SIZE, GRID_SIZE), '.', dtype=str)
333
+ for i in range(GRID_SIZE):
334
+ for j in range(GRID_SIZE):
335
+ state = (i, j)
336
+ if state in static_obstacles:
337
+ policy[i][j] = 'S'
338
+ elif state in Q:
339
+ best_action = np.argmax(Q[state])
340
+ policy[i][j] = ['↑', '↓', '←', '→'][best_action]
341
+ else:
342
+ policy[i][j] = ' '
343
+
344
+ print("\\nšŸ“ Learned Ambulance Dispatch Policy Grid:")
345
+ for row in policy:
346
+ print(' '.join(row))""",
347
+ 5: """import numpy as np
348
+ import random
349
+ import matplotlib.pyplot as plt
350
+
351
+ # Configuration
352
+ FLOORS = 5
353
+ ACTIONS = ['stay', 'up', 'down']
354
+ ACTION_SPACE = {0: 'stay', 1: 'up', 2: 'down'}
355
+ N_ACTIONS = len(ACTIONS)
356
+ GAMMA = 0.9 # Discount factor
357
+ ALPHA = 0.1 # Learning rate
358
+ EPSILON = 0.1 # Exploration rate
359
+ EPISODES = 10000
360
+ MAX_STEPS = 50
361
+ USE_SARSA = False # šŸ” Set to True to use SARSA; False for Q-Learning
362
+
363
+ # Initialize Q-table: Q[state][action]
364
+ Q = np.zeros((FLOORS, FLOORS, N_ACTIONS)) # [elevator_floor][request_floor][action]
365
+ episode_rewards = []
366
+
367
+ # Helper functions
368
+ def select_action(state):
369
+ ef, rf = state
370
+ if random.random() < EPSILON:
371
+ return random.randint(0, N_ACTIONS - 1)
372
+ return np.argmax(Q[ef][rf])
373
+
374
+ def take_action(ef, action):
375
+ if action == 0: # stay
376
+ return ef
377
+ elif action == 1: # up
378
+ return min(ef + 1, FLOORS - 1)
379
+ elif action == 2: # down
380
+ return max(ef - 1, 0)
381
+
382
+ def get_reward(ef, rf, next_ef):
383
+ if ef == rf and next_ef == rf:
384
+ return 10
385
+ elif abs(next_ef - rf) < abs(ef - rf):
386
+ return 1
387
+ elif abs(next_ef - rf) > abs(ef - rf):
388
+ return -2
389
+ elif next_ef == ef:
390
+ return -1
391
+ return -5
392
+
393
+ # Training loop
394
+ for ep in range(EPISODES):
395
+ ef = random.randint(0, FLOORS - 1) # elevator floor
396
+ rf = random.randint(0, FLOORS - 1) # request floor
397
+ state = (ef, rf)
398
+ total_reward = 0
399
+ action = select_action(state)
400
+
401
+ for step in range(MAX_STEPS):
402
+ next_ef = take_action(ef, action)
403
+ reward = get_reward(ef, rf, next_ef)
404
+ total_reward += reward
405
+ next_state = (next_ef, rf)
406
+ next_action = select_action(next_state)
407
+
408
+ if USE_SARSA:
409
+ Q[ef][rf][action] += ALPHA * (reward + GAMMA * Q[next_ef][rf][next_action] - Q[ef][rf][action])
410
+ else: # Q-Learning
411
+ Q[ef][rf][action] += ALPHA * (reward + GAMMA * np.max(Q[next_ef][rf]) - Q[ef][rf][action])
412
+
413
+ ef = next_ef
414
+ rf = rf # request remains until served
415
+ state = next_state
416
+ action = next_action if USE_SARSA else select_action(state)
417
+
418
+ if ef == rf:
419
+ break # request served
420
+
421
+ episode_rewards.append(total_reward)
422
+
423
+ print(f"šŸ¢ Training complete using {'SARSA' if USE_SARSA else 'Q-Learning'}.")
424
+ print("\\nšŸ“ Learned Elevator Policy:")
425
+ for ef in range(FLOORS):
426
+ for rf in range(FLOORS):
427
+ best_action = np.argmax(Q[ef][rf])
428
+ print(f"Elevator at {ef}, Request at {rf} → Action: {ACTION_SPACE[best_action]}")
429
+
430
+ # Plot learning curve
431
+ plt.plot(episode_rewards)
432
+ plt.title(f"Episode Rewards ({'SARSA' if USE_SARSA else 'Q-Learning'})")
433
+ plt.xlabel("Episode")
434
+ plt.ylabel("Total Reward")
435
+ plt.grid()
436
+ plt.show()""",
437
+ }
438
+
439
+ descriptions = {
440
+ 1: "core structure of Reinforcement Learning by simulating an agent-environment interaction loop in a simple Gridworld",
441
+ 2: "ε-Greedy strategy for solving the multi-armed bandit problem click-through rate (CTR)",
442
+ 3: "Markov Decision Process (MDP) and solve it using Value Iteration",
443
+ 4: "Monte Carlo Control for Emergency Ambulance Dispatch in Smart Cities",
444
+ 5: "Q-Learning for Intelligent Elevator Control in a Smart Building",
445
+ }
446
+
447
+ def info():
448
+ """Returns what the 5 codes are for."""
449
+ info_str = "This package contains 5 codes for the following:\\n"
450
+ for k, v in descriptions.items():
451
+ info_str += f"{k}. {v}\\n"
452
+ return info_str
453
+
454
+ def question(n):
455
+ """Returns the code for the given question number."""
456
+ if n in codes:
457
+ return codes[n]
458
+ else:
459
+ return "Invalid question number. Please choose between 1 and 5."
@@ -0,0 +1,15 @@
1
+ Metadata-Version: 2.4
2
+ Name: lab_copy
3
+ Version: 0.1.0
4
+ Summary: A package containing 5 specific codes.
5
+ Author: Your Name
6
+ Author-email: your.email@example.com
7
+ Classifier: Programming Language :: Python :: 3
8
+ Classifier: License :: OSI Approved :: MIT License
9
+ Classifier: Operating System :: OS Independent
10
+ Requires-Python: >=3.6
11
+ Dynamic: author
12
+ Dynamic: author-email
13
+ Dynamic: classifier
14
+ Dynamic: requires-python
15
+ Dynamic: summary
@@ -0,0 +1,5 @@
1
+ lab_copy/__init__.py,sha256=njwz-l6qoxdY-SB13mTF-H_5Ty4E0Z_gDN2W-kXaAOc,15013
2
+ lab_copy-0.1.0.dist-info/METADATA,sha256=pN1EFBNbYpJutNPgwK-tEH4WZUmyUncKfM4s6FjmSFk,439
3
+ lab_copy-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
4
+ lab_copy-0.1.0.dist-info/top_level.txt,sha256=DbsIFHxtoewjU2SlS4x128s4_37Ou1PeQGXsB9GCXoU,9
5
+ lab_copy-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1 @@
1
+ lab_copy