@game_ryo/lsji 0.1.0 → 0.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +4 -1
- package/docs/README.md +0 -43
- package/docs/blog/2019-05-28-first-blog-post.mdx +0 -12
- package/docs/blog/2019-05-29-long-blog-post.mdx +0 -44
- package/docs/blog/2021-08-01-mdx-blog-post.mdx +0 -24
- package/docs/blog/2021-08-26-welcome/docusaurus-plushie-banner.jpeg +0 -0
- package/docs/blog/2021-08-26-welcome/index.mdx +0 -29
- package/docs/blog/authors.yml +0 -25
- package/docs/blog/tags.yml +0 -19
- package/docs/docs/api/agent.md +0 -151
- package/docs/docs/api/env.md +0 -133
- package/docs/docs/api/environments.md +0 -102
- package/docs/docs/api/qlearning.md +0 -138
- package/docs/docs/api/storage.md +0 -168
- package/docs/docs/architecture.md +0 -155
- package/docs/docs/cli.md +0 -210
- package/docs/docs/contributing.md +0 -162
- package/docs/docs/core-concepts.md +0 -152
- package/docs/docs/examples/advanced-training.md +0 -244
- package/docs/docs/examples/custom-environment.md +0 -198
- package/docs/docs/examples/custom-storage.md +0 -251
- package/docs/docs/getting-started.md +0 -91
- package/docs/docusaurus.config.ts +0 -149
- package/docs/package-lock.json +0 -19522
- package/docs/package.json +0 -49
- package/docs/sidebars.ts +0 -33
- package/docs/src/components/HomepageFeatures/index.tsx +0 -71
- package/docs/src/components/HomepageFeatures/styles.module.css +0 -11
- package/docs/src/css/custom.css +0 -79
- package/docs/src/pages/index.module.css +0 -23
- package/docs/src/pages/index.tsx +0 -44
- package/docs/src/pages/markdown-page.mdx +0 -7
- package/docs/static/.nojekyll +0 -0
- package/docs/static/img/docusaurus-social-card.jpg +0 -0
- package/docs/static/img/docusaurus.png +0 -0
- package/docs/static/img/favicon.ico +0 -0
- package/docs/static/img/logo.png +0 -0
- package/docs/static/img/undraw_docusaurus_mountain.svg +0 -171
- package/docs/static/img/undraw_docusaurus_react.svg +0 -170
- package/docs/static/img/undraw_docusaurus_tree.svg +0 -40
- package/docs/tsconfig.json +0 -12
- package/legacy/worker.js +0 -166
- package/legacy/wrangler.toml +0 -11
|
@@ -1,162 +0,0 @@
|
|
|
1
|
-
---
|
|
2
|
-
title: Contributing
|
|
3
|
-
description: How to contribute to LSJI
|
|
4
|
-
---
|
|
5
|
-
|
|
6
|
-
# Contributing
|
|
7
|
-
|
|
8
|
-
Thank you for your interest in contributing to LSJI!
|
|
9
|
-
|
|
10
|
-
## Development Setup
|
|
11
|
-
|
|
12
|
-
```bash
|
|
13
|
-
# Clone the repository
|
|
14
|
-
git clone https://github.com/ryotagtagtag-wq/LSJI.git
|
|
15
|
-
cd LSJI
|
|
16
|
-
|
|
17
|
-
# Install dependencies
|
|
18
|
-
npm install
|
|
19
|
-
|
|
20
|
-
# Run tests
|
|
21
|
-
npm test
|
|
22
|
-
|
|
23
|
-
# Build documentation
|
|
24
|
-
cd docs && npm run build
|
|
25
|
-
```
|
|
26
|
-
|
|
27
|
-
## Project Structure
|
|
28
|
-
|
|
29
|
-
```
|
|
30
|
-
LSJI/
|
|
31
|
-
├── src/ # Core library
|
|
32
|
-
│ ├── core/ # QLearning, Agent, Env
|
|
33
|
-
│ ├── storage/ # Storage backends
|
|
34
|
-
│ ├── envs/ # Built-in environments
|
|
35
|
-
│ ├── cli.ts # CLI
|
|
36
|
-
│ └── index.ts # Public exports
|
|
37
|
-
├── test/ # Vitest tests
|
|
38
|
-
├── docs/ # Docusaurus documentation
|
|
39
|
-
└── bin/ # CLI entry point
|
|
40
|
-
```
|
|
41
|
-
|
|
42
|
-
## Making Changes
|
|
43
|
-
|
|
44
|
-
### 1. Create a Branch
|
|
45
|
-
|
|
46
|
-
```bash
|
|
47
|
-
git checkout -b feature/my-feature
|
|
48
|
-
```
|
|
49
|
-
|
|
50
|
-
### 2. Make Changes
|
|
51
|
-
|
|
52
|
-
Follow the existing code style:
|
|
53
|
-
- TypeScript with JSDoc comments
|
|
54
|
-
- ESM imports/exports
|
|
55
|
-
- No external dependencies in core
|
|
56
|
-
|
|
57
|
-
### 3. Run Tests
|
|
58
|
-
|
|
59
|
-
```bash
|
|
60
|
-
npm test
|
|
61
|
-
```
|
|
62
|
-
|
|
63
|
-
### 4. Update Documentation
|
|
64
|
-
|
|
65
|
-
If you add new features, update relevant docs in `docs/docs/`.
|
|
66
|
-
|
|
67
|
-
### 5. Commit
|
|
68
|
-
|
|
69
|
-
```bash
|
|
70
|
-
git add .
|
|
71
|
-
git commit -m "feat: add my feature"
|
|
72
|
-
```
|
|
73
|
-
|
|
74
|
-
**Commit Message Format:**
|
|
75
|
-
- `feat:` — New feature
|
|
76
|
-
- `fix:` — Bug fix
|
|
77
|
-
- `docs:` — Documentation
|
|
78
|
-
- `refactor:` — Code refactoring
|
|
79
|
-
- `test:` — Tests
|
|
80
|
-
- `chore:` — Maintenance
|
|
81
|
-
|
|
82
|
-
### 6. Push and Create PR
|
|
83
|
-
|
|
84
|
-
```bash
|
|
85
|
-
git push origin feature/my-feature
|
|
86
|
-
```
|
|
87
|
-
|
|
88
|
-
## Adding a New Environment
|
|
89
|
-
|
|
90
|
-
1. Create `src/envs/my-env.ts` extending `Env`
|
|
91
|
-
2. Implement all abstract methods
|
|
92
|
-
3. Export from `src/index.ts`
|
|
93
|
-
4. Add documentation in `docs/docs/api/environments.md`
|
|
94
|
-
5. Add example in `docs/docs/examples/`
|
|
95
|
-
|
|
96
|
-
## Adding a New Storage Backend
|
|
97
|
-
|
|
98
|
-
1. Create `src/storage/my-backend.ts` extending `Storage`
|
|
99
|
-
2. Implement all abstract methods
|
|
100
|
-
3. Add to `createStorage` factory in `src/storage/index.ts`
|
|
101
|
-
3. Export from `src/index.ts`
|
|
102
|
-
4. Add tests in `test/storage/`
|
|
103
|
-
|
|
104
|
-
## Modifying Learning Algorithm
|
|
105
|
-
|
|
106
|
-
1. Extend `QLearning` class or create new class in `src/core/`
|
|
107
|
-
2. Maintain compatibility with `Agent` interface
|
|
108
|
-
3. Add tests for new algorithm
|
|
109
|
-
4. Document in `docs/docs/api/`
|
|
110
|
-
|
|
111
|
-
## Code Style
|
|
112
|
-
|
|
113
|
-
- **TypeScript** with strict mode
|
|
114
|
-
- **ESM** modules (`import`/`export`)
|
|
115
|
-
- **JSDoc** for all public APIs
|
|
116
|
-
- **No `any`** unless absolutely necessary
|
|
117
|
-
- **Async/await** for async operations
|
|
118
|
-
|
|
119
|
-
## Testing Guidelines
|
|
120
|
-
|
|
121
|
-
- Use `MemoryStorage` for unit tests
|
|
122
|
-
- Test both success and error cases
|
|
123
|
-
- Test edge cases (empty Q-table, terminal states)
|
|
124
|
-
- Keep tests fast and isolated
|
|
125
|
-
|
|
126
|
-
```typescript
|
|
127
|
-
// Example test structure
|
|
128
|
-
import { describe, it, expect, beforeEach } from 'vitest';
|
|
129
|
-
import { MyFeature } from '../src/core/my-feature';
|
|
130
|
-
import { MemoryStorage } from '../src/storage/memory';
|
|
131
|
-
|
|
132
|
-
describe('MyFeature', () => {
|
|
133
|
-
let storage;
|
|
134
|
-
let feature;
|
|
135
|
-
|
|
136
|
-
beforeEach(async () => {
|
|
137
|
-
storage = new MemoryStorage();
|
|
138
|
-
await storage.initialize();
|
|
139
|
-
feature = new MyFeature({ storage });
|
|
140
|
-
});
|
|
141
|
-
|
|
142
|
-
it('should do something', async () => {
|
|
143
|
-
const result = await feature.doSomething();
|
|
144
|
-
expect(result).toBe(expected);
|
|
145
|
-
});
|
|
146
|
-
});
|
|
147
|
-
```
|
|
148
|
-
|
|
149
|
-
## Documentation
|
|
150
|
-
|
|
151
|
-
- Update relevant `.md` files in `docs/docs/`
|
|
152
|
-
- Add JSDoc comments for new public APIs
|
|
153
|
-
- Include code examples
|
|
154
|
-
|
|
155
|
-
## License
|
|
156
|
-
|
|
157
|
-
By contributing, you agree that your contributions will be licensed under the Apache 2.0 License.
|
|
158
|
-
|
|
159
|
-
## Questions?
|
|
160
|
-
|
|
161
|
-
- Open a [GitHub Issue](https://github.com/ryotagtagtag-wq/LSJI/issues)
|
|
162
|
-
- Start a [Discussion](https://github.com/ryotagtagtag-wq/LSJI/discussions)
|
|
@@ -1,152 +0,0 @@
|
|
|
1
|
-
---
|
|
2
|
-
title: Core Concepts
|
|
3
|
-
description: Understand the core architecture of LSJI
|
|
4
|
-
---
|
|
5
|
-
|
|
6
|
-
# Core Concepts
|
|
7
|
-
|
|
8
|
-
LSJI is built around four core abstractions that work together to create a flexible reinforcement learning framework.
|
|
9
|
-
|
|
10
|
-
## Architecture Overview
|
|
11
|
-
|
|
12
|
-
```
|
|
13
|
-
┌─────────────────────────────────────────────────────────────┐
|
|
14
|
-
│ Agent │
|
|
15
|
-
│ ┌─────────────┐ ┌─────────────┐ ┌─────────────────────┐ │
|
|
16
|
-
│ │ QLearning │ │ Storage │ │ Env │ │
|
|
17
|
-
│ │ (Engine) │◄─┤ (Backend) │ │ (Environment) │ │
|
|
18
|
-
│ └─────────────┘ └─────────────┘ └─────────────────────┘ │
|
|
19
|
-
└─────────────────────────────────────────────────────────────┘
|
|
20
|
-
```
|
|
21
|
-
|
|
22
|
-
### 1. Environment (`Env`)
|
|
23
|
-
|
|
24
|
-
The `Env` interface defines the problem domain. Any RL environment must implement:
|
|
25
|
-
|
|
26
|
-
```typescript
|
|
27
|
-
abstract class Env {
|
|
28
|
-
getState(): string; // Current state representation
|
|
29
|
-
step(action: number): Promise<StepResult>; // Execute action
|
|
30
|
-
actionSize(): number; // Number of possible actions
|
|
31
|
-
reset(): Promise<string>; // Reset to initial state
|
|
32
|
-
}
|
|
33
|
-
```
|
|
34
|
-
|
|
35
|
-
**StepResult** contains:
|
|
36
|
-
- `state` — New state after action
|
|
37
|
-
- `reward` — Reward received (-1, 0, 1)
|
|
38
|
-
- `done` — Whether episode ended
|
|
39
|
-
- `info` — Additional diagnostic info
|
|
40
|
-
|
|
41
|
-
### 2. Q-Learning Engine (`QLearning`)
|
|
42
|
-
|
|
43
|
-
Tabular Q-Learning with Temporal Difference (TD) updates:
|
|
44
|
-
|
|
45
|
-
```typescript
|
|
46
|
-
class QLearning {
|
|
47
|
-
constructor({ alpha, gamma, epsilon, storage });
|
|
48
|
-
|
|
49
|
-
// Epsilon-greedy action selection
|
|
50
|
-
async act(state: string, actionSize: number): Promise<number>;
|
|
51
|
-
|
|
52
|
-
// Full TD update: Q(s,a) ← Q(s,a) + α[r + γ·max Q(s',a') - Q(s,a)]
|
|
53
|
-
async learn(state, action, reward, nextState, nextActionSize);
|
|
54
|
-
|
|
55
|
-
// Simplified update (terminal states): Q(s,a) ← Q(s,a) + α[r - Q(s,a)]
|
|
56
|
-
async learnSimple(state, action, reward);
|
|
57
|
-
|
|
58
|
-
// Get all Q-values for inspection
|
|
59
|
-
async getFullQTable(): Promise<QTableRecord[]>;
|
|
60
|
-
}
|
|
61
|
-
```
|
|
62
|
-
|
|
63
|
-
**Hyperparameters:**
|
|
64
|
-
- `alpha` (0.1) — Learning rate
|
|
65
|
-
- `gamma` (0.9) — Discount factor
|
|
66
|
-
- `epsilon` (0.1) — Exploration rate
|
|
67
|
-
|
|
68
|
-
### 3. Storage Backend (`Storage`)
|
|
69
|
-
|
|
70
|
-
Pluggable persistence layer with three implementations:
|
|
71
|
-
|
|
72
|
-
| Backend | Package | Use Case |
|
|
73
|
-
|---------|---------|----------|
|
|
74
|
-
| `SqliteStorage` | `node:sqlite` (built-in) | **Recommended** — Zero dependencies |
|
|
75
|
-
| `BetterSqliteStorage` | `better-sqlite3` | High-performance synchronous access |
|
|
76
|
-
| `MemoryStorage` | Built-in | Testing, CI, ephemeral workloads |
|
|
77
|
-
|
|
78
|
-
All implement the same interface:
|
|
79
|
-
```typescript
|
|
80
|
-
interface Storage {
|
|
81
|
-
initialize(): Promise<void>;
|
|
82
|
-
close(): Promise<void>;
|
|
83
|
-
getSetting(key): Promise<Setting>;
|
|
84
|
-
setSetting(key, value): Promise<void>;
|
|
85
|
-
getQTable(): Promise<QTableRecord[]>;
|
|
86
|
-
updateQ(state, action, qValue): Promise<void>;
|
|
87
|
-
addBattle(record): Promise<void>;
|
|
88
|
-
getTodayBattleCount(): Promise<number>;
|
|
89
|
-
getPerformanceStats(): Promise<PerformanceStat[]>;
|
|
90
|
-
}
|
|
91
|
-
```
|
|
92
|
-
|
|
93
|
-
### 4. Agent (`Agent`)
|
|
94
|
-
|
|
95
|
-
High-level orchestration combining all components:
|
|
96
|
-
|
|
97
|
-
```typescript
|
|
98
|
-
class Agent {
|
|
99
|
-
constructor({ qlearning, storage, env });
|
|
100
|
-
|
|
101
|
-
async train({ episodes, actionSelector, batchSize });
|
|
102
|
-
async play(options?): Promise<PlayResult>;
|
|
103
|
-
async status(): Promise<StatusInfo>;
|
|
104
|
-
async start(): Promise<{status, message}>;
|
|
105
|
-
async stop(): Promise<{status, message}>;
|
|
106
|
-
setEnvironment(env): void;
|
|
107
|
-
}
|
|
108
|
-
```
|
|
109
|
-
|
|
110
|
-
## Data Flow
|
|
111
|
-
|
|
112
|
-
### Training Loop
|
|
113
|
-
```
|
|
114
|
-
for each episode:
|
|
115
|
-
1. Get current state from Env
|
|
116
|
-
2. Select action via QLearning.act() (ε-greedy)
|
|
117
|
-
3. Execute action in Env → StepResult
|
|
118
|
-
4. Update Q-table via QLearning.learnSimple()
|
|
119
|
-
5. Persist battle record to Storage
|
|
120
|
-
6. Batch DB writes for performance
|
|
121
|
-
```
|
|
122
|
-
|
|
123
|
-
### Play Loop
|
|
124
|
-
```
|
|
125
|
-
1. Get current state from Env
|
|
126
|
-
2. Select best action via QLearning.act() (ε=0 for exploitation)
|
|
127
|
-
3. Execute action in Env
|
|
128
|
-
4. Update Q-table with result
|
|
129
|
-
5. Record battle to Storage
|
|
130
|
-
6. Return result
|
|
131
|
-
```
|
|
132
|
-
|
|
133
|
-
## Reward System (RPS Example)
|
|
134
|
-
|
|
135
|
-
| Outcome | Judge Formula | Reward |
|
|
136
|
-
|---------|---------------|--------|
|
|
137
|
-
| Win | (ai - user + 3) % 3 = 2 | +1 |
|
|
138
|
-
| Lose | (ai - user + 3) % 3 = 1 | -1 |
|
|
139
|
-
| Draw | (ai - user + 3) % 3 = 0 | 0 |
|
|
140
|
-
|
|
141
|
-
## Training Patterns
|
|
142
|
-
|
|
143
|
-
Built-in patterns for the RPS environment:
|
|
144
|
-
|
|
145
|
-
| Pattern | ID | Description |
|
|
146
|
-
|---------|-----|-------------|
|
|
147
|
-
| Random | 0 | Uniform random actions |
|
|
148
|
-
| Always Rock | 1 | Always play action 0 |
|
|
149
|
-
| Counter | 2 | Play counter to previous action |
|
|
150
|
-
| Sequential | 3 | Cycle through 0,1,2,0,1,2... |
|
|
151
|
-
|
|
152
|
-
Custom patterns can be implemented via `actionSelector` function.
|
|
@@ -1,244 +0,0 @@
|
|
|
1
|
-
---
|
|
2
|
-
title: Advanced Training
|
|
3
|
-
description: Custom training patterns and techniques
|
|
4
|
-
---
|
|
5
|
-
|
|
6
|
-
# Advanced Training
|
|
7
|
-
|
|
8
|
-
Learn advanced training techniques for better agent performance.
|
|
9
|
-
|
|
10
|
-
## Custom Action Selectors
|
|
11
|
-
|
|
12
|
-
The `train()` method accepts an `actionSelector` function for custom training patterns.
|
|
13
|
-
|
|
14
|
-
```typescript
|
|
15
|
-
const result = await agent.train({
|
|
16
|
-
episodes: 1000,
|
|
17
|
-
actionSelector: (episode, lastAction) => {
|
|
18
|
-
// Your custom logic here
|
|
19
|
-
return action;
|
|
20
|
-
}
|
|
21
|
-
});
|
|
22
|
-
```
|
|
23
|
-
|
|
24
|
-
### Epsilon-Greedy with Decay
|
|
25
|
-
|
|
26
|
-
```typescript
|
|
27
|
-
let epsilon = 1.0;
|
|
28
|
-
const minEpsilon = 0.01;
|
|
29
|
-
const decayRate = 0.9995;
|
|
30
|
-
|
|
31
|
-
const result = await agent.train({
|
|
32
|
-
episodes: 10000,
|
|
33
|
-
actionSelector: (episode, lastAction) => {
|
|
34
|
-
epsilon = Math.max(minEpsilon, epsilon * decayRate);
|
|
35
|
-
|
|
36
|
-
if (Math.random() < epsilon) {
|
|
37
|
-
return Math.floor(Math.random() * 3); // Explore
|
|
38
|
-
}
|
|
39
|
-
|
|
40
|
-
// Exploit: use agent's Q-learning
|
|
41
|
-
const state = await agent.env.getState();
|
|
42
|
-
return agent.qlearning.act(state, 3);
|
|
43
|
-
}
|
|
44
|
-
});
|
|
45
|
-
```
|
|
46
|
-
|
|
47
|
-
### Curriculum Learning
|
|
48
|
-
|
|
49
|
-
Start with easy opponents, progress to harder ones.
|
|
50
|
-
|
|
51
|
-
```typescript
|
|
52
|
-
const opponents = ['always_rock', 'sequential', 'counter', 'random'];
|
|
53
|
-
const episodesPerStage = 250;
|
|
54
|
-
|
|
55
|
-
for (const opponent of opponents) {
|
|
56
|
-
const env = new RockPaperScissorsEnv({ opponent });
|
|
57
|
-
agent.setEnvironment(env);
|
|
58
|
-
|
|
59
|
-
console.log(`Training against ${opponent}...`);
|
|
60
|
-
await agent.train({ episodes: episodesPerStage });
|
|
61
|
-
|
|
62
|
-
const status = await agent.status();
|
|
63
|
-
console.log(`Win rate: ${status.performance.find(p => p.mode === 'train')?.win_rate}%`);
|
|
64
|
-
}
|
|
65
|
-
```
|
|
66
|
-
|
|
67
|
-
### Self-Play Training
|
|
68
|
-
|
|
69
|
-
Train agent against itself.
|
|
70
|
-
|
|
71
|
-
```typescript
|
|
72
|
-
// Create two agents sharing the same Q-table
|
|
73
|
-
const storage = await createStorage('sqlite', { path: './selfplay.db' });
|
|
74
|
-
const qlearning = new QLearning({ alpha: 0.1, gamma: 0.9, epsilon: 0.1, storage });
|
|
75
|
-
|
|
76
|
-
const env1 = new RockPaperScissorsEnv({ opponent: 'random' });
|
|
77
|
-
const env2 = new RockPaperScissorsEnv({ opponent: 'random' });
|
|
78
|
-
|
|
79
|
-
const agent1 = new Agent({ qlearning, storage, env: env1 });
|
|
80
|
-
const agent2 = new Agent({ qlearning, storage, env: env2 });
|
|
81
|
-
|
|
82
|
-
// Alternate training
|
|
83
|
-
for (let i = 0; i < 100; i++) {
|
|
84
|
-
await agent1.train({ episodes: 50 });
|
|
85
|
-
await agent2.train({ episodes: 50 });
|
|
86
|
-
|
|
87
|
-
if (i % 10 === 0) {
|
|
88
|
-
const status = await agent1.status();
|
|
89
|
-
console.log(`Iteration ${i}: ${status.performance[0].win_rate}% win rate`);
|
|
90
|
-
}
|
|
91
|
-
}
|
|
92
|
-
```
|
|
93
|
-
|
|
94
|
-
## Hyperparameter Tuning
|
|
95
|
-
|
|
96
|
-
### Grid Search
|
|
97
|
-
|
|
98
|
-
```typescript
|
|
99
|
-
const configs = [
|
|
100
|
-
{ alpha: 0.05, gamma: 0.9, epsilon: 0.1 },
|
|
101
|
-
{ alpha: 0.1, gamma: 0.9, epsilon: 0.1 },
|
|
102
|
-
{ alpha: 0.2, gamma: 0.9, epsilon: 0.1 },
|
|
103
|
-
{ alpha: 0.1, gamma: 0.95, epsilon: 0.1 },
|
|
104
|
-
{ alpha: 0.1, gamma: 0.9, epsilon: 0.2 },
|
|
105
|
-
];
|
|
106
|
-
|
|
107
|
-
for (const config of configs) {
|
|
108
|
-
const storage = await createStorage('memory');
|
|
109
|
-
const qlearning = new QLearning({ ...config, storage });
|
|
110
|
-
const env = new RockPaperScissorsEnv({ opponent: 'random' });
|
|
111
|
-
const agent = new Agent({ qlearning, storage, env });
|
|
112
|
-
|
|
113
|
-
await agent.train({ episodes: 2000 });
|
|
114
|
-
const status = await agent.status();
|
|
115
|
-
const winRate = status.performance.find(p => p.mode === 'train')?.win_rate || 0;
|
|
116
|
-
|
|
117
|
-
console.log(`${JSON.stringify(config)} => ${winRate}%`);
|
|
118
|
-
await storage.close();
|
|
119
|
-
}
|
|
120
|
-
```
|
|
121
|
-
|
|
122
|
-
### Bayesian Optimization
|
|
123
|
-
|
|
124
|
-
Use libraries like `bayes-opt` for efficient hyperparameter search.
|
|
125
|
-
|
|
126
|
-
## Evaluation Techniques
|
|
127
|
-
|
|
128
|
-
### Fixed Opponent Evaluation
|
|
129
|
-
|
|
130
|
-
```typescript
|
|
131
|
-
async function evaluate(agent, opponent, games = 100) {
|
|
132
|
-
const env = new RockPaperScissorsEnv({ opponent });
|
|
133
|
-
agent.setEnvironment(env);
|
|
134
|
-
|
|
135
|
-
let wins = 0, losses = 0, draws = 0;
|
|
136
|
-
|
|
137
|
-
for (let i = 0; i < games; i++) {
|
|
138
|
-
const result = await agent.play(Math.floor(Math.random() * 3));
|
|
139
|
-
if (result.reward > 0) wins++;
|
|
140
|
-
else if (result.reward < 0) losses++;
|
|
141
|
-
else draws++;
|
|
142
|
-
}
|
|
143
|
-
|
|
144
|
-
return { wins, losses, draws, winRate: wins / games };
|
|
145
|
-
}
|
|
146
|
-
|
|
147
|
-
const agents = {
|
|
148
|
-
random: await evaluate(agent, 'random'),
|
|
149
|
-
alwaysRock: await evaluate(agent, 'always_rock'),
|
|
150
|
-
counter: await evaluate(agent, 'counter'),
|
|
151
|
-
sequential: await evaluate(agent, 'sequential'),
|
|
152
|
-
};
|
|
153
|
-
|
|
154
|
-
console.table(agents);
|
|
155
|
-
```
|
|
156
|
-
|
|
157
|
-
### Cross-Validation
|
|
158
|
-
|
|
159
|
-
```typescript
|
|
160
|
-
async function crossValidate(config, folds = 5, episodesPerFold = 1000) {
|
|
161
|
-
const results = [];
|
|
162
|
-
|
|
163
|
-
for (let fold = 0; fold < folds; fold++) {
|
|
164
|
-
const storage = await createStorage('memory');
|
|
165
|
-
const qlearning = new QLearning({ ...config, storage });
|
|
166
|
-
const env = new RockPaperScissorsEnv({ opponent: 'random' });
|
|
167
|
-
const agent = new Agent({ qlearning, storage, env });
|
|
168
|
-
|
|
169
|
-
await agent.train({ episodes: episodesPerFold });
|
|
170
|
-
const evalResult = await evaluate(agent, 'random', 200);
|
|
171
|
-
results.push(evalResult.winRate);
|
|
172
|
-
|
|
173
|
-
await storage.close();
|
|
174
|
-
}
|
|
175
|
-
|
|
176
|
-
const mean = results.reduce((a, b) => a + b, 0) / results.length;
|
|
177
|
-
const std = Math.sqrt(results.reduce((a, b) => a + (b - mean) ** 2, 0) / results.length);
|
|
178
|
-
|
|
179
|
-
return { mean, std, results };
|
|
180
|
-
}
|
|
181
|
-
```
|
|
182
|
-
|
|
183
|
-
## Checkpointing and Resuming
|
|
184
|
-
|
|
185
|
-
```typescript
|
|
186
|
-
// Save Q-table periodically
|
|
187
|
-
async function trainWithCheckpoints(agent, episodes, checkpointEvery = 100) {
|
|
188
|
-
for (let i = 0; i < episodes; i += checkpointEvery) {
|
|
189
|
-
const batch = Math.min(checkpointEvery, episodes - i);
|
|
190
|
-
await agent.train({ episodes: batch });
|
|
191
|
-
|
|
192
|
-
// Q-table automatically persisted to storage
|
|
193
|
-
const status = await agent.status();
|
|
194
|
-
console.log(`Checkpoint ${i + batch}: ${status.aiBrain.length} Q-values`);
|
|
195
|
-
}
|
|
196
|
-
}
|
|
197
|
-
|
|
198
|
-
// Resume from existing Q-table
|
|
199
|
-
const storage = await createStorage('sqlite', { path: './existing.db' });
|
|
200
|
-
const qlearning = new QLearning({ alpha: 0.1, gamma: 0.9, epsilon: 0.1, storage });
|
|
201
|
-
// Q-table loads automatically on first use
|
|
202
|
-
```
|
|
203
|
-
|
|
204
|
-
## Distributed Training
|
|
205
|
-
|
|
206
|
-
Run multiple training processes with shared storage.
|
|
207
|
-
|
|
208
|
-
```bash
|
|
209
|
-
# Terminal 1
|
|
210
|
-
lsji train --episodes 500 --db-path ./shared.db --storage sqlite
|
|
211
|
-
|
|
212
|
-
# Terminal 2 (same database)
|
|
213
|
-
lsji train --episodes 500 --db-path ./shared.db --storage sqlite
|
|
214
|
-
|
|
215
|
-
# Terminal 3
|
|
216
|
-
lsji train --episodes 500 --db-path ./shared.db --storage sqlite
|
|
217
|
-
```
|
|
218
|
-
|
|
219
|
-
All processes read/write to the same SQLite database, enabling parallel training.
|
|
220
|
-
|
|
221
|
-
## Monitoring Training Progress
|
|
222
|
-
|
|
223
|
-
```typescript
|
|
224
|
-
async function trainWithLogging(agent, episodes) {
|
|
225
|
-
const history = [];
|
|
226
|
-
|
|
227
|
-
for (let i = 0; i < episodes; i += 100) {
|
|
228
|
-
await agent.train({ episodes: 100 });
|
|
229
|
-
|
|
230
|
-
const status = await agent.status();
|
|
231
|
-
const trainStat = status.performance.find(p => p.mode === 'train');
|
|
232
|
-
|
|
233
|
-
history.push({
|
|
234
|
-
episode: i + 100,
|
|
235
|
-
winRate: trainStat?.win_rate || 0,
|
|
236
|
-
qTableSize: status.aiBrain.length
|
|
237
|
-
});
|
|
238
|
-
|
|
239
|
-
console.log(`Episode ${i + 100}: ${history[history.length - 1].winRate}% win rate`);
|
|
240
|
-
}
|
|
241
|
-
|
|
242
|
-
return history;
|
|
243
|
-
}
|
|
244
|
-
```
|