File size: 10,318 Bytes
a9d8689
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
"""
Training script for NGO Multi-Agent Coordination Environment
Trains agents using simple Q-learning approach
"""

import numpy as np
import matplotlib.pyplot as plt
from typing import List, Tuple, Dict
import json
from ngo_coordination_env import NGOCoordinationEnv


class SimpleRLAgent:
    """
    Simple Q-learning based agent for demonstration
    In production, you'd use PPO, SAC, or MADDPG from Stable-Baselines3
    """
    
    def __init__(self, action_dim: int = 3, learning_rate: float = 0.01):
        self.lr = learning_rate
        self.action_dim = action_dim
        
        # Initialize policy parameters (simple neural network weights)
        self.theta = np.random.randn(action_dim) * 0.1
        
        # Exploration parameters
        self.epsilon = 1.0  # Start with high exploration
        self.epsilon_decay = 0.995
        self.epsilon_min = 0.05
    
    def select_action(self, observation: Dict, explore: bool = True) -> np.ndarray:
        """Select action using epsilon-greedy strategy"""
        # Extract state features
        urgency = observation['urgency'] / 10.0
        resources = observation['available_resources'][0] / 100.0
        people = observation['people_affected'][0] / 300.0
        
        # Simple feature vector
        state_features = np.array([urgency, resources, people])
        
        if explore and np.random.random() < self.epsilon:
            # Explore: random action
            action = np.random.uniform(0, 1, size=self.action_dim)
        else:
            # Exploit: use learned policy
            # Simple linear policy: action = sigmoid(theta * features)
            raw_action = self.theta * np.mean(state_features)
            action = 1.0 / (1.0 + np.exp(-raw_action))  # Sigmoid
            action = np.clip(action, 0, 1)
        
        return action
    
    def update(self, observation: Dict, action: np.ndarray, reward: float, next_observation: Dict):
        """Update policy based on experience"""
        # Simple policy gradient update
        # In practice, use proper RL algorithms (PPO, SAC, etc.)
        
        # Gradient direction: increase probability of actions that gave high reward
        gradient = reward * action * 0.01
        self.theta += self.lr * np.mean(gradient)
        
        # Decay exploration
        self.epsilon = max(self.epsilon * self.epsilon_decay, self.epsilon_min)


def train_multi_agent_system(
    num_episodes: int = 100,
    max_steps: int = 50,
    num_agents: int = 3
) -> Tuple[List[float], List[Dict]]:
    """
    Train multiple agents in the NGO coordination environment
    
    Returns:
        rewards_history: List of total rewards per episode
        info_history: Detailed logs for analysis
    """
    # Create environment
    env = NGOCoordinationEnv(num_agents=num_agents, max_steps=max_steps)
    
    # Create agents
    agents = [SimpleRLAgent(action_dim=3) for _ in range(num_agents)]
    
    # Training history
    rewards_history = []
    info_history = []
    
    print("Starting Multi-Agent Training...")
    print(f"Episodes: {num_episodes}, Max Steps: {max_steps}, Agents: {num_agents}\n")
    
    for episode in range(num_episodes):
        observation, info = env.reset()
        episode_reward = 0
        episode_info = {
            'episode': episode + 1,
            'steps': [],
            'task_type': info['task_type']
        }
        
        for step in range(max_steps):
            # Each agent selects action
            actions = np.vstack([agent.select_action(observation, explore=True) 
                                for agent in agents])
            
            # Environment step
            next_observation, reward, terminated, truncated, step_info = env.step(actions)
            
            # Update agents
            for i, agent in enumerate(agents):
                agent.update(observation, actions[i], reward, next_observation)
            
            # Track progress
            episode_reward += reward
            episode_info['steps'].append({
                'step': step + 1,
                'reward': reward,
                'allocations': step_info['allocations'].tolist()
            })
            
            observation = next_observation
            
            if terminated or truncated:
                break
        
        rewards_history.append(episode_reward)
        info_history.append(episode_info)
        
        # Print progress
        if (episode + 1) % 10 == 0:
            avg_reward = np.mean(rewards_history[-10:])
            print(f"Episode {episode + 1}/{num_episodes} | "
                  f"Avg Reward (last 10): {avg_reward:.2f} | "
                  f"Task: {info['task_type']}")
    
    print("\nTraining Complete!")
    return rewards_history, info_history


def plot_learning_curves(rewards_history: List[float], info_history: List[Dict]):
    """Generate the 3 required visualizations"""
    
    # Calculate cumulative steps
    cumulative_steps = []
    total = 0
    for info in info_history:
        for step_info in info['steps']:
            total += 1
            cumulative_steps.append(total)
    
    # Extract all step rewards
    all_rewards = []
    for info in info_history:
        for step_info in info['steps']:
            all_rewards.append(step_info['reward'])
    
    # 1. OVERALL LEARNING CURVE
    plt.figure(figsize=(12, 6))
    plt.scatter(cumulative_steps, all_rewards, alpha=0.5, s=20, label='Step Rewards', color='green')
    
    # Smooth curve
    from scipy.ndimage import gaussian_filter1d
    if len(all_rewards) > 10:
        smoothed = gaussian_filter1d(all_rewards, sigma=5)
        plt.plot(cumulative_steps, smoothed, color='darkgreen', linewidth=2, label='Learning Curve')
    
    # Trend line
    z = np.polyfit(cumulative_steps, all_rewards, 1)
    p = np.poly1d(z)
    plt.plot(cumulative_steps, p(cumulative_steps), "--", color='blue', linewidth=2, label='Trend')
    
    plt.xlabel('Step Number', fontsize=12)
    plt.ylabel('Reward', fontsize=12)
    plt.title('Multi-Agent Learning Curve - Overall Progress', fontsize=14, fontweight='bold')
    plt.legend()
    plt.grid(alpha=0.3)
    plt.tight_layout()
    plt.savefig('results/overall_learning_curve.png', dpi=150)
    print("Saved: results/overall_learning_curve.png")
    
    # 2. TASK COMPARISON (4 subplots)
    fig, axes = plt.subplots(2, 2, figsize=(14, 10))
    axes = axes.flatten()
    
    task_types = ['cooperation', 'competition', 'negotiation', 'coalition']
    colors = ['blue', 'green', 'orange', 'red']
    
    for idx, task_type in enumerate(task_types):
        # Filter episodes for this task
        task_episodes = [info for info in info_history if info['task_type'] == task_type]
        
        task_rewards = []
        task_steps = []
        step_counter = 0
        
        for ep_info in task_episodes:
            for step_info in ep_info['steps']:
                task_rewards.append(step_info['reward'])
                task_steps.append(step_counter)
                step_counter += 1
        
        if task_rewards:
            axes[idx].scatter(task_steps, task_rewards, alpha=0.6, s=15, color=colors[idx])
            
            if len(task_rewards) > 5:
                smoothed = gaussian_filter1d(task_rewards, sigma=3)
                axes[idx].plot(task_steps, smoothed, color=colors[idx], linewidth=2)
            
            axes[idx].set_title(f'Task: {task_type.capitalize()}', fontweight='bold')
            axes[idx].set_xlabel('Step')
            axes[idx].set_ylabel('Reward')
            axes[idx].grid(alpha=0.3)
    
    plt.tight_layout()
    plt.savefig('results/task_comparison.png', dpi=150)
    print("Saved: results/task_comparison.png")
    
    # 3. TASK PROGRESSION (bar chart)
    plt.figure(figsize=(10, 6))
    
    task_avg_rewards = []
    for task_type in task_types:
        task_rewards = []
        for info in info_history:
            if info['task_type'] == task_type:
                for step_info in info['steps']:
                    task_rewards.append(step_info['reward'])
        task_avg_rewards.append(np.mean(task_rewards) if task_rewards else 0)
    
    bars = plt.bar(range(len(task_types)), task_avg_rewards, color=colors, alpha=0.7, edgecolor='black')
    
    # Add value labels on bars
    for i, (bar, val) in enumerate(zip(bars, task_avg_rewards)):
        plt.text(bar.get_x() + bar.get_width()/2, bar.get_height() + 0.5,
                f'{val:.2f}', ha='center', fontweight='bold')
    
    plt.xlabel('Task Type', fontsize=12)
    plt.ylabel('Average Reward', fontsize=12)
    plt.title('Task Progression - Average Rewards', fontsize=14, fontweight='bold')
    plt.xticks(range(len(task_types)), [t.capitalize() for t in task_types])
    plt.grid(axis='y', alpha=0.3)
    plt.tight_layout()
    plt.savefig('results/task_progression.png', dpi=150)
    print("Saved: results/task_progression.png")
    
    plt.close('all')


def main():
    """Main execution"""
    
    # Create results directory
    import os
    os.makedirs('results', exist_ok=True)
    
    # Train agents
    rewards_history, info_history = train_multi_agent_system(
        num_episodes=100,  # 100 episodes total
        max_steps=50,      # 50 steps per episode
        num_agents=3       # 3 agents
    )
    
    # Generate visualizations
    print("\nGenerating Visualizations...")
    plot_learning_curves(rewards_history, info_history)
    
    # Save training data
    with open('results/training_results.json', 'w') as f:
        json.dump({
            'rewards_history': rewards_history,
            'info_history': info_history
        }, f, indent=2)
    print("Saved: results/training_results.json")
    
    # Print summary statistics
    print("\n" + "="*60)
    print("TRAINING SUMMARY")
    print("="*60)
    print(f"Total Episodes: {len(rewards_history)}")
    print(f"First 10 Episodes Avg Reward: {np.mean(rewards_history[:10]):.2f}")
    print(f"Last 10 Episodes Avg Reward: {np.mean(rewards_history[-10:]):.2f}")
    improvement = ((np.mean(rewards_history[-10:]) - np.mean(rewards_history[:10])) / 
                   np.mean(rewards_history[:10]) * 100)
    print(f"Improvement: {improvement:.1f}%")
    print("="*60)


if __name__ == "__main__":
    main()