File size: 3,738 Bytes
ba7b9c7
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
d8c8e20
ba7b9c7
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
import random
import torch

AU = 149.6e9
G = 6.67428e-11
Scale = 250 / AU   # 1AU = 100 pixels
Step_p_frame = 1
TIMESTEP = 3600 # 1 hour
WIDTH, HEIGHT =  800, 600
FPS = 60e3
ACCELERATING_RATE = 1_000_000


WHITE = (255, 255, 255)
JUP = (255,255,75)
YELLOW = (255, 255, 0)
BLUE = (100, 149, 237)
RED = (188, 39, 50)
DARK_GREY = (80, 78, 81)
LIGHT_GREY = (175,173,179)
GREEN = (100,230,72)

random.random()

DEVICE = "cpu"

# Weights for reward components 

# --- Constants for Target Distance Reward Function ---
TARGET_DISTANCE_FACTOR = 5e3       # Target distance = TARGET_DISTANCE_FACTOR * planet_radius
# Scale factor for error (e.g., 1 million km = 1e9 meters, or adjust based on typical orbital scales)
DISTANCE_REWARD_SCALE_FACTOR = 1e9
# Alpha controls the sharpness of the peak reward (applied to scaled error)
REWARD_ALPHA = 10.0
# Coeff determines the max bonus reward at the target distance
REWARD_COEFF = 50.0     
""" TESTS ON REWARD DISTANCE
# REWARD_POTENTIAL_SCALE = 1e13
"""
REWARD_DISTANCE_SCALE = 1e-9  # Scales the reward/penalty for change in distance


REWARD_SPEED_SCALE = 1e1    # Scales the reward for being close to orbital speed
NO_THRUST_REWARD = 25         # Penalty for firing any thruster (action > 0)
TIME_PENALTY = 0.01           # Small penalty for each time step taken
GOAL_REWARD = 100.0           # Large reward for achieving stable orbit with motors off
ORBIT_SPEED_TOLERANCE = 100   # Speed difference tolerance (m/s) to be considered 'in orbit' for reward


# --- DQN Hyperparameters ---
STATE_SIZE = 4          # State vector size: [dx, dy, dvx, dvy] -> Update if state definition changes
ACTION_SIZE = 7         # Actions: 0: none, 1: right, 2: left, 3: up, 4:right+up, 5:left+up, 6:right+left
MEMORY_CAPACITY = 50000 # Replay memory size (adjust based on RAM)
BATCH_SIZE = 128        # Number of experiences to sample for each learning step
GAMMA = 0.99            # Discount factor for future rewards
EPS_START = 0.05        # Starting value for epsilon (exploration rate)
EPS_END = 0.01          # Minimum value for epsilon
EPS_DECAY = 20000       # Controls how fast epsilon decreases (higher means slower decay)
TAU = 0.005             # Soft update parameter for target network weights
LR = 5e-4               # Learning rate for the Adam optimizer
TARGET_UPDATE_FREQ = 10 # How many steps between hard updates of target network (if not using soft updates)
                        # If using soft updates (TAU > 0), this can be ignored or set to 1 for updates every step.

# --- Att Hyperparameters ---
# Define constants (consider moving these to Const.py)
SEQ_LENGTH_DEFAULT = 50
# Velocity normalization factor used in the original get_state [cite: 124]
NORM_FACTOR_VEL = 5e4
PADDING_VALUE = 0.0 # Default padding for numerical sequences
ACTION_PADDING_VALUE = 0 # Default padding for action sequence
# These should match the output of get_state_sequence
NUM_FEATURES = 5  # action, rel_x, rel_y, rel_vx, rel_vy
SEQ_LENGTH = 50   # Example sequence length, should be configurable
MAX_HISTORY_LEN = 5000 # Max length for deques (can be larger than SEQ_LENGTH)
MODEL_NAME = "AttentionDQN" # Name for saving/loading

# --- Simulation Settings for RL ---
CRASH_DISTANCE_THRESHOLD = 1.0E6 # Example: Define a crash distance from planet surface (meters) - needs tuning
OUT_OF_BOUNDS_DISTANCE = 25 * AU # Example: Max distance from Sun before episode ends

Core_position = {
                0:'Sun',
                1:'Mercury',
                2:'Venus',
                3:'Earth',
                4:'Moon',
                5:'Mars',
                6:'Jupiter',
                7:'Saturn',
                8:'Uranus',
                9:'Neptune',
                10:'Pluto',
                }