File size: 5,048 Bytes
546e1b0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
{
  "schema_version": 1,
  "status": "complete",
  "config_fingerprint": "0667d4ea7a9eeb08101909c8f4b52860faadd491ce6d4ceb8094405e04a5f75e",
  "dataset_fingerprint": "6cf9eb3d777a338eb007dcc5606b2aac6605bf899a191ea399ccd77d5527acfb",
  "cursor": {
    "global_step": 1530,
    "micro_step": 12275,
    "dataset_epoch": 0,
    "sample_offset": 24480,
    "accumulation_step": 0
  },
  "extra": {
    "last_metrics": {
      "global_step": 1530,
      "behavior_loss": 0.01784470462007448,
      "behavior_microbatches": 8,
      "behavior_singleton_microbatches": 0,
      "oom_replayed_as_singletons": false,
      "gradient_norm": 0.017304301261901855,
      "learning_rate_used": 5e-06,
      "step_seconds": 12.875,
      "peak_reserved_gib": 24.904,
      "optimizer_state_device": "cpu_between_updates",
      "weighted_behavior_objective": 0.01784470462007448,
      "validation_behavior_loss": 0.017060593152418733,
      "lr_window_phase": "confirming",
      "lr_window_start_step": 1480,
      "lr_window_end_step": 1530,
      "lr_window_point_count": 6,
      "lr_window_log_slope_per_step": 2.7408283244737912e-05,
      "lr_window_fitted_log_descent": -0.0013704141622368955,
      "lr_window_relative_descent": -0.00137135360881957,
      "lr_window_residual_mad_scale": 0.0010623870441984341,
      "lr_window_descent_to_noise": -1.2899386995733428,
      "lr_window_accepted": false,
      "lr_window_elbow_step": 1490,
      "lr_training_window_start_step": 1490,
      "lr_training_window_end_step": 1530,
      "lr_training_window_point_count": 5,
      "lr_training_window_log_slope_per_step": 0.003067318613296203,
      "lr_training_window_fitted_log_descent": -0.12269274453184813,
      "lr_training_window_relative_descent": -0.13053700390093237,
      "lr_training_window_residual_mad_scale": 0.0651870427677517,
      "lr_training_window_descent_to_noise": -1.8821646039225566,
      "lr_training_window_accepted": false,
      "lr_training_window_elbow_step": 1500,
      "lr_window_any_signal_accepted": false,
      "validation_interval": 10,
      "next_validation_step": 1540,
      "learning_rate": 5e-06
    },
    "reason": "periodic",
    "parent": null,
    "lr_control": {
      "version": 3,
      "config": {
        "factor": 0.5,
        "minimum_learning_rate": 1e-06,
        "window_steps": 50,
        "probe_every_steps": 10,
        "confirmation_steps": 20,
        "minimum_descent_to_noise": 1.0,
        "minimum_relative_descent": 0.0
      },
      "state": {
        "learning_rate": 5e-06,
        "points": [
          {
            "step": 1480,
            "loss": 0.0170914699556306
          },
          {
            "step": 1490,
            "loss": 0.017028517997823657
          },
          {
            "step": 1500,
            "loss": 0.017046570871025325
          },
          {
            "step": 1510,
            "loss": 0.017040204368531705
          },
          {
            "step": 1520,
            "loss": 0.017080724351108075
          },
          {
            "step": 1530,
            "loss": 0.017060593152418733
          }
        ],
        "training_points": [
          {
            "step": 1490,
            "loss": 0.01357189457048662
          },
          {
            "step": 1500,
            "loss": 0.013392652114271186
          },
          {
            "step": 1510,
            "loss": 0.014630921845673583
          },
          {
            "step": 1520,
            "loss": 0.01602509914082475
          },
          {
            "step": 1530,
            "loss": 0.014382915897294879
          }
        ],
        "training_loss_sum": 0.0,
        "training_loss_count": 0,
        "last_training_step": 1530,
        "window_start_step": 1480,
        "next_validation_step": 1540,
        "last_analysis": {
          "start_step": 1480,
          "end_step": 1530,
          "point_count": 6,
          "log_slope_per_step": 2.7408283244737912e-05,
          "fitted_log_descent": -0.0013704141622368955,
          "relative_descent": -0.00137135360881957,
          "residual_mad_scale": 0.0010623870441984341,
          "descent_to_noise": -1.2899386995733428,
          "accepted": false,
          "elbow_step": 1490
        },
        "last_training_analysis": {
          "start_step": 1490,
          "end_step": 1530,
          "point_count": 5,
          "log_slope_per_step": 0.003067318613296203,
          "fitted_log_descent": -0.12269274453184813,
          "relative_descent": -0.13053700390093237,
          "residual_mad_scale": 0.0651870427677517,
          "descent_to_noise": -1.8821646039225566,
          "accepted": false,
          "elbow_step": 1500
        },
        "confirming": true,
        "pending_rollback": null,
        "reductions": 6
      }
    },
    "rollback_replay": null
  },
  "files": {
    "adapters.safetensors": "eb53557a4d1476bf2b62ac6946476bc4fcbf8282034ab04df1e6a33feceaa96c",
    "training_state.pt": "1fc980b573bfc358cc596786e8c5162a4d6cb44a4dfdd945923cf498cd58b447"
  }
}