Training in progress, step 20, checkpoint

Browse files

Files changed (4) hide show

last-checkpoint/optimizer.pt +1 -1
last-checkpoint/rng_state.pth +1 -1
last-checkpoint/scheduler.pt +1 -1
last-checkpoint/trainer_state.json +91 -5

last-checkpoint/optimizer.pt CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:f8e63b5c0db5c6e0c4a520687f3f6c4f0247aead4a1d3766794a2661f98b2e9b
 size 41459700

 version https://git-lfs.github.com/spec/v1
+oid sha256:656e55f9ff1d64cf8ea8f60e7adba37bcb805bab3053670c22a3660aa6555f2c
 size 41459700

last-checkpoint/rng_state.pth CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:11f775c1ccf3d22e28827fb97e42f46af086c555f27d776c68d4071ccefdb3a7
 size 14244

 version https://git-lfs.github.com/spec/v1
+oid sha256:5225e37d7d543a33e698142d3eeff29567a056e9f14e596173f4a3fbf89f1212
 size 14244

last-checkpoint/scheduler.pt CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:2b92c48cfe63570ecdc76f9963627895dd577d4f8ab23942155a3e930aa789f3
 size 1064

 version https://git-lfs.github.com/spec/v1
+oid sha256:1c9cf7b2578a838553bdc8215a789e53a772a64b9586c9497a256787be75da01
 size 1064

last-checkpoint/trainer_state.json CHANGED Viewed

@@ -1,9 +1,9 @@
 {
   "best_metric": NaN,
   "best_model_checkpoint": "miner_id_24/checkpoint-10",
-  "epoch": 0.0056409533211112676,
   "eval_steps": 5,
-  "global_step": 10,
   "is_hyper_param_search": false,
   "is_local_process_zero": true,
   "is_world_process_zero": true,
@@ -101,6 +101,92 @@
       "eval_samples_per_second": 1.612,
       "eval_steps_per_second": 0.807,
       "step": 10
     }
   ],
   "logging_steps": 1,
@@ -115,7 +201,7 @@
         "early_stopping_threshold": 0.0
       },
       "attributes": {
-        "early_stopping_patience_counter": 0
       }
     },
     "TrainerControl": {
@@ -124,12 +210,12 @@
         "should_evaluate": false,
         "should_log": false,
         "should_save": true,
-        "should_training_stop": false
       },
       "attributes": {}
     }
   },
-  "total_flos": 1742636046090240.0,
   "train_batch_size": 2,
   "trial_name": null,
   "trial_params": null

 {
   "best_metric": NaN,
   "best_model_checkpoint": "miner_id_24/checkpoint-10",
+  "epoch": 0.011281906642222535,
   "eval_steps": 5,
+  "global_step": 20,
   "is_hyper_param_search": false,
   "is_local_process_zero": true,
   "is_world_process_zero": true,
       "eval_samples_per_second": 1.612,
       "eval_steps_per_second": 0.807,
       "step": 10
+    },
+    {
+      "epoch": 0.006205048653222395,
+      "grad_norm": NaN,
+      "learning_rate": 4.988329086794122e-05,
+      "loss": 0.0,
+      "step": 11
+    },
+    {
+      "epoch": 0.006769143985333521,
+      "grad_norm": NaN,
+      "learning_rate": 4.984119057295783e-05,
+      "loss": 0.0,
+      "step": 12
+    },
+    {
+      "epoch": 0.007333239317444648,
+      "grad_norm": NaN,
+      "learning_rate": 4.979264274553905e-05,
+      "loss": 0.0,
+      "step": 13
+    },
+    {
+      "epoch": 0.007897334649555774,
+      "grad_norm": NaN,
+      "learning_rate": 4.973765998627628e-05,
+      "loss": 0.0,
+      "step": 14
+    },
+    {
+      "epoch": 0.008461429981666902,
+      "grad_norm": NaN,
+      "learning_rate": 4.967625656594782e-05,
+      "loss": 0.0,
+      "step": 15
+    },
+    {
+      "epoch": 0.008461429981666902,
+      "eval_loss": NaN,
+      "eval_runtime": 432.8515,
+      "eval_samples_per_second": 1.726,
+      "eval_steps_per_second": 0.864,
+      "step": 15
+    },
+    {
+      "epoch": 0.009025525313778029,
+      "grad_norm": NaN,
+      "learning_rate": 4.960844842181494e-05,
+      "loss": 0.0,
+      "step": 16
+    },
+    {
+      "epoch": 0.009589620645889155,
+      "grad_norm": NaN,
+      "learning_rate": 4.953425315348534e-05,
+      "loss": 0.0,
+      "step": 17
+    },
+    {
+      "epoch": 0.010153715978000282,
+      "grad_norm": NaN,
+      "learning_rate": 4.9453690018345144e-05,
+      "loss": 0.0,
+      "step": 18
+    },
+    {
+      "epoch": 0.01071781131011141,
+      "grad_norm": NaN,
+      "learning_rate": 4.93667799265607e-05,
+      "loss": 0.0,
+      "step": 19
+    },
+    {
+      "epoch": 0.011281906642222535,
+      "grad_norm": NaN,
+      "learning_rate": 4.92735454356513e-05,
+      "loss": 0.0,
+      "step": 20
+    },
+    {
+      "epoch": 0.011281906642222535,
+      "eval_loss": NaN,
+      "eval_runtime": 416.148,
+      "eval_samples_per_second": 1.795,
+      "eval_steps_per_second": 0.899,
+      "step": 20
     }
   ],
   "logging_steps": 1,
         "early_stopping_threshold": 0.0
       },
       "attributes": {
+        "early_stopping_patience_counter": 2
       }
     },
     "TrainerControl": {
         "should_evaluate": false,
         "should_log": false,
         "should_save": true,
+        "should_training_stop": true
       },
       "attributes": {}
     }
   },
+  "total_flos": 3485272092180480.0,
   "train_batch_size": 2,
   "trial_name": null,
   "trial_params": null