mubi commited on
Commit
78b0cca
·
verified ·
1 Parent(s): c135c33

Training in progress, step 6486, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:72410fcfce11116c642dad2ac1f851954613ae9ca9f37e106b99cf4cc8560a58
3
  size 161533192
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ea754247180acec31b5b60c9ee3e1d6fd7ce56dba7f4c60f16005dbd744581c2
3
  size 161533192
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:ae85ff0a08cdb19000c96789f83d4d2f0f08f9eb5bc57e5e57eff3b37fe31f52
3
  size 83485139
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ae85412c313971e6aec21cfd5503ef8a93f0211094610414e36c27ee1e29c78c
3
  size 83485139
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b76a7ba81bcbc059ee6a62bdea93f43091f236ec991103bc5ca0ea04659b8444
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1e01861a9dacc9c9f1e775212e57b52c6613522e0feab6ac1e96f98ea0ab030a
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:859b1a1ccc982c4b50f6001dac47e34317244bd40e686328c0369a7b85017494
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b8f204bc8ad3a4181d86b396ae73f7291ac1fabef2c4c1f2beb205fa474346cc
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,9 +2,9 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 2.9834624725338266,
6
  "eval_steps": 500,
7
- "global_step": 6450,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -4523,6 +4523,27 @@
4523
  "learning_rate": 1.1753789050417569e-06,
4524
  "loss": 0.1940107226371765,
4525
  "step": 6450
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
4526
  }
4527
  ],
4528
  "logging_steps": 10,
@@ -4537,12 +4558,12 @@
4537
  "should_evaluate": false,
4538
  "should_log": false,
4539
  "should_save": true,
4540
- "should_training_stop": false
4541
  },
4542
  "attributes": {}
4543
  }
4544
  },
4545
- "total_flos": 1.202617054555607e+17,
4546
  "train_batch_size": 2,
4547
  "trial_name": null,
4548
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 3.0,
6
  "eval_steps": 500,
7
+ "global_step": 6486,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
4523
  "learning_rate": 1.1753789050417569e-06,
4524
  "loss": 0.1940107226371765,
4525
  "step": 6450
4526
+ },
4527
+ {
4528
+ "epoch": 2.9880883543425467,
4529
+ "grad_norm": 0.30115655064582825,
4530
+ "learning_rate": 8.660686668728735e-07,
4531
+ "loss": 0.1959635615348816,
4532
+ "step": 6460
4533
+ },
4534
+ {
4535
+ "epoch": 2.9927142361512664,
4536
+ "grad_norm": 0.29174426198005676,
4537
+ "learning_rate": 5.567584287039901e-07,
4538
+ "loss": 0.19444901943206788,
4539
+ "step": 6470
4540
+ },
4541
+ {
4542
+ "epoch": 2.997340117959986,
4543
+ "grad_norm": 0.4074135422706604,
4544
+ "learning_rate": 2.474481905351067e-07,
4545
+ "loss": 0.20223517417907716,
4546
+ "step": 6480
4547
  }
4548
  ],
4549
  "logging_steps": 10,
 
4558
  "should_evaluate": false,
4559
  "should_log": false,
4560
  "should_save": true,
4561
+ "should_training_stop": true
4562
  },
4563
  "attributes": {}
4564
  }
4565
  },
4566
+ "total_flos": 1.2092503276056269e+17,
4567
  "train_batch_size": 2,
4568
  "trial_name": null,
4569
  "trial_params": null