ygaci commited on
Commit
c3e44e9
·
verified ·
1 Parent(s): 03bee23

Training in progress, epoch 2, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:36c4183b3f6ae7f8de61238edc780abaf030dcb3be9b32af3b20cb6c1492fe04
3
  size 146815928
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:352aaca1805beff8b3dca9410c6799fba0c5d2ebef2b9843d0cc983b32b85269
3
  size 146815928
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d22e73536337c711d81d48701cdc1584bca29db7505507292c31dd295916a203
3
  size 74680890
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cef92f4f1cc5379d8df11ef1f05734dc3bf1aa7ad0721e7ac4489c256d8e4d2c
3
  size 74680890
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6968d40cf0c71d4e3ee9bf2859204537af78d1178eb0820c06fa4101d008adb9
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e7c2b9c568ef9032dbbe63731fc47d6288db403b82ba8fc7fe6f0e4bbf73753d
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:cd0ba268c49e105d03b88e15b67f1fa9af614b44c781f77e09e8b26a00bd96e7
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0a3d6766978e7b23f56ff8b154d43c7f45dbb5076c006b6787ac495916d5ce0c
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
- "epoch": 1.0,
5
  "eval_steps": 500,
6
- "global_step": 350,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -36,6 +36,42 @@
36
  "eval_samples_per_second": 2.109,
37
  "eval_steps_per_second": 0.527,
38
  "step": 350
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
39
  }
40
  ],
41
  "logging_steps": 100,
@@ -55,7 +91,7 @@
55
  "attributes": {}
56
  }
57
  },
58
- "total_flos": 8531255003971584.0,
59
  "train_batch_size": 4,
60
  "trial_name": null,
61
  "trial_params": null
 
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
+ "epoch": 2.0,
5
  "eval_steps": 500,
6
+ "global_step": 700,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
36
  "eval_samples_per_second": 2.109,
37
  "eval_steps_per_second": 0.527,
38
  "step": 350
39
+ },
40
+ {
41
+ "epoch": 1.1428571428571428,
42
+ "grad_norm": 1.8639676570892334,
43
+ "learning_rate": 0.00016,
44
+ "loss": 0.5839,
45
+ "step": 400
46
+ },
47
+ {
48
+ "epoch": 1.4285714285714286,
49
+ "grad_norm": 1.6524343490600586,
50
+ "learning_rate": 0.0002,
51
+ "loss": 0.4143,
52
+ "step": 500
53
+ },
54
+ {
55
+ "epoch": 1.7142857142857144,
56
+ "grad_norm": 1.1418366432189941,
57
+ "learning_rate": 0.00019333333333333333,
58
+ "loss": 0.3445,
59
+ "step": 600
60
+ },
61
+ {
62
+ "epoch": 2.0,
63
+ "grad_norm": 1.0962567329406738,
64
+ "learning_rate": 0.0001866666666666667,
65
+ "loss": 0.3012,
66
+ "step": 700
67
+ },
68
+ {
69
+ "epoch": 2.0,
70
+ "eval_loss": 0.2971023917198181,
71
+ "eval_runtime": 331.597,
72
+ "eval_samples_per_second": 2.111,
73
+ "eval_steps_per_second": 0.528,
74
+ "step": 700
75
  }
76
  ],
77
  "logging_steps": 100,
 
91
  "attributes": {}
92
  }
93
  },
94
+ "total_flos": 1.7093931183374336e+16,
95
  "train_batch_size": 4,
96
  "trial_name": null,
97
  "trial_params": null