kevinoli commited on
Commit
ac689a9
1 Parent(s): 75a5a28

Training in progress, step 5500, checkpoint

Browse files
checkpoint-5500/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:729a6a85e257b4c739b3ce105faee865913d25a537e71e9d9f091dfb1a3e5581
3
  size 1711848436
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9590c3d526f56c677e9ede89377608894aea6d14838d46126713f9329fc29629
3
  size 1711848436
checkpoint-5500/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2ec0874d9a6d5e1ff6c3dc10f3363ed475ba4c93b58ea632bcb1c816c9257ee5
3
  size 3424043887
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5c6192412b2e4e817cdf864acf47c8442158445b0d6c888300bf9f6b194682f5
3
  size 3424043887
checkpoint-5500/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:644d67403e188f6cf15854a8681855db17137190a385d9cf4e1f53e0e4ed045d
3
  size 14503
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:84d7386eb687578a262eee2929a289c77af23249d40cfd0b37a991f11419006b
3
  size 14503
checkpoint-5500/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:54280fd1c95daf5adabdb168b633e673edd4fb6899a62b3a3dd0b9b25686d7ad
3
  size 623
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4bc752e4e3d898fda3a115c9a6fb577f2d79dce43465c9e8a7d0ae405f494fc8
3
  size 623
checkpoint-5500/trainer_state.json CHANGED
@@ -1,7 +1,7 @@
1
  {
2
- "best_metric": 1.0355629920959473,
3
  "best_model_checkpoint": "./output/clip-finetuned-csu-p14-336-e4l58-l/checkpoint-5500",
4
- "epoch": 1.013264554163596,
5
  "eval_steps": 500,
6
  "global_step": 5500,
7
  "is_hyper_param_search": false,
@@ -9,173 +9,173 @@
9
  "is_world_process_zero": true,
10
  "log_history": [
11
  {
12
- "epoch": 0.09211495946941783,
13
- "grad_norm": 20.04043960571289,
14
- "learning_rate": 4.8848563006632274e-08,
15
- "loss": 0.3758,
16
  "step": 500
17
  },
18
  {
19
- "epoch": 0.09211495946941783,
20
- "eval_loss": 1.4184927940368652,
21
- "eval_runtime": 74.3562,
22
- "eval_samples_per_second": 16.233,
23
- "eval_steps_per_second": 2.031,
24
  "step": 500
25
  },
26
  {
27
- "epoch": 0.18422991893883567,
28
- "grad_norm": 74.25025939941406,
29
- "learning_rate": 4.7697126013264556e-08,
30
- "loss": 0.4103,
31
  "step": 1000
32
  },
33
  {
34
- "epoch": 0.18422991893883567,
35
- "eval_loss": 1.3500770330429077,
36
- "eval_runtime": 75.7013,
37
- "eval_samples_per_second": 15.944,
38
- "eval_steps_per_second": 1.995,
39
  "step": 1000
40
  },
41
  {
42
- "epoch": 0.2763448784082535,
43
- "grad_norm": 0.5102410912513733,
44
- "learning_rate": 4.6545689019896826e-08,
45
- "loss": 0.433,
46
  "step": 1500
47
  },
48
  {
49
- "epoch": 0.2763448784082535,
50
- "eval_loss": 1.2885024547576904,
51
- "eval_runtime": 75.197,
52
- "eval_samples_per_second": 16.051,
53
- "eval_steps_per_second": 2.008,
54
  "step": 1500
55
  },
56
  {
57
- "epoch": 0.36845983787767134,
58
- "grad_norm": 0.1667678952217102,
59
- "learning_rate": 4.539425202652911e-08,
60
- "loss": 0.3424,
61
  "step": 2000
62
  },
63
  {
64
- "epoch": 0.36845983787767134,
65
- "eval_loss": 1.239119052886963,
66
- "eval_runtime": 75.387,
67
- "eval_samples_per_second": 16.011,
68
- "eval_steps_per_second": 2.003,
69
  "step": 2000
70
  },
71
  {
72
- "epoch": 0.46057479734708917,
73
- "grad_norm": 343.8175048828125,
74
- "learning_rate": 4.4242815033161385e-08,
75
- "loss": 0.3645,
76
  "step": 2500
77
  },
78
  {
79
- "epoch": 0.46057479734708917,
80
- "eval_loss": 1.1902339458465576,
81
- "eval_runtime": 75.9899,
82
- "eval_samples_per_second": 15.884,
83
- "eval_steps_per_second": 1.987,
84
  "step": 2500
85
  },
86
  {
87
- "epoch": 0.552689756816507,
88
- "grad_norm": 305.5626525878906,
89
- "learning_rate": 4.309137803979366e-08,
90
- "loss": 0.3172,
91
  "step": 3000
92
  },
93
  {
94
- "epoch": 0.552689756816507,
95
- "eval_loss": 1.1506118774414062,
96
- "eval_runtime": 74.8655,
97
- "eval_samples_per_second": 16.122,
98
- "eval_steps_per_second": 2.017,
99
  "step": 3000
100
  },
101
  {
102
- "epoch": 0.6448047162859248,
103
- "grad_norm": 65.59048461914062,
104
- "learning_rate": 4.193994104642594e-08,
105
- "loss": 0.2751,
106
  "step": 3500
107
  },
108
  {
109
- "epoch": 0.6448047162859248,
110
- "eval_loss": 1.1169357299804688,
111
- "eval_runtime": 76.2936,
112
- "eval_samples_per_second": 15.82,
113
- "eval_steps_per_second": 1.979,
114
  "step": 3500
115
  },
116
  {
117
- "epoch": 0.7369196757553427,
118
- "grad_norm": 18.106201171875,
119
- "learning_rate": 4.078850405305821e-08,
120
- "loss": 0.2919,
121
  "step": 4000
122
  },
123
  {
124
- "epoch": 0.7369196757553427,
125
- "eval_loss": 1.0920671224594116,
126
- "eval_runtime": 74.9101,
127
- "eval_samples_per_second": 16.113,
128
- "eval_steps_per_second": 2.016,
129
  "step": 4000
130
  },
131
  {
132
- "epoch": 0.8290346352247605,
133
- "grad_norm": 506.40570068359375,
134
- "learning_rate": 3.9637067059690496e-08,
135
- "loss": 0.2583,
136
  "step": 4500
137
  },
138
  {
139
- "epoch": 0.8290346352247605,
140
- "eval_loss": 1.0721209049224854,
141
- "eval_runtime": 75.8053,
142
- "eval_samples_per_second": 15.922,
143
- "eval_steps_per_second": 1.992,
144
  "step": 4500
145
  },
146
  {
147
- "epoch": 0.9211495946941783,
148
- "grad_norm": 1.1508910655975342,
149
- "learning_rate": 3.848563006632277e-08,
150
- "loss": 0.2679,
151
  "step": 5000
152
  },
153
  {
154
- "epoch": 0.9211495946941783,
155
- "eval_loss": 1.0519349575042725,
156
- "eval_runtime": 75.226,
157
- "eval_samples_per_second": 16.045,
158
- "eval_steps_per_second": 2.007,
159
  "step": 5000
160
  },
161
  {
162
- "epoch": 1.013264554163596,
163
- "grad_norm": 0.1393290013074875,
164
- "learning_rate": 3.733419307295505e-08,
165
- "loss": 0.2472,
166
  "step": 5500
167
  },
168
  {
169
- "epoch": 1.013264554163596,
170
- "eval_loss": 1.0355629920959473,
171
- "eval_runtime": 74.6535,
172
- "eval_samples_per_second": 16.168,
173
- "eval_steps_per_second": 2.023,
174
  "step": 5500
175
  }
176
  ],
177
  "logging_steps": 500,
178
- "max_steps": 21712,
179
  "num_input_tokens_seen": 0,
180
  "num_train_epochs": 4,
181
  "save_steps": 500,
 
1
  {
2
+ "best_metric": 1.0982334613800049,
3
  "best_model_checkpoint": "./output/clip-finetuned-csu-p14-336-e4l58-l/checkpoint-5500",
4
+ "epoch": 0.58660409556314,
5
  "eval_steps": 500,
6
  "global_step": 5500,
7
  "is_hyper_param_search": false,
 
9
  "is_world_process_zero": true,
10
  "log_history": [
11
  {
12
+ "epoch": 0.05332764505119454,
13
+ "grad_norm": 217.55360412597656,
14
+ "learning_rate": 4.933340443686007e-08,
15
+ "loss": 0.4667,
16
  "step": 500
17
  },
18
  {
19
+ "epoch": 0.05332764505119454,
20
+ "eval_loss": 1.4426143169403076,
21
+ "eval_runtime": 61.9901,
22
+ "eval_samples_per_second": 15.922,
23
+ "eval_steps_per_second": 2.0,
24
  "step": 500
25
  },
26
  {
27
+ "epoch": 0.10665529010238908,
28
+ "grad_norm": 73.73340606689453,
29
+ "learning_rate": 4.8666808873720136e-08,
30
+ "loss": 0.4532,
31
  "step": 1000
32
  },
33
  {
34
+ "epoch": 0.10665529010238908,
35
+ "eval_loss": 1.3815597295761108,
36
+ "eval_runtime": 62.4624,
37
+ "eval_samples_per_second": 15.801,
38
+ "eval_steps_per_second": 1.985,
39
  "step": 1000
40
  },
41
  {
42
+ "epoch": 0.1599829351535836,
43
+ "grad_norm": 471.4011535644531,
44
+ "learning_rate": 4.80002133105802e-08,
45
+ "loss": 0.3749,
46
  "step": 1500
47
  },
48
  {
49
+ "epoch": 0.1599829351535836,
50
+ "eval_loss": 1.3310539722442627,
51
+ "eval_runtime": 63.3918,
52
+ "eval_samples_per_second": 15.57,
53
+ "eval_steps_per_second": 1.956,
54
  "step": 1500
55
  },
56
  {
57
+ "epoch": 0.21331058020477817,
58
+ "grad_norm": 11.962693214416504,
59
+ "learning_rate": 4.733361774744027e-08,
60
+ "loss": 0.336,
61
  "step": 2000
62
  },
63
  {
64
+ "epoch": 0.21331058020477817,
65
+ "eval_loss": 1.2890639305114746,
66
+ "eval_runtime": 63.177,
67
+ "eval_samples_per_second": 15.623,
68
+ "eval_steps_per_second": 1.963,
69
  "step": 2000
70
  },
71
  {
72
+ "epoch": 0.2666382252559727,
73
+ "grad_norm": 375.1609191894531,
74
+ "learning_rate": 4.666702218430034e-08,
75
+ "loss": 0.3585,
76
  "step": 2500
77
  },
78
  {
79
+ "epoch": 0.2666382252559727,
80
+ "eval_loss": 1.2536433935165405,
81
+ "eval_runtime": 63.1512,
82
+ "eval_samples_per_second": 15.629,
83
+ "eval_steps_per_second": 1.964,
84
  "step": 2500
85
  },
86
  {
87
+ "epoch": 0.3199658703071672,
88
+ "grad_norm": 0.0022936267778277397,
89
+ "learning_rate": 4.600042662116041e-08,
90
+ "loss": 0.303,
91
  "step": 3000
92
  },
93
  {
94
+ "epoch": 0.3199658703071672,
95
+ "eval_loss": 1.2202869653701782,
96
+ "eval_runtime": 63.5857,
97
+ "eval_samples_per_second": 15.522,
98
+ "eval_steps_per_second": 1.95,
99
  "step": 3000
100
  },
101
  {
102
+ "epoch": 0.37329351535836175,
103
+ "grad_norm": 15.805087089538574,
104
+ "learning_rate": 4.5333831058020476e-08,
105
+ "loss": 0.3242,
106
  "step": 3500
107
  },
108
  {
109
+ "epoch": 0.37329351535836175,
110
+ "eval_loss": 1.195627212524414,
111
+ "eval_runtime": 63.4024,
112
+ "eval_samples_per_second": 15.567,
113
+ "eval_steps_per_second": 1.956,
114
  "step": 3500
115
  },
116
  {
117
+ "epoch": 0.42662116040955633,
118
+ "grad_norm": 5.377908229827881,
119
+ "learning_rate": 4.4667235494880546e-08,
120
+ "loss": 0.2427,
121
  "step": 4000
122
  },
123
  {
124
+ "epoch": 0.42662116040955633,
125
+ "eval_loss": 1.169384241104126,
126
+ "eval_runtime": 63.5082,
127
+ "eval_samples_per_second": 15.541,
128
+ "eval_steps_per_second": 1.953,
129
  "step": 4000
130
  },
131
  {
132
+ "epoch": 0.47994880546075086,
133
+ "grad_norm": 150.5334930419922,
134
+ "learning_rate": 4.4000639931740615e-08,
135
+ "loss": 0.2993,
136
  "step": 4500
137
  },
138
  {
139
+ "epoch": 0.47994880546075086,
140
+ "eval_loss": 1.145558476448059,
141
+ "eval_runtime": 62.3888,
142
+ "eval_samples_per_second": 15.82,
143
+ "eval_steps_per_second": 1.988,
144
  "step": 4500
145
  },
146
  {
147
+ "epoch": 0.5332764505119454,
148
+ "grad_norm": 0.12246542423963547,
149
+ "learning_rate": 4.333404436860068e-08,
150
+ "loss": 0.3183,
151
  "step": 5000
152
  },
153
  {
154
+ "epoch": 0.5332764505119454,
155
+ "eval_loss": 1.1201218366622925,
156
+ "eval_runtime": 63.7154,
157
+ "eval_samples_per_second": 15.491,
158
+ "eval_steps_per_second": 1.946,
159
  "step": 5000
160
  },
161
  {
162
+ "epoch": 0.58660409556314,
163
+ "grad_norm": 0.3306196630001068,
164
+ "learning_rate": 4.266744880546075e-08,
165
+ "loss": 0.307,
166
  "step": 5500
167
  },
168
  {
169
+ "epoch": 0.58660409556314,
170
+ "eval_loss": 1.0982334613800049,
171
+ "eval_runtime": 62.4882,
172
+ "eval_samples_per_second": 15.795,
173
+ "eval_steps_per_second": 1.984,
174
  "step": 5500
175
  }
176
  ],
177
  "logging_steps": 500,
178
+ "max_steps": 37504,
179
  "num_input_tokens_seen": 0,
180
  "num_train_epochs": 4,
181
  "save_steps": 500,
checkpoint-5500/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9a29c2374e59ce3f57b34db9cb20d2d1bdc870c3dad128e9a4d462636c37e8b0
3
  size 4847
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:87e35dca972e13467c750106c19301b098fbd2918fda1087487e227bac3aca37
3
  size 4847