ramdhanfirdaus commited on
Commit
e5c5089
1 Parent(s): c4db3b1

Training in progress, step 800, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b6786a7e41ebfcfd9932b459935a215f8ebec4f7b7d4b51c05869e30aff97a40
3
  size 9444296
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:dd60c41843428e1b79d82ed5e8a6ca5a63df384a3d4ae70610ab91e9d7622128
3
  size 9444296
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a03b074570a0ff3a9c1a1337dceb1382505c03a5c0e313013a67abe37e5618db
3
  size 18902665
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4022a9ac3c7308cd04fc492323d56144cd4d417816030f86fd2ff356f1b6a85d
3
  size 18902665
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7393dbde4bffa4ea759a39a2e6dd5d0164b7e91c9e8ab3bfffc0ca38d5daac71
3
  size 14575
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4cf66005e5e6a4f3553669f4e9894b69e1bc5f9207c5bf2c3d2ce3aab9394607
3
  size 14575
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:216f76b8039f833c337db298c81f13b12082d5fd4f9d866cecd34b2ca7550b37
3
  size 627
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:de7840bcb72f2f480fd301578d289cdfa174589e831b0d33e5772f3956b6beae
3
  size 627
last-checkpoint/tokenizer.json CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:17a208233d2ee8d8c83b23bc214df737c44806a1919f444e89b31e586cd956ba
3
- size 14500471
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5d81d9b2c9d9db79ea02c00d4c7e79bb77a718dc57ab01f5f3b1cd6649f08993
3
+ size 14500569
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
- "best_metric": 2.6424405574798584,
3
- "best_model_checkpoint": "./outputs/checkpoint-100",
4
- "epoch": 0.07285974499089254,
5
  "eval_steps": 100,
6
- "global_step": 100,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -11,23 +11,121 @@
11
  {
12
  "epoch": 0.07,
13
  "learning_rate": 0.0002,
14
- "loss": 2.7406,
15
  "step": 100
16
  },
17
  {
18
  "epoch": 0.07,
19
- "eval_loss": 2.6424405574798584,
20
- "eval_runtime": 206.8728,
21
- "eval_samples_per_second": 30.328,
22
- "eval_steps_per_second": 3.795,
23
  "step": 100
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
24
  }
25
  ],
26
  "logging_steps": 100,
27
  "max_steps": 4116,
28
  "num_train_epochs": 3,
29
  "save_steps": 100,
30
- "total_flos": 2917794121482240.0,
31
  "trial_name": null,
32
  "trial_params": null
33
  }
 
1
  {
2
+ "best_metric": 2.4231719970703125,
3
+ "best_model_checkpoint": "./outputs/checkpoint-800",
4
+ "epoch": 0.5828779599271403,
5
  "eval_steps": 100,
6
+ "global_step": 800,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
11
  {
12
  "epoch": 0.07,
13
  "learning_rate": 0.0002,
14
+ "loss": 2.7399,
15
  "step": 100
16
  },
17
  {
18
  "epoch": 0.07,
19
+ "eval_loss": 2.6418099403381348,
20
+ "eval_runtime": 347.6157,
21
+ "eval_samples_per_second": 18.049,
22
+ "eval_steps_per_second": 2.258,
23
  "step": 100
24
+ },
25
+ {
26
+ "epoch": 0.15,
27
+ "learning_rate": 0.0002,
28
+ "loss": 2.6052,
29
+ "step": 200
30
+ },
31
+ {
32
+ "epoch": 0.15,
33
+ "eval_loss": 2.5918312072753906,
34
+ "eval_runtime": 333.731,
35
+ "eval_samples_per_second": 18.8,
36
+ "eval_steps_per_second": 2.352,
37
+ "step": 200
38
+ },
39
+ {
40
+ "epoch": 0.22,
41
+ "learning_rate": 0.0002,
42
+ "loss": 2.5622,
43
+ "step": 300
44
+ },
45
+ {
46
+ "epoch": 0.22,
47
+ "eval_loss": 2.551574468612671,
48
+ "eval_runtime": 204.9306,
49
+ "eval_samples_per_second": 30.615,
50
+ "eval_steps_per_second": 3.831,
51
+ "step": 300
52
+ },
53
+ {
54
+ "epoch": 0.29,
55
+ "learning_rate": 0.0002,
56
+ "loss": 2.5366,
57
+ "step": 400
58
+ },
59
+ {
60
+ "epoch": 0.29,
61
+ "eval_loss": 2.517575263977051,
62
+ "eval_runtime": 204.3925,
63
+ "eval_samples_per_second": 30.696,
64
+ "eval_steps_per_second": 3.841,
65
+ "step": 400
66
+ },
67
+ {
68
+ "epoch": 0.36,
69
+ "learning_rate": 0.0002,
70
+ "loss": 2.4946,
71
+ "step": 500
72
+ },
73
+ {
74
+ "epoch": 0.36,
75
+ "eval_loss": 2.4924821853637695,
76
+ "eval_runtime": 204.4035,
77
+ "eval_samples_per_second": 30.694,
78
+ "eval_steps_per_second": 3.84,
79
+ "step": 500
80
+ },
81
+ {
82
+ "epoch": 0.44,
83
+ "learning_rate": 0.0002,
84
+ "loss": 2.4686,
85
+ "step": 600
86
+ },
87
+ {
88
+ "epoch": 0.44,
89
+ "eval_loss": 2.4666266441345215,
90
+ "eval_runtime": 207.3453,
91
+ "eval_samples_per_second": 30.259,
92
+ "eval_steps_per_second": 3.786,
93
+ "step": 600
94
+ },
95
+ {
96
+ "epoch": 0.51,
97
+ "learning_rate": 0.0002,
98
+ "loss": 2.4503,
99
+ "step": 700
100
+ },
101
+ {
102
+ "epoch": 0.51,
103
+ "eval_loss": 2.4440107345581055,
104
+ "eval_runtime": 205.5485,
105
+ "eval_samples_per_second": 30.523,
106
+ "eval_steps_per_second": 3.819,
107
+ "step": 700
108
+ },
109
+ {
110
+ "epoch": 0.58,
111
+ "learning_rate": 0.0002,
112
+ "loss": 2.4271,
113
+ "step": 800
114
+ },
115
+ {
116
+ "epoch": 0.58,
117
+ "eval_loss": 2.4231719970703125,
118
+ "eval_runtime": 204.3763,
119
+ "eval_samples_per_second": 30.698,
120
+ "eval_steps_per_second": 3.841,
121
+ "step": 800
122
  }
123
  ],
124
  "logging_steps": 100,
125
  "max_steps": 4116,
126
  "num_train_epochs": 3,
127
  "save_steps": 100,
128
+ "total_flos": 2.333865205776384e+16,
129
  "trial_name": null,
130
  "trial_params": null
131
  }
last-checkpoint/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f6bbb4e93666410d7a938dd3851f7194e97be075889249a5ac42f4cb3cfacdd3
3
  size 4219
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:faed5845c6fa602a9a75d6d7a3c4d37017580998e72f3835a07c1c95f579635b
3
  size 4219