clinno commited on
Commit
815fde3
·
verified ·
1 Parent(s): 51d7c14

Delete checkpoint-3300

Browse files
checkpoint-3300/config.json DELETED
@@ -1,29 +0,0 @@
1
- {
2
- "_name_or_path": "NousResearch/Meta-Llama-3-8B-Instruct",
3
- "architectures": [
4
- "LlamaForCausalLM"
5
- ],
6
- "attention_bias": false,
7
- "attention_dropout": 0.0,
8
- "bos_token_id": 128000,
9
- "eos_token_id": 128009,
10
- "hidden_act": "silu",
11
- "hidden_size": 4096,
12
- "initializer_range": 0.02,
13
- "intermediate_size": 14336,
14
- "max_position_embeddings": 8192,
15
- "mlp_bias": false,
16
- "model_type": "llama",
17
- "num_attention_heads": 32,
18
- "num_hidden_layers": 32,
19
- "num_key_value_heads": 8,
20
- "pretraining_tp": 1,
21
- "rms_norm_eps": 1e-05,
22
- "rope_scaling": null,
23
- "rope_theta": 500000.0,
24
- "tie_word_embeddings": false,
25
- "torch_dtype": "bfloat16",
26
- "transformers_version": "4.44.2",
27
- "use_cache": false,
28
- "vocab_size": 128256
29
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
checkpoint-3300/generation_config.json DELETED
@@ -1,12 +0,0 @@
1
- {
2
- "bos_token_id": 128000,
3
- "do_sample": true,
4
- "eos_token_id": [
5
- 128001,
6
- 128009
7
- ],
8
- "max_length": 4096,
9
- "temperature": 0.6,
10
- "top_p": 0.9,
11
- "transformers_version": "4.44.2"
12
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
checkpoint-3300/model-00001-of-00004.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:6ce8310054018ee51b78299d3d660d133731acacca439d5c43ac36a22021a4a1
3
- size 4976698672
 
 
 
 
checkpoint-3300/model-00002-of-00004.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:34f6cbceea7eee5ecca69f4e97863818b1850d40e1c0b9d27f0a13af55ed0380
3
- size 4999802720
 
 
 
 
checkpoint-3300/model-00003-of-00004.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:a3154807ed0c4bf3fcc841323e69c082c60b8e1d12779c1e8d363dc8eb986519
3
- size 4915916176
 
 
 
 
checkpoint-3300/model-00004-of-00004.safetensors DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:fa64f16e7b8a8f83d11ef628f49a1d6986b67729c41b08bd03a8e5788a215fd6
3
- size 1168138808
 
 
 
 
checkpoint-3300/model.safetensors.index.json DELETED
@@ -1,298 +0,0 @@
1
- {
2
- "metadata": {
3
- "total_size": 16060522496
4
- },
5
- "weight_map": {
6
- "lm_head.weight": "model-00004-of-00004.safetensors",
7
- "model.embed_tokens.weight": "model-00001-of-00004.safetensors",
8
- "model.layers.0.input_layernorm.weight": "model-00001-of-00004.safetensors",
9
- "model.layers.0.mlp.down_proj.weight": "model-00001-of-00004.safetensors",
10
- "model.layers.0.mlp.gate_proj.weight": "model-00001-of-00004.safetensors",
11
- "model.layers.0.mlp.up_proj.weight": "model-00001-of-00004.safetensors",
12
- "model.layers.0.post_attention_layernorm.weight": "model-00001-of-00004.safetensors",
13
- "model.layers.0.self_attn.k_proj.weight": "model-00001-of-00004.safetensors",
14
- "model.layers.0.self_attn.o_proj.weight": "model-00001-of-00004.safetensors",
15
- "model.layers.0.self_attn.q_proj.weight": "model-00001-of-00004.safetensors",
16
- "model.layers.0.self_attn.v_proj.weight": "model-00001-of-00004.safetensors",
17
- "model.layers.1.input_layernorm.weight": "model-00001-of-00004.safetensors",
18
- "model.layers.1.mlp.down_proj.weight": "model-00001-of-00004.safetensors",
19
- "model.layers.1.mlp.gate_proj.weight": "model-00001-of-00004.safetensors",
20
- "model.layers.1.mlp.up_proj.weight": "model-00001-of-00004.safetensors",
21
- "model.layers.1.post_attention_layernorm.weight": "model-00001-of-00004.safetensors",
22
- "model.layers.1.self_attn.k_proj.weight": "model-00001-of-00004.safetensors",
23
- "model.layers.1.self_attn.o_proj.weight": "model-00001-of-00004.safetensors",
24
- "model.layers.1.self_attn.q_proj.weight": "model-00001-of-00004.safetensors",
25
- "model.layers.1.self_attn.v_proj.weight": "model-00001-of-00004.safetensors",
26
- "model.layers.10.input_layernorm.weight": "model-00002-of-00004.safetensors",
27
- "model.layers.10.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
28
- "model.layers.10.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
29
- "model.layers.10.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
30
- "model.layers.10.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
31
- "model.layers.10.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
32
- "model.layers.10.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
33
- "model.layers.10.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
34
- "model.layers.10.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
35
- "model.layers.11.input_layernorm.weight": "model-00002-of-00004.safetensors",
36
- "model.layers.11.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
37
- "model.layers.11.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
38
- "model.layers.11.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
39
- "model.layers.11.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
40
- "model.layers.11.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
41
- "model.layers.11.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
42
- "model.layers.11.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
43
- "model.layers.11.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
44
- "model.layers.12.input_layernorm.weight": "model-00002-of-00004.safetensors",
45
- "model.layers.12.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
46
- "model.layers.12.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
47
- "model.layers.12.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
48
- "model.layers.12.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
49
- "model.layers.12.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
50
- "model.layers.12.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
51
- "model.layers.12.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
52
- "model.layers.12.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
53
- "model.layers.13.input_layernorm.weight": "model-00002-of-00004.safetensors",
54
- "model.layers.13.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
55
- "model.layers.13.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
56
- "model.layers.13.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
57
- "model.layers.13.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
58
- "model.layers.13.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
59
- "model.layers.13.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
60
- "model.layers.13.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
61
- "model.layers.13.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
62
- "model.layers.14.input_layernorm.weight": "model-00002-of-00004.safetensors",
63
- "model.layers.14.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
64
- "model.layers.14.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
65
- "model.layers.14.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
66
- "model.layers.14.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
67
- "model.layers.14.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
68
- "model.layers.14.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
69
- "model.layers.14.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
70
- "model.layers.14.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
71
- "model.layers.15.input_layernorm.weight": "model-00002-of-00004.safetensors",
72
- "model.layers.15.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
73
- "model.layers.15.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
74
- "model.layers.15.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
75
- "model.layers.15.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
76
- "model.layers.15.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
77
- "model.layers.15.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
78
- "model.layers.15.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
79
- "model.layers.15.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
80
- "model.layers.16.input_layernorm.weight": "model-00002-of-00004.safetensors",
81
- "model.layers.16.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
82
- "model.layers.16.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
83
- "model.layers.16.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
84
- "model.layers.16.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
85
- "model.layers.16.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
86
- "model.layers.16.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
87
- "model.layers.16.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
88
- "model.layers.16.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
89
- "model.layers.17.input_layernorm.weight": "model-00002-of-00004.safetensors",
90
- "model.layers.17.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
91
- "model.layers.17.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
92
- "model.layers.17.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
93
- "model.layers.17.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
94
- "model.layers.17.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
95
- "model.layers.17.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
96
- "model.layers.17.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
97
- "model.layers.17.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
98
- "model.layers.18.input_layernorm.weight": "model-00002-of-00004.safetensors",
99
- "model.layers.18.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
100
- "model.layers.18.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
101
- "model.layers.18.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
102
- "model.layers.18.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
103
- "model.layers.18.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
104
- "model.layers.18.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
105
- "model.layers.18.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
106
- "model.layers.18.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
107
- "model.layers.19.input_layernorm.weight": "model-00002-of-00004.safetensors",
108
- "model.layers.19.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
109
- "model.layers.19.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
110
- "model.layers.19.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
111
- "model.layers.19.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
112
- "model.layers.19.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
113
- "model.layers.19.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
114
- "model.layers.19.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
115
- "model.layers.19.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
116
- "model.layers.2.input_layernorm.weight": "model-00001-of-00004.safetensors",
117
- "model.layers.2.mlp.down_proj.weight": "model-00001-of-00004.safetensors",
118
- "model.layers.2.mlp.gate_proj.weight": "model-00001-of-00004.safetensors",
119
- "model.layers.2.mlp.up_proj.weight": "model-00001-of-00004.safetensors",
120
- "model.layers.2.post_attention_layernorm.weight": "model-00001-of-00004.safetensors",
121
- "model.layers.2.self_attn.k_proj.weight": "model-00001-of-00004.safetensors",
122
- "model.layers.2.self_attn.o_proj.weight": "model-00001-of-00004.safetensors",
123
- "model.layers.2.self_attn.q_proj.weight": "model-00001-of-00004.safetensors",
124
- "model.layers.2.self_attn.v_proj.weight": "model-00001-of-00004.safetensors",
125
- "model.layers.20.input_layernorm.weight": "model-00003-of-00004.safetensors",
126
- "model.layers.20.mlp.down_proj.weight": "model-00003-of-00004.safetensors",
127
- "model.layers.20.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
128
- "model.layers.20.mlp.up_proj.weight": "model-00003-of-00004.safetensors",
129
- "model.layers.20.post_attention_layernorm.weight": "model-00003-of-00004.safetensors",
130
- "model.layers.20.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
131
- "model.layers.20.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
132
- "model.layers.20.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
133
- "model.layers.20.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
134
- "model.layers.21.input_layernorm.weight": "model-00003-of-00004.safetensors",
135
- "model.layers.21.mlp.down_proj.weight": "model-00003-of-00004.safetensors",
136
- "model.layers.21.mlp.gate_proj.weight": "model-00003-of-00004.safetensors",
137
- "model.layers.21.mlp.up_proj.weight": "model-00003-of-00004.safetensors",
138
- "model.layers.21.post_attention_layernorm.weight": "model-00003-of-00004.safetensors",
139
- "model.layers.21.self_attn.k_proj.weight": "model-00003-of-00004.safetensors",
140
- "model.layers.21.self_attn.o_proj.weight": "model-00003-of-00004.safetensors",
141
- "model.layers.21.self_attn.q_proj.weight": "model-00003-of-00004.safetensors",
142
- "model.layers.21.self_attn.v_proj.weight": "model-00003-of-00004.safetensors",
143
- "model.layers.22.input_layernorm.weight": "model-00003-of-00004.safetensors",
144
- "model.layers.22.mlp.down_proj.weight": "model-00003-of-00004.safetensors",
145
- "model.layers.22.mlp.gate_proj.weight": "model-00003-of-00004.safetensors",
146
- "model.layers.22.mlp.up_proj.weight": "model-00003-of-00004.safetensors",
147
- "model.layers.22.post_attention_layernorm.weight": "model-00003-of-00004.safetensors",
148
- "model.layers.22.self_attn.k_proj.weight": "model-00003-of-00004.safetensors",
149
- "model.layers.22.self_attn.o_proj.weight": "model-00003-of-00004.safetensors",
150
- "model.layers.22.self_attn.q_proj.weight": "model-00003-of-00004.safetensors",
151
- "model.layers.22.self_attn.v_proj.weight": "model-00003-of-00004.safetensors",
152
- "model.layers.23.input_layernorm.weight": "model-00003-of-00004.safetensors",
153
- "model.layers.23.mlp.down_proj.weight": "model-00003-of-00004.safetensors",
154
- "model.layers.23.mlp.gate_proj.weight": "model-00003-of-00004.safetensors",
155
- "model.layers.23.mlp.up_proj.weight": "model-00003-of-00004.safetensors",
156
- "model.layers.23.post_attention_layernorm.weight": "model-00003-of-00004.safetensors",
157
- "model.layers.23.self_attn.k_proj.weight": "model-00003-of-00004.safetensors",
158
- "model.layers.23.self_attn.o_proj.weight": "model-00003-of-00004.safetensors",
159
- "model.layers.23.self_attn.q_proj.weight": "model-00003-of-00004.safetensors",
160
- "model.layers.23.self_attn.v_proj.weight": "model-00003-of-00004.safetensors",
161
- "model.layers.24.input_layernorm.weight": "model-00003-of-00004.safetensors",
162
- "model.layers.24.mlp.down_proj.weight": "model-00003-of-00004.safetensors",
163
- "model.layers.24.mlp.gate_proj.weight": "model-00003-of-00004.safetensors",
164
- "model.layers.24.mlp.up_proj.weight": "model-00003-of-00004.safetensors",
165
- "model.layers.24.post_attention_layernorm.weight": "model-00003-of-00004.safetensors",
166
- "model.layers.24.self_attn.k_proj.weight": "model-00003-of-00004.safetensors",
167
- "model.layers.24.self_attn.o_proj.weight": "model-00003-of-00004.safetensors",
168
- "model.layers.24.self_attn.q_proj.weight": "model-00003-of-00004.safetensors",
169
- "model.layers.24.self_attn.v_proj.weight": "model-00003-of-00004.safetensors",
170
- "model.layers.25.input_layernorm.weight": "model-00003-of-00004.safetensors",
171
- "model.layers.25.mlp.down_proj.weight": "model-00003-of-00004.safetensors",
172
- "model.layers.25.mlp.gate_proj.weight": "model-00003-of-00004.safetensors",
173
- "model.layers.25.mlp.up_proj.weight": "model-00003-of-00004.safetensors",
174
- "model.layers.25.post_attention_layernorm.weight": "model-00003-of-00004.safetensors",
175
- "model.layers.25.self_attn.k_proj.weight": "model-00003-of-00004.safetensors",
176
- "model.layers.25.self_attn.o_proj.weight": "model-00003-of-00004.safetensors",
177
- "model.layers.25.self_attn.q_proj.weight": "model-00003-of-00004.safetensors",
178
- "model.layers.25.self_attn.v_proj.weight": "model-00003-of-00004.safetensors",
179
- "model.layers.26.input_layernorm.weight": "model-00003-of-00004.safetensors",
180
- "model.layers.26.mlp.down_proj.weight": "model-00003-of-00004.safetensors",
181
- "model.layers.26.mlp.gate_proj.weight": "model-00003-of-00004.safetensors",
182
- "model.layers.26.mlp.up_proj.weight": "model-00003-of-00004.safetensors",
183
- "model.layers.26.post_attention_layernorm.weight": "model-00003-of-00004.safetensors",
184
- "model.layers.26.self_attn.k_proj.weight": "model-00003-of-00004.safetensors",
185
- "model.layers.26.self_attn.o_proj.weight": "model-00003-of-00004.safetensors",
186
- "model.layers.26.self_attn.q_proj.weight": "model-00003-of-00004.safetensors",
187
- "model.layers.26.self_attn.v_proj.weight": "model-00003-of-00004.safetensors",
188
- "model.layers.27.input_layernorm.weight": "model-00003-of-00004.safetensors",
189
- "model.layers.27.mlp.down_proj.weight": "model-00003-of-00004.safetensors",
190
- "model.layers.27.mlp.gate_proj.weight": "model-00003-of-00004.safetensors",
191
- "model.layers.27.mlp.up_proj.weight": "model-00003-of-00004.safetensors",
192
- "model.layers.27.post_attention_layernorm.weight": "model-00003-of-00004.safetensors",
193
- "model.layers.27.self_attn.k_proj.weight": "model-00003-of-00004.safetensors",
194
- "model.layers.27.self_attn.o_proj.weight": "model-00003-of-00004.safetensors",
195
- "model.layers.27.self_attn.q_proj.weight": "model-00003-of-00004.safetensors",
196
- "model.layers.27.self_attn.v_proj.weight": "model-00003-of-00004.safetensors",
197
- "model.layers.28.input_layernorm.weight": "model-00003-of-00004.safetensors",
198
- "model.layers.28.mlp.down_proj.weight": "model-00003-of-00004.safetensors",
199
- "model.layers.28.mlp.gate_proj.weight": "model-00003-of-00004.safetensors",
200
- "model.layers.28.mlp.up_proj.weight": "model-00003-of-00004.safetensors",
201
- "model.layers.28.post_attention_layernorm.weight": "model-00003-of-00004.safetensors",
202
- "model.layers.28.self_attn.k_proj.weight": "model-00003-of-00004.safetensors",
203
- "model.layers.28.self_attn.o_proj.weight": "model-00003-of-00004.safetensors",
204
- "model.layers.28.self_attn.q_proj.weight": "model-00003-of-00004.safetensors",
205
- "model.layers.28.self_attn.v_proj.weight": "model-00003-of-00004.safetensors",
206
- "model.layers.29.input_layernorm.weight": "model-00003-of-00004.safetensors",
207
- "model.layers.29.mlp.down_proj.weight": "model-00003-of-00004.safetensors",
208
- "model.layers.29.mlp.gate_proj.weight": "model-00003-of-00004.safetensors",
209
- "model.layers.29.mlp.up_proj.weight": "model-00003-of-00004.safetensors",
210
- "model.layers.29.post_attention_layernorm.weight": "model-00003-of-00004.safetensors",
211
- "model.layers.29.self_attn.k_proj.weight": "model-00003-of-00004.safetensors",
212
- "model.layers.29.self_attn.o_proj.weight": "model-00003-of-00004.safetensors",
213
- "model.layers.29.self_attn.q_proj.weight": "model-00003-of-00004.safetensors",
214
- "model.layers.29.self_attn.v_proj.weight": "model-00003-of-00004.safetensors",
215
- "model.layers.3.input_layernorm.weight": "model-00001-of-00004.safetensors",
216
- "model.layers.3.mlp.down_proj.weight": "model-00001-of-00004.safetensors",
217
- "model.layers.3.mlp.gate_proj.weight": "model-00001-of-00004.safetensors",
218
- "model.layers.3.mlp.up_proj.weight": "model-00001-of-00004.safetensors",
219
- "model.layers.3.post_attention_layernorm.weight": "model-00001-of-00004.safetensors",
220
- "model.layers.3.self_attn.k_proj.weight": "model-00001-of-00004.safetensors",
221
- "model.layers.3.self_attn.o_proj.weight": "model-00001-of-00004.safetensors",
222
- "model.layers.3.self_attn.q_proj.weight": "model-00001-of-00004.safetensors",
223
- "model.layers.3.self_attn.v_proj.weight": "model-00001-of-00004.safetensors",
224
- "model.layers.30.input_layernorm.weight": "model-00003-of-00004.safetensors",
225
- "model.layers.30.mlp.down_proj.weight": "model-00003-of-00004.safetensors",
226
- "model.layers.30.mlp.gate_proj.weight": "model-00003-of-00004.safetensors",
227
- "model.layers.30.mlp.up_proj.weight": "model-00003-of-00004.safetensors",
228
- "model.layers.30.post_attention_layernorm.weight": "model-00003-of-00004.safetensors",
229
- "model.layers.30.self_attn.k_proj.weight": "model-00003-of-00004.safetensors",
230
- "model.layers.30.self_attn.o_proj.weight": "model-00003-of-00004.safetensors",
231
- "model.layers.30.self_attn.q_proj.weight": "model-00003-of-00004.safetensors",
232
- "model.layers.30.self_attn.v_proj.weight": "model-00003-of-00004.safetensors",
233
- "model.layers.31.input_layernorm.weight": "model-00004-of-00004.safetensors",
234
- "model.layers.31.mlp.down_proj.weight": "model-00004-of-00004.safetensors",
235
- "model.layers.31.mlp.gate_proj.weight": "model-00003-of-00004.safetensors",
236
- "model.layers.31.mlp.up_proj.weight": "model-00003-of-00004.safetensors",
237
- "model.layers.31.post_attention_layernorm.weight": "model-00004-of-00004.safetensors",
238
- "model.layers.31.self_attn.k_proj.weight": "model-00003-of-00004.safetensors",
239
- "model.layers.31.self_attn.o_proj.weight": "model-00003-of-00004.safetensors",
240
- "model.layers.31.self_attn.q_proj.weight": "model-00003-of-00004.safetensors",
241
- "model.layers.31.self_attn.v_proj.weight": "model-00003-of-00004.safetensors",
242
- "model.layers.4.input_layernorm.weight": "model-00001-of-00004.safetensors",
243
- "model.layers.4.mlp.down_proj.weight": "model-00001-of-00004.safetensors",
244
- "model.layers.4.mlp.gate_proj.weight": "model-00001-of-00004.safetensors",
245
- "model.layers.4.mlp.up_proj.weight": "model-00001-of-00004.safetensors",
246
- "model.layers.4.post_attention_layernorm.weight": "model-00001-of-00004.safetensors",
247
- "model.layers.4.self_attn.k_proj.weight": "model-00001-of-00004.safetensors",
248
- "model.layers.4.self_attn.o_proj.weight": "model-00001-of-00004.safetensors",
249
- "model.layers.4.self_attn.q_proj.weight": "model-00001-of-00004.safetensors",
250
- "model.layers.4.self_attn.v_proj.weight": "model-00001-of-00004.safetensors",
251
- "model.layers.5.input_layernorm.weight": "model-00001-of-00004.safetensors",
252
- "model.layers.5.mlp.down_proj.weight": "model-00001-of-00004.safetensors",
253
- "model.layers.5.mlp.gate_proj.weight": "model-00001-of-00004.safetensors",
254
- "model.layers.5.mlp.up_proj.weight": "model-00001-of-00004.safetensors",
255
- "model.layers.5.post_attention_layernorm.weight": "model-00001-of-00004.safetensors",
256
- "model.layers.5.self_attn.k_proj.weight": "model-00001-of-00004.safetensors",
257
- "model.layers.5.self_attn.o_proj.weight": "model-00001-of-00004.safetensors",
258
- "model.layers.5.self_attn.q_proj.weight": "model-00001-of-00004.safetensors",
259
- "model.layers.5.self_attn.v_proj.weight": "model-00001-of-00004.safetensors",
260
- "model.layers.6.input_layernorm.weight": "model-00001-of-00004.safetensors",
261
- "model.layers.6.mlp.down_proj.weight": "model-00001-of-00004.safetensors",
262
- "model.layers.6.mlp.gate_proj.weight": "model-00001-of-00004.safetensors",
263
- "model.layers.6.mlp.up_proj.weight": "model-00001-of-00004.safetensors",
264
- "model.layers.6.post_attention_layernorm.weight": "model-00001-of-00004.safetensors",
265
- "model.layers.6.self_attn.k_proj.weight": "model-00001-of-00004.safetensors",
266
- "model.layers.6.self_attn.o_proj.weight": "model-00001-of-00004.safetensors",
267
- "model.layers.6.self_attn.q_proj.weight": "model-00001-of-00004.safetensors",
268
- "model.layers.6.self_attn.v_proj.weight": "model-00001-of-00004.safetensors",
269
- "model.layers.7.input_layernorm.weight": "model-00001-of-00004.safetensors",
270
- "model.layers.7.mlp.down_proj.weight": "model-00001-of-00004.safetensors",
271
- "model.layers.7.mlp.gate_proj.weight": "model-00001-of-00004.safetensors",
272
- "model.layers.7.mlp.up_proj.weight": "model-00001-of-00004.safetensors",
273
- "model.layers.7.post_attention_layernorm.weight": "model-00001-of-00004.safetensors",
274
- "model.layers.7.self_attn.k_proj.weight": "model-00001-of-00004.safetensors",
275
- "model.layers.7.self_attn.o_proj.weight": "model-00001-of-00004.safetensors",
276
- "model.layers.7.self_attn.q_proj.weight": "model-00001-of-00004.safetensors",
277
- "model.layers.7.self_attn.v_proj.weight": "model-00001-of-00004.safetensors",
278
- "model.layers.8.input_layernorm.weight": "model-00001-of-00004.safetensors",
279
- "model.layers.8.mlp.down_proj.weight": "model-00001-of-00004.safetensors",
280
- "model.layers.8.mlp.gate_proj.weight": "model-00001-of-00004.safetensors",
281
- "model.layers.8.mlp.up_proj.weight": "model-00001-of-00004.safetensors",
282
- "model.layers.8.post_attention_layernorm.weight": "model-00001-of-00004.safetensors",
283
- "model.layers.8.self_attn.k_proj.weight": "model-00001-of-00004.safetensors",
284
- "model.layers.8.self_attn.o_proj.weight": "model-00001-of-00004.safetensors",
285
- "model.layers.8.self_attn.q_proj.weight": "model-00001-of-00004.safetensors",
286
- "model.layers.8.self_attn.v_proj.weight": "model-00001-of-00004.safetensors",
287
- "model.layers.9.input_layernorm.weight": "model-00002-of-00004.safetensors",
288
- "model.layers.9.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
289
- "model.layers.9.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
290
- "model.layers.9.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
291
- "model.layers.9.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
292
- "model.layers.9.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
293
- "model.layers.9.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
294
- "model.layers.9.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
295
- "model.layers.9.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
296
- "model.norm.weight": "model-00004-of-00004.safetensors"
297
- }
298
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
checkpoint-3300/optimizer.pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:043bb3427e47a8e55c513c150c7b0ff4c9a1f5266a54719c93ba405940876832
3
- size 1744904195
 
 
 
 
checkpoint-3300/rng_state.pth DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:5328f04f222a66b45931d6bc246721e0747decf9d78d167903d0547a248f78f0
3
- size 14244
 
 
 
 
checkpoint-3300/scheduler.pt DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:b5ce95da8b1f2b5b24d133763d92b0b36b11c87b0d99042ec4fafad2db811c4c
3
- size 1064
 
 
 
 
checkpoint-3300/special_tokens_map.json DELETED
@@ -1,17 +0,0 @@
1
- {
2
- "bos_token": {
3
- "content": "<|begin_of_text|>",
4
- "lstrip": false,
5
- "normalized": false,
6
- "rstrip": false,
7
- "single_word": false
8
- },
9
- "eos_token": {
10
- "content": "<|eot_id|>",
11
- "lstrip": false,
12
- "normalized": false,
13
- "rstrip": false,
14
- "single_word": false
15
- },
16
- "pad_token": "<|eot_id|>"
17
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
checkpoint-3300/tokenizer.json DELETED
The diff for this file is too large to render. See raw diff
 
checkpoint-3300/tokenizer_config.json DELETED
@@ -1,2065 +0,0 @@
1
- {
2
- "added_tokens_decoder": {
3
- "128000": {
4
- "content": "<|begin_of_text|>",
5
- "lstrip": false,
6
- "normalized": false,
7
- "rstrip": false,
8
- "single_word": false,
9
- "special": true
10
- },
11
- "128001": {
12
- "content": "<|end_of_text|>",
13
- "lstrip": false,
14
- "normalized": false,
15
- "rstrip": false,
16
- "single_word": false,
17
- "special": true
18
- },
19
- "128002": {
20
- "content": "<|reserved_special_token_0|>",
21
- "lstrip": false,
22
- "normalized": false,
23
- "rstrip": false,
24
- "single_word": false,
25
- "special": true
26
- },
27
- "128003": {
28
- "content": "<|reserved_special_token_1|>",
29
- "lstrip": false,
30
- "normalized": false,
31
- "rstrip": false,
32
- "single_word": false,
33
- "special": true
34
- },
35
- "128004": {
36
- "content": "<|reserved_special_token_2|>",
37
- "lstrip": false,
38
- "normalized": false,
39
- "rstrip": false,
40
- "single_word": false,
41
- "special": true
42
- },
43
- "128005": {
44
- "content": "<|reserved_special_token_3|>",
45
- "lstrip": false,
46
- "normalized": false,
47
- "rstrip": false,
48
- "single_word": false,
49
- "special": true
50
- },
51
- "128006": {
52
- "content": "<|start_header_id|>",
53
- "lstrip": false,
54
- "normalized": false,
55
- "rstrip": false,
56
- "single_word": false,
57
- "special": true
58
- },
59
- "128007": {
60
- "content": "<|end_header_id|>",
61
- "lstrip": false,
62
- "normalized": false,
63
- "rstrip": false,
64
- "single_word": false,
65
- "special": true
66
- },
67
- "128008": {
68
- "content": "<|reserved_special_token_4|>",
69
- "lstrip": false,
70
- "normalized": false,
71
- "rstrip": false,
72
- "single_word": false,
73
- "special": true
74
- },
75
- "128009": {
76
- "content": "<|eot_id|>",
77
- "lstrip": false,
78
- "normalized": false,
79
- "rstrip": false,
80
- "single_word": false,
81
- "special": true
82
- },
83
- "128010": {
84
- "content": "<|reserved_special_token_5|>",
85
- "lstrip": false,
86
- "normalized": false,
87
- "rstrip": false,
88
- "single_word": false,
89
- "special": true
90
- },
91
- "128011": {
92
- "content": "<|reserved_special_token_6|>",
93
- "lstrip": false,
94
- "normalized": false,
95
- "rstrip": false,
96
- "single_word": false,
97
- "special": true
98
- },
99
- "128012": {
100
- "content": "<|reserved_special_token_7|>",
101
- "lstrip": false,
102
- "normalized": false,
103
- "rstrip": false,
104
- "single_word": false,
105
- "special": true
106
- },
107
- "128013": {
108
- "content": "<|reserved_special_token_8|>",
109
- "lstrip": false,
110
- "normalized": false,
111
- "rstrip": false,
112
- "single_word": false,
113
- "special": true
114
- },
115
- "128014": {
116
- "content": "<|reserved_special_token_9|>",
117
- "lstrip": false,
118
- "normalized": false,
119
- "rstrip": false,
120
- "single_word": false,
121
- "special": true
122
- },
123
- "128015": {
124
- "content": "<|reserved_special_token_10|>",
125
- "lstrip": false,
126
- "normalized": false,
127
- "rstrip": false,
128
- "single_word": false,
129
- "special": true
130
- },
131
- "128016": {
132
- "content": "<|reserved_special_token_11|>",
133
- "lstrip": false,
134
- "normalized": false,
135
- "rstrip": false,
136
- "single_word": false,
137
- "special": true
138
- },
139
- "128017": {
140
- "content": "<|reserved_special_token_12|>",
141
- "lstrip": false,
142
- "normalized": false,
143
- "rstrip": false,
144
- "single_word": false,
145
- "special": true
146
- },
147
- "128018": {
148
- "content": "<|reserved_special_token_13|>",
149
- "lstrip": false,
150
- "normalized": false,
151
- "rstrip": false,
152
- "single_word": false,
153
- "special": true
154
- },
155
- "128019": {
156
- "content": "<|reserved_special_token_14|>",
157
- "lstrip": false,
158
- "normalized": false,
159
- "rstrip": false,
160
- "single_word": false,
161
- "special": true
162
- },
163
- "128020": {
164
- "content": "<|reserved_special_token_15|>",
165
- "lstrip": false,
166
- "normalized": false,
167
- "rstrip": false,
168
- "single_word": false,
169
- "special": true
170
- },
171
- "128021": {
172
- "content": "<|reserved_special_token_16|>",
173
- "lstrip": false,
174
- "normalized": false,
175
- "rstrip": false,
176
- "single_word": false,
177
- "special": true
178
- },
179
- "128022": {
180
- "content": "<|reserved_special_token_17|>",
181
- "lstrip": false,
182
- "normalized": false,
183
- "rstrip": false,
184
- "single_word": false,
185
- "special": true
186
- },
187
- "128023": {
188
- "content": "<|reserved_special_token_18|>",
189
- "lstrip": false,
190
- "normalized": false,
191
- "rstrip": false,
192
- "single_word": false,
193
- "special": true
194
- },
195
- "128024": {
196
- "content": "<|reserved_special_token_19|>",
197
- "lstrip": false,
198
- "normalized": false,
199
- "rstrip": false,
200
- "single_word": false,
201
- "special": true
202
- },
203
- "128025": {
204
- "content": "<|reserved_special_token_20|>",
205
- "lstrip": false,
206
- "normalized": false,
207
- "rstrip": false,
208
- "single_word": false,
209
- "special": true
210
- },
211
- "128026": {
212
- "content": "<|reserved_special_token_21|>",
213
- "lstrip": false,
214
- "normalized": false,
215
- "rstrip": false,
216
- "single_word": false,
217
- "special": true
218
- },
219
- "128027": {
220
- "content": "<|reserved_special_token_22|>",
221
- "lstrip": false,
222
- "normalized": false,
223
- "rstrip": false,
224
- "single_word": false,
225
- "special": true
226
- },
227
- "128028": {
228
- "content": "<|reserved_special_token_23|>",
229
- "lstrip": false,
230
- "normalized": false,
231
- "rstrip": false,
232
- "single_word": false,
233
- "special": true
234
- },
235
- "128029": {
236
- "content": "<|reserved_special_token_24|>",
237
- "lstrip": false,
238
- "normalized": false,
239
- "rstrip": false,
240
- "single_word": false,
241
- "special": true
242
- },
243
- "128030": {
244
- "content": "<|reserved_special_token_25|>",
245
- "lstrip": false,
246
- "normalized": false,
247
- "rstrip": false,
248
- "single_word": false,
249
- "special": true
250
- },
251
- "128031": {
252
- "content": "<|reserved_special_token_26|>",
253
- "lstrip": false,
254
- "normalized": false,
255
- "rstrip": false,
256
- "single_word": false,
257
- "special": true
258
- },
259
- "128032": {
260
- "content": "<|reserved_special_token_27|>",
261
- "lstrip": false,
262
- "normalized": false,
263
- "rstrip": false,
264
- "single_word": false,
265
- "special": true
266
- },
267
- "128033": {
268
- "content": "<|reserved_special_token_28|>",
269
- "lstrip": false,
270
- "normalized": false,
271
- "rstrip": false,
272
- "single_word": false,
273
- "special": true
274
- },
275
- "128034": {
276
- "content": "<|reserved_special_token_29|>",
277
- "lstrip": false,
278
- "normalized": false,
279
- "rstrip": false,
280
- "single_word": false,
281
- "special": true
282
- },
283
- "128035": {
284
- "content": "<|reserved_special_token_30|>",
285
- "lstrip": false,
286
- "normalized": false,
287
- "rstrip": false,
288
- "single_word": false,
289
- "special": true
290
- },
291
- "128036": {
292
- "content": "<|reserved_special_token_31|>",
293
- "lstrip": false,
294
- "normalized": false,
295
- "rstrip": false,
296
- "single_word": false,
297
- "special": true
298
- },
299
- "128037": {
300
- "content": "<|reserved_special_token_32|>",
301
- "lstrip": false,
302
- "normalized": false,
303
- "rstrip": false,
304
- "single_word": false,
305
- "special": true
306
- },
307
- "128038": {
308
- "content": "<|reserved_special_token_33|>",
309
- "lstrip": false,
310
- "normalized": false,
311
- "rstrip": false,
312
- "single_word": false,
313
- "special": true
314
- },
315
- "128039": {
316
- "content": "<|reserved_special_token_34|>",
317
- "lstrip": false,
318
- "normalized": false,
319
- "rstrip": false,
320
- "single_word": false,
321
- "special": true
322
- },
323
- "128040": {
324
- "content": "<|reserved_special_token_35|>",
325
- "lstrip": false,
326
- "normalized": false,
327
- "rstrip": false,
328
- "single_word": false,
329
- "special": true
330
- },
331
- "128041": {
332
- "content": "<|reserved_special_token_36|>",
333
- "lstrip": false,
334
- "normalized": false,
335
- "rstrip": false,
336
- "single_word": false,
337
- "special": true
338
- },
339
- "128042": {
340
- "content": "<|reserved_special_token_37|>",
341
- "lstrip": false,
342
- "normalized": false,
343
- "rstrip": false,
344
- "single_word": false,
345
- "special": true
346
- },
347
- "128043": {
348
- "content": "<|reserved_special_token_38|>",
349
- "lstrip": false,
350
- "normalized": false,
351
- "rstrip": false,
352
- "single_word": false,
353
- "special": true
354
- },
355
- "128044": {
356
- "content": "<|reserved_special_token_39|>",
357
- "lstrip": false,
358
- "normalized": false,
359
- "rstrip": false,
360
- "single_word": false,
361
- "special": true
362
- },
363
- "128045": {
364
- "content": "<|reserved_special_token_40|>",
365
- "lstrip": false,
366
- "normalized": false,
367
- "rstrip": false,
368
- "single_word": false,
369
- "special": true
370
- },
371
- "128046": {
372
- "content": "<|reserved_special_token_41|>",
373
- "lstrip": false,
374
- "normalized": false,
375
- "rstrip": false,
376
- "single_word": false,
377
- "special": true
378
- },
379
- "128047": {
380
- "content": "<|reserved_special_token_42|>",
381
- "lstrip": false,
382
- "normalized": false,
383
- "rstrip": false,
384
- "single_word": false,
385
- "special": true
386
- },
387
- "128048": {
388
- "content": "<|reserved_special_token_43|>",
389
- "lstrip": false,
390
- "normalized": false,
391
- "rstrip": false,
392
- "single_word": false,
393
- "special": true
394
- },
395
- "128049": {
396
- "content": "<|reserved_special_token_44|>",
397
- "lstrip": false,
398
- "normalized": false,
399
- "rstrip": false,
400
- "single_word": false,
401
- "special": true
402
- },
403
- "128050": {
404
- "content": "<|reserved_special_token_45|>",
405
- "lstrip": false,
406
- "normalized": false,
407
- "rstrip": false,
408
- "single_word": false,
409
- "special": true
410
- },
411
- "128051": {
412
- "content": "<|reserved_special_token_46|>",
413
- "lstrip": false,
414
- "normalized": false,
415
- "rstrip": false,
416
- "single_word": false,
417
- "special": true
418
- },
419
- "128052": {
420
- "content": "<|reserved_special_token_47|>",
421
- "lstrip": false,
422
- "normalized": false,
423
- "rstrip": false,
424
- "single_word": false,
425
- "special": true
426
- },
427
- "128053": {
428
- "content": "<|reserved_special_token_48|>",
429
- "lstrip": false,
430
- "normalized": false,
431
- "rstrip": false,
432
- "single_word": false,
433
- "special": true
434
- },
435
- "128054": {
436
- "content": "<|reserved_special_token_49|>",
437
- "lstrip": false,
438
- "normalized": false,
439
- "rstrip": false,
440
- "single_word": false,
441
- "special": true
442
- },
443
- "128055": {
444
- "content": "<|reserved_special_token_50|>",
445
- "lstrip": false,
446
- "normalized": false,
447
- "rstrip": false,
448
- "single_word": false,
449
- "special": true
450
- },
451
- "128056": {
452
- "content": "<|reserved_special_token_51|>",
453
- "lstrip": false,
454
- "normalized": false,
455
- "rstrip": false,
456
- "single_word": false,
457
- "special": true
458
- },
459
- "128057": {
460
- "content": "<|reserved_special_token_52|>",
461
- "lstrip": false,
462
- "normalized": false,
463
- "rstrip": false,
464
- "single_word": false,
465
- "special": true
466
- },
467
- "128058": {
468
- "content": "<|reserved_special_token_53|>",
469
- "lstrip": false,
470
- "normalized": false,
471
- "rstrip": false,
472
- "single_word": false,
473
- "special": true
474
- },
475
- "128059": {
476
- "content": "<|reserved_special_token_54|>",
477
- "lstrip": false,
478
- "normalized": false,
479
- "rstrip": false,
480
- "single_word": false,
481
- "special": true
482
- },
483
- "128060": {
484
- "content": "<|reserved_special_token_55|>",
485
- "lstrip": false,
486
- "normalized": false,
487
- "rstrip": false,
488
- "single_word": false,
489
- "special": true
490
- },
491
- "128061": {
492
- "content": "<|reserved_special_token_56|>",
493
- "lstrip": false,
494
- "normalized": false,
495
- "rstrip": false,
496
- "single_word": false,
497
- "special": true
498
- },
499
- "128062": {
500
- "content": "<|reserved_special_token_57|>",
501
- "lstrip": false,
502
- "normalized": false,
503
- "rstrip": false,
504
- "single_word": false,
505
- "special": true
506
- },
507
- "128063": {
508
- "content": "<|reserved_special_token_58|>",
509
- "lstrip": false,
510
- "normalized": false,
511
- "rstrip": false,
512
- "single_word": false,
513
- "special": true
514
- },
515
- "128064": {
516
- "content": "<|reserved_special_token_59|>",
517
- "lstrip": false,
518
- "normalized": false,
519
- "rstrip": false,
520
- "single_word": false,
521
- "special": true
522
- },
523
- "128065": {
524
- "content": "<|reserved_special_token_60|>",
525
- "lstrip": false,
526
- "normalized": false,
527
- "rstrip": false,
528
- "single_word": false,
529
- "special": true
530
- },
531
- "128066": {
532
- "content": "<|reserved_special_token_61|>",
533
- "lstrip": false,
534
- "normalized": false,
535
- "rstrip": false,
536
- "single_word": false,
537
- "special": true
538
- },
539
- "128067": {
540
- "content": "<|reserved_special_token_62|>",
541
- "lstrip": false,
542
- "normalized": false,
543
- "rstrip": false,
544
- "single_word": false,
545
- "special": true
546
- },
547
- "128068": {
548
- "content": "<|reserved_special_token_63|>",
549
- "lstrip": false,
550
- "normalized": false,
551
- "rstrip": false,
552
- "single_word": false,
553
- "special": true
554
- },
555
- "128069": {
556
- "content": "<|reserved_special_token_64|>",
557
- "lstrip": false,
558
- "normalized": false,
559
- "rstrip": false,
560
- "single_word": false,
561
- "special": true
562
- },
563
- "128070": {
564
- "content": "<|reserved_special_token_65|>",
565
- "lstrip": false,
566
- "normalized": false,
567
- "rstrip": false,
568
- "single_word": false,
569
- "special": true
570
- },
571
- "128071": {
572
- "content": "<|reserved_special_token_66|>",
573
- "lstrip": false,
574
- "normalized": false,
575
- "rstrip": false,
576
- "single_word": false,
577
- "special": true
578
- },
579
- "128072": {
580
- "content": "<|reserved_special_token_67|>",
581
- "lstrip": false,
582
- "normalized": false,
583
- "rstrip": false,
584
- "single_word": false,
585
- "special": true
586
- },
587
- "128073": {
588
- "content": "<|reserved_special_token_68|>",
589
- "lstrip": false,
590
- "normalized": false,
591
- "rstrip": false,
592
- "single_word": false,
593
- "special": true
594
- },
595
- "128074": {
596
- "content": "<|reserved_special_token_69|>",
597
- "lstrip": false,
598
- "normalized": false,
599
- "rstrip": false,
600
- "single_word": false,
601
- "special": true
602
- },
603
- "128075": {
604
- "content": "<|reserved_special_token_70|>",
605
- "lstrip": false,
606
- "normalized": false,
607
- "rstrip": false,
608
- "single_word": false,
609
- "special": true
610
- },
611
- "128076": {
612
- "content": "<|reserved_special_token_71|>",
613
- "lstrip": false,
614
- "normalized": false,
615
- "rstrip": false,
616
- "single_word": false,
617
- "special": true
618
- },
619
- "128077": {
620
- "content": "<|reserved_special_token_72|>",
621
- "lstrip": false,
622
- "normalized": false,
623
- "rstrip": false,
624
- "single_word": false,
625
- "special": true
626
- },
627
- "128078": {
628
- "content": "<|reserved_special_token_73|>",
629
- "lstrip": false,
630
- "normalized": false,
631
- "rstrip": false,
632
- "single_word": false,
633
- "special": true
634
- },
635
- "128079": {
636
- "content": "<|reserved_special_token_74|>",
637
- "lstrip": false,
638
- "normalized": false,
639
- "rstrip": false,
640
- "single_word": false,
641
- "special": true
642
- },
643
- "128080": {
644
- "content": "<|reserved_special_token_75|>",
645
- "lstrip": false,
646
- "normalized": false,
647
- "rstrip": false,
648
- "single_word": false,
649
- "special": true
650
- },
651
- "128081": {
652
- "content": "<|reserved_special_token_76|>",
653
- "lstrip": false,
654
- "normalized": false,
655
- "rstrip": false,
656
- "single_word": false,
657
- "special": true
658
- },
659
- "128082": {
660
- "content": "<|reserved_special_token_77|>",
661
- "lstrip": false,
662
- "normalized": false,
663
- "rstrip": false,
664
- "single_word": false,
665
- "special": true
666
- },
667
- "128083": {
668
- "content": "<|reserved_special_token_78|>",
669
- "lstrip": false,
670
- "normalized": false,
671
- "rstrip": false,
672
- "single_word": false,
673
- "special": true
674
- },
675
- "128084": {
676
- "content": "<|reserved_special_token_79|>",
677
- "lstrip": false,
678
- "normalized": false,
679
- "rstrip": false,
680
- "single_word": false,
681
- "special": true
682
- },
683
- "128085": {
684
- "content": "<|reserved_special_token_80|>",
685
- "lstrip": false,
686
- "normalized": false,
687
- "rstrip": false,
688
- "single_word": false,
689
- "special": true
690
- },
691
- "128086": {
692
- "content": "<|reserved_special_token_81|>",
693
- "lstrip": false,
694
- "normalized": false,
695
- "rstrip": false,
696
- "single_word": false,
697
- "special": true
698
- },
699
- "128087": {
700
- "content": "<|reserved_special_token_82|>",
701
- "lstrip": false,
702
- "normalized": false,
703
- "rstrip": false,
704
- "single_word": false,
705
- "special": true
706
- },
707
- "128088": {
708
- "content": "<|reserved_special_token_83|>",
709
- "lstrip": false,
710
- "normalized": false,
711
- "rstrip": false,
712
- "single_word": false,
713
- "special": true
714
- },
715
- "128089": {
716
- "content": "<|reserved_special_token_84|>",
717
- "lstrip": false,
718
- "normalized": false,
719
- "rstrip": false,
720
- "single_word": false,
721
- "special": true
722
- },
723
- "128090": {
724
- "content": "<|reserved_special_token_85|>",
725
- "lstrip": false,
726
- "normalized": false,
727
- "rstrip": false,
728
- "single_word": false,
729
- "special": true
730
- },
731
- "128091": {
732
- "content": "<|reserved_special_token_86|>",
733
- "lstrip": false,
734
- "normalized": false,
735
- "rstrip": false,
736
- "single_word": false,
737
- "special": true
738
- },
739
- "128092": {
740
- "content": "<|reserved_special_token_87|>",
741
- "lstrip": false,
742
- "normalized": false,
743
- "rstrip": false,
744
- "single_word": false,
745
- "special": true
746
- },
747
- "128093": {
748
- "content": "<|reserved_special_token_88|>",
749
- "lstrip": false,
750
- "normalized": false,
751
- "rstrip": false,
752
- "single_word": false,
753
- "special": true
754
- },
755
- "128094": {
756
- "content": "<|reserved_special_token_89|>",
757
- "lstrip": false,
758
- "normalized": false,
759
- "rstrip": false,
760
- "single_word": false,
761
- "special": true
762
- },
763
- "128095": {
764
- "content": "<|reserved_special_token_90|>",
765
- "lstrip": false,
766
- "normalized": false,
767
- "rstrip": false,
768
- "single_word": false,
769
- "special": true
770
- },
771
- "128096": {
772
- "content": "<|reserved_special_token_91|>",
773
- "lstrip": false,
774
- "normalized": false,
775
- "rstrip": false,
776
- "single_word": false,
777
- "special": true
778
- },
779
- "128097": {
780
- "content": "<|reserved_special_token_92|>",
781
- "lstrip": false,
782
- "normalized": false,
783
- "rstrip": false,
784
- "single_word": false,
785
- "special": true
786
- },
787
- "128098": {
788
- "content": "<|reserved_special_token_93|>",
789
- "lstrip": false,
790
- "normalized": false,
791
- "rstrip": false,
792
- "single_word": false,
793
- "special": true
794
- },
795
- "128099": {
796
- "content": "<|reserved_special_token_94|>",
797
- "lstrip": false,
798
- "normalized": false,
799
- "rstrip": false,
800
- "single_word": false,
801
- "special": true
802
- },
803
- "128100": {
804
- "content": "<|reserved_special_token_95|>",
805
- "lstrip": false,
806
- "normalized": false,
807
- "rstrip": false,
808
- "single_word": false,
809
- "special": true
810
- },
811
- "128101": {
812
- "content": "<|reserved_special_token_96|>",
813
- "lstrip": false,
814
- "normalized": false,
815
- "rstrip": false,
816
- "single_word": false,
817
- "special": true
818
- },
819
- "128102": {
820
- "content": "<|reserved_special_token_97|>",
821
- "lstrip": false,
822
- "normalized": false,
823
- "rstrip": false,
824
- "single_word": false,
825
- "special": true
826
- },
827
- "128103": {
828
- "content": "<|reserved_special_token_98|>",
829
- "lstrip": false,
830
- "normalized": false,
831
- "rstrip": false,
832
- "single_word": false,
833
- "special": true
834
- },
835
- "128104": {
836
- "content": "<|reserved_special_token_99|>",
837
- "lstrip": false,
838
- "normalized": false,
839
- "rstrip": false,
840
- "single_word": false,
841
- "special": true
842
- },
843
- "128105": {
844
- "content": "<|reserved_special_token_100|>",
845
- "lstrip": false,
846
- "normalized": false,
847
- "rstrip": false,
848
- "single_word": false,
849
- "special": true
850
- },
851
- "128106": {
852
- "content": "<|reserved_special_token_101|>",
853
- "lstrip": false,
854
- "normalized": false,
855
- "rstrip": false,
856
- "single_word": false,
857
- "special": true
858
- },
859
- "128107": {
860
- "content": "<|reserved_special_token_102|>",
861
- "lstrip": false,
862
- "normalized": false,
863
- "rstrip": false,
864
- "single_word": false,
865
- "special": true
866
- },
867
- "128108": {
868
- "content": "<|reserved_special_token_103|>",
869
- "lstrip": false,
870
- "normalized": false,
871
- "rstrip": false,
872
- "single_word": false,
873
- "special": true
874
- },
875
- "128109": {
876
- "content": "<|reserved_special_token_104|>",
877
- "lstrip": false,
878
- "normalized": false,
879
- "rstrip": false,
880
- "single_word": false,
881
- "special": true
882
- },
883
- "128110": {
884
- "content": "<|reserved_special_token_105|>",
885
- "lstrip": false,
886
- "normalized": false,
887
- "rstrip": false,
888
- "single_word": false,
889
- "special": true
890
- },
891
- "128111": {
892
- "content": "<|reserved_special_token_106|>",
893
- "lstrip": false,
894
- "normalized": false,
895
- "rstrip": false,
896
- "single_word": false,
897
- "special": true
898
- },
899
- "128112": {
900
- "content": "<|reserved_special_token_107|>",
901
- "lstrip": false,
902
- "normalized": false,
903
- "rstrip": false,
904
- "single_word": false,
905
- "special": true
906
- },
907
- "128113": {
908
- "content": "<|reserved_special_token_108|>",
909
- "lstrip": false,
910
- "normalized": false,
911
- "rstrip": false,
912
- "single_word": false,
913
- "special": true
914
- },
915
- "128114": {
916
- "content": "<|reserved_special_token_109|>",
917
- "lstrip": false,
918
- "normalized": false,
919
- "rstrip": false,
920
- "single_word": false,
921
- "special": true
922
- },
923
- "128115": {
924
- "content": "<|reserved_special_token_110|>",
925
- "lstrip": false,
926
- "normalized": false,
927
- "rstrip": false,
928
- "single_word": false,
929
- "special": true
930
- },
931
- "128116": {
932
- "content": "<|reserved_special_token_111|>",
933
- "lstrip": false,
934
- "normalized": false,
935
- "rstrip": false,
936
- "single_word": false,
937
- "special": true
938
- },
939
- "128117": {
940
- "content": "<|reserved_special_token_112|>",
941
- "lstrip": false,
942
- "normalized": false,
943
- "rstrip": false,
944
- "single_word": false,
945
- "special": true
946
- },
947
- "128118": {
948
- "content": "<|reserved_special_token_113|>",
949
- "lstrip": false,
950
- "normalized": false,
951
- "rstrip": false,
952
- "single_word": false,
953
- "special": true
954
- },
955
- "128119": {
956
- "content": "<|reserved_special_token_114|>",
957
- "lstrip": false,
958
- "normalized": false,
959
- "rstrip": false,
960
- "single_word": false,
961
- "special": true
962
- },
963
- "128120": {
964
- "content": "<|reserved_special_token_115|>",
965
- "lstrip": false,
966
- "normalized": false,
967
- "rstrip": false,
968
- "single_word": false,
969
- "special": true
970
- },
971
- "128121": {
972
- "content": "<|reserved_special_token_116|>",
973
- "lstrip": false,
974
- "normalized": false,
975
- "rstrip": false,
976
- "single_word": false,
977
- "special": true
978
- },
979
- "128122": {
980
- "content": "<|reserved_special_token_117|>",
981
- "lstrip": false,
982
- "normalized": false,
983
- "rstrip": false,
984
- "single_word": false,
985
- "special": true
986
- },
987
- "128123": {
988
- "content": "<|reserved_special_token_118|>",
989
- "lstrip": false,
990
- "normalized": false,
991
- "rstrip": false,
992
- "single_word": false,
993
- "special": true
994
- },
995
- "128124": {
996
- "content": "<|reserved_special_token_119|>",
997
- "lstrip": false,
998
- "normalized": false,
999
- "rstrip": false,
1000
- "single_word": false,
1001
- "special": true
1002
- },
1003
- "128125": {
1004
- "content": "<|reserved_special_token_120|>",
1005
- "lstrip": false,
1006
- "normalized": false,
1007
- "rstrip": false,
1008
- "single_word": false,
1009
- "special": true
1010
- },
1011
- "128126": {
1012
- "content": "<|reserved_special_token_121|>",
1013
- "lstrip": false,
1014
- "normalized": false,
1015
- "rstrip": false,
1016
- "single_word": false,
1017
- "special": true
1018
- },
1019
- "128127": {
1020
- "content": "<|reserved_special_token_122|>",
1021
- "lstrip": false,
1022
- "normalized": false,
1023
- "rstrip": false,
1024
- "single_word": false,
1025
- "special": true
1026
- },
1027
- "128128": {
1028
- "content": "<|reserved_special_token_123|>",
1029
- "lstrip": false,
1030
- "normalized": false,
1031
- "rstrip": false,
1032
- "single_word": false,
1033
- "special": true
1034
- },
1035
- "128129": {
1036
- "content": "<|reserved_special_token_124|>",
1037
- "lstrip": false,
1038
- "normalized": false,
1039
- "rstrip": false,
1040
- "single_word": false,
1041
- "special": true
1042
- },
1043
- "128130": {
1044
- "content": "<|reserved_special_token_125|>",
1045
- "lstrip": false,
1046
- "normalized": false,
1047
- "rstrip": false,
1048
- "single_word": false,
1049
- "special": true
1050
- },
1051
- "128131": {
1052
- "content": "<|reserved_special_token_126|>",
1053
- "lstrip": false,
1054
- "normalized": false,
1055
- "rstrip": false,
1056
- "single_word": false,
1057
- "special": true
1058
- },
1059
- "128132": {
1060
- "content": "<|reserved_special_token_127|>",
1061
- "lstrip": false,
1062
- "normalized": false,
1063
- "rstrip": false,
1064
- "single_word": false,
1065
- "special": true
1066
- },
1067
- "128133": {
1068
- "content": "<|reserved_special_token_128|>",
1069
- "lstrip": false,
1070
- "normalized": false,
1071
- "rstrip": false,
1072
- "single_word": false,
1073
- "special": true
1074
- },
1075
- "128134": {
1076
- "content": "<|reserved_special_token_129|>",
1077
- "lstrip": false,
1078
- "normalized": false,
1079
- "rstrip": false,
1080
- "single_word": false,
1081
- "special": true
1082
- },
1083
- "128135": {
1084
- "content": "<|reserved_special_token_130|>",
1085
- "lstrip": false,
1086
- "normalized": false,
1087
- "rstrip": false,
1088
- "single_word": false,
1089
- "special": true
1090
- },
1091
- "128136": {
1092
- "content": "<|reserved_special_token_131|>",
1093
- "lstrip": false,
1094
- "normalized": false,
1095
- "rstrip": false,
1096
- "single_word": false,
1097
- "special": true
1098
- },
1099
- "128137": {
1100
- "content": "<|reserved_special_token_132|>",
1101
- "lstrip": false,
1102
- "normalized": false,
1103
- "rstrip": false,
1104
- "single_word": false,
1105
- "special": true
1106
- },
1107
- "128138": {
1108
- "content": "<|reserved_special_token_133|>",
1109
- "lstrip": false,
1110
- "normalized": false,
1111
- "rstrip": false,
1112
- "single_word": false,
1113
- "special": true
1114
- },
1115
- "128139": {
1116
- "content": "<|reserved_special_token_134|>",
1117
- "lstrip": false,
1118
- "normalized": false,
1119
- "rstrip": false,
1120
- "single_word": false,
1121
- "special": true
1122
- },
1123
- "128140": {
1124
- "content": "<|reserved_special_token_135|>",
1125
- "lstrip": false,
1126
- "normalized": false,
1127
- "rstrip": false,
1128
- "single_word": false,
1129
- "special": true
1130
- },
1131
- "128141": {
1132
- "content": "<|reserved_special_token_136|>",
1133
- "lstrip": false,
1134
- "normalized": false,
1135
- "rstrip": false,
1136
- "single_word": false,
1137
- "special": true
1138
- },
1139
- "128142": {
1140
- "content": "<|reserved_special_token_137|>",
1141
- "lstrip": false,
1142
- "normalized": false,
1143
- "rstrip": false,
1144
- "single_word": false,
1145
- "special": true
1146
- },
1147
- "128143": {
1148
- "content": "<|reserved_special_token_138|>",
1149
- "lstrip": false,
1150
- "normalized": false,
1151
- "rstrip": false,
1152
- "single_word": false,
1153
- "special": true
1154
- },
1155
- "128144": {
1156
- "content": "<|reserved_special_token_139|>",
1157
- "lstrip": false,
1158
- "normalized": false,
1159
- "rstrip": false,
1160
- "single_word": false,
1161
- "special": true
1162
- },
1163
- "128145": {
1164
- "content": "<|reserved_special_token_140|>",
1165
- "lstrip": false,
1166
- "normalized": false,
1167
- "rstrip": false,
1168
- "single_word": false,
1169
- "special": true
1170
- },
1171
- "128146": {
1172
- "content": "<|reserved_special_token_141|>",
1173
- "lstrip": false,
1174
- "normalized": false,
1175
- "rstrip": false,
1176
- "single_word": false,
1177
- "special": true
1178
- },
1179
- "128147": {
1180
- "content": "<|reserved_special_token_142|>",
1181
- "lstrip": false,
1182
- "normalized": false,
1183
- "rstrip": false,
1184
- "single_word": false,
1185
- "special": true
1186
- },
1187
- "128148": {
1188
- "content": "<|reserved_special_token_143|>",
1189
- "lstrip": false,
1190
- "normalized": false,
1191
- "rstrip": false,
1192
- "single_word": false,
1193
- "special": true
1194
- },
1195
- "128149": {
1196
- "content": "<|reserved_special_token_144|>",
1197
- "lstrip": false,
1198
- "normalized": false,
1199
- "rstrip": false,
1200
- "single_word": false,
1201
- "special": true
1202
- },
1203
- "128150": {
1204
- "content": "<|reserved_special_token_145|>",
1205
- "lstrip": false,
1206
- "normalized": false,
1207
- "rstrip": false,
1208
- "single_word": false,
1209
- "special": true
1210
- },
1211
- "128151": {
1212
- "content": "<|reserved_special_token_146|>",
1213
- "lstrip": false,
1214
- "normalized": false,
1215
- "rstrip": false,
1216
- "single_word": false,
1217
- "special": true
1218
- },
1219
- "128152": {
1220
- "content": "<|reserved_special_token_147|>",
1221
- "lstrip": false,
1222
- "normalized": false,
1223
- "rstrip": false,
1224
- "single_word": false,
1225
- "special": true
1226
- },
1227
- "128153": {
1228
- "content": "<|reserved_special_token_148|>",
1229
- "lstrip": false,
1230
- "normalized": false,
1231
- "rstrip": false,
1232
- "single_word": false,
1233
- "special": true
1234
- },
1235
- "128154": {
1236
- "content": "<|reserved_special_token_149|>",
1237
- "lstrip": false,
1238
- "normalized": false,
1239
- "rstrip": false,
1240
- "single_word": false,
1241
- "special": true
1242
- },
1243
- "128155": {
1244
- "content": "<|reserved_special_token_150|>",
1245
- "lstrip": false,
1246
- "normalized": false,
1247
- "rstrip": false,
1248
- "single_word": false,
1249
- "special": true
1250
- },
1251
- "128156": {
1252
- "content": "<|reserved_special_token_151|>",
1253
- "lstrip": false,
1254
- "normalized": false,
1255
- "rstrip": false,
1256
- "single_word": false,
1257
- "special": true
1258
- },
1259
- "128157": {
1260
- "content": "<|reserved_special_token_152|>",
1261
- "lstrip": false,
1262
- "normalized": false,
1263
- "rstrip": false,
1264
- "single_word": false,
1265
- "special": true
1266
- },
1267
- "128158": {
1268
- "content": "<|reserved_special_token_153|>",
1269
- "lstrip": false,
1270
- "normalized": false,
1271
- "rstrip": false,
1272
- "single_word": false,
1273
- "special": true
1274
- },
1275
- "128159": {
1276
- "content": "<|reserved_special_token_154|>",
1277
- "lstrip": false,
1278
- "normalized": false,
1279
- "rstrip": false,
1280
- "single_word": false,
1281
- "special": true
1282
- },
1283
- "128160": {
1284
- "content": "<|reserved_special_token_155|>",
1285
- "lstrip": false,
1286
- "normalized": false,
1287
- "rstrip": false,
1288
- "single_word": false,
1289
- "special": true
1290
- },
1291
- "128161": {
1292
- "content": "<|reserved_special_token_156|>",
1293
- "lstrip": false,
1294
- "normalized": false,
1295
- "rstrip": false,
1296
- "single_word": false,
1297
- "special": true
1298
- },
1299
- "128162": {
1300
- "content": "<|reserved_special_token_157|>",
1301
- "lstrip": false,
1302
- "normalized": false,
1303
- "rstrip": false,
1304
- "single_word": false,
1305
- "special": true
1306
- },
1307
- "128163": {
1308
- "content": "<|reserved_special_token_158|>",
1309
- "lstrip": false,
1310
- "normalized": false,
1311
- "rstrip": false,
1312
- "single_word": false,
1313
- "special": true
1314
- },
1315
- "128164": {
1316
- "content": "<|reserved_special_token_159|>",
1317
- "lstrip": false,
1318
- "normalized": false,
1319
- "rstrip": false,
1320
- "single_word": false,
1321
- "special": true
1322
- },
1323
- "128165": {
1324
- "content": "<|reserved_special_token_160|>",
1325
- "lstrip": false,
1326
- "normalized": false,
1327
- "rstrip": false,
1328
- "single_word": false,
1329
- "special": true
1330
- },
1331
- "128166": {
1332
- "content": "<|reserved_special_token_161|>",
1333
- "lstrip": false,
1334
- "normalized": false,
1335
- "rstrip": false,
1336
- "single_word": false,
1337
- "special": true
1338
- },
1339
- "128167": {
1340
- "content": "<|reserved_special_token_162|>",
1341
- "lstrip": false,
1342
- "normalized": false,
1343
- "rstrip": false,
1344
- "single_word": false,
1345
- "special": true
1346
- },
1347
- "128168": {
1348
- "content": "<|reserved_special_token_163|>",
1349
- "lstrip": false,
1350
- "normalized": false,
1351
- "rstrip": false,
1352
- "single_word": false,
1353
- "special": true
1354
- },
1355
- "128169": {
1356
- "content": "<|reserved_special_token_164|>",
1357
- "lstrip": false,
1358
- "normalized": false,
1359
- "rstrip": false,
1360
- "single_word": false,
1361
- "special": true
1362
- },
1363
- "128170": {
1364
- "content": "<|reserved_special_token_165|>",
1365
- "lstrip": false,
1366
- "normalized": false,
1367
- "rstrip": false,
1368
- "single_word": false,
1369
- "special": true
1370
- },
1371
- "128171": {
1372
- "content": "<|reserved_special_token_166|>",
1373
- "lstrip": false,
1374
- "normalized": false,
1375
- "rstrip": false,
1376
- "single_word": false,
1377
- "special": true
1378
- },
1379
- "128172": {
1380
- "content": "<|reserved_special_token_167|>",
1381
- "lstrip": false,
1382
- "normalized": false,
1383
- "rstrip": false,
1384
- "single_word": false,
1385
- "special": true
1386
- },
1387
- "128173": {
1388
- "content": "<|reserved_special_token_168|>",
1389
- "lstrip": false,
1390
- "normalized": false,
1391
- "rstrip": false,
1392
- "single_word": false,
1393
- "special": true
1394
- },
1395
- "128174": {
1396
- "content": "<|reserved_special_token_169|>",
1397
- "lstrip": false,
1398
- "normalized": false,
1399
- "rstrip": false,
1400
- "single_word": false,
1401
- "special": true
1402
- },
1403
- "128175": {
1404
- "content": "<|reserved_special_token_170|>",
1405
- "lstrip": false,
1406
- "normalized": false,
1407
- "rstrip": false,
1408
- "single_word": false,
1409
- "special": true
1410
- },
1411
- "128176": {
1412
- "content": "<|reserved_special_token_171|>",
1413
- "lstrip": false,
1414
- "normalized": false,
1415
- "rstrip": false,
1416
- "single_word": false,
1417
- "special": true
1418
- },
1419
- "128177": {
1420
- "content": "<|reserved_special_token_172|>",
1421
- "lstrip": false,
1422
- "normalized": false,
1423
- "rstrip": false,
1424
- "single_word": false,
1425
- "special": true
1426
- },
1427
- "128178": {
1428
- "content": "<|reserved_special_token_173|>",
1429
- "lstrip": false,
1430
- "normalized": false,
1431
- "rstrip": false,
1432
- "single_word": false,
1433
- "special": true
1434
- },
1435
- "128179": {
1436
- "content": "<|reserved_special_token_174|>",
1437
- "lstrip": false,
1438
- "normalized": false,
1439
- "rstrip": false,
1440
- "single_word": false,
1441
- "special": true
1442
- },
1443
- "128180": {
1444
- "content": "<|reserved_special_token_175|>",
1445
- "lstrip": false,
1446
- "normalized": false,
1447
- "rstrip": false,
1448
- "single_word": false,
1449
- "special": true
1450
- },
1451
- "128181": {
1452
- "content": "<|reserved_special_token_176|>",
1453
- "lstrip": false,
1454
- "normalized": false,
1455
- "rstrip": false,
1456
- "single_word": false,
1457
- "special": true
1458
- },
1459
- "128182": {
1460
- "content": "<|reserved_special_token_177|>",
1461
- "lstrip": false,
1462
- "normalized": false,
1463
- "rstrip": false,
1464
- "single_word": false,
1465
- "special": true
1466
- },
1467
- "128183": {
1468
- "content": "<|reserved_special_token_178|>",
1469
- "lstrip": false,
1470
- "normalized": false,
1471
- "rstrip": false,
1472
- "single_word": false,
1473
- "special": true
1474
- },
1475
- "128184": {
1476
- "content": "<|reserved_special_token_179|>",
1477
- "lstrip": false,
1478
- "normalized": false,
1479
- "rstrip": false,
1480
- "single_word": false,
1481
- "special": true
1482
- },
1483
- "128185": {
1484
- "content": "<|reserved_special_token_180|>",
1485
- "lstrip": false,
1486
- "normalized": false,
1487
- "rstrip": false,
1488
- "single_word": false,
1489
- "special": true
1490
- },
1491
- "128186": {
1492
- "content": "<|reserved_special_token_181|>",
1493
- "lstrip": false,
1494
- "normalized": false,
1495
- "rstrip": false,
1496
- "single_word": false,
1497
- "special": true
1498
- },
1499
- "128187": {
1500
- "content": "<|reserved_special_token_182|>",
1501
- "lstrip": false,
1502
- "normalized": false,
1503
- "rstrip": false,
1504
- "single_word": false,
1505
- "special": true
1506
- },
1507
- "128188": {
1508
- "content": "<|reserved_special_token_183|>",
1509
- "lstrip": false,
1510
- "normalized": false,
1511
- "rstrip": false,
1512
- "single_word": false,
1513
- "special": true
1514
- },
1515
- "128189": {
1516
- "content": "<|reserved_special_token_184|>",
1517
- "lstrip": false,
1518
- "normalized": false,
1519
- "rstrip": false,
1520
- "single_word": false,
1521
- "special": true
1522
- },
1523
- "128190": {
1524
- "content": "<|reserved_special_token_185|>",
1525
- "lstrip": false,
1526
- "normalized": false,
1527
- "rstrip": false,
1528
- "single_word": false,
1529
- "special": true
1530
- },
1531
- "128191": {
1532
- "content": "<|reserved_special_token_186|>",
1533
- "lstrip": false,
1534
- "normalized": false,
1535
- "rstrip": false,
1536
- "single_word": false,
1537
- "special": true
1538
- },
1539
- "128192": {
1540
- "content": "<|reserved_special_token_187|>",
1541
- "lstrip": false,
1542
- "normalized": false,
1543
- "rstrip": false,
1544
- "single_word": false,
1545
- "special": true
1546
- },
1547
- "128193": {
1548
- "content": "<|reserved_special_token_188|>",
1549
- "lstrip": false,
1550
- "normalized": false,
1551
- "rstrip": false,
1552
- "single_word": false,
1553
- "special": true
1554
- },
1555
- "128194": {
1556
- "content": "<|reserved_special_token_189|>",
1557
- "lstrip": false,
1558
- "normalized": false,
1559
- "rstrip": false,
1560
- "single_word": false,
1561
- "special": true
1562
- },
1563
- "128195": {
1564
- "content": "<|reserved_special_token_190|>",
1565
- "lstrip": false,
1566
- "normalized": false,
1567
- "rstrip": false,
1568
- "single_word": false,
1569
- "special": true
1570
- },
1571
- "128196": {
1572
- "content": "<|reserved_special_token_191|>",
1573
- "lstrip": false,
1574
- "normalized": false,
1575
- "rstrip": false,
1576
- "single_word": false,
1577
- "special": true
1578
- },
1579
- "128197": {
1580
- "content": "<|reserved_special_token_192|>",
1581
- "lstrip": false,
1582
- "normalized": false,
1583
- "rstrip": false,
1584
- "single_word": false,
1585
- "special": true
1586
- },
1587
- "128198": {
1588
- "content": "<|reserved_special_token_193|>",
1589
- "lstrip": false,
1590
- "normalized": false,
1591
- "rstrip": false,
1592
- "single_word": false,
1593
- "special": true
1594
- },
1595
- "128199": {
1596
- "content": "<|reserved_special_token_194|>",
1597
- "lstrip": false,
1598
- "normalized": false,
1599
- "rstrip": false,
1600
- "single_word": false,
1601
- "special": true
1602
- },
1603
- "128200": {
1604
- "content": "<|reserved_special_token_195|>",
1605
- "lstrip": false,
1606
- "normalized": false,
1607
- "rstrip": false,
1608
- "single_word": false,
1609
- "special": true
1610
- },
1611
- "128201": {
1612
- "content": "<|reserved_special_token_196|>",
1613
- "lstrip": false,
1614
- "normalized": false,
1615
- "rstrip": false,
1616
- "single_word": false,
1617
- "special": true
1618
- },
1619
- "128202": {
1620
- "content": "<|reserved_special_token_197|>",
1621
- "lstrip": false,
1622
- "normalized": false,
1623
- "rstrip": false,
1624
- "single_word": false,
1625
- "special": true
1626
- },
1627
- "128203": {
1628
- "content": "<|reserved_special_token_198|>",
1629
- "lstrip": false,
1630
- "normalized": false,
1631
- "rstrip": false,
1632
- "single_word": false,
1633
- "special": true
1634
- },
1635
- "128204": {
1636
- "content": "<|reserved_special_token_199|>",
1637
- "lstrip": false,
1638
- "normalized": false,
1639
- "rstrip": false,
1640
- "single_word": false,
1641
- "special": true
1642
- },
1643
- "128205": {
1644
- "content": "<|reserved_special_token_200|>",
1645
- "lstrip": false,
1646
- "normalized": false,
1647
- "rstrip": false,
1648
- "single_word": false,
1649
- "special": true
1650
- },
1651
- "128206": {
1652
- "content": "<|reserved_special_token_201|>",
1653
- "lstrip": false,
1654
- "normalized": false,
1655
- "rstrip": false,
1656
- "single_word": false,
1657
- "special": true
1658
- },
1659
- "128207": {
1660
- "content": "<|reserved_special_token_202|>",
1661
- "lstrip": false,
1662
- "normalized": false,
1663
- "rstrip": false,
1664
- "single_word": false,
1665
- "special": true
1666
- },
1667
- "128208": {
1668
- "content": "<|reserved_special_token_203|>",
1669
- "lstrip": false,
1670
- "normalized": false,
1671
- "rstrip": false,
1672
- "single_word": false,
1673
- "special": true
1674
- },
1675
- "128209": {
1676
- "content": "<|reserved_special_token_204|>",
1677
- "lstrip": false,
1678
- "normalized": false,
1679
- "rstrip": false,
1680
- "single_word": false,
1681
- "special": true
1682
- },
1683
- "128210": {
1684
- "content": "<|reserved_special_token_205|>",
1685
- "lstrip": false,
1686
- "normalized": false,
1687
- "rstrip": false,
1688
- "single_word": false,
1689
- "special": true
1690
- },
1691
- "128211": {
1692
- "content": "<|reserved_special_token_206|>",
1693
- "lstrip": false,
1694
- "normalized": false,
1695
- "rstrip": false,
1696
- "single_word": false,
1697
- "special": true
1698
- },
1699
- "128212": {
1700
- "content": "<|reserved_special_token_207|>",
1701
- "lstrip": false,
1702
- "normalized": false,
1703
- "rstrip": false,
1704
- "single_word": false,
1705
- "special": true
1706
- },
1707
- "128213": {
1708
- "content": "<|reserved_special_token_208|>",
1709
- "lstrip": false,
1710
- "normalized": false,
1711
- "rstrip": false,
1712
- "single_word": false,
1713
- "special": true
1714
- },
1715
- "128214": {
1716
- "content": "<|reserved_special_token_209|>",
1717
- "lstrip": false,
1718
- "normalized": false,
1719
- "rstrip": false,
1720
- "single_word": false,
1721
- "special": true
1722
- },
1723
- "128215": {
1724
- "content": "<|reserved_special_token_210|>",
1725
- "lstrip": false,
1726
- "normalized": false,
1727
- "rstrip": false,
1728
- "single_word": false,
1729
- "special": true
1730
- },
1731
- "128216": {
1732
- "content": "<|reserved_special_token_211|>",
1733
- "lstrip": false,
1734
- "normalized": false,
1735
- "rstrip": false,
1736
- "single_word": false,
1737
- "special": true
1738
- },
1739
- "128217": {
1740
- "content": "<|reserved_special_token_212|>",
1741
- "lstrip": false,
1742
- "normalized": false,
1743
- "rstrip": false,
1744
- "single_word": false,
1745
- "special": true
1746
- },
1747
- "128218": {
1748
- "content": "<|reserved_special_token_213|>",
1749
- "lstrip": false,
1750
- "normalized": false,
1751
- "rstrip": false,
1752
- "single_word": false,
1753
- "special": true
1754
- },
1755
- "128219": {
1756
- "content": "<|reserved_special_token_214|>",
1757
- "lstrip": false,
1758
- "normalized": false,
1759
- "rstrip": false,
1760
- "single_word": false,
1761
- "special": true
1762
- },
1763
- "128220": {
1764
- "content": "<|reserved_special_token_215|>",
1765
- "lstrip": false,
1766
- "normalized": false,
1767
- "rstrip": false,
1768
- "single_word": false,
1769
- "special": true
1770
- },
1771
- "128221": {
1772
- "content": "<|reserved_special_token_216|>",
1773
- "lstrip": false,
1774
- "normalized": false,
1775
- "rstrip": false,
1776
- "single_word": false,
1777
- "special": true
1778
- },
1779
- "128222": {
1780
- "content": "<|reserved_special_token_217|>",
1781
- "lstrip": false,
1782
- "normalized": false,
1783
- "rstrip": false,
1784
- "single_word": false,
1785
- "special": true
1786
- },
1787
- "128223": {
1788
- "content": "<|reserved_special_token_218|>",
1789
- "lstrip": false,
1790
- "normalized": false,
1791
- "rstrip": false,
1792
- "single_word": false,
1793
- "special": true
1794
- },
1795
- "128224": {
1796
- "content": "<|reserved_special_token_219|>",
1797
- "lstrip": false,
1798
- "normalized": false,
1799
- "rstrip": false,
1800
- "single_word": false,
1801
- "special": true
1802
- },
1803
- "128225": {
1804
- "content": "<|reserved_special_token_220|>",
1805
- "lstrip": false,
1806
- "normalized": false,
1807
- "rstrip": false,
1808
- "single_word": false,
1809
- "special": true
1810
- },
1811
- "128226": {
1812
- "content": "<|reserved_special_token_221|>",
1813
- "lstrip": false,
1814
- "normalized": false,
1815
- "rstrip": false,
1816
- "single_word": false,
1817
- "special": true
1818
- },
1819
- "128227": {
1820
- "content": "<|reserved_special_token_222|>",
1821
- "lstrip": false,
1822
- "normalized": false,
1823
- "rstrip": false,
1824
- "single_word": false,
1825
- "special": true
1826
- },
1827
- "128228": {
1828
- "content": "<|reserved_special_token_223|>",
1829
- "lstrip": false,
1830
- "normalized": false,
1831
- "rstrip": false,
1832
- "single_word": false,
1833
- "special": true
1834
- },
1835
- "128229": {
1836
- "content": "<|reserved_special_token_224|>",
1837
- "lstrip": false,
1838
- "normalized": false,
1839
- "rstrip": false,
1840
- "single_word": false,
1841
- "special": true
1842
- },
1843
- "128230": {
1844
- "content": "<|reserved_special_token_225|>",
1845
- "lstrip": false,
1846
- "normalized": false,
1847
- "rstrip": false,
1848
- "single_word": false,
1849
- "special": true
1850
- },
1851
- "128231": {
1852
- "content": "<|reserved_special_token_226|>",
1853
- "lstrip": false,
1854
- "normalized": false,
1855
- "rstrip": false,
1856
- "single_word": false,
1857
- "special": true
1858
- },
1859
- "128232": {
1860
- "content": "<|reserved_special_token_227|>",
1861
- "lstrip": false,
1862
- "normalized": false,
1863
- "rstrip": false,
1864
- "single_word": false,
1865
- "special": true
1866
- },
1867
- "128233": {
1868
- "content": "<|reserved_special_token_228|>",
1869
- "lstrip": false,
1870
- "normalized": false,
1871
- "rstrip": false,
1872
- "single_word": false,
1873
- "special": true
1874
- },
1875
- "128234": {
1876
- "content": "<|reserved_special_token_229|>",
1877
- "lstrip": false,
1878
- "normalized": false,
1879
- "rstrip": false,
1880
- "single_word": false,
1881
- "special": true
1882
- },
1883
- "128235": {
1884
- "content": "<|reserved_special_token_230|>",
1885
- "lstrip": false,
1886
- "normalized": false,
1887
- "rstrip": false,
1888
- "single_word": false,
1889
- "special": true
1890
- },
1891
- "128236": {
1892
- "content": "<|reserved_special_token_231|>",
1893
- "lstrip": false,
1894
- "normalized": false,
1895
- "rstrip": false,
1896
- "single_word": false,
1897
- "special": true
1898
- },
1899
- "128237": {
1900
- "content": "<|reserved_special_token_232|>",
1901
- "lstrip": false,
1902
- "normalized": false,
1903
- "rstrip": false,
1904
- "single_word": false,
1905
- "special": true
1906
- },
1907
- "128238": {
1908
- "content": "<|reserved_special_token_233|>",
1909
- "lstrip": false,
1910
- "normalized": false,
1911
- "rstrip": false,
1912
- "single_word": false,
1913
- "special": true
1914
- },
1915
- "128239": {
1916
- "content": "<|reserved_special_token_234|>",
1917
- "lstrip": false,
1918
- "normalized": false,
1919
- "rstrip": false,
1920
- "single_word": false,
1921
- "special": true
1922
- },
1923
- "128240": {
1924
- "content": "<|reserved_special_token_235|>",
1925
- "lstrip": false,
1926
- "normalized": false,
1927
- "rstrip": false,
1928
- "single_word": false,
1929
- "special": true
1930
- },
1931
- "128241": {
1932
- "content": "<|reserved_special_token_236|>",
1933
- "lstrip": false,
1934
- "normalized": false,
1935
- "rstrip": false,
1936
- "single_word": false,
1937
- "special": true
1938
- },
1939
- "128242": {
1940
- "content": "<|reserved_special_token_237|>",
1941
- "lstrip": false,
1942
- "normalized": false,
1943
- "rstrip": false,
1944
- "single_word": false,
1945
- "special": true
1946
- },
1947
- "128243": {
1948
- "content": "<|reserved_special_token_238|>",
1949
- "lstrip": false,
1950
- "normalized": false,
1951
- "rstrip": false,
1952
- "single_word": false,
1953
- "special": true
1954
- },
1955
- "128244": {
1956
- "content": "<|reserved_special_token_239|>",
1957
- "lstrip": false,
1958
- "normalized": false,
1959
- "rstrip": false,
1960
- "single_word": false,
1961
- "special": true
1962
- },
1963
- "128245": {
1964
- "content": "<|reserved_special_token_240|>",
1965
- "lstrip": false,
1966
- "normalized": false,
1967
- "rstrip": false,
1968
- "single_word": false,
1969
- "special": true
1970
- },
1971
- "128246": {
1972
- "content": "<|reserved_special_token_241|>",
1973
- "lstrip": false,
1974
- "normalized": false,
1975
- "rstrip": false,
1976
- "single_word": false,
1977
- "special": true
1978
- },
1979
- "128247": {
1980
- "content": "<|reserved_special_token_242|>",
1981
- "lstrip": false,
1982
- "normalized": false,
1983
- "rstrip": false,
1984
- "single_word": false,
1985
- "special": true
1986
- },
1987
- "128248": {
1988
- "content": "<|reserved_special_token_243|>",
1989
- "lstrip": false,
1990
- "normalized": false,
1991
- "rstrip": false,
1992
- "single_word": false,
1993
- "special": true
1994
- },
1995
- "128249": {
1996
- "content": "<|reserved_special_token_244|>",
1997
- "lstrip": false,
1998
- "normalized": false,
1999
- "rstrip": false,
2000
- "single_word": false,
2001
- "special": true
2002
- },
2003
- "128250": {
2004
- "content": "<|reserved_special_token_245|>",
2005
- "lstrip": false,
2006
- "normalized": false,
2007
- "rstrip": false,
2008
- "single_word": false,
2009
- "special": true
2010
- },
2011
- "128251": {
2012
- "content": "<|reserved_special_token_246|>",
2013
- "lstrip": false,
2014
- "normalized": false,
2015
- "rstrip": false,
2016
- "single_word": false,
2017
- "special": true
2018
- },
2019
- "128252": {
2020
- "content": "<|reserved_special_token_247|>",
2021
- "lstrip": false,
2022
- "normalized": false,
2023
- "rstrip": false,
2024
- "single_word": false,
2025
- "special": true
2026
- },
2027
- "128253": {
2028
- "content": "<|reserved_special_token_248|>",
2029
- "lstrip": false,
2030
- "normalized": false,
2031
- "rstrip": false,
2032
- "single_word": false,
2033
- "special": true
2034
- },
2035
- "128254": {
2036
- "content": "<|reserved_special_token_249|>",
2037
- "lstrip": false,
2038
- "normalized": false,
2039
- "rstrip": false,
2040
- "single_word": false,
2041
- "special": true
2042
- },
2043
- "128255": {
2044
- "content": "<|reserved_special_token_250|>",
2045
- "lstrip": false,
2046
- "normalized": false,
2047
- "rstrip": false,
2048
- "single_word": false,
2049
- "special": true
2050
- }
2051
- },
2052
- "bos_token": "<|begin_of_text|>",
2053
- "chat_template": "{% set loop_messages = messages %}{% for message in loop_messages %}{% set content = '<|start_header_id|>' + message['role'] + '<|end_header_id|>\n\n'+ message['content'] | trim + '<|eot_id|>' %}{% if loop.index0 == 0 %}{% set content = bos_token + content %}{% endif %}{{ content }}{% endfor %}{% if add_generation_prompt %}{{ '<|start_header_id|>assistant<|end_header_id|>\n\n' }}{% endif %}",
2054
- "clean_up_tokenization_spaces": true,
2055
- "eos_token": "<|eot_id|>",
2056
- "model_input_names": [
2057
- "input_ids",
2058
- "attention_mask"
2059
- ],
2060
- "model_max_length": 1000000000000000019884624838656,
2061
- "pad_token": "<|eot_id|>",
2062
- "padding_side": "right",
2063
- "split_special_tokens": false,
2064
- "tokenizer_class": "PreTrainedTokenizerFast"
2065
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
checkpoint-3300/trainer_state.json DELETED
@@ -1,2367 +0,0 @@
1
- {
2
- "best_metric": null,
3
- "best_model_checkpoint": null,
4
- "epoch": 11.989100817438691,
5
- "eval_steps": 1000,
6
- "global_step": 3300,
7
- "is_hyper_param_search": false,
8
- "is_local_process_zero": true,
9
- "is_world_process_zero": true,
10
- "log_history": [
11
- {
12
- "epoch": 0.03633060853769301,
13
- "grad_norm": 5.34375,
14
- "learning_rate": 3.0303030303030305e-07,
15
- "loss": 2.0607,
16
- "step": 10
17
- },
18
- {
19
- "epoch": 0.07266121707538602,
20
- "grad_norm": 5.90625,
21
- "learning_rate": 6.060606060606061e-07,
22
- "loss": 2.033,
23
- "step": 20
24
- },
25
- {
26
- "epoch": 0.10899182561307902,
27
- "grad_norm": 5.46875,
28
- "learning_rate": 9.090909090909091e-07,
29
- "loss": 2.0622,
30
- "step": 30
31
- },
32
- {
33
- "epoch": 0.14532243415077203,
34
- "grad_norm": 4.96875,
35
- "learning_rate": 1.2121212121212122e-06,
36
- "loss": 1.9896,
37
- "step": 40
38
- },
39
- {
40
- "epoch": 0.18165304268846502,
41
- "grad_norm": 15.625,
42
- "learning_rate": 1.5151515151515152e-06,
43
- "loss": 1.9998,
44
- "step": 50
45
- },
46
- {
47
- "epoch": 0.21798365122615804,
48
- "grad_norm": 18.0,
49
- "learning_rate": 1.8181818181818183e-06,
50
- "loss": 1.9608,
51
- "step": 60
52
- },
53
- {
54
- "epoch": 0.254314259763851,
55
- "grad_norm": 15.625,
56
- "learning_rate": 2.1212121212121216e-06,
57
- "loss": 1.9272,
58
- "step": 70
59
- },
60
- {
61
- "epoch": 0.29064486830154407,
62
- "grad_norm": 15.5625,
63
- "learning_rate": 2.4242424242424244e-06,
64
- "loss": 1.9031,
65
- "step": 80
66
- },
67
- {
68
- "epoch": 0.32697547683923706,
69
- "grad_norm": 14.1875,
70
- "learning_rate": 2.7272727272727272e-06,
71
- "loss": 1.8012,
72
- "step": 90
73
- },
74
- {
75
- "epoch": 0.36330608537693004,
76
- "grad_norm": 3.421875,
77
- "learning_rate": 3.0303030303030305e-06,
78
- "loss": 1.7756,
79
- "step": 100
80
- },
81
- {
82
- "epoch": 0.3996366939146231,
83
- "grad_norm": 3.09375,
84
- "learning_rate": 3.3333333333333333e-06,
85
- "loss": 1.6836,
86
- "step": 110
87
- },
88
- {
89
- "epoch": 0.4359673024523161,
90
- "grad_norm": 2.46875,
91
- "learning_rate": 3.6363636363636366e-06,
92
- "loss": 1.7013,
93
- "step": 120
94
- },
95
- {
96
- "epoch": 0.47229791099000906,
97
- "grad_norm": 2.859375,
98
- "learning_rate": 3.93939393939394e-06,
99
- "loss": 1.6328,
100
- "step": 130
101
- },
102
- {
103
- "epoch": 0.508628519527702,
104
- "grad_norm": 2.6875,
105
- "learning_rate": 4.242424242424243e-06,
106
- "loss": 1.6888,
107
- "step": 140
108
- },
109
- {
110
- "epoch": 0.5449591280653951,
111
- "grad_norm": 2.828125,
112
- "learning_rate": 4.5454545454545455e-06,
113
- "loss": 1.6583,
114
- "step": 150
115
- },
116
- {
117
- "epoch": 0.5812897366030881,
118
- "grad_norm": 2.0625,
119
- "learning_rate": 4.848484848484849e-06,
120
- "loss": 1.5485,
121
- "step": 160
122
- },
123
- {
124
- "epoch": 0.6176203451407811,
125
- "grad_norm": 2.1875,
126
- "learning_rate": 5.151515151515152e-06,
127
- "loss": 1.5464,
128
- "step": 170
129
- },
130
- {
131
- "epoch": 0.6539509536784741,
132
- "grad_norm": 2.25,
133
- "learning_rate": 5.4545454545454545e-06,
134
- "loss": 1.5506,
135
- "step": 180
136
- },
137
- {
138
- "epoch": 0.6902815622161671,
139
- "grad_norm": 2.4375,
140
- "learning_rate": 5.7575757575757586e-06,
141
- "loss": 1.5301,
142
- "step": 190
143
- },
144
- {
145
- "epoch": 0.7266121707538601,
146
- "grad_norm": 2.4375,
147
- "learning_rate": 6.060606060606061e-06,
148
- "loss": 1.5116,
149
- "step": 200
150
- },
151
- {
152
- "epoch": 0.7629427792915532,
153
- "grad_norm": 2.609375,
154
- "learning_rate": 6.363636363636364e-06,
155
- "loss": 1.4874,
156
- "step": 210
157
- },
158
- {
159
- "epoch": 0.7992733878292462,
160
- "grad_norm": 2.3125,
161
- "learning_rate": 6.666666666666667e-06,
162
- "loss": 1.5056,
163
- "step": 220
164
- },
165
- {
166
- "epoch": 0.8356039963669392,
167
- "grad_norm": 2.21875,
168
- "learning_rate": 6.969696969696971e-06,
169
- "loss": 1.5036,
170
- "step": 230
171
- },
172
- {
173
- "epoch": 0.8719346049046321,
174
- "grad_norm": 2.359375,
175
- "learning_rate": 7.272727272727273e-06,
176
- "loss": 1.4625,
177
- "step": 240
178
- },
179
- {
180
- "epoch": 0.9082652134423251,
181
- "grad_norm": 1.921875,
182
- "learning_rate": 7.5757575757575764e-06,
183
- "loss": 1.4439,
184
- "step": 250
185
- },
186
- {
187
- "epoch": 0.9445958219800181,
188
- "grad_norm": 2.265625,
189
- "learning_rate": 7.87878787878788e-06,
190
- "loss": 1.4025,
191
- "step": 260
192
- },
193
- {
194
- "epoch": 0.9809264305177112,
195
- "grad_norm": 2.609375,
196
- "learning_rate": 8.181818181818183e-06,
197
- "loss": 1.4267,
198
- "step": 270
199
- },
200
- {
201
- "epoch": 1.017257039055404,
202
- "grad_norm": 1.84375,
203
- "learning_rate": 8.484848484848486e-06,
204
- "loss": 1.3484,
205
- "step": 280
206
- },
207
- {
208
- "epoch": 1.0535876475930972,
209
- "grad_norm": 2.0,
210
- "learning_rate": 8.787878787878788e-06,
211
- "loss": 1.3809,
212
- "step": 290
213
- },
214
- {
215
- "epoch": 1.0899182561307903,
216
- "grad_norm": 2.1875,
217
- "learning_rate": 9.090909090909091e-06,
218
- "loss": 1.3901,
219
- "step": 300
220
- },
221
- {
222
- "epoch": 1.1262488646684832,
223
- "grad_norm": 1.8984375,
224
- "learning_rate": 9.393939393939396e-06,
225
- "loss": 1.3203,
226
- "step": 310
227
- },
228
- {
229
- "epoch": 1.1625794732061763,
230
- "grad_norm": 1.9296875,
231
- "learning_rate": 9.696969696969698e-06,
232
- "loss": 1.3295,
233
- "step": 320
234
- },
235
- {
236
- "epoch": 1.1989100817438691,
237
- "grad_norm": 1.9140625,
238
- "learning_rate": 1e-05,
239
- "loss": 1.3936,
240
- "step": 330
241
- },
242
- {
243
- "epoch": 1.2352406902815622,
244
- "grad_norm": 2.5,
245
- "learning_rate": 9.999720280459576e-06,
246
- "loss": 1.3483,
247
- "step": 340
248
- },
249
- {
250
- "epoch": 1.2715712988192553,
251
- "grad_norm": 1.7890625,
252
- "learning_rate": 9.99888115313551e-06,
253
- "loss": 1.3384,
254
- "step": 350
255
- },
256
- {
257
- "epoch": 1.3079019073569482,
258
- "grad_norm": 1.921875,
259
- "learning_rate": 9.997482711915926e-06,
260
- "loss": 1.2888,
261
- "step": 360
262
- },
263
- {
264
- "epoch": 1.344232515894641,
265
- "grad_norm": 1.9453125,
266
- "learning_rate": 9.99552511326936e-06,
267
- "loss": 1.3144,
268
- "step": 370
269
- },
270
- {
271
- "epoch": 1.3805631244323342,
272
- "grad_norm": 1.796875,
273
- "learning_rate": 9.993008576227248e-06,
274
- "loss": 1.2905,
275
- "step": 380
276
- },
277
- {
278
- "epoch": 1.4168937329700273,
279
- "grad_norm": 1.8203125,
280
- "learning_rate": 9.989933382359423e-06,
281
- "loss": 1.2816,
282
- "step": 390
283
- },
284
- {
285
- "epoch": 1.4532243415077202,
286
- "grad_norm": 1.6953125,
287
- "learning_rate": 9.986299875742612e-06,
288
- "loss": 1.3315,
289
- "step": 400
290
- },
291
- {
292
- "epoch": 1.4895549500454133,
293
- "grad_norm": 1.7734375,
294
- "learning_rate": 9.982108462921938e-06,
295
- "loss": 1.2984,
296
- "step": 410
297
- },
298
- {
299
- "epoch": 1.5258855585831061,
300
- "grad_norm": 2.765625,
301
- "learning_rate": 9.977359612865424e-06,
302
- "loss": 1.2677,
303
- "step": 420
304
- },
305
- {
306
- "epoch": 1.5622161671207992,
307
- "grad_norm": 1.859375,
308
- "learning_rate": 9.972053856911534e-06,
309
- "loss": 1.3098,
310
- "step": 430
311
- },
312
- {
313
- "epoch": 1.5985467756584923,
314
- "grad_norm": 1.84375,
315
- "learning_rate": 9.966191788709716e-06,
316
- "loss": 1.3503,
317
- "step": 440
318
- },
319
- {
320
- "epoch": 1.6348773841961854,
321
- "grad_norm": 2.609375,
322
- "learning_rate": 9.959774064153977e-06,
323
- "loss": 1.3301,
324
- "step": 450
325
- },
326
- {
327
- "epoch": 1.6712079927338783,
328
- "grad_norm": 1.8359375,
329
- "learning_rate": 9.952801401309504e-06,
330
- "loss": 1.2454,
331
- "step": 460
332
- },
333
- {
334
- "epoch": 1.7075386012715712,
335
- "grad_norm": 1.8515625,
336
- "learning_rate": 9.945274580332316e-06,
337
- "loss": 1.239,
338
- "step": 470
339
- },
340
- {
341
- "epoch": 1.7438692098092643,
342
- "grad_norm": 1.8046875,
343
- "learning_rate": 9.937194443381972e-06,
344
- "loss": 1.2823,
345
- "step": 480
346
- },
347
- {
348
- "epoch": 1.7801998183469574,
349
- "grad_norm": 2.203125,
350
- "learning_rate": 9.928561894527354e-06,
351
- "loss": 1.2774,
352
- "step": 490
353
- },
354
- {
355
- "epoch": 1.8165304268846503,
356
- "grad_norm": 2.15625,
357
- "learning_rate": 9.919377899645497e-06,
358
- "loss": 1.2215,
359
- "step": 500
360
- },
361
- {
362
- "epoch": 1.8528610354223434,
363
- "grad_norm": 1.671875,
364
- "learning_rate": 9.909643486313533e-06,
365
- "loss": 1.2509,
366
- "step": 510
367
- },
368
- {
369
- "epoch": 1.8891916439600362,
370
- "grad_norm": 1.8828125,
371
- "learning_rate": 9.899359743693715e-06,
372
- "loss": 1.2656,
373
- "step": 520
374
- },
375
- {
376
- "epoch": 1.9255222524977293,
377
- "grad_norm": 1.6328125,
378
- "learning_rate": 9.888527822411543e-06,
379
- "loss": 1.186,
380
- "step": 530
381
- },
382
- {
383
- "epoch": 1.9618528610354224,
384
- "grad_norm": 1.6953125,
385
- "learning_rate": 9.877148934427037e-06,
386
- "loss": 1.2468,
387
- "step": 540
388
- },
389
- {
390
- "epoch": 1.9981834695731153,
391
- "grad_norm": 1.59375,
392
- "learning_rate": 9.86522435289912e-06,
393
- "loss": 1.2471,
394
- "step": 550
395
- },
396
- {
397
- "epoch": 2.034514078110808,
398
- "grad_norm": 1.9140625,
399
- "learning_rate": 9.85275541204318e-06,
400
- "loss": 1.1688,
401
- "step": 560
402
- },
403
- {
404
- "epoch": 2.0708446866485013,
405
- "grad_norm": 2.046875,
406
- "learning_rate": 9.839743506981783e-06,
407
- "loss": 1.1216,
408
- "step": 570
409
- },
410
- {
411
- "epoch": 2.1071752951861944,
412
- "grad_norm": 1.6328125,
413
- "learning_rate": 9.826190093588564e-06,
414
- "loss": 1.1333,
415
- "step": 580
416
- },
417
- {
418
- "epoch": 2.1435059037238875,
419
- "grad_norm": 1.6953125,
420
- "learning_rate": 9.812096688325354e-06,
421
- "loss": 1.1148,
422
- "step": 590
423
- },
424
- {
425
- "epoch": 2.1798365122615806,
426
- "grad_norm": 1.796875,
427
- "learning_rate": 9.797464868072489e-06,
428
- "loss": 1.1331,
429
- "step": 600
430
- },
431
- {
432
- "epoch": 2.2161671207992732,
433
- "grad_norm": 1.890625,
434
- "learning_rate": 9.78229626995238e-06,
435
- "loss": 1.1931,
436
- "step": 610
437
- },
438
- {
439
- "epoch": 2.2524977293369663,
440
- "grad_norm": 1.578125,
441
- "learning_rate": 9.766592591146353e-06,
442
- "loss": 1.146,
443
- "step": 620
444
- },
445
- {
446
- "epoch": 2.2888283378746594,
447
- "grad_norm": 1.7890625,
448
- "learning_rate": 9.750355588704728e-06,
449
- "loss": 1.1397,
450
- "step": 630
451
- },
452
- {
453
- "epoch": 2.3251589464123525,
454
- "grad_norm": 1.8359375,
455
- "learning_rate": 9.733587079350254e-06,
456
- "loss": 1.0941,
457
- "step": 640
458
- },
459
- {
460
- "epoch": 2.3614895549500456,
461
- "grad_norm": 1.7265625,
462
- "learning_rate": 9.716288939274818e-06,
463
- "loss": 1.101,
464
- "step": 650
465
- },
466
- {
467
- "epoch": 2.3978201634877383,
468
- "grad_norm": 2.703125,
469
- "learning_rate": 9.698463103929542e-06,
470
- "loss": 1.0632,
471
- "step": 660
472
- },
473
- {
474
- "epoch": 2.4341507720254314,
475
- "grad_norm": 2.5625,
476
- "learning_rate": 9.680111567808212e-06,
477
- "loss": 1.1315,
478
- "step": 670
479
- },
480
- {
481
- "epoch": 2.4704813805631245,
482
- "grad_norm": 1.890625,
483
- "learning_rate": 9.66123638422413e-06,
484
- "loss": 1.0989,
485
- "step": 680
486
- },
487
- {
488
- "epoch": 2.5068119891008176,
489
- "grad_norm": 1.8828125,
490
- "learning_rate": 9.641839665080363e-06,
491
- "loss": 1.0829,
492
- "step": 690
493
- },
494
- {
495
- "epoch": 2.5431425976385107,
496
- "grad_norm": 1.9140625,
497
- "learning_rate": 9.621923580633462e-06,
498
- "loss": 1.1155,
499
- "step": 700
500
- },
501
- {
502
- "epoch": 2.5794732061762033,
503
- "grad_norm": 1.5546875,
504
- "learning_rate": 9.601490359250616e-06,
505
- "loss": 1.1248,
506
- "step": 710
507
- },
508
- {
509
- "epoch": 2.6158038147138964,
510
- "grad_norm": 1.7890625,
511
- "learning_rate": 9.580542287160348e-06,
512
- "loss": 1.1391,
513
- "step": 720
514
- },
515
- {
516
- "epoch": 2.6521344232515895,
517
- "grad_norm": 1.734375,
518
- "learning_rate": 9.559081708196696e-06,
519
- "loss": 1.0538,
520
- "step": 730
521
- },
522
- {
523
- "epoch": 2.688465031789282,
524
- "grad_norm": 1.953125,
525
- "learning_rate": 9.537111023536973e-06,
526
- "loss": 1.055,
527
- "step": 740
528
- },
529
- {
530
- "epoch": 2.7247956403269757,
531
- "grad_norm": 1.6328125,
532
- "learning_rate": 9.514632691433108e-06,
533
- "loss": 1.1769,
534
- "step": 750
535
- },
536
- {
537
- "epoch": 2.7611262488646684,
538
- "grad_norm": 2.609375,
539
- "learning_rate": 9.491649226936586e-06,
540
- "loss": 1.0949,
541
- "step": 760
542
- },
543
- {
544
- "epoch": 2.7974568574023615,
545
- "grad_norm": 1.9609375,
546
- "learning_rate": 9.468163201617063e-06,
547
- "loss": 1.1308,
548
- "step": 770
549
- },
550
- {
551
- "epoch": 2.8337874659400546,
552
- "grad_norm": 1.953125,
553
- "learning_rate": 9.444177243274619e-06,
554
- "loss": 1.1283,
555
- "step": 780
556
- },
557
- {
558
- "epoch": 2.8701180744777472,
559
- "grad_norm": 1.703125,
560
- "learning_rate": 9.419694035645753e-06,
561
- "loss": 1.0907,
562
- "step": 790
563
- },
564
- {
565
- "epoch": 2.9064486830154403,
566
- "grad_norm": 1.5,
567
- "learning_rate": 9.394716318103098e-06,
568
- "loss": 1.1043,
569
- "step": 800
570
- },
571
- {
572
- "epoch": 2.9427792915531334,
573
- "grad_norm": 2.3125,
574
- "learning_rate": 9.369246885348926e-06,
575
- "loss": 1.0676,
576
- "step": 810
577
- },
578
- {
579
- "epoch": 2.9791099000908265,
580
- "grad_norm": 1.65625,
581
- "learning_rate": 9.343288587102444e-06,
582
- "loss": 1.0642,
583
- "step": 820
584
- },
585
- {
586
- "epoch": 3.0154405086285196,
587
- "grad_norm": 1.5234375,
588
- "learning_rate": 9.316844327780955e-06,
589
- "loss": 1.0437,
590
- "step": 830
591
- },
592
- {
593
- "epoch": 3.0517711171662127,
594
- "grad_norm": 1.6953125,
595
- "learning_rate": 9.289917066174887e-06,
596
- "loss": 0.9874,
597
- "step": 840
598
- },
599
- {
600
- "epoch": 3.0881017257039054,
601
- "grad_norm": 1.578125,
602
- "learning_rate": 9.262509815116732e-06,
603
- "loss": 0.9582,
604
- "step": 850
605
- },
606
- {
607
- "epoch": 3.1244323342415985,
608
- "grad_norm": 1.8046875,
609
- "learning_rate": 9.234625641143962e-06,
610
- "loss": 0.9471,
611
- "step": 860
612
- },
613
- {
614
- "epoch": 3.1607629427792916,
615
- "grad_norm": 2.015625,
616
- "learning_rate": 9.206267664155906e-06,
617
- "loss": 0.9424,
618
- "step": 870
619
- },
620
- {
621
- "epoch": 3.1970935513169847,
622
- "grad_norm": 1.625,
623
- "learning_rate": 9.177439057064684e-06,
624
- "loss": 0.972,
625
- "step": 880
626
- },
627
- {
628
- "epoch": 3.2334241598546773,
629
- "grad_norm": 1.4921875,
630
- "learning_rate": 9.148143045440181e-06,
631
- "loss": 0.9729,
632
- "step": 890
633
- },
634
- {
635
- "epoch": 3.2697547683923704,
636
- "grad_norm": 1.8203125,
637
- "learning_rate": 9.118382907149164e-06,
638
- "loss": 0.9558,
639
- "step": 900
640
- },
641
- {
642
- "epoch": 3.3060853769300635,
643
- "grad_norm": 1.453125,
644
- "learning_rate": 9.088161971988517e-06,
645
- "loss": 0.9518,
646
- "step": 910
647
- },
648
- {
649
- "epoch": 3.3424159854677566,
650
- "grad_norm": 1.3671875,
651
- "learning_rate": 9.057483621312671e-06,
652
- "loss": 0.9694,
653
- "step": 920
654
- },
655
- {
656
- "epoch": 3.3787465940054497,
657
- "grad_norm": 1.15625,
658
- "learning_rate": 9.026351287655294e-06,
659
- "loss": 0.9457,
660
- "step": 930
661
- },
662
- {
663
- "epoch": 3.4150772025431424,
664
- "grad_norm": 1.3359375,
665
- "learning_rate": 8.994768454345207e-06,
666
- "loss": 0.9585,
667
- "step": 940
668
- },
669
- {
670
- "epoch": 3.4514078110808355,
671
- "grad_norm": 1.3203125,
672
- "learning_rate": 8.96273865511666e-06,
673
- "loss": 0.9479,
674
- "step": 950
675
- },
676
- {
677
- "epoch": 3.4877384196185286,
678
- "grad_norm": 1.3203125,
679
- "learning_rate": 8.930265473713939e-06,
680
- "loss": 0.9105,
681
- "step": 960
682
- },
683
- {
684
- "epoch": 3.5240690281562217,
685
- "grad_norm": 1.3828125,
686
- "learning_rate": 8.897352543490396e-06,
687
- "loss": 0.9795,
688
- "step": 970
689
- },
690
- {
691
- "epoch": 3.560399636693915,
692
- "grad_norm": 1.21875,
693
- "learning_rate": 8.864003547001916e-06,
694
- "loss": 0.9403,
695
- "step": 980
696
- },
697
- {
698
- "epoch": 3.5967302452316074,
699
- "grad_norm": 1.3203125,
700
- "learning_rate": 8.83022221559489e-06,
701
- "loss": 0.925,
702
- "step": 990
703
- },
704
- {
705
- "epoch": 3.6330608537693005,
706
- "grad_norm": 1.25,
707
- "learning_rate": 8.796012328988716e-06,
708
- "loss": 0.9225,
709
- "step": 1000
710
- },
711
- {
712
- "epoch": 3.6330608537693005,
713
- "eval_loss": 1.1292678117752075,
714
- "eval_runtime": 10.2075,
715
- "eval_samples_per_second": 24.002,
716
- "eval_steps_per_second": 24.002,
717
- "step": 1000
718
- },
719
- {
720
- "epoch": 3.6693914623069936,
721
- "grad_norm": 1.171875,
722
- "learning_rate": 8.7613777148529e-06,
723
- "loss": 0.9778,
724
- "step": 1010
725
- },
726
- {
727
- "epoch": 3.7057220708446867,
728
- "grad_norm": 1.234375,
729
- "learning_rate": 8.726322248378775e-06,
730
- "loss": 0.8997,
731
- "step": 1020
732
- },
733
- {
734
- "epoch": 3.74205267938238,
735
- "grad_norm": 1.1328125,
736
- "learning_rate": 8.690849851845933e-06,
737
- "loss": 0.9874,
738
- "step": 1030
739
- },
740
- {
741
- "epoch": 3.7783832879200725,
742
- "grad_norm": 1.453125,
743
- "learning_rate": 8.65496449418336e-06,
744
- "loss": 0.9101,
745
- "step": 1040
746
- },
747
- {
748
- "epoch": 3.8147138964577656,
749
- "grad_norm": 1.3515625,
750
- "learning_rate": 8.61867019052535e-06,
751
- "loss": 0.9101,
752
- "step": 1050
753
- },
754
- {
755
- "epoch": 3.8510445049954587,
756
- "grad_norm": 1.25,
757
- "learning_rate": 8.581971001762287e-06,
758
- "loss": 0.9549,
759
- "step": 1060
760
- },
761
- {
762
- "epoch": 3.887375113533152,
763
- "grad_norm": 1.375,
764
- "learning_rate": 8.54487103408625e-06,
765
- "loss": 0.9486,
766
- "step": 1070
767
- },
768
- {
769
- "epoch": 3.923705722070845,
770
- "grad_norm": 1.1953125,
771
- "learning_rate": 8.507374438531606e-06,
772
- "loss": 0.9476,
773
- "step": 1080
774
- },
775
- {
776
- "epoch": 3.9600363306085375,
777
- "grad_norm": 1.2421875,
778
- "learning_rate": 8.469485410510545e-06,
779
- "loss": 0.9283,
780
- "step": 1090
781
- },
782
- {
783
- "epoch": 3.9963669391462306,
784
- "grad_norm": 1.2109375,
785
- "learning_rate": 8.43120818934367e-06,
786
- "loss": 0.9616,
787
- "step": 1100
788
- },
789
- {
790
- "epoch": 4.032697547683924,
791
- "grad_norm": 1.5,
792
- "learning_rate": 8.392547057785662e-06,
793
- "loss": 0.8816,
794
- "step": 1110
795
- },
796
- {
797
- "epoch": 4.069028156221616,
798
- "grad_norm": 1.0546875,
799
- "learning_rate": 8.353506341546106e-06,
800
- "loss": 0.8704,
801
- "step": 1120
802
- },
803
- {
804
- "epoch": 4.10535876475931,
805
- "grad_norm": 1.0859375,
806
- "learning_rate": 8.314090408805481e-06,
807
- "loss": 0.8515,
808
- "step": 1130
809
- },
810
- {
811
- "epoch": 4.141689373297003,
812
- "grad_norm": 1.125,
813
- "learning_rate": 8.274303669726427e-06,
814
- "loss": 0.7986,
815
- "step": 1140
816
- },
817
- {
818
- "epoch": 4.178019981834696,
819
- "grad_norm": 0.9765625,
820
- "learning_rate": 8.234150575960288e-06,
821
- "loss": 0.859,
822
- "step": 1150
823
- },
824
- {
825
- "epoch": 4.214350590372389,
826
- "grad_norm": 1.0625,
827
- "learning_rate": 8.193635620149041e-06,
828
- "loss": 0.8537,
829
- "step": 1160
830
- },
831
- {
832
- "epoch": 4.2506811989100814,
833
- "grad_norm": 1.0625,
834
- "learning_rate": 8.152763335422612e-06,
835
- "loss": 0.8513,
836
- "step": 1170
837
- },
838
- {
839
- "epoch": 4.287011807447775,
840
- "grad_norm": 1.0234375,
841
- "learning_rate": 8.111538294891684e-06,
842
- "loss": 0.8318,
843
- "step": 1180
844
- },
845
- {
846
- "epoch": 4.323342415985468,
847
- "grad_norm": 1.03125,
848
- "learning_rate": 8.06996511113601e-06,
849
- "loss": 0.846,
850
- "step": 1190
851
- },
852
- {
853
- "epoch": 4.359673024523161,
854
- "grad_norm": 1.0,
855
- "learning_rate": 8.028048435688333e-06,
856
- "loss": 0.8667,
857
- "step": 1200
858
- },
859
- {
860
- "epoch": 4.396003633060854,
861
- "grad_norm": 0.9921875,
862
- "learning_rate": 7.985792958513932e-06,
863
- "loss": 0.8715,
864
- "step": 1210
865
- },
866
- {
867
- "epoch": 4.4323342415985465,
868
- "grad_norm": 0.9296875,
869
- "learning_rate": 7.943203407485864e-06,
870
- "loss": 0.848,
871
- "step": 1220
872
- },
873
- {
874
- "epoch": 4.46866485013624,
875
- "grad_norm": 0.94921875,
876
- "learning_rate": 7.900284547855992e-06,
877
- "loss": 0.8859,
878
- "step": 1230
879
- },
880
- {
881
- "epoch": 4.504995458673933,
882
- "grad_norm": 0.98046875,
883
- "learning_rate": 7.857041181721788e-06,
884
- "loss": 0.8575,
885
- "step": 1240
886
- },
887
- {
888
- "epoch": 4.541326067211626,
889
- "grad_norm": 0.94140625,
890
- "learning_rate": 7.813478147489052e-06,
891
- "loss": 0.8496,
892
- "step": 1250
893
- },
894
- {
895
- "epoch": 4.577656675749319,
896
- "grad_norm": 0.92578125,
897
- "learning_rate": 7.769600319330553e-06,
898
- "loss": 0.8166,
899
- "step": 1260
900
- },
901
- {
902
- "epoch": 4.6139872842870115,
903
- "grad_norm": 1.15625,
904
- "learning_rate": 7.725412606640658e-06,
905
- "loss": 0.8191,
906
- "step": 1270
907
- },
908
- {
909
- "epoch": 4.650317892824705,
910
- "grad_norm": 1.046875,
911
- "learning_rate": 7.680919953486047e-06,
912
- "loss": 0.833,
913
- "step": 1280
914
- },
915
- {
916
- "epoch": 4.686648501362398,
917
- "grad_norm": 1.0390625,
918
- "learning_rate": 7.636127338052513e-06,
919
- "loss": 0.839,
920
- "step": 1290
921
- },
922
- {
923
- "epoch": 4.722979109900091,
924
- "grad_norm": 1.2421875,
925
- "learning_rate": 7.5910397720879785e-06,
926
- "loss": 0.807,
927
- "step": 1300
928
- },
929
- {
930
- "epoch": 4.759309718437784,
931
- "grad_norm": 0.984375,
932
- "learning_rate": 7.545662300341736e-06,
933
- "loss": 0.8215,
934
- "step": 1310
935
- },
936
- {
937
- "epoch": 4.795640326975477,
938
- "grad_norm": 1.03125,
939
- "learning_rate": 7.500000000000001e-06,
940
- "loss": 0.8584,
941
- "step": 1320
942
- },
943
- {
944
- "epoch": 4.83197093551317,
945
- "grad_norm": 1.0546875,
946
- "learning_rate": 7.454057980117842e-06,
947
- "loss": 0.7955,
948
- "step": 1330
949
- },
950
- {
951
- "epoch": 4.868301544050863,
952
- "grad_norm": 1.140625,
953
- "learning_rate": 7.407841381047533e-06,
954
- "loss": 0.8482,
955
- "step": 1340
956
- },
957
- {
958
- "epoch": 4.904632152588556,
959
- "grad_norm": 1.234375,
960
- "learning_rate": 7.361355373863415e-06,
961
- "loss": 0.8457,
962
- "step": 1350
963
- },
964
- {
965
- "epoch": 4.940962761126249,
966
- "grad_norm": 0.91796875,
967
- "learning_rate": 7.314605159783313e-06,
968
- "loss": 0.8705,
969
- "step": 1360
970
- },
971
- {
972
- "epoch": 4.977293369663942,
973
- "grad_norm": 1.0234375,
974
- "learning_rate": 7.2675959695865896e-06,
975
- "loss": 0.829,
976
- "step": 1370
977
- },
978
- {
979
- "epoch": 5.013623978201635,
980
- "grad_norm": 0.9375,
981
- "learning_rate": 7.2203330630288714e-06,
982
- "loss": 0.8391,
983
- "step": 1380
984
- },
985
- {
986
- "epoch": 5.049954586739328,
987
- "grad_norm": 0.9921875,
988
- "learning_rate": 7.172821728253563e-06,
989
- "loss": 0.7915,
990
- "step": 1390
991
- },
992
- {
993
- "epoch": 5.0862851952770205,
994
- "grad_norm": 1.1171875,
995
- "learning_rate": 7.1250672812001505e-06,
996
- "loss": 0.8199,
997
- "step": 1400
998
- },
999
- {
1000
- "epoch": 5.122615803814714,
1001
- "grad_norm": 1.0,
1002
- "learning_rate": 7.0770750650094335e-06,
1003
- "loss": 0.8065,
1004
- "step": 1410
1005
- },
1006
- {
1007
- "epoch": 5.158946412352407,
1008
- "grad_norm": 1.2578125,
1009
- "learning_rate": 7.02885044942567e-06,
1010
- "loss": 0.8098,
1011
- "step": 1420
1012
- },
1013
- {
1014
- "epoch": 5.1952770208901,
1015
- "grad_norm": 1.0625,
1016
- "learning_rate": 6.980398830195785e-06,
1017
- "loss": 0.7675,
1018
- "step": 1430
1019
- },
1020
- {
1021
- "epoch": 5.231607629427793,
1022
- "grad_norm": 1.125,
1023
- "learning_rate": 6.931725628465643e-06,
1024
- "loss": 0.7747,
1025
- "step": 1440
1026
- },
1027
- {
1028
- "epoch": 5.2679382379654855,
1029
- "grad_norm": 1.375,
1030
- "learning_rate": 6.882836290173493e-06,
1031
- "loss": 0.7623,
1032
- "step": 1450
1033
- },
1034
- {
1035
- "epoch": 5.304268846503179,
1036
- "grad_norm": 1.2109375,
1037
- "learning_rate": 6.833736285440632e-06,
1038
- "loss": 0.8113,
1039
- "step": 1460
1040
- },
1041
- {
1042
- "epoch": 5.340599455040872,
1043
- "grad_norm": 1.3359375,
1044
- "learning_rate": 6.78443110795936e-06,
1045
- "loss": 0.8411,
1046
- "step": 1470
1047
- },
1048
- {
1049
- "epoch": 5.376930063578565,
1050
- "grad_norm": 1.1171875,
1051
- "learning_rate": 6.734926274378313e-06,
1052
- "loss": 0.7592,
1053
- "step": 1480
1054
- },
1055
- {
1056
- "epoch": 5.413260672116258,
1057
- "grad_norm": 1.234375,
1058
- "learning_rate": 6.685227323685209e-06,
1059
- "loss": 0.7776,
1060
- "step": 1490
1061
- },
1062
- {
1063
- "epoch": 5.449591280653951,
1064
- "grad_norm": 1.46875,
1065
- "learning_rate": 6.635339816587109e-06,
1066
- "loss": 0.7727,
1067
- "step": 1500
1068
- },
1069
- {
1070
- "epoch": 5.485921889191644,
1071
- "grad_norm": 1.40625,
1072
- "learning_rate": 6.5852693348882345e-06,
1073
- "loss": 0.7539,
1074
- "step": 1510
1075
- },
1076
- {
1077
- "epoch": 5.522252497729337,
1078
- "grad_norm": 1.4609375,
1079
- "learning_rate": 6.535021480865439e-06,
1080
- "loss": 0.8292,
1081
- "step": 1520
1082
- },
1083
- {
1084
- "epoch": 5.55858310626703,
1085
- "grad_norm": 1.8359375,
1086
- "learning_rate": 6.484601876641375e-06,
1087
- "loss": 0.7463,
1088
- "step": 1530
1089
- },
1090
- {
1091
- "epoch": 5.594913714804723,
1092
- "grad_norm": 1.3515625,
1093
- "learning_rate": 6.434016163555452e-06,
1094
- "loss": 0.7744,
1095
- "step": 1540
1096
- },
1097
- {
1098
- "epoch": 5.631244323342416,
1099
- "grad_norm": 2.578125,
1100
- "learning_rate": 6.383270001532636e-06,
1101
- "loss": 0.7868,
1102
- "step": 1550
1103
- },
1104
- {
1105
- "epoch": 5.667574931880109,
1106
- "grad_norm": 2.65625,
1107
- "learning_rate": 6.332369068450175e-06,
1108
- "loss": 0.7485,
1109
- "step": 1560
1110
- },
1111
- {
1112
- "epoch": 5.703905540417802,
1113
- "grad_norm": 2.34375,
1114
- "learning_rate": 6.2813190595023135e-06,
1115
- "loss": 0.7783,
1116
- "step": 1570
1117
- },
1118
- {
1119
- "epoch": 5.740236148955495,
1120
- "grad_norm": 2.40625,
1121
- "learning_rate": 6.230125686563068e-06,
1122
- "loss": 0.7769,
1123
- "step": 1580
1124
- },
1125
- {
1126
- "epoch": 5.776566757493188,
1127
- "grad_norm": 2.390625,
1128
- "learning_rate": 6.178794677547138e-06,
1129
- "loss": 0.7977,
1130
- "step": 1590
1131
- },
1132
- {
1133
- "epoch": 5.812897366030881,
1134
- "grad_norm": 3.65625,
1135
- "learning_rate": 6.127331775769023e-06,
1136
- "loss": 0.8101,
1137
- "step": 1600
1138
- },
1139
- {
1140
- "epoch": 5.849227974568574,
1141
- "grad_norm": 5.4375,
1142
- "learning_rate": 6.07574273930042e-06,
1143
- "loss": 0.8397,
1144
- "step": 1610
1145
- },
1146
- {
1147
- "epoch": 5.885558583106267,
1148
- "grad_norm": 4.78125,
1149
- "learning_rate": 6.024033340325954e-06,
1150
- "loss": 0.8191,
1151
- "step": 1620
1152
- },
1153
- {
1154
- "epoch": 5.9218891916439595,
1155
- "grad_norm": 4.625,
1156
- "learning_rate": 5.972209364497355e-06,
1157
- "loss": 0.7843,
1158
- "step": 1630
1159
- },
1160
- {
1161
- "epoch": 5.958219800181653,
1162
- "grad_norm": 5.4375,
1163
- "learning_rate": 5.920276610286102e-06,
1164
- "loss": 0.8099,
1165
- "step": 1640
1166
- },
1167
- {
1168
- "epoch": 5.994550408719346,
1169
- "grad_norm": 10.875,
1170
- "learning_rate": 5.8682408883346535e-06,
1171
- "loss": 0.8025,
1172
- "step": 1650
1173
- },
1174
- {
1175
- "epoch": 6.030881017257039,
1176
- "grad_norm": 10.25,
1177
- "learning_rate": 5.816108020806297e-06,
1178
- "loss": 0.7426,
1179
- "step": 1660
1180
- },
1181
- {
1182
- "epoch": 6.067211625794732,
1183
- "grad_norm": 9.8125,
1184
- "learning_rate": 5.763883840733736e-06,
1185
- "loss": 0.7439,
1186
- "step": 1670
1187
- },
1188
- {
1189
- "epoch": 6.1035422343324255,
1190
- "grad_norm": 8.4375,
1191
- "learning_rate": 5.711574191366427e-06,
1192
- "loss": 0.7617,
1193
- "step": 1680
1194
- },
1195
- {
1196
- "epoch": 6.139872842870118,
1197
- "grad_norm": 8.4375,
1198
- "learning_rate": 5.659184925516802e-06,
1199
- "loss": 0.7333,
1200
- "step": 1690
1201
- },
1202
- {
1203
- "epoch": 6.176203451407811,
1204
- "grad_norm": 2.453125,
1205
- "learning_rate": 5.60672190490541e-06,
1206
- "loss": 0.7747,
1207
- "step": 1700
1208
- },
1209
- {
1210
- "epoch": 6.212534059945504,
1211
- "grad_norm": 2.5625,
1212
- "learning_rate": 5.5541909995050554e-06,
1213
- "loss": 0.7469,
1214
- "step": 1710
1215
- },
1216
- {
1217
- "epoch": 6.248864668483197,
1218
- "grad_norm": 3.015625,
1219
- "learning_rate": 5.5015980868840254e-06,
1220
- "loss": 0.7507,
1221
- "step": 1720
1222
- },
1223
- {
1224
- "epoch": 6.28519527702089,
1225
- "grad_norm": 2.546875,
1226
- "learning_rate": 5.448949051548459e-06,
1227
- "loss": 0.7774,
1228
- "step": 1730
1229
- },
1230
- {
1231
- "epoch": 6.321525885558583,
1232
- "grad_norm": 2.921875,
1233
- "learning_rate": 5.396249784283943e-06,
1234
- "loss": 0.773,
1235
- "step": 1740
1236
- },
1237
- {
1238
- "epoch": 6.357856494096276,
1239
- "grad_norm": 2.21875,
1240
- "learning_rate": 5.343506181496405e-06,
1241
- "loss": 0.7333,
1242
- "step": 1750
1243
- },
1244
- {
1245
- "epoch": 6.394187102633969,
1246
- "grad_norm": 2.28125,
1247
- "learning_rate": 5.290724144552379e-06,
1248
- "loss": 0.7352,
1249
- "step": 1760
1250
- },
1251
- {
1252
- "epoch": 6.430517711171662,
1253
- "grad_norm": 2.3125,
1254
- "learning_rate": 5.237909579118713e-06,
1255
- "loss": 0.68,
1256
- "step": 1770
1257
- },
1258
- {
1259
- "epoch": 6.466848319709355,
1260
- "grad_norm": 2.296875,
1261
- "learning_rate": 5.185068394501791e-06,
1262
- "loss": 0.7279,
1263
- "step": 1780
1264
- },
1265
- {
1266
- "epoch": 6.503178928247048,
1267
- "grad_norm": 2.53125,
1268
- "learning_rate": 5.132206502986368e-06,
1269
- "loss": 0.7547,
1270
- "step": 1790
1271
- },
1272
- {
1273
- "epoch": 6.539509536784741,
1274
- "grad_norm": 2.421875,
1275
- "learning_rate": 5.07932981917404e-06,
1276
- "loss": 0.7015,
1277
- "step": 1800
1278
- },
1279
- {
1280
- "epoch": 6.575840145322434,
1281
- "grad_norm": 2.4375,
1282
- "learning_rate": 5.026444259321489e-06,
1283
- "loss": 0.7614,
1284
- "step": 1810
1285
- },
1286
- {
1287
- "epoch": 6.612170753860127,
1288
- "grad_norm": 2.171875,
1289
- "learning_rate": 4.973555740678512e-06,
1290
- "loss": 0.7953,
1291
- "step": 1820
1292
- },
1293
- {
1294
- "epoch": 6.64850136239782,
1295
- "grad_norm": 2.40625,
1296
- "learning_rate": 4.9206701808259605e-06,
1297
- "loss": 0.736,
1298
- "step": 1830
1299
- },
1300
- {
1301
- "epoch": 6.684831970935513,
1302
- "grad_norm": 2.28125,
1303
- "learning_rate": 4.867793497013634e-06,
1304
- "loss": 0.7502,
1305
- "step": 1840
1306
- },
1307
- {
1308
- "epoch": 6.721162579473206,
1309
- "grad_norm": 2.390625,
1310
- "learning_rate": 4.81493160549821e-06,
1311
- "loss": 0.7496,
1312
- "step": 1850
1313
- },
1314
- {
1315
- "epoch": 6.7574931880108995,
1316
- "grad_norm": 2.359375,
1317
- "learning_rate": 4.762090420881289e-06,
1318
- "loss": 0.7888,
1319
- "step": 1860
1320
- },
1321
- {
1322
- "epoch": 6.793823796548592,
1323
- "grad_norm": 2.078125,
1324
- "learning_rate": 4.7092758554476215e-06,
1325
- "loss": 0.7238,
1326
- "step": 1870
1327
- },
1328
- {
1329
- "epoch": 6.830154405086285,
1330
- "grad_norm": 2.1875,
1331
- "learning_rate": 4.6564938185035954e-06,
1332
- "loss": 0.7484,
1333
- "step": 1880
1334
- },
1335
- {
1336
- "epoch": 6.866485013623978,
1337
- "grad_norm": 2.0625,
1338
- "learning_rate": 4.603750215716057e-06,
1339
- "loss": 0.7835,
1340
- "step": 1890
1341
- },
1342
- {
1343
- "epoch": 6.902815622161671,
1344
- "grad_norm": 2.53125,
1345
- "learning_rate": 4.551050948451542e-06,
1346
- "loss": 0.6878,
1347
- "step": 1900
1348
- },
1349
- {
1350
- "epoch": 6.9391462306993645,
1351
- "grad_norm": 1.953125,
1352
- "learning_rate": 4.498401913115975e-06,
1353
- "loss": 0.7528,
1354
- "step": 1910
1355
- },
1356
- {
1357
- "epoch": 6.975476839237057,
1358
- "grad_norm": 2.0625,
1359
- "learning_rate": 4.445809000494945e-06,
1360
- "loss": 0.7486,
1361
- "step": 1920
1362
- },
1363
- {
1364
- "epoch": 7.01180744777475,
1365
- "grad_norm": 2.15625,
1366
- "learning_rate": 4.393278095094591e-06,
1367
- "loss": 0.7948,
1368
- "step": 1930
1369
- },
1370
- {
1371
- "epoch": 7.048138056312443,
1372
- "grad_norm": 2.03125,
1373
- "learning_rate": 4.340815074483199e-06,
1374
- "loss": 0.6872,
1375
- "step": 1940
1376
- },
1377
- {
1378
- "epoch": 7.084468664850136,
1379
- "grad_norm": 1.953125,
1380
- "learning_rate": 4.2884258086335755e-06,
1381
- "loss": 0.7289,
1382
- "step": 1950
1383
- },
1384
- {
1385
- "epoch": 7.12079927338783,
1386
- "grad_norm": 2.15625,
1387
- "learning_rate": 4.2361161592662655e-06,
1388
- "loss": 0.612,
1389
- "step": 1960
1390
- },
1391
- {
1392
- "epoch": 7.157129881925522,
1393
- "grad_norm": 2.1875,
1394
- "learning_rate": 4.183891979193703e-06,
1395
- "loss": 0.7135,
1396
- "step": 1970
1397
- },
1398
- {
1399
- "epoch": 7.193460490463215,
1400
- "grad_norm": 2.28125,
1401
- "learning_rate": 4.131759111665349e-06,
1402
- "loss": 0.6932,
1403
- "step": 1980
1404
- },
1405
- {
1406
- "epoch": 7.229791099000908,
1407
- "grad_norm": 1.9453125,
1408
- "learning_rate": 4.079723389713899e-06,
1409
- "loss": 0.705,
1410
- "step": 1990
1411
- },
1412
- {
1413
- "epoch": 7.266121707538601,
1414
- "grad_norm": 2.09375,
1415
- "learning_rate": 4.027790635502646e-06,
1416
- "loss": 0.7145,
1417
- "step": 2000
1418
- },
1419
- {
1420
- "epoch": 7.266121707538601,
1421
- "eval_loss": 1.1055254936218262,
1422
- "eval_runtime": 10.2576,
1423
- "eval_samples_per_second": 23.885,
1424
- "eval_steps_per_second": 23.885,
1425
- "step": 2000
1426
- },
1427
- {
1428
- "epoch": 7.302452316076295,
1429
- "grad_norm": 1.921875,
1430
- "learning_rate": 3.975966659674048e-06,
1431
- "loss": 0.6724,
1432
- "step": 2010
1433
- },
1434
- {
1435
- "epoch": 7.338782924613987,
1436
- "grad_norm": 2.3125,
1437
- "learning_rate": 3.924257260699583e-06,
1438
- "loss": 0.6468,
1439
- "step": 2020
1440
- },
1441
- {
1442
- "epoch": 7.37511353315168,
1443
- "grad_norm": 2.140625,
1444
- "learning_rate": 3.872668224230979e-06,
1445
- "loss": 0.6792,
1446
- "step": 2030
1447
- },
1448
- {
1449
- "epoch": 7.4114441416893735,
1450
- "grad_norm": 2.265625,
1451
- "learning_rate": 3.821205322452863e-06,
1452
- "loss": 0.723,
1453
- "step": 2040
1454
- },
1455
- {
1456
- "epoch": 7.447774750227066,
1457
- "grad_norm": 1.984375,
1458
- "learning_rate": 3.769874313436933e-06,
1459
- "loss": 0.6608,
1460
- "step": 2050
1461
- },
1462
- {
1463
- "epoch": 7.48410535876476,
1464
- "grad_norm": 2.21875,
1465
- "learning_rate": 3.7186809404976877e-06,
1466
- "loss": 0.6679,
1467
- "step": 2060
1468
- },
1469
- {
1470
- "epoch": 7.520435967302452,
1471
- "grad_norm": 2.078125,
1472
- "learning_rate": 3.667630931549826e-06,
1473
- "loss": 0.6944,
1474
- "step": 2070
1475
- },
1476
- {
1477
- "epoch": 7.556766575840145,
1478
- "grad_norm": 2.15625,
1479
- "learning_rate": 3.6167299984673655e-06,
1480
- "loss": 0.7172,
1481
- "step": 2080
1482
- },
1483
- {
1484
- "epoch": 7.5930971843778385,
1485
- "grad_norm": 2.265625,
1486
- "learning_rate": 3.5659838364445505e-06,
1487
- "loss": 0.7029,
1488
- "step": 2090
1489
- },
1490
- {
1491
- "epoch": 7.629427792915531,
1492
- "grad_norm": 2.078125,
1493
- "learning_rate": 3.5153981233586277e-06,
1494
- "loss": 0.6549,
1495
- "step": 2100
1496
- },
1497
- {
1498
- "epoch": 7.665758401453225,
1499
- "grad_norm": 2.1875,
1500
- "learning_rate": 3.4649785191345613e-06,
1501
- "loss": 0.7342,
1502
- "step": 2110
1503
- },
1504
- {
1505
- "epoch": 7.702089009990917,
1506
- "grad_norm": 2.09375,
1507
- "learning_rate": 3.4147306651117663e-06,
1508
- "loss": 0.6246,
1509
- "step": 2120
1510
- },
1511
- {
1512
- "epoch": 7.73841961852861,
1513
- "grad_norm": 1.90625,
1514
- "learning_rate": 3.3646601834128924e-06,
1515
- "loss": 0.6587,
1516
- "step": 2130
1517
- },
1518
- {
1519
- "epoch": 7.774750227066304,
1520
- "grad_norm": 1.953125,
1521
- "learning_rate": 3.3147726763147913e-06,
1522
- "loss": 0.6682,
1523
- "step": 2140
1524
- },
1525
- {
1526
- "epoch": 7.811080835603996,
1527
- "grad_norm": 2.421875,
1528
- "learning_rate": 3.2650737256216885e-06,
1529
- "loss": 0.6854,
1530
- "step": 2150
1531
- },
1532
- {
1533
- "epoch": 7.84741144414169,
1534
- "grad_norm": 2.1875,
1535
- "learning_rate": 3.2155688920406415e-06,
1536
- "loss": 0.6908,
1537
- "step": 2160
1538
- },
1539
- {
1540
- "epoch": 7.883742052679382,
1541
- "grad_norm": 2.046875,
1542
- "learning_rate": 3.16626371455937e-06,
1543
- "loss": 0.6875,
1544
- "step": 2170
1545
- },
1546
- {
1547
- "epoch": 7.920072661217075,
1548
- "grad_norm": 2.28125,
1549
- "learning_rate": 3.1171637098265063e-06,
1550
- "loss": 0.6732,
1551
- "step": 2180
1552
- },
1553
- {
1554
- "epoch": 7.956403269754769,
1555
- "grad_norm": 1.9765625,
1556
- "learning_rate": 3.0682743715343565e-06,
1557
- "loss": 0.742,
1558
- "step": 2190
1559
- },
1560
- {
1561
- "epoch": 7.992733878292461,
1562
- "grad_norm": 2.796875,
1563
- "learning_rate": 3.019601169804216e-06,
1564
- "loss": 0.6791,
1565
- "step": 2200
1566
- },
1567
- {
1568
- "epoch": 8.029064486830155,
1569
- "grad_norm": 1.796875,
1570
- "learning_rate": 2.9711495505743317e-06,
1571
- "loss": 0.654,
1572
- "step": 2210
1573
- },
1574
- {
1575
- "epoch": 8.065395095367847,
1576
- "grad_norm": 2.0,
1577
- "learning_rate": 2.9229249349905686e-06,
1578
- "loss": 0.705,
1579
- "step": 2220
1580
- },
1581
- {
1582
- "epoch": 8.10172570390554,
1583
- "grad_norm": 2.0,
1584
- "learning_rate": 2.8749327187998516e-06,
1585
- "loss": 0.6067,
1586
- "step": 2230
1587
- },
1588
- {
1589
- "epoch": 8.138056312443233,
1590
- "grad_norm": 2.03125,
1591
- "learning_rate": 2.8271782717464413e-06,
1592
- "loss": 0.6043,
1593
- "step": 2240
1594
- },
1595
- {
1596
- "epoch": 8.174386920980927,
1597
- "grad_norm": 1.6875,
1598
- "learning_rate": 2.7796669369711294e-06,
1599
- "loss": 0.6292,
1600
- "step": 2250
1601
- },
1602
- {
1603
- "epoch": 8.21071752951862,
1604
- "grad_norm": 2.0625,
1605
- "learning_rate": 2.7324040304134125e-06,
1606
- "loss": 0.6041,
1607
- "step": 2260
1608
- },
1609
- {
1610
- "epoch": 8.247048138056313,
1611
- "grad_norm": 2.296875,
1612
- "learning_rate": 2.685394840216688e-06,
1613
- "loss": 0.6222,
1614
- "step": 2270
1615
- },
1616
- {
1617
- "epoch": 8.283378746594005,
1618
- "grad_norm": 2.046875,
1619
- "learning_rate": 2.6386446261365874e-06,
1620
- "loss": 0.6586,
1621
- "step": 2280
1622
- },
1623
- {
1624
- "epoch": 8.319709355131698,
1625
- "grad_norm": 2.015625,
1626
- "learning_rate": 2.5921586189524694e-06,
1627
- "loss": 0.6612,
1628
- "step": 2290
1629
- },
1630
- {
1631
- "epoch": 8.356039963669392,
1632
- "grad_norm": 2.0,
1633
- "learning_rate": 2.5459420198821604e-06,
1634
- "loss": 0.6693,
1635
- "step": 2300
1636
- },
1637
- {
1638
- "epoch": 8.392370572207085,
1639
- "grad_norm": 2.125,
1640
- "learning_rate": 2.5000000000000015e-06,
1641
- "loss": 0.6244,
1642
- "step": 2310
1643
- },
1644
- {
1645
- "epoch": 8.428701180744778,
1646
- "grad_norm": 2.09375,
1647
- "learning_rate": 2.454337699658267e-06,
1648
- "loss": 0.6312,
1649
- "step": 2320
1650
- },
1651
- {
1652
- "epoch": 8.46503178928247,
1653
- "grad_norm": 2.09375,
1654
- "learning_rate": 2.4089602279120224e-06,
1655
- "loss": 0.6729,
1656
- "step": 2330
1657
- },
1658
- {
1659
- "epoch": 8.501362397820163,
1660
- "grad_norm": 1.8125,
1661
- "learning_rate": 2.363872661947488e-06,
1662
- "loss": 0.6276,
1663
- "step": 2340
1664
- },
1665
- {
1666
- "epoch": 8.537693006357856,
1667
- "grad_norm": 1.90625,
1668
- "learning_rate": 2.319080046513954e-06,
1669
- "loss": 0.6292,
1670
- "step": 2350
1671
- },
1672
- {
1673
- "epoch": 8.57402361489555,
1674
- "grad_norm": 2.125,
1675
- "learning_rate": 2.274587393359342e-06,
1676
- "loss": 0.6593,
1677
- "step": 2360
1678
- },
1679
- {
1680
- "epoch": 8.610354223433243,
1681
- "grad_norm": 1.875,
1682
- "learning_rate": 2.230399680669449e-06,
1683
- "loss": 0.6302,
1684
- "step": 2370
1685
- },
1686
- {
1687
- "epoch": 8.646684831970935,
1688
- "grad_norm": 1.796875,
1689
- "learning_rate": 2.1865218525109496e-06,
1690
- "loss": 0.6481,
1691
- "step": 2380
1692
- },
1693
- {
1694
- "epoch": 8.683015440508628,
1695
- "grad_norm": 2.15625,
1696
- "learning_rate": 2.1429588182782147e-06,
1697
- "loss": 0.6175,
1698
- "step": 2390
1699
- },
1700
- {
1701
- "epoch": 8.719346049046322,
1702
- "grad_norm": 1.953125,
1703
- "learning_rate": 2.09971545214401e-06,
1704
- "loss": 0.6144,
1705
- "step": 2400
1706
- },
1707
- {
1708
- "epoch": 8.755676657584015,
1709
- "grad_norm": 1.6875,
1710
- "learning_rate": 2.0567965925141366e-06,
1711
- "loss": 0.6572,
1712
- "step": 2410
1713
- },
1714
- {
1715
- "epoch": 8.792007266121708,
1716
- "grad_norm": 1.890625,
1717
- "learning_rate": 2.0142070414860704e-06,
1718
- "loss": 0.654,
1719
- "step": 2420
1720
- },
1721
- {
1722
- "epoch": 8.8283378746594,
1723
- "grad_norm": 1.8359375,
1724
- "learning_rate": 1.971951564311668e-06,
1725
- "loss": 0.621,
1726
- "step": 2430
1727
- },
1728
- {
1729
- "epoch": 8.864668483197093,
1730
- "grad_norm": 1.78125,
1731
- "learning_rate": 1.9300348888639915e-06,
1732
- "loss": 0.62,
1733
- "step": 2440
1734
- },
1735
- {
1736
- "epoch": 8.900999091734786,
1737
- "grad_norm": 1.6640625,
1738
- "learning_rate": 1.8884617051083183e-06,
1739
- "loss": 0.5858,
1740
- "step": 2450
1741
- },
1742
- {
1743
- "epoch": 8.93732970027248,
1744
- "grad_norm": 1.578125,
1745
- "learning_rate": 1.8472366645773892e-06,
1746
- "loss": 0.6467,
1747
- "step": 2460
1748
- },
1749
- {
1750
- "epoch": 8.973660308810173,
1751
- "grad_norm": 1.921875,
1752
- "learning_rate": 1.8063643798509594e-06,
1753
- "loss": 0.6138,
1754
- "step": 2470
1755
- },
1756
- {
1757
- "epoch": 9.009990917347865,
1758
- "grad_norm": 1.65625,
1759
- "learning_rate": 1.7658494240397127e-06,
1760
- "loss": 0.6388,
1761
- "step": 2480
1762
- },
1763
- {
1764
- "epoch": 9.046321525885558,
1765
- "grad_norm": 1.4453125,
1766
- "learning_rate": 1.7256963302735752e-06,
1767
- "loss": 0.5733,
1768
- "step": 2490
1769
- },
1770
- {
1771
- "epoch": 9.082652134423252,
1772
- "grad_norm": 1.421875,
1773
- "learning_rate": 1.68590959119452e-06,
1774
- "loss": 0.6066,
1775
- "step": 2500
1776
- },
1777
- {
1778
- "epoch": 9.118982742960945,
1779
- "grad_norm": 1.2890625,
1780
- "learning_rate": 1.646493658453896e-06,
1781
- "loss": 0.5849,
1782
- "step": 2510
1783
- },
1784
- {
1785
- "epoch": 9.155313351498638,
1786
- "grad_norm": 1.34375,
1787
- "learning_rate": 1.6074529422143398e-06,
1788
- "loss": 0.6136,
1789
- "step": 2520
1790
- },
1791
- {
1792
- "epoch": 9.19164396003633,
1793
- "grad_norm": 1.34375,
1794
- "learning_rate": 1.5687918106563326e-06,
1795
- "loss": 0.5891,
1796
- "step": 2530
1797
- },
1798
- {
1799
- "epoch": 9.227974568574023,
1800
- "grad_norm": 1.5234375,
1801
- "learning_rate": 1.5305145894894547e-06,
1802
- "loss": 0.6065,
1803
- "step": 2540
1804
- },
1805
- {
1806
- "epoch": 9.264305177111716,
1807
- "grad_norm": 1.4453125,
1808
- "learning_rate": 1.4926255614683931e-06,
1809
- "loss": 0.5854,
1810
- "step": 2550
1811
- },
1812
- {
1813
- "epoch": 9.30063578564941,
1814
- "grad_norm": 1.5703125,
1815
- "learning_rate": 1.4551289659137497e-06,
1816
- "loss": 0.6023,
1817
- "step": 2560
1818
- },
1819
- {
1820
- "epoch": 9.336966394187103,
1821
- "grad_norm": 1.4296875,
1822
- "learning_rate": 1.4180289982377138e-06,
1823
- "loss": 0.6052,
1824
- "step": 2570
1825
- },
1826
- {
1827
- "epoch": 9.373297002724795,
1828
- "grad_norm": 1.4453125,
1829
- "learning_rate": 1.3813298094746491e-06,
1830
- "loss": 0.5806,
1831
- "step": 2580
1832
- },
1833
- {
1834
- "epoch": 9.409627611262488,
1835
- "grad_norm": 1.421875,
1836
- "learning_rate": 1.345035505816642e-06,
1837
- "loss": 0.6058,
1838
- "step": 2590
1839
- },
1840
- {
1841
- "epoch": 9.44595821980018,
1842
- "grad_norm": 1.3671875,
1843
- "learning_rate": 1.3091501481540676e-06,
1844
- "loss": 0.6339,
1845
- "step": 2600
1846
- },
1847
- {
1848
- "epoch": 9.482288828337875,
1849
- "grad_norm": 1.2734375,
1850
- "learning_rate": 1.2736777516212267e-06,
1851
- "loss": 0.6058,
1852
- "step": 2610
1853
- },
1854
- {
1855
- "epoch": 9.518619436875568,
1856
- "grad_norm": 1.546875,
1857
- "learning_rate": 1.238622285147103e-06,
1858
- "loss": 0.6186,
1859
- "step": 2620
1860
- },
1861
- {
1862
- "epoch": 9.55495004541326,
1863
- "grad_norm": 1.3203125,
1864
- "learning_rate": 1.2039876710112847e-06,
1865
- "loss": 0.5913,
1866
- "step": 2630
1867
- },
1868
- {
1869
- "epoch": 9.591280653950953,
1870
- "grad_norm": 1.359375,
1871
- "learning_rate": 1.1697777844051105e-06,
1872
- "loss": 0.663,
1873
- "step": 2640
1874
- },
1875
- {
1876
- "epoch": 9.627611262488646,
1877
- "grad_norm": 1.2109375,
1878
- "learning_rate": 1.135996452998085e-06,
1879
- "loss": 0.6278,
1880
- "step": 2650
1881
- },
1882
- {
1883
- "epoch": 9.66394187102634,
1884
- "grad_norm": 1.390625,
1885
- "learning_rate": 1.1026474565096068e-06,
1886
- "loss": 0.6074,
1887
- "step": 2660
1888
- },
1889
- {
1890
- "epoch": 9.700272479564033,
1891
- "grad_norm": 1.4609375,
1892
- "learning_rate": 1.0697345262860638e-06,
1893
- "loss": 0.6545,
1894
- "step": 2670
1895
- },
1896
- {
1897
- "epoch": 9.736603088101726,
1898
- "grad_norm": 1.421875,
1899
- "learning_rate": 1.0372613448833429e-06,
1900
- "loss": 0.6141,
1901
- "step": 2680
1902
- },
1903
- {
1904
- "epoch": 9.772933696639418,
1905
- "grad_norm": 1.3125,
1906
- "learning_rate": 1.0052315456547934e-06,
1907
- "loss": 0.5699,
1908
- "step": 2690
1909
- },
1910
- {
1911
- "epoch": 9.809264305177111,
1912
- "grad_norm": 1.296875,
1913
- "learning_rate": 9.73648712344707e-07,
1914
- "loss": 0.579,
1915
- "step": 2700
1916
- },
1917
- {
1918
- "epoch": 9.845594913714805,
1919
- "grad_norm": 1.2421875,
1920
- "learning_rate": 9.425163786873292e-07,
1921
- "loss": 0.61,
1922
- "step": 2710
1923
- },
1924
- {
1925
- "epoch": 9.881925522252498,
1926
- "grad_norm": 1.296875,
1927
- "learning_rate": 9.118380280114858e-07,
1928
- "loss": 0.6106,
1929
- "step": 2720
1930
- },
1931
- {
1932
- "epoch": 9.91825613079019,
1933
- "grad_norm": 1.296875,
1934
- "learning_rate": 8.816170928508367e-07,
1935
- "loss": 0.6066,
1936
- "step": 2730
1937
- },
1938
- {
1939
- "epoch": 9.954586739327883,
1940
- "grad_norm": 1.0703125,
1941
- "learning_rate": 8.518569545598198e-07,
1942
- "loss": 0.6094,
1943
- "step": 2740
1944
- },
1945
- {
1946
- "epoch": 9.990917347865576,
1947
- "grad_norm": 1.1640625,
1948
- "learning_rate": 8.225609429353187e-07,
1949
- "loss": 0.6383,
1950
- "step": 2750
1951
- },
1952
- {
1953
- "epoch": 10.02724795640327,
1954
- "grad_norm": 1.203125,
1955
- "learning_rate": 7.937323358440935e-07,
1956
- "loss": 0.6017,
1957
- "step": 2760
1958
- },
1959
- {
1960
- "epoch": 10.063578564940963,
1961
- "grad_norm": 1.03125,
1962
- "learning_rate": 7.653743588560387e-07,
1963
- "loss": 0.6005,
1964
- "step": 2770
1965
- },
1966
- {
1967
- "epoch": 10.099909173478656,
1968
- "grad_norm": 1.03125,
1969
- "learning_rate": 7.374901848832683e-07,
1970
- "loss": 0.647,
1971
- "step": 2780
1972
- },
1973
- {
1974
- "epoch": 10.136239782016348,
1975
- "grad_norm": 1.0625,
1976
- "learning_rate": 7.100829338251147e-07,
1977
- "loss": 0.5894,
1978
- "step": 2790
1979
- },
1980
- {
1981
- "epoch": 10.172570390554041,
1982
- "grad_norm": 0.953125,
1983
- "learning_rate": 6.831556722190453e-07,
1984
- "loss": 0.6075,
1985
- "step": 2800
1986
- },
1987
- {
1988
- "epoch": 10.208900999091735,
1989
- "grad_norm": 0.97265625,
1990
- "learning_rate": 6.567114128975571e-07,
1991
- "loss": 0.5531,
1992
- "step": 2810
1993
- },
1994
- {
1995
- "epoch": 10.245231607629428,
1996
- "grad_norm": 1.1171875,
1997
- "learning_rate": 6.307531146510754e-07,
1998
- "loss": 0.6052,
1999
- "step": 2820
2000
- },
2001
- {
2002
- "epoch": 10.28156221616712,
2003
- "grad_norm": 1.0234375,
2004
- "learning_rate": 6.052836818969027e-07,
2005
- "loss": 0.5833,
2006
- "step": 2830
2007
- },
2008
- {
2009
- "epoch": 10.317892824704813,
2010
- "grad_norm": 1.1484375,
2011
- "learning_rate": 5.803059643542491e-07,
2012
- "loss": 0.5934,
2013
- "step": 2840
2014
- },
2015
- {
2016
- "epoch": 10.354223433242506,
2017
- "grad_norm": 0.9609375,
2018
- "learning_rate": 5.558227567253832e-07,
2019
- "loss": 0.5931,
2020
- "step": 2850
2021
- },
2022
- {
2023
- "epoch": 10.3905540417802,
2024
- "grad_norm": 1.15625,
2025
- "learning_rate": 5.318367983829393e-07,
2026
- "loss": 0.6253,
2027
- "step": 2860
2028
- },
2029
- {
2030
- "epoch": 10.426884650317893,
2031
- "grad_norm": 1.0390625,
2032
- "learning_rate": 5.083507730634152e-07,
2033
- "loss": 0.61,
2034
- "step": 2870
2035
- },
2036
- {
2037
- "epoch": 10.463215258855586,
2038
- "grad_norm": 1.0234375,
2039
- "learning_rate": 4.853673085668947e-07,
2040
- "loss": 0.6329,
2041
- "step": 2880
2042
- },
2043
- {
2044
- "epoch": 10.499545867393278,
2045
- "grad_norm": 0.96484375,
2046
- "learning_rate": 4.628889764630279e-07,
2047
- "loss": 0.5733,
2048
- "step": 2890
2049
- },
2050
- {
2051
- "epoch": 10.535876475930971,
2052
- "grad_norm": 1.0,
2053
- "learning_rate": 4.4091829180330503e-07,
2054
- "loss": 0.5922,
2055
- "step": 2900
2056
- },
2057
- {
2058
- "epoch": 10.572207084468666,
2059
- "grad_norm": 1.0078125,
2060
- "learning_rate": 4.194577128396521e-07,
2061
- "loss": 0.5843,
2062
- "step": 2910
2063
- },
2064
- {
2065
- "epoch": 10.608537693006358,
2066
- "grad_norm": 1.0546875,
2067
- "learning_rate": 3.985096407493838e-07,
2068
- "loss": 0.6028,
2069
- "step": 2920
2070
- },
2071
- {
2072
- "epoch": 10.64486830154405,
2073
- "grad_norm": 0.984375,
2074
- "learning_rate": 3.7807641936653984e-07,
2075
- "loss": 0.5767,
2076
- "step": 2930
2077
- },
2078
- {
2079
- "epoch": 10.681198910081743,
2080
- "grad_norm": 1.0625,
2081
- "learning_rate": 3.581603349196372e-07,
2082
- "loss": 0.6062,
2083
- "step": 2940
2084
- },
2085
- {
2086
- "epoch": 10.717529518619436,
2087
- "grad_norm": 1.0,
2088
- "learning_rate": 3.3876361577587115e-07,
2089
- "loss": 0.5978,
2090
- "step": 2950
2091
- },
2092
- {
2093
- "epoch": 10.75386012715713,
2094
- "grad_norm": 1.046875,
2095
- "learning_rate": 3.1988843219178776e-07,
2096
- "loss": 0.5984,
2097
- "step": 2960
2098
- },
2099
- {
2100
- "epoch": 10.790190735694823,
2101
- "grad_norm": 1.1953125,
2102
- "learning_rate": 3.015368960704584e-07,
2103
- "loss": 0.5761,
2104
- "step": 2970
2105
- },
2106
- {
2107
- "epoch": 10.826521344232516,
2108
- "grad_norm": 1.0078125,
2109
- "learning_rate": 2.8371106072518194e-07,
2110
- "loss": 0.5988,
2111
- "step": 2980
2112
- },
2113
- {
2114
- "epoch": 10.862851952770209,
2115
- "grad_norm": 1.015625,
2116
- "learning_rate": 2.664129206497479e-07,
2117
- "loss": 0.593,
2118
- "step": 2990
2119
- },
2120
- {
2121
- "epoch": 10.899182561307901,
2122
- "grad_norm": 1.125,
2123
- "learning_rate": 2.4964441129527337e-07,
2124
- "loss": 0.6401,
2125
- "step": 3000
2126
- },
2127
- {
2128
- "epoch": 10.899182561307901,
2129
- "eval_loss": 1.12338387966156,
2130
- "eval_runtime": 10.3307,
2131
- "eval_samples_per_second": 23.716,
2132
- "eval_steps_per_second": 23.716,
2133
- "step": 3000
2134
- },
2135
- {
2136
- "epoch": 10.935513169845596,
2137
- "grad_norm": 0.921875,
2138
- "learning_rate": 2.3340740885364922e-07,
2139
- "loss": 0.5798,
2140
- "step": 3010
2141
- },
2142
- {
2143
- "epoch": 10.971843778383288,
2144
- "grad_norm": 1.0546875,
2145
- "learning_rate": 2.1770373004762035e-07,
2146
- "loss": 0.5684,
2147
- "step": 3020
2148
- },
2149
- {
2150
- "epoch": 11.008174386920981,
2151
- "grad_norm": 1.171875,
2152
- "learning_rate": 2.0253513192751374e-07,
2153
- "loss": 0.619,
2154
- "step": 3030
2155
- },
2156
- {
2157
- "epoch": 11.044504995458674,
2158
- "grad_norm": 1.125,
2159
- "learning_rate": 1.8790331167464758e-07,
2160
- "loss": 0.5739,
2161
- "step": 3040
2162
- },
2163
- {
2164
- "epoch": 11.080835603996366,
2165
- "grad_norm": 1.140625,
2166
- "learning_rate": 1.738099064114368e-07,
2167
- "loss": 0.6039,
2168
- "step": 3050
2169
- },
2170
- {
2171
- "epoch": 11.11716621253406,
2172
- "grad_norm": 1.3359375,
2173
- "learning_rate": 1.6025649301821877e-07,
2174
- "loss": 0.6057,
2175
- "step": 3060
2176
- },
2177
- {
2178
- "epoch": 11.153496821071753,
2179
- "grad_norm": 1.2109375,
2180
- "learning_rate": 1.4724458795681962e-07,
2181
- "loss": 0.5966,
2182
- "step": 3070
2183
- },
2184
- {
2185
- "epoch": 11.189827429609446,
2186
- "grad_norm": 1.109375,
2187
- "learning_rate": 1.3477564710088097e-07,
2188
- "loss": 0.5987,
2189
- "step": 3080
2190
- },
2191
- {
2192
- "epoch": 11.226158038147139,
2193
- "grad_norm": 1.1796875,
2194
- "learning_rate": 1.2285106557296479e-07,
2195
- "loss": 0.614,
2196
- "step": 3090
2197
- },
2198
- {
2199
- "epoch": 11.262488646684831,
2200
- "grad_norm": 1.25,
2201
- "learning_rate": 1.1147217758845752e-07,
2202
- "loss": 0.5854,
2203
- "step": 3100
2204
- },
2205
- {
2206
- "epoch": 11.298819255222526,
2207
- "grad_norm": 1.2734375,
2208
- "learning_rate": 1.0064025630628583e-07,
2209
- "loss": 0.5866,
2210
- "step": 3110
2211
- },
2212
- {
2213
- "epoch": 11.335149863760218,
2214
- "grad_norm": 1.25,
2215
- "learning_rate": 9.035651368646647e-08,
2216
- "loss": 0.6412,
2217
- "step": 3120
2218
- },
2219
- {
2220
- "epoch": 11.371480472297911,
2221
- "grad_norm": 1.203125,
2222
- "learning_rate": 8.06221003545038e-08,
2223
- "loss": 0.5935,
2224
- "step": 3130
2225
- },
2226
- {
2227
- "epoch": 11.407811080835604,
2228
- "grad_norm": 1.3125,
2229
- "learning_rate": 7.143810547264762e-08,
2230
- "loss": 0.6097,
2231
- "step": 3140
2232
- },
2233
- {
2234
- "epoch": 11.444141689373296,
2235
- "grad_norm": 2.1875,
2236
- "learning_rate": 6.280555661802857e-08,
2237
- "loss": 0.5737,
2238
- "step": 3150
2239
- },
2240
- {
2241
- "epoch": 11.48047229791099,
2242
- "grad_norm": 2.09375,
2243
- "learning_rate": 5.472541966768552e-08,
2244
- "loss": 0.5908,
2245
- "step": 3160
2246
- },
2247
- {
2248
- "epoch": 11.516802906448683,
2249
- "grad_norm": 2.078125,
2250
- "learning_rate": 4.719859869049659e-08,
2251
- "loss": 0.5884,
2252
- "step": 3170
2253
- },
2254
- {
2255
- "epoch": 11.553133514986376,
2256
- "grad_norm": 2.125,
2257
- "learning_rate": 4.02259358460233e-08,
2258
- "loss": 0.6072,
2259
- "step": 3180
2260
- },
2261
- {
2262
- "epoch": 11.589464123524069,
2263
- "grad_norm": 2.1875,
2264
- "learning_rate": 3.3808211290284886e-08,
2265
- "loss": 0.6049,
2266
- "step": 3190
2267
- },
2268
- {
2269
- "epoch": 11.625794732061761,
2270
- "grad_norm": 3.734375,
2271
- "learning_rate": 2.7946143088466437e-08,
2272
- "loss": 0.6021,
2273
- "step": 3200
2274
- },
2275
- {
2276
- "epoch": 11.662125340599456,
2277
- "grad_norm": 4.34375,
2278
- "learning_rate": 2.264038713457706e-08,
2279
- "loss": 0.6059,
2280
- "step": 3210
2281
- },
2282
- {
2283
- "epoch": 11.698455949137148,
2284
- "grad_norm": 3.9375,
2285
- "learning_rate": 1.789153707806357e-08,
2286
- "loss": 0.601,
2287
- "step": 3220
2288
- },
2289
- {
2290
- "epoch": 11.734786557674841,
2291
- "grad_norm": 4.0625,
2292
- "learning_rate": 1.3700124257388092e-08,
2293
- "loss": 0.5657,
2294
- "step": 3230
2295
- },
2296
- {
2297
- "epoch": 11.771117166212534,
2298
- "grad_norm": 4.25,
2299
- "learning_rate": 1.006661764057837e-08,
2300
- "loss": 0.6255,
2301
- "step": 3240
2302
- },
2303
- {
2304
- "epoch": 11.807447774750226,
2305
- "grad_norm": 7.65625,
2306
- "learning_rate": 6.991423772753636e-09,
2307
- "loss": 0.6209,
2308
- "step": 3250
2309
- },
2310
- {
2311
- "epoch": 11.84377838328792,
2312
- "grad_norm": 7.90625,
2313
- "learning_rate": 4.474886730641004e-09,
2314
- "loss": 0.6111,
2315
- "step": 3260
2316
- },
2317
- {
2318
- "epoch": 11.880108991825614,
2319
- "grad_norm": 8.5625,
2320
- "learning_rate": 2.5172880840745873e-09,
2321
- "loss": 0.6328,
2322
- "step": 3270
2323
- },
2324
- {
2325
- "epoch": 11.916439600363306,
2326
- "grad_norm": 8.375,
2327
- "learning_rate": 1.118846864490708e-09,
2328
- "loss": 0.5511,
2329
- "step": 3280
2330
- },
2331
- {
2332
- "epoch": 11.952770208900999,
2333
- "grad_norm": 8.4375,
2334
- "learning_rate": 2.797195404247166e-10,
2335
- "loss": 0.5795,
2336
- "step": 3290
2337
- },
2338
- {
2339
- "epoch": 11.989100817438691,
2340
- "grad_norm": 2.59375,
2341
- "learning_rate": 0.0,
2342
- "loss": 0.5714,
2343
- "step": 3300
2344
- }
2345
- ],
2346
- "logging_steps": 10,
2347
- "max_steps": 3300,
2348
- "num_input_tokens_seen": 0,
2349
- "num_train_epochs": 12,
2350
- "save_steps": 0,
2351
- "stateful_callbacks": {
2352
- "TrainerControl": {
2353
- "args": {
2354
- "should_epoch_stop": false,
2355
- "should_evaluate": false,
2356
- "should_log": false,
2357
- "should_save": true,
2358
- "should_training_stop": true
2359
- },
2360
- "attributes": {}
2361
- }
2362
- },
2363
- "total_flos": 3.144153968877896e+17,
2364
- "train_batch_size": 1,
2365
- "trial_name": null,
2366
- "trial_params": null
2367
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
checkpoint-3300/training_args.bin DELETED
@@ -1,3 +0,0 @@
1
- version https://git-lfs.github.com/spec/v1
2
- oid sha256:4bf2707d2b44041692fc48569fafa359449408ba1404e7dcad3487b612ff0546
3
- size 5304