floriangardin
/

model

@@ -11,9 +11,9 @@ should probably proofread and complete it, then remove this comment. -->
 # model
-This model is a fine-tuned version of [](https://huggingface.co/) on the None dataset.
 It achieves the following results on the evaluation set:
-- Loss: 0.9350
 ## Model description
@@ -46,20 +46,19 @@ The following hyperparameters were used during training:
 | Training Loss | Epoch | Step  | Validation Loss |
 |:-------------:|:-----:|:-----:|:---------------:|
-| 2.799         | 0.07  | 2000  | 2.6594          |
-| 1.5819        | 0.13  | 4000  | 1.4878          |
-| 1.3252        | 0.2   | 6000  | 1.2718          |
-| 1.2293        | 0.27  | 8000  | 1.1748          |
-| 1.141         | 0.34  | 10000 | 1.1004          |
-| 1.093         | 0.4   | 12000 | 1.0582          |
-| 1.0601        | 0.47  | 14000 | 1.0282          |
-| 1.0285        | 0.54  | 16000 | 0.9957          |
-| 1.002         | 0.61  | 18000 | 0.9794          |
-| 0.9876        | 0.67  | 20000 | 0.9605          |
-| 0.9903        | 0.74  | 22000 | 0.9489          |
-| 0.9698        | 0.81  | 24000 | 0.9418          |
-| 0.962         | 0.88  | 26000 | 0.9370          |
-| 0.9598        | 0.94  | 28000 | 0.9350          |
 ### Framework versions

 # model
+This model is a fine-tuned version of [](https://huggingface.co/) on an unknown dataset.
 It achieves the following results on the evaluation set:
+- Loss: 1.0360
 ## Model description
 | Training Loss | Epoch | Step  | Validation Loss |
 |:-------------:|:-----:|:-----:|:---------------:|
+| 3.1667        | 0.07  | 2000  | 3.1054          |
+| 1.8298        | 0.14  | 4000  | 1.7209          |
+| 1.4726        | 0.22  | 6000  | 1.4237          |
+| 1.3446        | 0.29  | 8000  | 1.2875          |
+| 1.2647        | 0.36  | 10000 | 1.2120          |
+| 1.2023        | 0.43  | 12000 | 1.1621          |
+| 1.185         | 0.51  | 14000 | 1.1240          |
+| 1.1308        | 0.58  | 16000 | 1.0957          |
+| 1.1057        | 0.65  | 18000 | 1.0736          |
+| 1.0894        | 0.72  | 20000 | 1.0555          |
+| 1.087         | 0.8   | 22000 | 1.0439          |
+| 1.0829        | 0.87  | 24000 | 1.0372          |
+| 1.0566        | 0.94  | 26000 | 1.0360          |
 ### Framework versions

config.json CHANGED Viewed

@@ -4,9 +4,9 @@
     "GPT2LMHeadModel"
   ],
   "attn_pdrop": 0.1,
-  "bos_token_id": 30000,
   "embd_pdrop": 0.1,
-  "eos_token_id": 30000,
   "initializer_range": 0.02,
   "layer_norm_epsilon": 1e-05,
   "model_type": "gpt2",
@@ -15,7 +15,7 @@
   "n_inner": null,
   "n_layer": 10,
   "n_positions": 4096,
-  "padding_token_id": 30000,
   "reorder_and_upcast_attn": false,
   "resid_pdrop": 0.1,
   "scale_attn_by_inverse_layer_idx": false,
@@ -28,5 +28,5 @@
   "torch_dtype": "float32",
   "transformers_version": "4.37.2",
   "use_cache": true,
-  "vocab_size": 30001
 }

     "GPT2LMHeadModel"
   ],
   "attn_pdrop": 0.1,
+  "bos_token_id": 1,
   "embd_pdrop": 0.1,
+  "eos_token_id": 1,
   "initializer_range": 0.02,
   "layer_norm_epsilon": 1e-05,
   "model_type": "gpt2",
   "n_inner": null,
   "n_layer": 10,
   "n_positions": 4096,
+  "padding_token_id": 1,
   "reorder_and_upcast_attn": false,
   "resid_pdrop": 0.1,
   "scale_attn_by_inverse_layer_idx": false,
   "torch_dtype": "float32",
   "transformers_version": "4.37.2",
   "use_cache": true,
+  "vocab_size": 30000
 }

generation_config.json CHANGED Viewed

@@ -1,6 +1,6 @@
 {
   "_from_model_config": true,
-  "bos_token_id": 30000,
-  "eos_token_id": 30000,
   "transformers_version": "4.37.2"
 }

 {
   "_from_model_config": true,
+  "bos_token_id": 1,
+  "eos_token_id": 1,
   "transformers_version": "4.37.2"
 }

model.safetensors CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:22726d5d220ff7c2d8b4bda44044f456543da712c6f13abceb7ebfbdf291e29c
-size 254962056

 version https://git-lfs.github.com/spec/v1
+oid sha256:6e91ec2e5db8defd484c25027a4f99b039f5e416917f7cbfade856f1fcf1154a
+size 254959656

training_args.bin CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:db2daf1ba377eb851dad0905d389222e965f048ff071e14d1039b7e459c05c67
 size 4664

 version https://git-lfs.github.com/spec/v1
+oid sha256:84fd46c4d64452c3ae2ae1094579da98c0405b2f53b95a5f1906bbe659ae8c78
 size 4664