Commit
·
7d2c093
1
Parent(s):
c0d8ae5
Remove eval
Browse filesThis view is limited to 50 files because it contains too many changes.
See raw diff
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Answer_Given_options/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Choose_Story_Ending/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Generate_Ending/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Novel_Correct_Ending/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Story_Continuation_and_Options/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Answer_Given_options/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Choose_Story_Ending/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Generate_Ending/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Novel_Correct_Ending/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Story_Continuation_and_Options/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Answer_Given_options/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Choose_Story_Ending/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Generate_Ending/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Novel_Correct_Ending/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Story_Continuation_and_Options/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Answer_Given_options/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Choose_Story_Ending/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Generate_Ending/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Novel_Correct_Ending/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Story_Continuation_and_Options/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Answer_Given_options/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Choose_Story_Ending/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Generate_Ending/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Novel_Correct_Ending/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Story_Continuation_and_Options/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Answer_Given_options/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Choose_Story_Ending/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Generate_Ending/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Novel_Correct_Ending/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Story_Continuation_and_Options/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/Replace/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/True_or_False/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/does_underscore_refer_to/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/stand_for/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/underscore_refer_to/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/Replace/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/True_or_False/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/does_underscore_refer_to/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/stand_for/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/underscore_refer_to/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/Replace/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/True_or_False/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/does_underscore_refer_to/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/stand_for/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/underscore_refer_to/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/zh/Replace/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/zh/True_or_False/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/zh/does_underscore_refer_to/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/zh/stand_for/results.json +0 -9
- evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/zh/underscore_refer_to/results.json +0 -9
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Answer_Given_options/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "ar",
|
4 |
-
"template_name": "Answer Given options",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.7968232958305758
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Choose_Story_Ending/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "ar",
|
4 |
-
"template_name": "Choose Story Ending",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.9232296492389146
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Generate_Ending/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "ar",
|
4 |
-
"template_name": "Generate Ending",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.6677696889477167
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Novel_Correct_Ending/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "ar",
|
4 |
-
"template_name": "Novel Correct Ending",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.9265387160820648
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/ar/Story_Continuation_and_Options/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "ar",
|
4 |
-
"template_name": "Story Continuation and Options",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.9126406353408338
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='ar', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Answer_Given_options/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "es",
|
4 |
-
"template_name": "Answer Given options",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.8729318332230311
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Choose_Story_Ending/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "es",
|
4 |
-
"template_name": "Choose Story Ending",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.9417604235605559
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Generate_Ending/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "es",
|
4 |
-
"template_name": "Generate Ending",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.7359364659166115
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Novel_Correct_Ending/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "es",
|
4 |
-
"template_name": "Novel Correct Ending",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.9430840502978161
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/es/Story_Continuation_and_Options/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "es",
|
4 |
-
"template_name": "Story Continuation and Options",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.9318332230311053
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='es', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Answer_Given_options/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "eu",
|
4 |
-
"template_name": "Answer Given options",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.7054930509596293
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Choose_Story_Ending/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "eu",
|
4 |
-
"template_name": "Choose Story Ending",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.8663136995367307
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Generate_Ending/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "eu",
|
4 |
-
"template_name": "Generate Ending",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.6320317670416943
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Novel_Correct_Ending/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "eu",
|
4 |
-
"template_name": "Novel Correct Ending",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.8689609530112509
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/eu/Story_Continuation_and_Options/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "eu",
|
4 |
-
"template_name": "Story Continuation and Options",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.8524156187954997
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='eu', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Answer_Given_options/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "hi",
|
4 |
-
"template_name": "Answer Given options",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.798808735936466
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Choose_Story_Ending/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "hi",
|
4 |
-
"template_name": "Choose Story Ending",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.8702845797485109
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Generate_Ending/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "hi",
|
4 |
-
"template_name": "Generate Ending",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.6604897418927862
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Novel_Correct_Ending/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "hi",
|
4 |
-
"template_name": "Novel Correct Ending",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.8788881535407015
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/hi/Story_Continuation_and_Options/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "hi",
|
4 |
-
"template_name": "Story Continuation and Options",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.870946393117141
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='hi', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Answer_Given_options/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "id",
|
4 |
-
"template_name": "Answer Given options",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.8557246856386499
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Choose_Story_Ending/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "id",
|
4 |
-
"template_name": "Choose Story Ending",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.9212442091330245
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Generate_Ending/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "id",
|
4 |
-
"template_name": "Generate Ending",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.7041694242223693
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Novel_Correct_Ending/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "id",
|
4 |
-
"template_name": "Novel Correct Ending",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.9205823957643945
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/id/Story_Continuation_and_Options/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "id",
|
4 |
-
"template_name": "Story Continuation and Options",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.9066843150231635
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='id', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Answer_Given_options/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "zh",
|
4 |
-
"template_name": "Answer Given options",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.900066181336863
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Answer Given options', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Choose_Story_Ending/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "zh",
|
4 |
-
"template_name": "Choose Story Ending",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.9232296492389146
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Choose Story Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Generate_Ending/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "zh",
|
4 |
-
"template_name": "Generate Ending",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.684976836532098
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Generate Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Novel_Correct_Ending/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "zh",
|
4 |
-
"template_name": "Novel Correct Ending",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.9311714096624751
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Novel Correct Ending', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xstory_cloze/zh/Story_Continuation_and_Options/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xstory_cloze",
|
3 |
-
"dataset_config_name": "zh",
|
4 |
-
"template_name": "Story Continuation and Options",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.9199205823957644
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xstory_cloze', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Story Continuation and Options', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/Replace/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
3 |
-
"dataset_config_name": "en",
|
4 |
-
"template_name": "Replace",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.6847311827956989
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Replace', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/True_or_False/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
3 |
-
"dataset_config_name": "en",
|
4 |
-
"template_name": "True or False",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.5135483870967742
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/does_underscore_refer_to/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
3 |
-
"dataset_config_name": "en",
|
4 |
-
"template_name": "does underscore refer to",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.6787096774193548
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/stand_for/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
3 |
-
"dataset_config_name": "en",
|
4 |
-
"template_name": "stand for",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.5053763440860215
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='stand for', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/en/underscore_refer_to/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
3 |
-
"dataset_config_name": "en",
|
4 |
-
"template_name": "underscore refer to",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.690752688172043
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='en', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/Replace/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
3 |
-
"dataset_config_name": "fr",
|
4 |
-
"template_name": "Replace",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.6506024096385542
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Replace', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/True_or_False/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
3 |
-
"dataset_config_name": "fr",
|
4 |
-
"template_name": "True or False",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.4939759036144578
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/does_underscore_refer_to/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
3 |
-
"dataset_config_name": "fr",
|
4 |
-
"template_name": "does underscore refer to",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.6867469879518072
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/stand_for/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
3 |
-
"dataset_config_name": "fr",
|
4 |
-
"template_name": "stand for",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.46987951807228917
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='stand for', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/fr/underscore_refer_to/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
3 |
-
"dataset_config_name": "fr",
|
4 |
-
"template_name": "underscore refer to",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.6626506024096386
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='fr', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/Replace/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
3 |
-
"dataset_config_name": "pt",
|
4 |
-
"template_name": "Replace",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.6349809885931559
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='Replace', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/True_or_False/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
3 |
-
"dataset_config_name": "pt",
|
4 |
-
"template_name": "True or False",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.4866920152091255
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/does_underscore_refer_to/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
3 |
-
"dataset_config_name": "pt",
|
4 |
-
"template_name": "does underscore refer to",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.6387832699619772
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/stand_for/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
3 |
-
"dataset_config_name": "pt",
|
4 |
-
"template_name": "stand for",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.49429657794676807
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='stand for', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/pt/underscore_refer_to/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
3 |
-
"dataset_config_name": "pt",
|
4 |
-
"template_name": "underscore refer to",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.6425855513307985
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='pt', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='test', target_max_length=256, template_config_name='en', template_name='underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/zh/Replace/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
3 |
-
"dataset_config_name": "zh",
|
4 |
-
"template_name": "Replace",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.6865079365079365
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='Replace', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/zh/True_or_False/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
3 |
-
"dataset_config_name": "zh",
|
4 |
-
"template_name": "True or False",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.5277777777777778
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='True or False', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/zh/does_underscore_refer_to/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
3 |
-
"dataset_config_name": "zh",
|
4 |
-
"template_name": "does underscore refer to",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.6884920634920635
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='does underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/zh/stand_for/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
3 |
-
"dataset_config_name": "zh",
|
4 |
-
"template_name": "stand for",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.4861111111111111
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='stand for', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
evaluation_bloomz/evaluation_l1/Muennighoff_xwinograd/zh/underscore_refer_to/results.json
DELETED
@@ -1,9 +0,0 @@
|
|
1 |
-
{
|
2 |
-
"dataset_name": "Muennighoff/xwinograd",
|
3 |
-
"dataset_config_name": "zh",
|
4 |
-
"template_name": "underscore refer to",
|
5 |
-
"evaluation": {
|
6 |
-
"accuracy": 0.6904761904761905
|
7 |
-
},
|
8 |
-
"arguments": "Namespace(config_name=None, dataset_config_name='zh', dataset_name='Muennighoff/xwinograd', debug=False, dtype='bfloat16', max_length=2048, model_name_or_path='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498', output_dir='/gpfsscratch/rech/six/commun/commun/experiments/muennighoff/bloomckpt/176bt0/xp3capmixnewcodelonglossseq_global_step498/evaluation', pad_to_max_length=False, per_device_eval_batch_size=8, prefixlm=False, split='validation', target_max_length=256, template_config_name='en', template_name='underscore refer to', tokenizer_name=None, use_slow_tokenizer=False)"
|
9 |
-
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|