Added language model

Browse files

Files changed (13) hide show

README.md +68 -3
added_tokens.json +4 -0
alphabet.json +1 -0
config.json +109 -0
language_model/5gram.bin +3 -0
language_model/attrs.json +1 -0
language_model/unigrams.txt +0 -0
model.safetensors +3 -0
preprocessor_config.json +10 -0
special_tokens_map.json +30 -0
tokenizer_config.json +48 -0
training_args.bin +3 -0
vocab.json +37 -0

README.md CHANGED Viewed

@@ -1,3 +1,68 @@
----
-license: apache-2.0
----

+---
+license: apache-2.0
+base_model: facebook/wav2vec2-xls-r-300m
+tags:
+- generated_from_trainer
+metrics:
+- wer
+model-index:
+- name: bambara-1-hours-bambara-asr-hf
+ results: []
+---
+<!-- This model card has been generated automatically according to the information the Trainer had access to. You
+should probably proofread and complete it, then remove this comment. -->
+# bambara-1-hours-bambara-asr-hf
+This model is a fine-tuned version of [facebook/wav2vec2-xls-r-300m](https://huggingface.co/facebook/wav2vec2-xls-r-300m) on an unknown dataset.
+It achieves the following results on the evaluation set:
+- Loss: 1.7420
+- Wer: 0.6528
+- Cer: 0.2937
+## Model description
+More information needed
+## Intended uses & limitations
+More information needed
+## Training and evaluation data
+More information needed
+## Training procedure
+### Training hyperparameters
+The following hyperparameters were used during training:
+- learning_rate: 0.0003
+- train_batch_size: 16
+- eval_batch_size: 16
+- seed: 42
+- gradient_accumulation_steps: 2
+- total_train_batch_size: 32
+- optimizer: Adam with betas=(0.9,0.999) and epsilon=1e-08
+- lr_scheduler_type: linear
+- lr_scheduler_warmup_steps: 500
+- num_epochs: 30
+### Training results
+| Training Loss | Epoch | Step | Validation Loss | Wer | Cer |
+|:-------------:|:-----:|:----:|:---------------:|:------:|:------:|
+| 1.1128 | 5.71 | 200 | 1.1737 | 0.7170 | 0.3240 |
+| 0.9466 | 11.43 | 400 | 1.3777 | 0.6729 | 0.3018 |
+| 0.7191 | 17.14 | 600 | 1.5054 | 0.6833 | 0.3096 |
+| 0.624 | 22.86 | 800 | 1.6121 | 0.6704 | 0.3018 |
+| 0.4667 | 28.57 | 1000 | 1.7420 | 0.6528 | 0.2937 |
+### Framework versions
+- Transformers 4.38.1
+- Pytorch 2.1.0+cu118
+- Datasets 2.17.0
+- Tokenizers 0.15.2

added_tokens.json ADDED Viewed

	@@ -0,0 +1,4 @@

+{
+ "</s>": 36,
+ "<s>": 35
+}

alphabet.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"labels": [" ", "a", "b", "c", "d", "e", "f", "g", "h", "i", "j", "k", "l", "m", "n", "o", "p", "q", "r", "s", "t", "u", "v", "w", "x", "y", "z", "\u00e7", "\u00ea", "\u014b", "\u0254", "\u025b", "\u0272", "\u2047", "", "<s>", "</s>"], "is_bpe": false}

config.json ADDED Viewed

	@@ -0,0 +1,109 @@

+{
+ "_name_or_path": "facebook/wav2vec2-xls-r-300m",
+ "activation_dropout": 0.0,
+ "adapter_attn_dim": null,
+ "adapter_kernel_size": 3,
+ "adapter_stride": 2,
+ "add_adapter": false,
+ "apply_spec_augment": true,
+ "architectures": [
+ "Wav2Vec2ForCTC"
+ ],
+ "attention_dropout": 0.1,
+ "bos_token_id": 1,
+ "classifier_proj_size": 256,
+ "codevector_dim": 768,
+ "contrastive_logits_temperature": 0.1,
+ "conv_bias": true,
+ "conv_dim": [
+ 512,
+ 512,
+ 512,
+ 512,
+ 512,
+ 512,
+ 512
+ ],
+ "conv_kernel": [
+ 10,
+ 3,
+ 3,
+ 3,
+ 3,
+ 2,
+ 2
+ ],
+ "conv_stride": [
+ 5,
+ 2,
+ 2,
+ 2,
+ 2,
+ 2,
+ 2
+ ],
+ "ctc_loss_reduction": "mean",
+ "ctc_zero_infinity": true,
+ "diversity_loss_weight": 0.1,
+ "do_stable_layer_norm": true,
+ "eos_token_id": 2,
+ "feat_extract_activation": "gelu",
+ "feat_extract_dropout": 0.0,
+ "feat_extract_norm": "layer",
+ "feat_proj_dropout": 0.1,
+ "feat_quantizer_dropout": 0.0,
+ "final_dropout": 0.0,
+ "gradient_checkpointing": false,
+ "hidden_act": "gelu",
+ "hidden_dropout": 0.1,
+ "hidden_size": 1024,
+ "initializer_range": 0.02,
+ "intermediate_size": 4096,
+ "layer_norm_eps": 1e-05,
+ "layerdrop": 0.1,
+ "mask_feature_length": 10,
+ "mask_feature_min_masks": 0,
+ "mask_feature_prob": 0.0,
+ "mask_time_length": 10,
+ "mask_time_min_masks": 2,
+ "mask_time_prob": 0.05,
+ "model_type": "wav2vec2",
+ "num_adapter_layers": 3,
+ "num_attention_heads": 16,
+ "num_codevector_groups": 2,
+ "num_codevectors_per_group": 320,
+ "num_conv_pos_embedding_groups": 16,
+ "num_conv_pos_embeddings": 128,
+ "num_feat_extract_layers": 7,
+ "num_hidden_layers": 24,
+ "num_negatives": 100,
+ "output_hidden_size": 1024,
+ "pad_token_id": 34,
+ "proj_codevector_dim": 768,
+ "tdnn_dilation": [
+ 1,
+ 2,
+ 3,
+ 1,
+ 1
+ ],
+ "tdnn_dim": [
+ 512,
+ 512,
+ 512,
+ 512,
+ 1500
+ ],
+ "tdnn_kernel": [
+ 5,
+ 3,
+ 3,
+ 1,
+ 1
+ ],
+ "torch_dtype": "float32",
+ "transformers_version": "4.38.1",
+ "use_weighted_layer_sum": false,
+ "vocab_size": 37,
+ "xvector_output_dim": 512
+}

language_model/5gram.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:2a2935b1b23655f32bf0b2f3152f5281180498b0e52c19ef6da1286a31ab7d58
+size 16490821

language_model/attrs.json ADDED Viewed

	@@ -0,0 +1 @@


1	+ {"alpha": 0.5, "beta": 1.5, "unk_score_offset": -10.0, "score_boundary": true}

language_model/unigrams.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:688ee97e4e78032d93aaa7e46aeab964b37e15d29ae824c7b4c5dfa3bf41e446
+size 1261959180

preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,10 @@

+{
+ "do_normalize": true,
+ "feature_extractor_type": "Wav2Vec2FeatureExtractor",
+ "feature_size": 1,
+ "padding_side": "right",
+ "padding_value": 0.0,
+ "processor_class": "Wav2Vec2ProcessorWithLM",
+ "return_attention_mask": true,
+ "sampling_rate": 16000
+}

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,30 @@

+{
+ "bos_token": {
+ "content": "<s>",
+ "lstrip": false,
+ "normalized": false,
+ "rstrip": false,
+ "single_word": false
+ },
+ "eos_token": {
+ "content": "</s>",
+ "lstrip": false,
+ "normalized": false,
+ "rstrip": false,
+ "single_word": false
+ },
+ "pad_token": {
+ "content": "[PAD]",
+ "lstrip": true,
+ "normalized": false,
+ "rstrip": true,
+ "single_word": false
+ },
+ "unk_token": {
+ "content": "[UNK]",
+ "lstrip": true,
+ "normalized": false,
+ "rstrip": true,
+ "single_word": false
+ }
+}

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,48 @@

+{
+ "added_tokens_decoder": {
+ "33": {
+ "content": "[UNK]",
+ "lstrip": true,
+ "normalized": false,
+ "rstrip": true,
+ "single_word": false,
+ "special": false
+ },
+ "34": {
+ "content": "[PAD]",
+ "lstrip": true,
+ "normalized": false,
+ "rstrip": true,
+ "single_word": false,
+ "special": false
+ },
+ "35": {
+ "content": "<s>",
+ "lstrip": false,
+ "normalized": false,
+ "rstrip": false,
+ "single_word": false,
+ "special": true
+ },
+ "36": {
+ "content": "</s>",
+ "lstrip": false,
+ "normalized": false,
+ "rstrip": false,
+ "single_word": false,
+ "special": true
+ }
+ },
+ "bos_token": "<s>",
+ "clean_up_tokenization_spaces": true,
+ "do_lower_case": false,
+ "eos_token": "</s>",
+ "model_max_length": 1000000000000000019884624838656,
+ "pad_token": "[PAD]",
+ "processor_class": "Wav2Vec2ProcessorWithLM",
+ "replace_word_delimiter_char": " ",
+ "target_lang": null,
+ "tokenizer_class": "Wav2Vec2CTCTokenizer",
+ "unk_token": "[UNK]",
+ "word_delimiter_token": "|"
+}

training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:d9b2985a663eab6df1ccc603f89eb377003a1989e852d5f2e34a4e80195ac398
+size 4984

vocab.json ADDED Viewed

	@@ -0,0 +1,37 @@

+{
+ "[PAD]": 34,
+ "[UNK]": 33,
+ "a": 1,
+ "b": 2,
+ "c": 3,
+ "d": 4,
+ "e": 5,
+ "f": 6,
+ "g": 7,
+ "h": 8,
+ "i": 9,
+ "j": 10,
+ "k": 11,
+ "l": 12,
+ "m": 13,
+ "n": 14,
+ "o": 15,
+ "p": 16,
+ "q": 17,
+ "r": 18,
+ "s": 19,
+ "t": 20,
+ "u": 21,
+ "v": 22,
+ "w": 23,
+ "x": 24,
+ "y": 25,
+ "z": 26,
+ "|": 0,
+ "ç": 27,
+ "ê": 28,
+ "ŋ": 29,
+ "ɔ": 30,
+ "ɛ": 31,
+ "ɲ": 32
+}