prince-canuma commited on
Commit
e9c653a
1 Parent(s): 8c0e8c2

Upload folder using huggingface_hub (#1)

Browse files

- a3484704b0e07278cce8c8209abdc83e27973a658759791a791ececb3021f605 (fb74faee2e2d777344cd3775ace24ccf548f7b77)
- b9d430ac16c81a58aadbee843dfe227cc04ac4a70d64161d78912b28c074d805 (b9ef16312a6783694e091553cc578d2e4bb7be74)
- bc371e5826e3a57dcb7f590994fae904f7e8bf051943d1d973fa9d451e90ae57 (d05c4275c07f212e5654c3556280d761705aa68b)
- ad2143476f3511379453274952862d058c76567ac3f469f243dc876a1d656ad1 (49d8dc84755430beda0b6cf9e45ce821bea9dcee)
- 713eebc48829dcb9c9d231468d95d487a93e9db4cb3f1502cd93529ac02ed2ac (fd385fbe3164b03341d91fc82cadbe740328bdee)
- 9a5fae79e3fb0c25d05b58524cfe1937ee437b69e545dca34fc3667a00832815 (04d6f00d5a77f8386495d6f652ea0a61e5096ce8)
- 9c72a8f82dcdf84f76143ef68c8e90bf051875977da245b5775ffbfec99f8316 (78fcab42d3644940a546b3fd6bef9f478eb91668)

README.md ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ language:
3
+ - en
4
+ license: apache-2.0
5
+ tags:
6
+ - chat
7
+ - mlx
8
+ pipeline_tag: text-generation
9
+ ---
10
+
11
+ # mlx-community/Qwen2-57B-A14B-Instruct-4bit
12
+
13
+ The Model [mlx-community/Qwen2-57B-A14B-Instruct-4bit](https://huggingface.co/mlx-community/Qwen2-57B-A14B-Instruct-4bit) was converted to MLX format from [Qwen/Qwen2-57B-A14B-Instruct](https://huggingface.co/Qwen/Qwen2-57B-A14B-Instruct) using mlx-lm version **0.14.2**.
14
+
15
+ ## Use with mlx
16
+
17
+ ```bash
18
+ pip install mlx-lm
19
+ ```
20
+
21
+ ```python
22
+ from mlx_lm import load, generate
23
+
24
+ model, tokenizer = load("mlx-community/Qwen2-57B-A14B-Instruct-4bit")
25
+ response = generate(model, tokenizer, prompt="hello", verbose=True)
26
+ ```
added_tokens.json ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ {
2
+ "<|endoftext|>": 151643,
3
+ "<|im_end|>": 151645,
4
+ "<|im_start|>": 151644
5
+ }
config.json ADDED
@@ -0,0 +1,39 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen2MoeForCausalLM"
4
+ ],
5
+ "attention_dropout": 0.0,
6
+ "bos_token_id": 151643,
7
+ "decoder_sparse_step": 1,
8
+ "eos_token_id": 151643,
9
+ "hidden_act": "silu",
10
+ "hidden_size": 3584,
11
+ "initializer_range": 0.02,
12
+ "intermediate_size": 18944,
13
+ "max_position_embeddings": 32768,
14
+ "max_window_layers": 28,
15
+ "model_type": "qwen2_moe",
16
+ "moe_intermediate_size": 2560,
17
+ "norm_topk_prob": false,
18
+ "num_attention_heads": 28,
19
+ "num_experts": 64,
20
+ "num_experts_per_tok": 8,
21
+ "num_hidden_layers": 28,
22
+ "num_key_value_heads": 4,
23
+ "output_router_logits": false,
24
+ "quantization": {
25
+ "group_size": 64,
26
+ "bits": 4
27
+ },
28
+ "rms_norm_eps": 1e-06,
29
+ "rope_theta": 1000000.0,
30
+ "router_aux_loss_coef": 0.001,
31
+ "shared_expert_intermediate_size": 20480,
32
+ "sliding_window": 65536,
33
+ "tie_word_embeddings": false,
34
+ "torch_dtype": "bfloat16",
35
+ "transformers_version": "4.40.1",
36
+ "use_cache": true,
37
+ "use_sliding_window": false,
38
+ "vocab_size": 151936
39
+ }
merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
model-00001-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a35a0008940a83bfefb2bbee0c0ac88c2c0709c560bd9f69341a6296b1b946e3
3
+ size 5179027722
model-00002-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bc6aaf60fe337c2bb857fe15d7bab38c8e50749224f6ffecdbe30217cff9f5b6
3
+ size 5326907408
model-00003-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:68a74b3510080411af10059dd9c1a02ee479edb4699ae2585ba08b3da60caa0f
3
+ size 5186371187
model-00004-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2a2c87d6042282dad362dc86869ec44fd9bf177c8774ce43dc780e4243847ae7
3
+ size 5326907658
model-00005-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:262a370320d5b486a6c2a0af85dc91e69e9e6d93e0a6781f291a5ed461d953ac
3
+ size 5326907542
model-00006-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d6ad155c0d8626f50a3e90b0fd9cb0574bcd18d7ff7237f4f9f8f397057e12a8
3
+ size 5186371176
model-00007-of-00007.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:09e2dab260a604711837440739b30c6db3d85fc69f5d29bce4ed293553bacb7c
3
+ size 760493429
model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
special_tokens_map.json ADDED
@@ -0,0 +1,20 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "additional_special_tokens": [
3
+ "<|im_start|>",
4
+ "<|im_end|>"
5
+ ],
6
+ "eos_token": {
7
+ "content": "<|im_end|>",
8
+ "lstrip": false,
9
+ "normalized": false,
10
+ "rstrip": false,
11
+ "single_word": false
12
+ },
13
+ "pad_token": {
14
+ "content": "<|endoftext|>",
15
+ "lstrip": false,
16
+ "normalized": false,
17
+ "rstrip": false,
18
+ "single_word": false
19
+ }
20
+ }
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer_config.json ADDED
@@ -0,0 +1,43 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "added_tokens_decoder": {
4
+ "151643": {
5
+ "content": "<|endoftext|>",
6
+ "lstrip": false,
7
+ "normalized": false,
8
+ "rstrip": false,
9
+ "single_word": false,
10
+ "special": true
11
+ },
12
+ "151644": {
13
+ "content": "<|im_start|>",
14
+ "lstrip": false,
15
+ "normalized": false,
16
+ "rstrip": false,
17
+ "single_word": false,
18
+ "special": true
19
+ },
20
+ "151645": {
21
+ "content": "<|im_end|>",
22
+ "lstrip": false,
23
+ "normalized": false,
24
+ "rstrip": false,
25
+ "single_word": false,
26
+ "special": true
27
+ }
28
+ },
29
+ "additional_special_tokens": [
30
+ "<|im_start|>",
31
+ "<|im_end|>"
32
+ ],
33
+ "bos_token": null,
34
+ "chat_template": "{% for message in messages %}{% if loop.first and messages[0]['role'] != 'system' %}{{ '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}{% endif %}{{'<|im_start|>' + message['role'] + '\n' + message['content'] + '<|im_end|>' + '\n'}}{% endfor %}{% if add_generation_prompt %}{{ '<|im_start|>assistant\n' }}{% endif %}",
35
+ "clean_up_tokenization_spaces": false,
36
+ "eos_token": "<|im_end|>",
37
+ "errors": "replace",
38
+ "model_max_length": 65536,
39
+ "pad_token": "<|endoftext|>",
40
+ "split_special_tokens": false,
41
+ "tokenizer_class": "Qwen2Tokenizer",
42
+ "unk_token": null
43
+ }
vocab.json ADDED
The diff for this file is too large to render. See raw diff