MolGen
/

llama_ZINC_1B-raw_atomwise_SELFIES_948da50f

Model card Files Files and versions Community

kmchiti commited on Sep 20, 2024

Commit

79a5653

verified ·

1 Parent(s): 0ddb167

Training in progress, step 60000, checkpoint

Browse files

Files changed (6) hide show

tmp-spec-checkpoint-60000/config.json +30 -0
tmp-spec-checkpoint-60000/pytorch_model.bin +3 -0
tmp-spec-checkpoint-60000/special_tokens_map.json +30 -0
tmp-spec-checkpoint-60000/tokenizer.json +223 -0
tmp-spec-checkpoint-60000/tokenizer_config.json +43 -0
tmp-spec-checkpoint-60000/training_args.bin +3 -0

tmp-spec-checkpoint-60000/config.json ADDED Viewed

	@@ -0,0 +1,30 @@

+{
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "bos_token_id": 2,
+  "eos_token_id": 3,
+  "fused_bias_fc": false,
+  "fused_dropout_add_ln": false,
+  "fused_mlp": false,
+  "hidden_act": "silu",
+  "hidden_size": 512,
+  "initializer_range": 0.02,
+  "intermediate_size": 1024,
+  "max_position_embeddings": 2048,
+  "max_seq_length": 64,
+  "mlp_bias": false,
+  "model_type": "llama",
+  "num_attention_heads": 8,
+  "num_hidden_layers": 12,
+  "num_key_value_heads": 8,
+  "pretraining_tp": 1,
+  "residual_in_fp32": true,
+  "rms_norm_eps": 1e-06,
+  "rope_scaling": null,
+  "rope_theta": 10000.0,
+  "tie_word_embeddings": false,
+  "transformers_version": "4.43.4",
+  "use_cache": true,
+  "use_flash_attn": true,
+  "vocab_size": 104
+}

tmp-spec-checkpoint-60000/pytorch_model.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:d3e8b2e9cc6c80334dbd9a29dce9f5b9bd58eba56c68d18e2d3ee9152f657230
+size 63177295

tmp-spec-checkpoint-60000/special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,30 @@

+{
+  "bos_token": {
+    "content": "<bos>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "<eos>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "<pad>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "<unk>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

tmp-spec-checkpoint-60000/tokenizer.json ADDED Viewed

	@@ -0,0 +1,223 @@

+{
+  "version": "1.0",
+  "truncation": null,
+  "padding": null,
+  "added_tokens": [
+    {
+      "id": 0,
+      "content": "<unk>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 1,
+      "content": "<pad>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 2,
+      "content": "<bos>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 3,
+      "content": "<eos>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    }
+  ],
+  "normalizer": null,
+  "pre_tokenizer": {
+    "type": "Split",
+    "pattern": {
+      "Regex": "(\\[[^\\]]+]|Br?|Cl?|N|O|S|P|F|I|b|c|n|o|s|p|\\(|\\)|\\.|=|#|-|\\+|\\\\\\\\|\\/|:|~|@|\\?|>>?|\\*|\\$|\\%[0-9]{2}|[0-9])"
+    },
+    "behavior": "Isolated",
+    "invert": false
+  },
+  "post_processor": {
+    "type": "TemplateProcessing",
+    "single": [
+      {
+        "SpecialToken": {
+          "id": "<bos>",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      },
+      {
+        "SpecialToken": {
+          "id": "<eos>",
+          "type_id": 0
+        }
+      }
+    ],
+    "pair": [
+      {
+        "Sequence": {
+          "id": "A",
+          "type_id": 0
+        }
+      },
+      {
+        "Sequence": {
+          "id": "B",
+          "type_id": 1
+        }
+      }
+    ],
+    "special_tokens": {
+      "<bos>": {
+        "id": "<bos>",
+        "ids": [
+          2
+        ],
+        "tokens": [
+          "<bos>"
+        ]
+      },
+      "<eos>": {
+        "id": "<eos>",
+        "ids": [
+          3
+        ],
+        "tokens": [
+          "<eos>"
+        ]
+      }
+    }
+  },
+  "decoder": {
+    "type": "BPEDecoder",
+    "suffix": "</w>"
+  },
+  "model": {
+    "type": "WordLevel",
+    "vocab": {
+      "<unk>": 0,
+      "<pad>": 1,
+      "<bos>": 2,
+      "<eos>": 3,
+      "[C]": 4,
+      "[Ring1]": 5,
+      "[Branch1]": 6,
+      "[N]": 7,
+      "[=Branch1]": 8,
+      "[=C]": 9,
+      "[=O]": 10,
+      "[O]": 11,
+      "[Branch2]": 12,
+      "[Ring2]": 13,
+      "[C@H1]": 14,
+      "[C@@H1]": 15,
+      "[=N]": 16,
+      "[S]": 17,
+      "[F]": 18,
+      "[#Branch1]": 19,
+      "[=Branch2]": 20,
+      "[#C]": 21,
+      "[P]": 22,
+      "[#Branch2]": 23,
+      "[=Ring1]": 24,
+      "[Cl]": 25,
+      "[NH1]": 26,
+      "[C@]": 27,
+      "[C@@]": 28,
+      "[Br]": 29,
+      "[/C]": 30,
+      "[#N]": 31,
+      "[=Ring2]": 32,
+      "[O-1]": 33,
+      "[N+1]": 34,
+      "[=N+1]": 35,
+      "[I]": 36,
+      "[=N-1]": 37,
+      "[S@]": 38,
+      "[=S]": 39,
+      "[S@@]": 40,
+      "[N-1]": 41,
+      "[Si]": 42,
+      "[/C@H1]": 43,
+      "[/Cl]": 44,
+      "[/C@@H1]": 45,
+      "[S+1]": 46,
+      "[=S@]": 47,
+      "[=S@@]": 48,
+      "[B]": 49,
+      "[/Br]": 50,
+      "[/S]": 51,
+      "[P@]": 52,
+      "[/F]": 53,
+      "[P@@]": 54,
+      "[N@]": 55,
+      "[/N]": 56,
+      "[/O]": 57,
+      "[/N+1]": 58,
+      "[N@@]": 59,
+      "[=P]": 60,
+      "[/I]": 61,
+      "[B-1]": 62,
+      "[NH1+1]": 63,
+      "[N@@H1+1]": 64,
+      "[NH2+1]": 65,
+      "[N@H1+1]": 66,
+      "[OH0]": 67,
+      "[NH3+1]": 68,
+      "[PH1]": 69,
+      "[Si@]": 70,
+      "[Si@@]": 71,
+      "[/S@@]": 72,
+      "[=NH1+1]": 73,
+      "[N@+1]": 74,
+      "[/S@]": 75,
+      "[N@@+1]": 76,
+      "[/P]": 77,
+      "[Sn]": 78,
+      "[=Se]": 79,
+      ".": 80,
+      "[Cl-1]": 81,
+      "[#N+1]": 82,
+      "[=NH2+1]": 83,
+      "[/C@]": 84,
+      "[C-1]": 85,
+      "[=S+1]": 86,
+      "[CH0]": 87,
+      "[NH0]": 88,
+      "[=P@@]": 89,
+      "[S@@+1]": 90,
+      "[=NH0]": 91,
+      "[=P@]": 92,
+      "[/C@@]": 93,
+      "[/O-1]": 94,
+      "[=O+1]": 95,
+      "[Si@H1]": 96,
+      "[/Si]": 97,
+      "[=SH1]": 98,
+      "[O+1]": 99,
+      "[P+1]": 100,
+      "[P@@H1]": 101,
+      "[SH1]": 102,
+      "[Si@@H1]": 103
+    },
+    "unk_token": "<unk>"
+  }
+}

tmp-spec-checkpoint-60000/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,43 @@

+{
+  "added_tokens_decoder": {
+    "0": {
+      "content": "<unk>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<pad>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "<bos>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "3": {
+      "content": "<eos>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "<bos>",
+  "clean_up_tokenization_spaces": true,
+  "eos_token": "<eos>",
+  "model_max_length": 1000000000000000019884624838656,
+  "pad_token": "<pad>",
+  "tokenizer_class": "PreTrainedTokenizerFast",
+  "unk_token": "<unk>"
+}

tmp-spec-checkpoint-60000/training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:dd6a0ece896c7598dfe53dc759373ccab517bdc99b9d8cd508e8f0e1b9bad397
+size 6584