Upload 4 files

Files changed (4) hide show

added_tokens.json ADDED Viewed

+{
+  "<|im_end|>": 49153,
+  "<|im_start|>": 49152
+}

special_tokens_map.json CHANGED Viewed

@@ -1,58 +1,23 @@
 {
   "additional_special_tokens": [
-    "<|endoftext|>",
-    "<fim_prefix>",
-    "<fim_middle>",
-    "<fim_suffix>",
-    "<fim_pad>",
-    "<repo_name>",
-    "<file_sep>",
-    "<issue_start>",
-    "<issue_comment>",
-    "<issue_closed>",
-    "<jupyter_start>",
-    "<jupyter_text>",
-    "<jupyter_code>",
-    "<jupyter_output>",
-    "<jupyter_script>",
-    "<empty_output>",
-    "<code_to_intermediate>",
-    "<intermediate_to_code>",
-    "<pr>",
-    "<pr_status>",
-    "<pr_is_merged>",
-    "<pr_base>",
-    "<pr_file>",
-    "<pr_base_code>",
-    "<pr_diff>",
-    "<pr_diff_hunk>",
-    "<pr_comment>",
-    "<pr_event_id>",
-    "<pr_review>",
-    "<pr_review_state>",
-    "<pr_review_comment>",
-    "<pr_in_reply_to_review_id>",
-    "<pr_in_reply_to_comment_id>",
-    "<pr_diff_hunk_comment_line>",
-    "<NAME>",
-    "<EMAIL>",
-    "<KEY>",
-    "<PASSWORD>"
   ],
-  "bos_token": {
-    "content": "<|endoftext|>",
-    "lstrip": false,
-    "normalized": false,
-    "rstrip": false,
-    "single_word": false
-  },
-  "eos_token": {
-    "content": "<|endoftext|>",
-    "lstrip": false,
-    "normalized": false,
-    "rstrip": false,
-    "single_word": false
-  },
   "unk_token": {
     "content": "<|endoftext|>",
     "lstrip": false,

 {
   "additional_special_tokens": [
+    {
+      "content": "<|im_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "<|im_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    }
   ],
+  "bos_token": "<|im_start|>",
+  "eos_token": "<|im_end|>",
+  "pad_token": "<|im_end|>",
   "unk_token": {
     "content": "<|endoftext|>",
     "lstrip": false,

tokenizer.json CHANGED Viewed

@@ -344,6 +344,24 @@
       "rstrip": false,
       "normalized": false,
       "special": true
     }
   ],
   "normalizer": null,

       "rstrip": false,
       "normalized": false,
       "special": true
+    },
+    {
+      "id": 49152,
+      "content": "<|im_start|>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
+    },
+    {
+      "id": 49153,
+      "content": "<|im_end|>",
+      "single_word": false,
+      "lstrip": false,
+      "rstrip": false,
+      "normalized": false,
+      "special": true
     }
   ],
   "normalizer": null,

tokenizer_config.json CHANGED Viewed

@@ -304,52 +304,34 @@
       "rstrip": false,
       "single_word": false,
       "special": true
     }
   },
   "additional_special_tokens": [
-    "<|endoftext|>",
-    "<fim_prefix>",
-    "<fim_middle>",
-    "<fim_suffix>",
-    "<fim_pad>",
-    "<repo_name>",
-    "<file_sep>",
-    "<issue_start>",
-    "<issue_comment>",
-    "<issue_closed>",
-    "<jupyter_start>",
-    "<jupyter_text>",
-    "<jupyter_code>",
-    "<jupyter_output>",
-    "<jupyter_script>",
-    "<empty_output>",
-    "<code_to_intermediate>",
-    "<intermediate_to_code>",
-    "<pr>",
-    "<pr_status>",
-    "<pr_is_merged>",
-    "<pr_base>",
-    "<pr_file>",
-    "<pr_base_code>",
-    "<pr_diff>",
-    "<pr_diff_hunk>",
-    "<pr_comment>",
-    "<pr_event_id>",
-    "<pr_review>",
-    "<pr_review_state>",
-    "<pr_review_comment>",
-    "<pr_in_reply_to_review_id>",
-    "<pr_in_reply_to_comment_id>",
-    "<pr_diff_hunk_comment_line>",
-    "<NAME>",
-    "<EMAIL>",
-    "<KEY>",
-    "<PASSWORD>"
   ],
-  "bos_token": "<|endoftext|>",
   "clean_up_tokenization_spaces": true,
-  "eos_token": "<|endoftext|>",
   "model_max_length": 1000000000000000019884624838656,
   "tokenizer_class": "GPT2Tokenizer",
   "unk_token": "<|endoftext|>",
   "vocab_size": 49152

       "rstrip": false,
       "single_word": false,
       "special": true
+    },
+    "49152": {
+      "content": "<|im_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "49153": {
+      "content": "<|im_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
     }
   },
   "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>"
   ],
+  "bos_token": "<|im_start|>",
+  "chat_template": "{% for message in messages %}{{'<|im_start|>' + message['role'] + '\n' + message['content'] + '<|im_end|>' + '\n'}}{% endfor %}{% if add_generation_prompt %}{{ '<|im_start|>assistant\n' }}{% endif %}",
   "clean_up_tokenization_spaces": true,
+  "eos_token": "<|im_end|>",
   "model_max_length": 1000000000000000019884624838656,
+  "pad_token": "<|im_end|>",
   "tokenizer_class": "GPT2Tokenizer",
   "unk_token": "<|endoftext|>",
   "vocab_size": 49152