Revert PR 56 (#65)

- Revert "HF format chat template (#56)" (0dcbcc17e6493f61789c1379d9375f363d0c96ec)

Co-authored-by: Weijian Xu <[email protected]>

Files changed (5) hide show

chat_template.json DELETED Viewed

@@ -1,3 +0,0 @@
-{
-  "chat_template": "{% for message in messages %}{{ '<|' + message['role'] + '|>' }}{% if message['content'] is string %}{{ message['content'] }}{% else %}{% for content in message['content'] %}{% if content['type'] == 'image' %}{{ '<|image|>' }}{% elif content['type'] == 'audio' %}{{ '<|audio|>' }}{% elif content['type'] == 'text' %}{{ content['text'] }}{% endif %}{% endfor %}{% endif %}{% if message['role'] == 'system' and 'tools' in message and message['tools'] is not none %}{{ '<|tool|>' + message['tools'] + '<|/tool|>' + '<|end|>' }}{% endif %}{{ '<|end|>' }}{% endfor %}{% if add_generation_prompt %}{{ '<|assistant|>' }}{% else %}{{ eos_token }}{% endif %}"
-}

processor_config.json ADDED Viewed

+{
+  "auto_map": {
+    "AutoProcessor": "processing_phi4mm.Phi4MMProcessor"
+  },
+  "processor_class": "Phi4MMProcessor"
+}

tokenizer.json CHANGED Viewed

@@ -1,3 +1,3 @@
 version https://git-lfs.github.com/spec/v1
-oid sha256:94b9c8d751cbf0af78f9b03fea0ed6ea3e12a89d2c39c4ebb5bb10ce0a0310f3
-size 15524455

 version https://git-lfs.github.com/spec/v1
+oid sha256:4c1b9f641d4f8b7247b8d5007dd3b6a9f6a87cb5123134fe0d326f14d10c0585
+size 15524479

tokenizer_config.json CHANGED Viewed

@@ -2,7 +2,7 @@
   "add_prefix_space": false,
   "added_tokens_decoder": {
     "200010": {
-      "content": "<|image|>",
       "lstrip": false,
       "normalized": false,
       "rstrip": false,
@@ -10,7 +10,15 @@
       "special": true
     },
     "200011": {
-      "content": "<|audio|>",
       "lstrip": false,
       "normalized": false,
       "rstrip": false,
@@ -106,16 +114,10 @@
       "special": true
     }
   },
-  "audio_token": "<|audio|>",
   "bos_token": "<|endoftext|>",
   "chat_template": "{% for message in messages %}{% if message['role'] == 'system' and 'tools' in message and message['tools'] is not none %}{{ '<|' + message['role'] + '|>' + message['content'] + '<|tool|>' + message['tools'] + '<|/tool|>' + '<|end|>' }}{% else %}{{ '<|' + message['role'] + '|>' + message['content'] + '<|end|>' }}{% endif %}{% endfor %}{% if add_generation_prompt %}{{ '<|assistant|>' }}{% else %}{{ eos_token }}{% endif %}",
   "clean_up_tokenization_spaces": false,
   "eos_token": "<|endoftext|>",
-  "extra_special_tokens": {
-    "audio_token": "<|audio|>",
-    "image_token": "<|image|>"
-  },
-  "image_token": "<|image|>",
   "model_max_length": 131072,
   "pad_token": "<|endoftext|>",
   "tokenizer_class": "GPT2TokenizerFast",

   "add_prefix_space": false,
   "added_tokens_decoder": {
     "200010": {
+      "content": "<|endoftext10|>",
       "lstrip": false,
       "normalized": false,
       "rstrip": false,
       "special": true
     },
     "200011": {
+      "content": "<|endoftext11|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "199999": {
+      "content": "<|endoftext|>",
       "lstrip": false,
       "normalized": false,
       "rstrip": false,
       "special": true
     }
   },
   "bos_token": "<|endoftext|>",
   "chat_template": "{% for message in messages %}{% if message['role'] == 'system' and 'tools' in message and message['tools'] is not none %}{{ '<|' + message['role'] + '|>' + message['content'] + '<|tool|>' + message['tools'] + '<|/tool|>' + '<|end|>' }}{% else %}{{ '<|' + message['role'] + '|>' + message['content'] + '<|end|>' }}{% endif %}{% endfor %}{% if add_generation_prompt %}{{ '<|assistant|>' }}{% else %}{{ eos_token }}{% endif %}",
   "clean_up_tokenization_spaces": false,
   "eos_token": "<|endoftext|>",
   "model_max_length": 131072,
   "pad_token": "<|endoftext|>",
   "tokenizer_class": "GPT2TokenizerFast",

vocab.json CHANGED Viewed

The diff for this file is too large to render. See raw diff