Upload folder using huggingface_hub

Browse files

Files changed (7) hide show

README.md +142 -0
config.json +41 -0
generation_config.json +11 -0
model.safetensors +3 -0
special_tokens_map.json +30 -0
tokenizer.json +0 -0
tokenizer_config.json +132 -0

README.md ADDED Viewed

	@@ -0,0 +1,142 @@

+---
+library_name: transformers
+pipeline_tag: text-generation
+inference: true
+widget:
+  - text: Hello!
+    example_title: Hello world
+    group: Python
+base_model:
+- microsoft/Phi-tiny-MoE-instruct
+---
+This tiny model is for debugging. It is randomly initialized with the config adapted from [microsoft/Phi-tiny-MoE-instruct](https://huggingface.co/microsoft/Phi-tiny-MoE-instruct).
+### Example usage:
+```python
+import torch
+from transformers import AutoModelForCausalLM, AutoTokenizer, pipeline
+model_id = "tiny-random/phi-moe"
+tokenizer = AutoTokenizer.from_pretrained(model_id, trust_remote_code=True)
+model = AutoModelForCausalLM.from_pretrained(
+    model_id,
+    torch_dtype=torch.bfloat16,
+    trust_remote_code=True,
+)
+pipe = pipeline('text-generation', model=model, tokenizer=tokenizer, trust_remote_code=True)
+print(pipe('Write an article about Artificial Intelligence.'))
+```
+### Codes to create this repo:
+```python
+import json
+from pathlib import Path
+import torch
+import accelerate
+from huggingface_hub import file_exists, hf_hub_download
+from transformers import (
+    AutoConfig,
+    AutoModelForCausalLM,
+    AutoTokenizer,
+    GenerationConfig,
+    set_seed,
+)
+source_model_id = "microsoft/Phi-tiny-MoE-instruct"
+save_folder = "/tmp/tiny-random/phi-moe"
+processor = AutoTokenizer.from_pretrained(source_model_id)
+processor.save_pretrained(save_folder)
+with open(hf_hub_download(source_model_id, filename='config.json', repo_type='model'), 'r', encoding='utf-8') as f:
+    config_json = json.load(f)
+for k, v in config_json['auto_map'].items():
+    config_json['auto_map'][k] = f'{source_model_id}--{v}'
+config_json['head_dim'] = 32
+config_json['hidden_size'] = 64
+config_json['intermediate_size'] = 128
+config_json['num_attention_heads'] = 2
+config_json['num_experts_per_tok'] = 2
+config_json['num_hidden_layers'] = 2
+config_json['num_key_value_heads'] = 1
+config_json['num_local_experts'] = 8
+config_json['tie_word_embeddings'] = True
+with open(f"{save_folder}/config.json", "w", encoding='utf-8') as f:
+    json.dump(config_json, f, indent=2)
+config = AutoConfig.from_pretrained(
+    save_folder,
+    trust_remote_code=True,
+)
+print(config)
+automap = config_json['auto_map']
+torch.set_default_dtype(torch.bfloat16)
+model = AutoModelForCausalLM.from_config(config, trust_remote_code=True)
+torch.set_default_dtype(torch.float32)
+# according to source model, gat is in FP32
+for i in range(config.num_hidden_layers):
+    model.model.layers[i].block_sparse_moe.gate.float()
+if file_exists(filename="generation_config.json", repo_id=source_model_id, repo_type='model'):
+    model.generation_config = GenerationConfig.from_pretrained(
+        source_model_id, trust_remote_code=True,
+    )
+set_seed(42)
+model = model.cpu()  # cpu is more stable for random initialization across machines
+with torch.no_grad():
+    for name, p in sorted(model.named_parameters()):
+        torch.nn.init.normal_(p, 0, 0.2)
+        print(name, p.shape)
+model.save_pretrained(save_folder)
+print(model)
+with open(f"{save_folder}/config.json", "r", encoding='utf-8') as f:
+    config_json = json.load(f)
+    config_json['auto_map'] = automap
+with open(f"{save_folder}/config.json", "w", encoding='utf-8') as f:
+    json.dump(config_json, f, indent=2)
+for python_file in Path(save_folder).glob('*.py'):
+    python_file.unlink()
+```
+### Printing the model:
+```text
+PhiMoEForCausalLM(
+  (model): PhiMoEModel(
+    (embed_tokens): Embedding(32064, 64)
+    (layers): ModuleList(
+      (0-1): 2 x PhiMoEDecoderLayer(
+        (self_attn): PhiMoESdpaAttention(
+          (q_proj): Linear(in_features=64, out_features=64, bias=True)
+          (k_proj): Linear(in_features=64, out_features=32, bias=True)
+          (v_proj): Linear(in_features=64, out_features=32, bias=True)
+          (o_proj): Linear(in_features=64, out_features=64, bias=True)
+          (rotary_emb): PhiMoERotaryEmbedding()
+        )
+        (block_sparse_moe): PhiMoESparseMoeBlock(
+          (gate): Linear(in_features=64, out_features=8, bias=False)
+          (experts): ModuleList(
+            (0-7): 8 x PhiMoEBlockSparseTop2MLP(
+              (w1): Linear(in_features=64, out_features=128, bias=False)
+              (w2): Linear(in_features=128, out_features=64, bias=False)
+              (w3): Linear(in_features=64, out_features=128, bias=False)
+              (act_fn): SiLU()
+            )
+          )
+        )
+        (input_layernorm): LayerNorm((64,), eps=1e-05, elementwise_affine=True)
+        (post_attention_layernorm): LayerNorm((64,), eps=1e-05, elementwise_affine=True)
+      )
+    )
+    (norm): LayerNorm((64,), eps=1e-05, elementwise_affine=True)
+  )
+  (lm_head): Linear(in_features=64, out_features=32064, bias=True)
+)
+```

config.json ADDED Viewed

	@@ -0,0 +1,41 @@

+{
+  "architectures": [
+    "PhiMoEForCausalLM"
+  ],
+  "attention_bias": true,
+  "attention_dropout": 0.0,
+  "auto_map": {
+    "AutoConfig": "microsoft/Phi-tiny-MoE-instruct--configuration_slimmoe.PhiMoEConfig",
+    "AutoModelForCausalLM": "microsoft/Phi-tiny-MoE-instruct--modeling_slimmoe.PhiMoEForCausalLM"
+  },
+  "bos_token_id": 1,
+  "eos_token_id": 32000,
+  "expert_dropout": 0.0,
+  "head_dim": 32,
+  "hidden_act": "silu",
+  "hidden_dropout": 0.0,
+  "hidden_size": 64,
+  "initializer_range": 0.02,
+  "input_jitter_noise": 0.01,
+  "intermediate_size": 128,
+  "lm_head_bias": true,
+  "max_position_embeddings": 4096,
+  "model_type": "phimoe",
+  "num_attention_heads": 2,
+  "num_experts_per_tok": 2,
+  "num_hidden_layers": 2,
+  "num_key_value_heads": 1,
+  "num_local_experts": 8,
+  "output_router_logits": false,
+  "rms_norm_eps": 1e-05,
+  "rope_scaling": null,
+  "rope_theta": 10000.0,
+  "router_aux_loss_coef": 0.0,
+  "router_jitter_noise": 0.01,
+  "sliding_window": 2047,
+  "tie_word_embeddings": true,
+  "torch_dtype": "bfloat16",
+  "transformers_version": "4.51.3",
+  "use_cache": true,
+  "vocab_size": 32064
+}

generation_config.json ADDED Viewed

	@@ -0,0 +1,11 @@

+{
+  "_from_model_config": true,
+  "bos_token_id": 1,
+  "eos_token_id": [
+    32000,
+    32001,
+    32007
+  ],
+  "pad_token_id": 32000,
+  "transformers_version": "4.51.3"
+}

model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:bdf2f1fe515fa0b55cc3be523a7e19a36d27053254bfcdf6f0ecada43c393bcd
+size 5018960

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,30 @@

+{
+  "bos_token": {
+    "content": "<s>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "<unk>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,132 @@

+{
+  "add_bos_token": false,
+  "add_eos_token": false,
+  "add_prefix_space": null,
+  "added_tokens_decoder": {
+    "0": {
+      "content": "<unk>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "</s>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": true,
+      "single_word": false,
+      "special": false
+    },
+    "32000": {
+      "content": "<|endoftext|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "32001": {
+      "content": "<|assistant|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": true,
+      "single_word": false,
+      "special": true
+    },
+    "32002": {
+      "content": "<|placeholder1|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": true,
+      "single_word": false,
+      "special": true
+    },
+    "32003": {
+      "content": "<|placeholder2|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": true,
+      "single_word": false,
+      "special": true
+    },
+    "32004": {
+      "content": "<|placeholder3|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": true,
+      "single_word": false,
+      "special": true
+    },
+    "32005": {
+      "content": "<|placeholder4|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": true,
+      "single_word": false,
+      "special": true
+    },
+    "32006": {
+      "content": "<|system|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": true,
+      "single_word": false,
+      "special": true
+    },
+    "32007": {
+      "content": "<|end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": true,
+      "single_word": false,
+      "special": true
+    },
+    "32008": {
+      "content": "<|placeholder5|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": true,
+      "single_word": false,
+      "special": true
+    },
+    "32009": {
+      "content": "<|placeholder6|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": true,
+      "single_word": false,
+      "special": true
+    },
+    "32010": {
+      "content": "<|user|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": true,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "<s>",
+  "chat_template": "{% for message in messages %}{{'<|' + message['role'] + '|>' + '\n' + message['content'] + '<|end|>\n' }}{% endfor %}{% if add_generation_prompt %}{{ '<|assistant|>\n' }}{% else %}{{ eos_token }}{% endif %}",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<|endoftext|>",
+  "extra_special_tokens": {},
+  "legacy": false,
+  "model_max_length": 4096,
+  "pad_token": "<|endoftext|>",
+  "padding_side": "left",
+  "sp_model_kwargs": {},
+  "tokenizer_class": "LlamaTokenizerFast",
+  "unk_token": "<unk>",
+  "use_default_system_prompt": false
+}