Add llama instruct model weights with wasm conf

Files changed (6) hide show

smollm2-135m-qna-v1-q4f16_1-MLC/added_tokens.json +3 -0
smollm2-135m-qna-v1-q4f16_1-MLC/merges.txt +0 -0
smollm2-135m-qna-v1-q4f16_1-MLC/mlc-chat-config.json +81 -0
smollm2-135m-qna-v1-q4f16_1-MLC/tokenizer.json +0 -0
smollm2-135m-qna-v1-q4f16_1-MLC/tokenizer_config.json +166 -0
smollm2-135m-qna-v1-q4f16_1-MLC/vocab.json +0 -0

smollm2-135m-qna-v1-q4f16_1-MLC/added_tokens.json ADDED Viewed

	@@ -0,0 +1,3 @@

+{
+  "<|PAD_TOKEN|>": 49152
+}

smollm2-135m-qna-v1-q4f16_1-MLC/merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

smollm2-135m-qna-v1-q4f16_1-MLC/mlc-chat-config.json ADDED Viewed

	@@ -0,0 +1,81 @@

+{
+  "version": "0.1.0",
+  "model_type": "llama",
+  "quantization": "q4f16_1",
+  "model_config": {
+    "hidden_size": 576,
+    "intermediate_size": 1536,
+    "num_attention_heads": 9,
+    "num_hidden_layers": 30,
+    "rms_norm_eps": 1e-05,
+    "vocab_size": 49153,
+    "tie_word_embeddings": true,
+    "position_embedding_base": 100000,
+    "rope_scaling": null,
+    "context_window_size": 8192,
+    "prefill_chunk_size": 8192,
+    "num_key_value_heads": 3,
+    "head_dim": 64,
+    "tensor_parallel_shards": 1,
+    "pipeline_parallel_stages": 1,
+    "max_batch_size": 128,
+    "disaggregation": false
+  },
+  "vocab_size": 49153,
+  "context_window_size": 8192,
+  "sliding_window_size": -1,
+  "prefill_chunk_size": 8192,
+  "attention_sink_size": -1,
+  "tensor_parallel_shards": 1,
+  "pipeline_parallel_stages": 1,
+  "temperature": 1.0,
+  "presence_penalty": 0.0,
+  "frequency_penalty": 0.0,
+  "repetition_penalty": 1.0,
+  "top_p": 1.0,
+  "tokenizer_files": [
+    "tokenizer.json",
+    "vocab.json",
+    "merges.txt",
+    "added_tokens.json",
+    "tokenizer_config.json"
+  ],
+  "tokenizer_info": {
+    "token_postproc_method": "byte_level",
+    "prepend_space_in_encode": false,
+    "strip_space_in_decode": false
+  },
+  "conv_template": {
+    "name": "chatml",
+    "system_template": "<|im_start|>system\n{system_message}<|im_end|>\n",
+    "system_message": "A conversation between a user and an LLM-based AI assistant. The assistant gives helpful and honest answers.",
+    "system_prefix_token_ids": null,
+    "add_role_after_system_message": true,
+    "roles": {
+      "user": "<|im_start|>user",
+      "assistant": "<|im_start|>assistant"
+    },
+    "role_templates": {
+      "user": "{user_message}",
+      "assistant": "{assistant_message}",
+      "tool": "{tool_message}"
+    },
+    "messages": [],
+    "seps": [
+      "<|im_end|>\n"
+    ],
+    "role_content_sep": "\n",
+    "role_empty_sep": "\n",
+    "stop_str": [
+      "<|im_end|>"
+    ],
+    "stop_token_ids": [
+      2
+    ],
+    "function_string": "",
+    "use_function_calling": false
+  },
+  "pad_token_id": 49152,
+  "bos_token_id": 0,
+  "eos_token_id": 0
+}

smollm2-135m-qna-v1-q4f16_1-MLC/tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

smollm2-135m-qna-v1-q4f16_1-MLC/tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,166 @@

+{
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "0": {
+      "content": "<|im_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<|im_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "<|endoftext|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "3": {
+      "content": "<repo_name>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "4": {
+      "content": "<reponame>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "5": {
+      "content": "<file_sep>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "6": {
+      "content": "<filename>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "7": {
+      "content": "<gh_stars>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "8": {
+      "content": "<issue_start>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "9": {
+      "content": "<issue_comment>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "10": {
+      "content": "<issue_closed>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "11": {
+      "content": "<jupyter_start>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "12": {
+      "content": "<jupyter_text>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "13": {
+      "content": "<jupyter_code>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "14": {
+      "content": "<jupyter_output>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "15": {
+      "content": "<jupyter_script>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "16": {
+      "content": "<empty_output>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "24211": {
+      "content": "ï¿½",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "49152": {
+      "content": "<|PAD_TOKEN|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "<|endoftext|>",
+  "chat_template": "{% for message in messages %}{% if message['role'] == 'user' %}{{'<|im_start|>user\n' + message['content'] + '<|im_end|>\n'}}{% elif message['role'] == 'assistant' %}{{'<|im_start|>assistant\n' + message['content'] + '<|im_end|>\n' }}{% else %}{{ '<|im_start|>system\n' + message['content'] + '<|im_end|>\n' }}{% endif %}{% endfor %}{% if add_generation_prompt %}{{ '<|im_start|>assistant\n' }}{% endif %}",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<|im_end|>",
+  "extra_special_tokens": {},
+  "model_max_length": 1000000000000000019884624838656,
+  "pad_token": "<|PAD_TOKEN|>",
+  "tokenizer_class": "GPT2Tokenizer",
+  "unk_token": "ï¿½"
+}

smollm2-135m-qna-v1-q4f16_1-MLC/vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff