Add StableLM2 pre-tokenizer (#7349)

author Anas Ahouzi <redacted>

Sun, 19 May 2024 12:46:46 +0000 (14:46 +0200)

committer GitHub <redacted>

Sun, 19 May 2024 12:46:46 +0000 (22:46 +1000)
author Anas Ahouzi <redacted>
Sun, 19 May 2024 12:46:46 +0000 (14:46 +0200)
committer GitHub <redacted>
Sun, 19 May 2024 12:46:46 +0000 (22:46 +1000)
diff --git a/convert-hf-to-gguf-update.py b/convert-hf-to-gguf-update.py

index 27983fadf4ac522081da6f655486d213b64dbe99..45404b32b75ae6843ddf05779cbfe389509924b0 100755 (executable)
--- a/convert-hf-to-gguf-update.py
+++ b/convert-hf-to-gguf-update.py
@@ -72,6 +72,7 @@ models = [
      {"name": "mpt",            "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/mosaicml/mpt-7b", },
      {"name": "starcoder",      "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/bigcode/starcoder2-3b", },
      {"name": "gpt-2",          "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/openai-community/gpt2", },
+    {"name": "stablelm",       "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/stabilityai/stablelm-2-zephyr-1_6b", },
      {"name": "refact",         "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/smallcloudai/Refact-1_6-base", },
      {"name": "command-r",      "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/CohereForAI/c4ai-command-r-v01", },
      {"name": "qwen2",          "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/Qwen/Qwen1.5-7B", },
diff --git a/convert-hf-to-gguf.py b/convert-hf-to-gguf.py

index cd1750aa3f3ba27d605ec74876b16426e39d14ca..bd303150ae6b9da7b6b8eb64eaeb15b59da47901 100755 (executable)
--- a/convert-hf-to-gguf.py
+++ b/convert-hf-to-gguf.py
@@ -446,6 +446,9 @@ class Model:
          if chkhsh == "3ce83efda5659b07b1ad37ca97ca5797ea4285d9b9ab0dc679e4a720c9da7454":
              # ref: https://huggingface.co/openai-community/gpt2
              res = "gpt-2"
+        if chkhsh == "32d85c31273f8019248f2559fed492d929ea28b17e51d81d3bb36fff23ca72b3":
+            # ref: https://huggingface.co/stabilityai/stablelm-2-1_6b
+            res = "stablelm2"
          if chkhsh == "6221ad2852e85ce96f791f476e0b390cf9b474c9e3d1362f53a24a06dc8220ff":
              # ref: https://huggingface.co/smallcloudai/Refact-1_6-base
              res = "refact"
diff --git a/llama.cpp b/llama.cpp

index 1409a05da532837ab949fdd684652c939a6e9121..06ff4da617f5dc6c52ff2687d3138f5a8b5dc547 100644 (file)
--- a/llama.cpp
+++ b/llama.cpp
@@ -4463,6 +4463,9 @@ static void llm_load_vocab(
              } else if (
                  tokenizer_pre == "qwen2") {
                  vocab.type_pre = LLAMA_VOCAB_PRE_TYPE_QWEN2;
+            } else if (
+                tokenizer_pre == "stablelm2") {
+                vocab.type_pre = LLAMA_VOCAB_PRE_TYPE_STABLELM2;
              } else if (
                  tokenizer_pre == "olmo") {
                  vocab.type_pre = LLAMA_VOCAB_PRE_TYPE_OLMO;
@@ -12363,6 +12366,7 @@ struct llm_tokenizer_bpe {
                              "'s|'t|'re|'ve|'m|'ll|'d| ?\\p{L}+| ?\\p{N}+| ?[^\\s\\p{L}\\p{N}]+|\\s+(?!\\S)",
                          });
                          break;
+                    case LLAMA_VOCAB_PRE_TYPE_STABLELM2:
                      case LLAMA_VOCAB_PRE_TYPE_QWEN2:
                          word_collection = unicode_regex_split(text, {
                              // original regex from tokenizer.json
diff --git a/llama.h b/llama.h

index 612e32c4ea0587dacd2001a2b53ec411ccc02cad..b7bf2afcb403e0824e4d0594a8c6646e1775247d 100644 (file)
--- a/llama.h
+++ b/llama.h
@@ -81,9 +81,10 @@ extern "C" {
          LLAMA_VOCAB_PRE_TYPE_GPT2           = 7,
          LLAMA_VOCAB_PRE_TYPE_REFACT         = 8,
          LLAMA_VOCAB_PRE_TYPE_COMMAND_R      = 9,
-        LLAMA_VOCAB_PRE_TYPE_QWEN2          = 10,
-        LLAMA_VOCAB_PRE_TYPE_OLMO           = 11,
-        LLAMA_VOCAB_PRE_TYPE_DBRX           = 12,
+        LLAMA_VOCAB_PRE_TYPE_STABLELM2      = 10,
+        LLAMA_VOCAB_PRE_TYPE_QWEN2          = 11,
+        LLAMA_VOCAB_PRE_TYPE_OLMO           = 12,
+        LLAMA_VOCAB_PRE_TYPE_DBRX           = 13,
      };
  
      // note: these values should be synchronized with ggml_rope
author	Anas Ahouzi <redacted>
	Sun, 19 May 2024 12:46:46 +0000 (14:46 +0200)
committer	GitHub <redacted>
	Sun, 19 May 2024 12:46:46 +0000 (22:46 +1000)
convert-hf-to-gguf-update.py		patch \| blob \| history
convert-hf-to-gguf.py		patch \| blob \| history
llama.cpp		patch \| blob \| history
llama.h		patch \| blob \| history