py : add Phi-1.5/Phi-2 tokenizer (#9361)

author daminho <redacted>

Thu, 12 Sep 2024 11:28:20 +0000 (20:28 +0900)

committer GitHub <redacted>

Thu, 12 Sep 2024 11:28:20 +0000 (14:28 +0300)
author daminho <redacted>
Thu, 12 Sep 2024 11:28:20 +0000 (20:28 +0900)
committer GitHub <redacted>
Thu, 12 Sep 2024 11:28:20 +0000 (14:28 +0300)
diff --git a/convert_hf_to_gguf.py b/convert_hf_to_gguf.py

index f02c65026e0a691b877f16c11d64b38a2c738229..01a8a50a27cc630fed36a89866b395393bd0b777 100755 (executable)
--- a/convert_hf_to_gguf.py
+++ b/convert_hf_to_gguf.py
@@ -626,6 +626,9 @@ class Model:
          if chkhsh == "4e2b24cc4770243d65a2c9ec19770a72f08cffc161adbb73fcbb6b7dd45a0aae":
              # ref: https://huggingface.co/LGAI-EXAONE/EXAONE-3.0-7.8B-Instruct
              res = "exaone"
+        if chkhsh == "fcace8b9cac38ce847670c970cd5892031a753a1ef381abd1d9af00f713da085":
+            # ref: https://huggingface.co/microsoft/phi-2
+            res = "phi-2"
  
          if res is None:
              logger.warning("\n")
diff --git a/convert_hf_to_gguf_update.py b/convert_hf_to_gguf_update.py

index 59a0b81a188807009694ccad2fe0eeeafb4fb26f..021f65abdc45d12f0f792f97d3a0be21e409c6fd 100755 (executable)
--- a/convert_hf_to_gguf_update.py
+++ b/convert_hf_to_gguf_update.py
@@ -98,6 +98,7 @@ models = [
      {'name': "bloom",          "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/bigscience/bloom", },
      {'name': "gpt3-finnish",   "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/TurkuNLP/gpt3-finnish-small", },
      {"name": "exaone",         "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/LGAI-EXAONE/EXAONE-3.0-7.8B-Instruct", },
+    {"name": "phi-2",          "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/microsoft/phi-2", },
  ]
author	daminho <redacted>
	Thu, 12 Sep 2024 11:28:20 +0000 (20:28 +0900)
committer	GitHub <redacted>
	Thu, 12 Sep 2024 11:28:20 +0000 (14:28 +0300)
convert_hf_to_gguf.py		patch \| blob \| history
convert_hf_to_gguf_update.py		patch \| blob \| history