vocab: add Falcon-H1-Tiny-Coder FIM tokens (#19249)

author Alexey Dubrov <redacted>

Tue, 3 Feb 2026 06:31:01 +0000 (09:31 +0300)

committer GitHub <redacted>

Tue, 3 Feb 2026 06:31:01 +0000 (08:31 +0200)
author Alexey Dubrov <redacted>
Tue, 3 Feb 2026 06:31:01 +0000 (09:31 +0300)
committer GitHub <redacted>
Tue, 3 Feb 2026 06:31:01 +0000 (08:31 +0200)
diff --git a/src/llama-vocab.cpp b/src/llama-vocab.cpp

index 74a8496f9e04bdadbbc243761e13b6c821c832d7..38d03a8c39bce7d780c73a7c0a855ec40a9ad1eb 100644 (file)
--- a/src/llama-vocab.cpp
+++ b/src/llama-vocab.cpp
@@ -2262,6 +2262,7 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
                          || t.first == "<PRE>"
                          || t.first == "▁<PRE>"          // CodeLlama
                          || t.first == "<|code_prefix|>" // GLM-4.5
+                        || t.first == "<|prefix|>"      // Falcon-H1-Tiny-Coder
                          ) {
                      special_fim_pre_id = t.second;
                      if ((attr & LLAMA_TOKEN_ATTR_CONTROL) == 0) {
@@ -2282,6 +2283,7 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
                          || t.first == "<SUF>"
                          || t.first == "▁<SUF>"         // CodeLlama
                          || t.first == "<|code_suffix|>" // GLM-4.5
+                        || t.first == "<|suffix|>"      // Falcon-H1-Tiny-Coder
                          ) {
                      special_fim_suf_id = t.second;
                      if ((attr & LLAMA_TOKEN_ATTR_CONTROL) == 0) {
@@ -2302,6 +2304,7 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
                          || t.first == "<MID>"
                          || t.first == "▁<MID>"         // CodeLlama
                          || t.first == "<|code_middle|>" // GLM-4.5
+                        || t.first == "<|middle|>"      // Falcon-H1-Tiny-Coder
                          ) {
                      special_fim_mid_id = t.second;
                      if ((attr & LLAMA_TOKEN_ATTR_CONTROL) == 0) {
author	Alexey Dubrov <redacted>
	Tue, 3 Feb 2026 06:31:01 +0000 (09:31 +0300)
committer	GitHub <redacted>
	Tue, 3 Feb 2026 06:31:01 +0000 (08:31 +0200)