server : allow to specify tokens as strings in logit_bias (#5003)

author Alexey Parfenov <redacted>

Sun, 11 Feb 2024 13:38:14 +0000 (13:38 +0000)

committer GitHub <redacted>

Sun, 11 Feb 2024 13:38:14 +0000 (15:38 +0200)
author Alexey Parfenov <redacted>
Sun, 11 Feb 2024 13:38:14 +0000 (13:38 +0000)
committer GitHub <redacted>
Sun, 11 Feb 2024 13:38:14 +0000 (15:38 +0200)
diff --git a/examples/server/README.md b/examples/server/README.md

index 1db7cdf2191a7a26baae77d5ce921fdc772446fd..0f7373ae86204aa3b232d5b555ab9ffb526cc5d8 100644 (file)
--- a/examples/server/README.md
+++ b/examples/server/README.md
@@ -185,7 +185,7 @@ node index.js
  
      `ignore_eos`: Ignore end of stream token and continue generating (default: false).
  
-    `logit_bias`: Modify the likelihood of a token appearing in the generated text completion. For example, use `"logit_bias": [[15043,1.0]]` to increase the likelihood of the token 'Hello', or `"logit_bias": [[15043,-1.0]]` to decrease its likelihood. Setting the value to false, `"logit_bias": [[15043,false]]` ensures that the token `Hello` is never produced (default: []).
+    `logit_bias`: Modify the likelihood of a token appearing in the generated text completion. For example, use `"logit_bias": [[15043,1.0]]` to increase the likelihood of the token 'Hello', or `"logit_bias": [[15043,-1.0]]` to decrease its likelihood. Setting the value to false, `"logit_bias": [[15043,false]]` ensures that the token `Hello` is never produced. The tokens can also be represented as strings, e.g. `[["Hello, World!",-0.5]]` will reduce the likelihood of all the individual tokens that represent the string `Hello, World!`, just like the `presence_penalty` does. (default: []).
  
      `n_probs`: If greater than 0, the response also contains the probabilities of top N tokens for each generated token (default: 0)
  
diff --git a/examples/server/server.cpp b/examples/server/server.cpp

index 4d212f1f0e65cafcba10a65df38443a7e2ac6123..1699eb76b87404b4fe0e72b4eaa215793910f9d8 100644 (file)
--- a/examples/server/server.cpp
+++ b/examples/server/server.cpp
@@ -626,18 +626,36 @@ struct llama_server_context
              const int n_vocab = llama_n_vocab(model);
              for (const auto &el : *logit_bias)
              {
-                if (el.is_array() && el.size() == 2 && el[0].is_number_integer())
+                if (el.is_array() && el.size() == 2)
                  {
-                    llama_token tok = el[0].get<llama_token>();
-                    if (tok >= 0 && tok < n_vocab)
+                    float bias;
+                    if (el[1].is_number())
                      {
-                        if (el[1].is_number())
+                        bias = el[1].get<float>();
+                    }
+                    else if (el[1].is_boolean() && !el[1].get<bool>())
+                    {
+                        bias = -INFINITY;
+                    }
+                    else
+                    {
+                        continue;
+                    }
+
+                    if (el[0].is_number_integer())
+                    {
+                        llama_token tok = el[0].get<llama_token>();
+                        if (tok >= 0 && tok < n_vocab)
                          {
-                            slot->sparams.logit_bias[tok] = el[1].get<float>();
+                            slot->sparams.logit_bias[tok] = bias;
                          }
-                        else if (el[1].is_boolean() && !el[1].get<bool>())
+                    }
+                    else if (el[0].is_string())
+                    {
+                        auto toks = llama_tokenize(model, el[0].get<std::string>(), false);
+                        for (auto tok : toks)
                          {
-                            slot->sparams.logit_bias[tok] = -INFINITY;
+                            slot->sparams.logit_bias[tok] = bias;
                          }
                      }
                  }
author	Alexey Parfenov <redacted>
	Sun, 11 Feb 2024 13:38:14 +0000 (13:38 +0000)
committer	GitHub <redacted>
	Sun, 11 Feb 2024 13:38:14 +0000 (15:38 +0200)
examples/server/README.md		patch \| blob \| history
examples/server/server.cpp		patch \| blob \| history