server: allow to specify tokens as strings in logit_bias

z80maniac · z80maniac · commit 958660b9c0e7 · 2024-01-17T20:11:41.000+03:00
diff --git a/examples/server/README.md b/examples/server/README.md
@@ -168,7 +168,7 @@ node index.js
 
     `ignore_eos`: Ignore end of stream token and continue generating (default: false).
 
-    `logit_bias`: Modify the likelihood of a token appearing in the generated text completion. For example, use `"logit_bias": [[15043,1.0]]` to increase the likelihood of the token 'Hello', or `"logit_bias": [[15043,-1.0]]` to decrease its likelihood. Setting the value to false, `"logit_bias": [[15043,false]]` ensures that the token `Hello` is never produced (default: []).
+    `logit_bias`: Modify the likelihood of a token appearing in the generated text completion. For example, use `"logit_bias": [[15043,1.0]]` to increase the likelihood of the token 'Hello', or `"logit_bias": [[15043,-1.0]]` to decrease its likelihood. Setting the value to false, `"logit_bias": [[15043,false]]` ensures that the token `Hello` is never produced. The tokens can also be represented as strings, e.g. `[["Hello, World!",false]]` will ban all tokens that represent the string `Hello, World!`. (default: []).
 
     `n_probs`: If greater than 0, the response also contains the probabilities of top N tokens for each generated token (default: 0)
 
diff --git a/examples/server/server.cpp b/examples/server/server.cpp
@@ -828,18 +828,36 @@ struct llama_server_context
             const int n_vocab = llama_n_vocab(model);
             for (const auto &el : *logit_bias)
             {
-                if (el.is_array() && el.size() == 2 && el[0].is_number_integer())
+                if (el.is_array() && el.size() == 2)
                 {
-                    llama_token tok = el[0].get<llama_token>();
-                    if (tok >= 0 && tok < n_vocab)
+                    float bias;
+                    if (el[1].is_number())
                     {
-                        if (el[1].is_number())
+                        bias = el[1].get<float>();
+                    }
+                    else if (el[1].is_boolean() && !el[1].get<bool>())
+                    {
+                        bias = -INFINITY;
+                    }
+                    else
+                    {
+                        continue;
+                    }
+
+                    if(el[0].is_number_integer())
+                    {
+                        llama_token tok = el[0].get<llama_token>();
+                        if (tok >= 0 && tok < n_vocab)
                         {
-                            slot->sparams.logit_bias[tok] = el[1].get<float>();
+                            slot->sparams.logit_bias[tok] = bias;
                         }
-                        else if (el[1].is_boolean() && !el[1].get<bool>())
+                    }
+                    else if (el[0].is_string())
+                    {
+                        auto toks = llama_tokenize(model, el[0].get<std::string>(), false);
+                        for(auto tok : toks)
                         {
-                            slot->sparams.logit_bias[tok] = -INFINITY;
+                            slot->sparams.logit_bias[tok] = bias;
                         }
                     }
                 }

Original file line number	Diff line number	Diff line change
`@@ -828,18 +828,36 @@ struct llama_server_context`
`828`	`828`	`const int n_vocab = llama_n_vocab(model);`
`829`	`829`	`for (const auto &el : *logit_bias)`
`830`	`830`	`{`
`831`		`- if (el.is_array() && el.size() == 2 && el[0].is_number_integer())`
	`831`	`+ if (el.is_array() && el.size() == 2)`
`832`	`832`	`{`
`833`		`- llama_token tok = el[0].get<llama_token>();`
`834`		`- if (tok >= 0 && tok < n_vocab)`
	`833`	`+ float bias;`
	`834`	`+ if (el[1].is_number())`
`835`	`835`	`{`
`836`		`- if (el[1].is_number())`
	`836`	`+ bias = el[1].get<float>();`
	`837`	`+ }`
	`838`	`+ else if (el[1].is_boolean() && !el[1].get<bool>())`
	`839`	`+ {`
	`840`	`+ bias = -INFINITY;`
	`841`	`+ }`
	`842`	`+ else`
	`843`	`+ {`
	`844`	`+ continue;`
	`845`	`+ }`
	`846`	`+`
	`847`	`+ if(el[0].is_number_integer())`
	`848`	`+ {`
	`849`	`+ llama_token tok = el[0].get<llama_token>();`
	`850`	`+ if (tok >= 0 && tok < n_vocab)`
`837`	`851`	`{`
`838`		`- slot->sparams.logit_bias[tok] = el[1].get<float>();`
	`852`	`+ slot->sparams.logit_bias[tok] = bias;`
`839`	`853`	`}`
`840`		`- else if (el[1].is_boolean() && !el[1].get<bool>())`
	`854`	`+ }`
	`855`	`+ else if (el[0].is_string())`
	`856`	`+ {`
	`857`	`+ auto toks = llama_tokenize(model, el[0].get<std::string>(), false);`
	`858`	`+ for(auto tok : toks)`
`841`	`859`	`{`
`842`		`- slot->sparams.logit_bias[tok] = -INFINITY;`
	`860`	`+ slot->sparams.logit_bias[tok] = bias;`
`843`	`861`	`}`
`844`	`862`	`}`
`845`	`863`	`}`