server : honor --embd-normalize CLI arg (#23125)
The --embd-normalize flag was registered only for the embedding and debug examples, so llama-server rejected it and the /embedding handler used a hard-coded default of 2 (L2). Add LLAMA_EXAMPLE_SERVER to the flag's example set and read params.embd_normalize as the handler's default. The per-request "embd_normalize" body field continues to override.
This commit is contained in:
+1
-1
@@ -2808,7 +2808,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
|
|||||||
[](common_params & params, int value) {
|
[](common_params & params, int value) {
|
||||||
params.embd_normalize = value;
|
params.embd_normalize = value;
|
||||||
}
|
}
|
||||||
).set_examples({LLAMA_EXAMPLE_EMBEDDING, LLAMA_EXAMPLE_DEBUG}));
|
).set_examples({LLAMA_EXAMPLE_EMBEDDING, LLAMA_EXAMPLE_SERVER, LLAMA_EXAMPLE_DEBUG}));
|
||||||
add_opt(common_arg(
|
add_opt(common_arg(
|
||||||
{"--embd-output-format"}, "FORMAT",
|
{"--embd-output-format"}, "FORMAT",
|
||||||
"empty = default, \"array\" = [[],[]...], \"json\" = openai style, \"json+\" = same \"json\" + cosine similarity matrix, \"raw\" = plain whitespace-delimited output (one embedding per line)",
|
"empty = default, \"array\" = [[],[]...], \"json\" = openai style, \"json+\" = same \"json\" + cosine similarity matrix, \"raw\" = plain whitespace-delimited output (one embedding per line)",
|
||||||
|
|||||||
@@ -4527,7 +4527,7 @@ std::unique_ptr<server_res_generator> server_routes::handle_embeddings_impl(cons
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
int embd_normalize = 2; // default to Euclidean/L2 norm
|
int embd_normalize = params.embd_normalize;
|
||||||
if (body.count("embd_normalize") != 0) {
|
if (body.count("embd_normalize") != 0) {
|
||||||
embd_normalize = body.at("embd_normalize");
|
embd_normalize = body.at("embd_normalize");
|
||||||
if (meta->pooling_type == LLAMA_POOLING_TYPE_NONE) {
|
if (meta->pooling_type == LLAMA_POOLING_TYPE_NONE) {
|
||||||
|
|||||||
Reference in New Issue
Block a user