graph : cast KV to F16 when the KV cache is not used

ggerganov · ggerganov · commit 7cb9ae059ae4 · 2025-04-08T13:23:14.000+03:00
ggml-ci
diff --git a/examples/server_embd.py b/examples/server_embd.py
@@ -15,7 +15,7 @@ async def main():
     model_url = "http://127.0.0.1:6900"
     responses: list[requests.Response] = await asyncio.gather(*[requests_post_async(
         url= f"{model_url}/embedding",
-        json= {"content": str(0)*1024}
+        json= {"content": "a "*1022}
     ) for i in range(n)])
 
     for response in responses:
diff --git a/src/llama-graph.cpp b/src/llama-graph.cpp
@@ -1215,6 +1215,15 @@ ggml_tensor * llm_graph_context::build_attn_mha(
             v = ggml_transpose(ctx0, v);
         }
 
+        // this can happen when KV cache is not used (e.g. an embedding model with non-causal attn)
+        if (k->type == GGML_TYPE_F32) {
+            k = ggml_cast(ctx0, k, GGML_TYPE_F16);
+        }
+
+        if (v->type == GGML_TYPE_F32) {
+            v = ggml_cast(ctx0, v, GGML_TYPE_F16);
+        }
+
         cur = ggml_flash_attn_ext(ctx0, q, k, v, kq_mask, kq_scale, hparams.f_max_alibi_bias,
                                   hparams.attn_soft_cap ? hparams.f_attn_logit_softcapping : 0.0f);