ggml : add asserts for type conversion in fattn kernels (#9971)
ggml-ci
This commit is contained in:
1 parent
d5ebd79c76
commit
f594bc80ba
3 files changed
+8
-4
No files matched your search
+1
-1
@@ -19243,7 +19243,7 @@ struct llama_context * llama_new_context_with_model(
|
||||
params.flash_attn = false;
|
||||
}
|
||||
|
||||
if (params.type_v != GGML_TYPE_F16 && !params.flash_attn) {
|
||||
if (ggml_is_quantized(params.type_v) && !params.flash_attn) {
|
||||
LLAMA_LOG_ERROR("%s: V cache quantization requires flash_attn\n", __func__);
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
Reference in new issue
Block a user