q8_k_r8: fastest matrix multiplication known to human kind

We get PP-512(LLaMA-3.1-8B) = 370 t/s on a Ryzen-7950X!
2026-04-22 23:49:23 +00:00 · 2024-12-13 18:21:08 +02:00
parent 12f962dd24
commit 93a85c62bb
10 changed files with 242 additions and 7 deletions
--- a/include/llama.h
+++ b/include/llama.h
@@ -193,6 +193,7 @@ extern "C" {
        LLAMA_FTYPE_MOSTLY_Q6_0_R4       = 335, // except 1d tensors
        LLAMA_FTYPE_MOSTLY_IQ2_BN_R4     = 337, // except 1d tensors
        LLAMA_FTYPE_MOSTLY_IQ4_K_R4      = 340, // except 1d tensors
+        LLAMA_FTYPE_MOSTLY_Q8_K_R8       = 399, // except 1d tensors

        LLAMA_FTYPE_GUESSED = 1024, // not specified in the model file
    };