aboutsummaryrefslogtreecommitdiff
path: root/ggml.h
diff options
context:
space:
mode:
authorGeorgi Gerganov <ggerganov@gmail.com>2023-05-19 22:17:18 +0300
committerGitHub <noreply@github.com>2023-05-19 22:17:18 +0300
commit2d5db48371052087a83974abda3767d1aedec598 (patch)
treeca7e6ad4b2be21d96272aece6489b2f39c444ecb /ggml.h
parent6986c7835adc13ba3f9d933b95671bb1f3984dc6 (diff)
ggml : use F16 instead of F32 in Q4_0, Q4_1, Q8_0 (#1508)
* ggml : use F16 instead of F32 in Q4_0, Q4_1 and Q8_0 * llama : bump LLAMA_FILE_VERSION to 3 * cuda : update Q4 and Q8 dequantize kernels * ggml : fix AVX dot products * readme : update performance table + hot topics
Diffstat (limited to 'ggml.h')
-rw-r--r--ggml.h2
1 files changed, 1 insertions, 1 deletions
diff --git a/ggml.h b/ggml.h
index 255541d..dce5ca1 100644
--- a/ggml.h
+++ b/ggml.h
@@ -190,7 +190,7 @@
#define GGML_FILE_MAGIC 0x67676d6c // "ggml"
#define GGML_FILE_VERSION 1
-#define GGML_QNT_VERSION 1 // bump this on quantization format changes
+#define GGML_QNT_VERSION 2 // bump this on quantization format changes
#define GGML_QNT_VERSION_FACTOR 1000 // do not change this
#define GGML_MAX_DIMS 4