summaryrefslogtreecommitdiff
path: root/ggml/src/ggml-common.h
diff options
context:
space:
mode:
authorIwan Kawrakow <iwan.kawrakow@gmail.com>2024-08-01 16:08:32 +0300
committerKawrakow <48489457+ikawrakow@users.noreply.github.com>2024-08-09 16:00:31 +0200
commitcfb0410067be051dd8ea76c08280a3b04d5a5188 (patch)
tree64ffc06e18f75a8cfc4d04b98875f68cd7fa99d4 /ggml/src/ggml-common.h
parenta9f302ebe2373321c12b01d8760904901aa064a4 (diff)
iq6_k: WIP (nothing works)
Diffstat (limited to 'ggml/src/ggml-common.h')
-rw-r--r--ggml/src/ggml-common.h12
1 files changed, 12 insertions, 0 deletions
diff --git a/ggml/src/ggml-common.h b/ggml/src/ggml-common.h
index 5847d903..2fbac06a 100644
--- a/ggml/src/ggml-common.h
+++ b/ggml/src/ggml-common.h
@@ -142,6 +142,9 @@ typedef sycl::half2 ggml_half2;
#define QI5_XS (QK_K / (4*QR5_XS))
#define QR5_XS 2
+#define QI6_XS (QK_K / (4*QR6_XS))
+#define QR6_XS 2
+
#define QI3_S (QK_K / (4*QR3_S))
#define QR3_S 4
@@ -493,6 +496,15 @@ typedef struct {
} block_iq5_k;
static_assert(sizeof(block_iq5_k) == sizeof(ggml_half) + sizeof(uint16_t) + QK_K/2 + QK_K/8 + 3*QK_K/64, "wrong iq5_k block size/padding");
+typedef struct {
+ ggml_half d;
+ uint16_t extra;
+ int8_t scales[QK_K/16];
+ uint8_t qs[QK_K/2];
+ uint8_t qh[QK_K/4];
+} block_iq6_k;
+static_assert(sizeof(block_iq6_k) == sizeof(ggml_half) + sizeof(uint16_t) + QK_K/2 + QK_K/4 + QK_K/16, "wrong iq6_k block size/padding");
+
#endif // GGML_COMMON_DECL
#endif // GGML_COMMON_DECL