[SYCL] refactor (#6408)

* seperate lower precision GEMM from the main files * fix workgroup size hardcode
author: Meng, Hengyu <hengyu.meng@intel.com> 2024-06-19 09:11:51 +0800
committer: GitHub <noreply@github.com> 2024-06-19 09:11:51 +0800
commit: 623494a478134432fd2d7ee40135770a3340674f (patch)
tree: 234f90caed1bbe0601cd4f6239e0f2885c34bb6a /ggml-sycl/mmvq.hpp
parent: 37bef8943312d91183ff06d8f1214082a17344a5 (diff)
1 files changed, 27 insertions, 0 deletions
diff --git a/ggml-sycl/mmvq.hpp b/ggml-sycl/mmvq.hpp
new file mode 100644
index 00000000..049b43d4
--- /dev/null
+++ b/ggml-sycl/mmvq.hpp
@@ -0,0 +1,27 @@
+//
+// MIT license
+// Copyright (C) 2024 Intel Corporation
+// SPDX-License-Identifier: MIT
+//
+
+//
+// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
+// See https://llvm.org/LICENSE.txt for license information.
+// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
+//
+
+#ifndef GGML_SYCL_MMVQ_HPP
+#define GGML_SYCL_MMVQ_HPP
+
+#include "common.hpp"
+
+
+void ggml_sycl_op_mul_mat_vec_q(
+    ggml_backend_sycl_context & ctx,
+    const ggml_tensor *src0, const ggml_tensor *src1, ggml_tensor *dst,
+    const char *src0_dd_i, const float *src1_ddf_i, const char *src1_ddq_i,
+    float *dst_dd_i, const int64_t row_low, const int64_t row_high,
+    const int64_t src1_ncols, const int64_t src1_padded_row_size,
+    const dpct::queue_ptr &stream);
+
+#endif // GGML_SYCL_MMVQ_HPP
author	Meng, Hengyu <hengyu.meng@intel.com>	2024-06-19 09:11:51 +0800
committer	GitHub <noreply@github.com>	2024-06-19 09:11:51 +0800
commit	623494a478134432fd2d7ee40135770a3340674f (patch)
tree	234f90caed1bbe0601cd4f6239e0f2885c34bb6a /ggml-sycl/mmvq.hpp
parent	37bef8943312d91183ff06d8f1214082a17344a5 (diff)