replace the magic nunber 768 by max work group size to support iGPU (#19920)
Co-authored-by: Neo Zhang Jianyu <jianyu.zhang@intel.com>
This commit is contained in:
co-authored by
Neo Zhang Jianyu
parent
88cf781f51
commit
c17dce4f5c
@@ -55,7 +55,11 @@ void ggml_sycl_add_id(ggml_backend_sycl_context& ctx, ggml_tensor* dst) {
|
|||||||
const int32_t* src2_d = (const int32_t*)src2->data;
|
const int32_t* src2_d = (const int32_t*)src2->data;
|
||||||
float* dst_d = (float*)dst->data;
|
float* dst_d = (float*)dst->data;
|
||||||
|
|
||||||
int threads = std::min((int)ne00, 768); // cols
|
const unsigned int max_work_group_size = ggml_sycl_info().max_work_group_sizes[ctx.device];
|
||||||
|
assert(work_group_size % (WARP_SIZE * WARP_SIZE) == 0);
|
||||||
|
|
||||||
|
int threads = std::min((unsigned int)ne00, max_work_group_size); // cols
|
||||||
|
|
||||||
ctx.stream()->parallel_for(
|
ctx.stream()->parallel_for(
|
||||||
sycl::nd_range<3>(
|
sycl::nd_range<3>(
|
||||||
sycl::range<3>(1, ne02, ne01) * sycl::range<3>(1, 1, threads),
|
sycl::range<3>(1, ne02, ne01) * sycl::range<3>(1, 1, threads),
|
||||||
|
|||||||
Reference in New Issue
Block a user