From eaf34ba0cdcd7690f4b281b309c9bbc2f3e7b3e8 Mon Sep 17 00:00:00 2001 From: Georgi Gerganov Date: Fri, 14 Jun 2024 13:02:25 +0300 Subject: [PATCH] metal : utilize max shared memory for mul_mat_id --- ggml-metal.m | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/ggml-metal.m b/ggml-metal.m index ec9e95302..f894274ca 100644 --- a/ggml-metal.m +++ b/ggml-metal.m @@ -1862,9 +1862,10 @@ static enum ggml_status ggml_metal_graph_compute( // ne21 = n_rows const int dst_rows = ne20*ne21; const int dst_rows_min = n_as; + const int dst_rows_max = (ctx->device.maxThreadgroupMemoryLength - 32 - 8192)/4; // max size of the rowids array in the kernel shared buffer - GGML_ASSERT(dst_rows <= 2048); + GGML_ASSERT(dst_rows <= dst_rows_max); // for now the matrix-matrix multiplication kernel only works on A14+/M1+ SoCs // AMD GPU and older A-chips will reuse matrix-vector multiplication kernel