Skip to content

Commit 5ed8acd

Browse files
committed
cont : simplify
1 parent defeeb3 commit 5ed8acd

File tree

2 files changed

+35
-48
lines changed

2 files changed

+35
-48
lines changed

ggml/src/ggml-metal/ggml-metal-ops.cpp

Lines changed: 33 additions & 47 deletions
Original file line numberDiff line numberDiff line change
@@ -2009,11 +2009,20 @@ int ggml_metal_op_flash_attn_ext(ggml_metal_op_t ctx, int idx) {
20092009

20102010
GGML_ASSERT(ne01 < 65536);
20112011

2012+
ggml_metal_buffer_id bid_src0 = ggml_metal_get_buffer_id(op->src[0]);
2013+
ggml_metal_buffer_id bid_src1 = ggml_metal_get_buffer_id(op->src[1]);
2014+
ggml_metal_buffer_id bid_src2 = ggml_metal_get_buffer_id(op->src[2]);
2015+
ggml_metal_buffer_id bid_src3 = has_mask ? ggml_metal_get_buffer_id(op->src[3]) : bid_src0;
2016+
ggml_metal_buffer_id bid_src4 = has_sinks ? ggml_metal_get_buffer_id(op->src[4]) : bid_src0;
2017+
20122018
ggml_metal_buffer_id bid_dst = ggml_metal_get_buffer_id(op);
20132019

20142020
ggml_metal_buffer_id bid_pad = bid_dst;
20152021
bid_pad.offs += ggml_nbytes(op);
20162022

2023+
ggml_metal_buffer_id bid_tmp = bid_pad;
2024+
bid_tmp.offs += ggml_metal_op_flash_attn_ext_extra_pad(op);
2025+
20172026
if (!ggml_metal_op_flash_attn_ext_use_vec(op)) {
20182027
// half8x8 kernel
20192028
const int64_t nqptg = 8; // queries per threadgroup !! sync with kernel template arguments !!
@@ -2050,14 +2059,10 @@ int ggml_metal_op_flash_attn_ext(ggml_metal_op_t ctx, int idx) {
20502059

20512060
ggml_metal_encoder_set_pipeline(enc, pipeline0);
20522061
ggml_metal_encoder_set_bytes (enc, &args0, sizeof(args0), 0);
2053-
ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[1]), 1);
2054-
ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[2]), 2);
2055-
if (op->src[3]) {
2056-
ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(op->src[3]), 3);
2057-
} else {
2058-
ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(op->src[0]), 3);
2059-
}
2060-
ggml_metal_encoder_set_buffer (enc, bid_pad, 4);
2062+
ggml_metal_encoder_set_buffer (enc, bid_src1, 1);
2063+
ggml_metal_encoder_set_buffer (enc, bid_src2, 2);
2064+
ggml_metal_encoder_set_buffer (enc, bid_src3, 3);
2065+
ggml_metal_encoder_set_buffer (enc, bid_pad, 4);
20612066

20622067
assert(ne12 == ne22);
20632068
assert(ne13 == ne23);
@@ -2139,21 +2144,13 @@ int ggml_metal_op_flash_attn_ext(ggml_metal_op_t ctx, int idx) {
21392144

21402145
ggml_metal_encoder_set_pipeline(enc, pipeline);
21412146
ggml_metal_encoder_set_bytes (enc, &args, sizeof(args), 0);
2142-
ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[0]), 1);
2143-
ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[1]), 2);
2144-
ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[2]), 3);
2145-
if (op->src[3]) {
2146-
ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(op->src[3]), 4);
2147-
} else {
2148-
ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(op->src[0]), 4);
2149-
}
2150-
if (op->src[4]) {
2151-
ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(op->src[4]), 5);
2152-
} else {
2153-
ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(op->src[0]), 5);
2154-
}
2155-
ggml_metal_encoder_set_buffer (enc, bid_pad, 6);
2156-
ggml_metal_encoder_set_buffer (enc, bid_dst, 7);
2147+
ggml_metal_encoder_set_buffer (enc, bid_src0, 1);
2148+
ggml_metal_encoder_set_buffer (enc, bid_src1, 2);
2149+
ggml_metal_encoder_set_buffer (enc, bid_src2, 3);
2150+
ggml_metal_encoder_set_buffer (enc, bid_src3, 4);
2151+
ggml_metal_encoder_set_buffer (enc, bid_src4, 5);
2152+
ggml_metal_encoder_set_buffer (enc, bid_pad, 6);
2153+
ggml_metal_encoder_set_buffer (enc, bid_dst, 7);
21572154

21582155
ggml_metal_encoder_set_threadgroup_memory_size(enc, smem, 0);
21592156

@@ -2196,14 +2193,10 @@ int ggml_metal_op_flash_attn_ext(ggml_metal_op_t ctx, int idx) {
21962193

21972194
ggml_metal_encoder_set_pipeline(enc, pipeline0);
21982195
ggml_metal_encoder_set_bytes (enc, &args0, sizeof(args0), 0);
2199-
ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[1]), 1);
2200-
ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[2]), 2);
2201-
if (op->src[3]) {
2202-
ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(op->src[3]), 3);
2203-
} else {
2204-
ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(op->src[0]), 3);
2205-
}
2206-
ggml_metal_encoder_set_buffer (enc, bid_pad, 4);
2196+
ggml_metal_encoder_set_buffer (enc, bid_src1, 1);
2197+
ggml_metal_encoder_set_buffer (enc, bid_src2, 2);
2198+
ggml_metal_encoder_set_buffer (enc, bid_src3, 3);
2199+
ggml_metal_encoder_set_buffer (enc, bid_pad, 4);
22072200

22082201
assert(ne12 == ne22);
22092202
assert(ne13 == ne23);
@@ -2302,26 +2295,20 @@ int ggml_metal_op_flash_attn_ext(ggml_metal_op_t ctx, int idx) {
23022295

23032296
ggml_metal_encoder_set_pipeline(enc, pipeline);
23042297
ggml_metal_encoder_set_bytes (enc, &args, sizeof(args), 0);
2305-
ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[0]), 1);
2306-
ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[1]), 2);
2307-
ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[2]), 3);
2308-
if (op->src[3]) {
2309-
ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(op->src[3]), 4);
2310-
} else {
2311-
ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(op->src[0]), 4);
2312-
}
2313-
if (op->src[4]) {
2314-
ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(op->src[4]), 5);
2315-
} else {
2316-
ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(op->src[0]), 5);
2317-
}
2298+
ggml_metal_encoder_set_buffer (enc, bid_src0, 1);
2299+
ggml_metal_encoder_set_buffer (enc, bid_src1, 2);
2300+
ggml_metal_encoder_set_buffer (enc, bid_src2, 3);
2301+
ggml_metal_encoder_set_buffer (enc, bid_src3, 4);
2302+
ggml_metal_encoder_set_buffer (enc, bid_src4, 5);
23182303

23192304
const size_t smem = FATTN_SMEM(nsg);
23202305

23212306
//printf("smem: %zu, max: %zu, nsg = %d, nsgmax = %d\n", smem, props_dev->max_theadgroup_memory_size, (int) nsg, (int) nsgmax);
23222307
GGML_ASSERT(smem <= props_dev->max_theadgroup_memory_size);
23232308

23242309
if (nwg == 1) {
2310+
assert(ggml_metal_op_flash_attn_ext_extra_tmp(op) == 0);
2311+
23252312
// using 1 workgroup -> write the result directly into dst
23262313
ggml_metal_encoder_set_buffer(enc, bid_pad, 6);
23272314
ggml_metal_encoder_set_buffer(enc, bid_dst, 7);
@@ -2331,13 +2318,12 @@ int ggml_metal_op_flash_attn_ext(ggml_metal_op_t ctx, int idx) {
23312318
ggml_metal_encoder_dispatch_threadgroups(enc, (ne01 + nqptg - 1)/nqptg, ne02, ne03*nwg, 32, nsg, 1);
23322319
} else {
23332320
// sanity checks
2321+
assert(ggml_metal_op_flash_attn_ext_extra_tmp(op) != 0);
2322+
23342323
GGML_ASSERT(ne01*ne02*ne03 == ne1*ne2*ne3);
23352324
GGML_ASSERT((uint64_t)ne1*ne2*ne3 <= (1u << 31));
23362325

23372326
// write the results from each workgroup into a temp buffer
2338-
ggml_metal_buffer_id bid_tmp = bid_dst;
2339-
bid_tmp.offs += ggml_nbytes(op) + ggml_metal_op_flash_attn_ext_extra_pad(op);
2340-
23412327
ggml_metal_encoder_set_buffer(enc, bid_pad, 6);
23422328
ggml_metal_encoder_set_buffer(enc, bid_tmp, 7);
23432329

ggml/src/ggml-metal/ggml-metal.metal

Lines changed: 2 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -4624,7 +4624,6 @@ void kernel_flash_attn_ext_impl(
46244624

46254625
// mask storage in shared mem
46264626
threadgroup half2 * sm2 = (threadgroup half2 *) (shmem_f16 + Q*T + 2*C);
4627-
threadgroup half * sm = (threadgroup half *) (sm2);
46284627

46294628
// per-query mask pointers
46304629
device const half2 * pm2[NQ];
@@ -4709,6 +4708,8 @@ void kernel_flash_attn_ext_impl(
47094708
v += (ikv2 + ikv3*args.ne_12_2)*args.nb21*C;
47104709

47114710
if (!FC_flash_attn_ext_has_mask) {
4711+
threadgroup half * sm = (threadgroup half *) (sm2);
4712+
47124713
FOR_UNROLL (short jj = 0; jj < NQ; ++jj) {
47134714
const short j = jj*NSG + sgitg;
47144715

0 commit comments

Comments
 (0)