whisper : use flash attention (#2152)

* whisper : use flash attention in the encoder * whisper : add kv_pad * whisper : remove extra backend instance (huh?) * whisper : use FA for cross-attention * whisper : use FA for self-attention * whisper : simplify encoder FA * whisper : add flash_attn runtime parameter * scripts : add bench log * scripts : add M1 Pro bench log
2025-08-15 07:32:35 +02:00 · 2024-05-15 09:38:19 +03:00
parent 9d5771ae43
commit 7094ea5e75
13 changed files with 657 additions and 172 deletions
--- a/scripts/bench-all.sh
+++ b/scripts/bench-all.sh
@ -2,7 +2,7 @@

 # Helper script to run the bench tool on all models and print the results in share-able format

-printf "Usage: ./bench.sh [n_threads] [encoder-only]\n"
+printf "Usage: ./bench.sh [n_threads] [encoder-only] [flash-attn]\n"

 if [ -z "$1" ]; then
    n_threads=4
@ -11,12 +11,19 @@ else
 fi

 encoder_only=0
-if [ -z "$2" ]; then
+if [ -z "$2" ] || [ "$2" -eq 0 ]; then
    encoder_only=0
 else
    encoder_only=$2
 fi

+fattn=""
+if [ -z "$3" ] || [ "$3" -eq 0 ]; then
+    fattn=""
+else
+    fattn="-fa"
+fi
+
 models=(                                                                                                    \
      "tiny"     "tiny-q4_0"     "tiny-q4_1"     "tiny-q5_0"     "tiny-q5_1"     "tiny-q8_0"                \
      "base"     "base-q4_0"     "base-q4_1"     "base-q5_0"     "base-q5_1"     "base-q8_0"                \
@ -44,13 +51,19 @@ if [ "$encoder_only" -eq 0 ]; then
    printf "\n"
 fi

-printf "| %6s | %6s | %16s | %13s | %3s | %7s | %7s | %7s | %7s | %7s |\n" "CPU" "OS" "Config" "Model" "Th" "Enc." "Dec." "Bch5" "PP" "Commit"
-printf "| %6s | %6s | %16s | %13s | %3s | %7s | %7s | %7s | %7s | %7s |\n" "---" "---" "---" "---" "---" "---" "---" "---" "---" "---"
+if [ "$fattn" == "-fa" ]; then
+    fattn_i=1
+else
+    fattn_i=0
+fi
+
+printf "| %6s | %6s | %16s | %13s | %3s | %3s | %7s | %7s | %7s | %7s | %7s |\n" "CPU" "OS" "Config" "Model" "Th" "FA" "Enc." "Dec." "Bch5" "PP" "Commit"
+printf "| %6s | %6s | %16s | %13s | %3s | %3s | %7s | %7s | %7s | %7s | %7s |\n" "---" "---" "---" "---" "---" "---" "---" "---" "---" "---" "---"

 for model in "${models[@]}"; do
    # actual run
    # store stderr output in a variable in order to parse it later
-    output=$(./bench -m ./models/ggml-$model.bin -t $n_threads 2>&1)
+    output=$(./bench -m ./models/ggml-$model.bin -t $n_threads $fattn 2>&1)
    ret=$?

    # parse the output:
@ -95,6 +108,6 @@ for model in "${models[@]}"; do
    commit=$(git rev-parse --short HEAD)

    if [ $ret -eq 0 ]; then
-        printf "| <todo> | <todo> | %16s | %13s | %3s | %7s | %7s | %7s | %7s | %7s |\n" "$config" "$model" "$n_threads" "$encode_time" "$decode_time" "$batchd_time" "$prompt_time" "$commit"
+        printf "| <todo> | <todo> | %16s | %13s | %3s | %3s | %7s | %7s | %7s | %7s | %7s |\n" "$config" "$model" "$n_threads" "$fattn_i" "$encode_time" "$decode_time" "$batchd_time" "$prompt_time" "$commit"
    fi
 done