Benchmark Gemma-3

Guang Yang · Guang Yang · commit 0fc5801a13dc · 2025-06-25T12:12:21.000-07:00
diff --git a/.github/workflows/android-perf-private-device-experiment.yml b/.github/workflows/android-perf-private-device-experiment.yml
@@ -57,6 +57,6 @@ jobs:
       id-token: write
       contents: read
     with:
-      models: ${{ inputs.models || github.event_name == 'schedule' && 'Qwen/Qwen3-0.6B,HuggingFaceTB/SmolLM2-135M,meta-llama/Llama-3.2-1B,allenai/OLMo-1B-hf' || 'Qwen/Qwen3-0.6B' }}
+      models: ${{ inputs.models || github.event_name == 'schedule' && 'Qwen/Qwen3-0.6B,HuggingFaceTB/SmolLM2-135M,meta-llama/Llama-3.2-1B,allenai/OLMo-1B-hf,google/gemma-3-1b-it' || 'google/gemma-3-1b-it' }}
       devices: samsung_galaxy_s22_private
       benchmark_configs: ${{ inputs.benchmark_configs }}
diff --git a/.github/workflows/android-perf.yml b/.github/workflows/android-perf.yml
@@ -72,7 +72,7 @@ jobs:
           # Separate default values from the workflow dispatch. To ensure defaults are accessible
           # during scheduled runs and to provide flexibility for different defaults between
           # on-demand and periodic benchmarking.
-          CRON_DEFAULT_MODELS: ${{ github.event_name == 'schedule' && 'mv3,mv2,ic4,ic3,resnet50,edsr,mobilebert,w2l,meta-llama/Llama-3.2-1B,meta-llama/Llama-3.2-1B-Instruct-SpinQuant_INT4_EO8,meta-llama/Llama-3.2-1B-Instruct-QLORA_INT4_EO8,Qwen/Qwen3-0.6B,HuggingFaceTB/SmolLM2-135M,allenai/OLMo-1B-hf' || 'Qwen/Qwen3-0.6B' }}
+          CRON_DEFAULT_MODELS: ${{ github.event_name == 'schedule' && 'mv3,mv2,ic4,ic3,resnet50,edsr,mobilebert,w2l,meta-llama/Llama-3.2-1B,meta-llama/Llama-3.2-1B-Instruct-SpinQuant_INT4_EO8,meta-llama/Llama-3.2-1B-Instruct-QLORA_INT4_EO8,Qwen/Qwen3-0.6B,HuggingFaceTB/SmolLM2-135M,allenai/OLMo-1B-hf,google/gemma-3-1b-it' || 'Qwen/Qwen3-0.6B' }}
           CRON_DEFAULT_DEVICES: samsung_galaxy_s22
         run: |
           set -eux
@@ -344,7 +344,7 @@ jobs:
               git clone https://github.com/huggingface/optimum-executorch
               pushd optimum-executorch
               # There is no release yet, for CI stability, always test from the same commit on main
-              git checkout 4c3b18f6cca68c5ccff809131d570062723d7188
+              git checkout c2b20c9cec66655ce42a75636002b4176aa9644a
               python install_dev.py --skip_override_torch
               pip list
 
@@ -353,21 +353,12 @@ jobs:
                 "--task" "text-generation"
                 "--recipe" "xnnpack"
                 "--use_custom_sdpa"
+                "--use_custom_kv_cache"
                 "--qlinear"
                 "--qembedding"
                 "--output_dir" ".."
               )
 
-              # Add conditional arguments based on model
-              case "${HF_MODEL_REPO}" in
-                *"google/gemma-3-1b-it"*)
-                  echo "--use_custom_kv_cache can not be used for HybridCache"
-                  ;;
-                *)
-                  ARGS+=("--use_custom_kv_cache")
-                  ;;
-              esac
-
               optimum-cli export executorch "${ARGS[@]}"
               popd
 
diff --git a/.github/workflows/apple-perf-private-device-experiment.yml b/.github/workflows/apple-perf-private-device-experiment.yml
@@ -57,6 +57,6 @@ jobs:
       id-token: write
       contents: read
     with:
-      models: ${{ inputs.models || github.event_name == 'schedule' && 'Qwen/Qwen3-0.6B,HuggingFaceTB/SmolLM2-135M,meta-llama/Llama-3.2-1B,allenai/OLMo-1B-hf' || 'Qwen/Qwen3-0.6B' }}
+      models: ${{ inputs.models || github.event_name == 'schedule' && 'Qwen/Qwen3-0.6B,HuggingFaceTB/SmolLM2-135M,meta-llama/Llama-3.2-1B,allenai/OLMo-1B-hf,google/gemma-3-1b-it' || 'google/gemma-3-1b-it' }}
       devices: apple_iphone_15_private
       benchmark_configs: ${{ inputs.benchmark_configs }}
diff --git a/.github/workflows/apple-perf.yml b/.github/workflows/apple-perf.yml
@@ -72,7 +72,7 @@ jobs:
           # Separate default values from the workflow dispatch. To ensure defaults are accessible
           # during scheduled runs and to provide flexibility for different defaults between
           # on-demand and periodic benchmarking.
-          CRON_DEFAULT_MODELS: ${{ github.event_name == 'schedule' && 'mv3,mv2,ic4,ic3,resnet50,edsr,mobilebert,w2l,meta-llama/Llama-3.2-1B-Instruct-SpinQuant_INT4_EO8,meta-llama/Llama-3.2-1B-Instruct-QLORA_INT4_EO8,Qwen/Qwen3-0.6B,HuggingFaceTB/SmolLM2-135M,meta-llama/Llama-3.2-1B,allenai/OLMo-1B-hf' || 'Qwen/Qwen3-0.6B' }}
+          CRON_DEFAULT_MODELS: ${{ github.event_name == 'schedule' && 'mv3,mv2,ic4,ic3,resnet50,edsr,mobilebert,w2l,meta-llama/Llama-3.2-1B-Instruct-SpinQuant_INT4_EO8,meta-llama/Llama-3.2-1B-Instruct-QLORA_INT4_EO8,Qwen/Qwen3-0.6B,HuggingFaceTB/SmolLM2-135M,meta-llama/Llama-3.2-1B,allenai/OLMo-1B-hf,google/gemma-3-1b-it' || 'Qwen/Qwen3-0.6B' }}
           CRON_DEFAULT_DEVICES: apple_iphone_15
         run: |
           set -eux
@@ -349,7 +349,7 @@ jobs:
             git clone https://github.com/huggingface/optimum-executorch
             pushd optimum-executorch
             # There is no release yet, for CI stability, always test from the same commit on main
-            git checkout 4c3b18f6cca68c5ccff809131d570062723d7188
+            git checkout c2b20c9cec66655ce42a75636002b4176aa9644a
             ${CONDA_RUN} python install_dev.py --skip_override_torch
             pip list
 
@@ -358,21 +358,12 @@ jobs:
               "--task" "text-generation"
               "--recipe" "xnnpack"
               "--use_custom_sdpa"
+              "--use_custom_kv_cache"
               "--qlinear"
               "--qembedding"
               "--output_dir" ".."
             )
 
-            # Add conditional arguments based on model
-            case "${HF_MODEL_REPO}" in
-              *"google/gemma-3-1b-it"*)
-                echo "--use_custom_kv_cache can not be used for HybridCache"
-                ;;
-              *)
-                ARGS+=("--use_custom_kv_cache")
-                ;;
-            esac
-
             ${CONDA_RUN} optimum-cli export executorch "${ARGS[@]}"
             popd
 
diff --git a/.github/workflows/trunk.yml b/.github/workflows/trunk.yml
@@ -597,7 +597,7 @@ jobs:
         git clone https://github.com/huggingface/optimum-executorch
         pushd optimum-executorch
         # There is no release yet, for CI stability, always test from the same commit on main
-        git checkout 4c3b18f6cca68c5ccff809131d570062723d7188
+        git checkout c2b20c9cec66655ce42a75636002b4176aa9644a
         python install_dev.py --skip_override_torch
         popd
         pip list
@@ -614,21 +614,12 @@ jobs:
           "--task" "text-generation"
           "--recipe" "xnnpack"
           "--use_custom_sdpa"
+          "--use_custom_kv_cache"
           "--qlinear"
           "--qembedding"
           "--output_dir" "${OUTPUT_DIR}"
         )
 
-        # Add conditional arguments based on model
-        case "${MODEL_ID}" in
-          *"google/gemma-3-1b-it"*)
-            echo "--use_custom_kv_cache can not be used for HybridCache"
-            ;;
-          *)
-            ARGS+=("--use_custom_kv_cache")
-            ;;
-        esac
-
         optimum-cli export executorch "${ARGS[@]}"
 
         ls -FlAGhp ${OUTPUT_DIR}