From 0f8673851ceeca611839451750952ca19b52fee6 Mon Sep 17 00:00:00 2001 From: Khoa Pham Date: Mon, 8 Jun 2026 19:58:42 -0700 Subject: [PATCH] test: fix gemma GSM8K thresholds in nightly text eval (#27342) --- test/registered/eval/test_text_models_gsm8k_eval.py | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/test/registered/eval/test_text_models_gsm8k_eval.py b/test/registered/eval/test_text_models_gsm8k_eval.py index c2974439c..275430a01 100644 --- a/test/registered/eval/test_text_models_gsm8k_eval.py +++ b/test/registered/eval/test_text_models_gsm8k_eval.py @@ -30,7 +30,7 @@ MODEL_SCORE_THRESHOLDS = { "meta-llama/Llama-3.1-8B-Instruct": 0.80, # 84.5% - 5% "mistralai/Mistral-7B-Instruct-v0.3": 0.47, # 52.1% - 5% "deepseek-ai/DeepSeek-Coder-V2-Lite-Instruct": 0.81, # 86.4% - 5% - "google/gemma-2-27b-it": 0.86, # 90.7% - 5% + "google/gemma-2-27b-it": 0.81, # 85.5% measured - 5% "meta-llama/Llama-3.1-70B-Instruct": 0.89, # 94.1% - 5% "mistralai/Mixtral-8x7B-Instruct-v0.1": 0.69, # 74.4% - 5% "Qwen/Qwen2-57B-A14B-Instruct": 0.76, # 80.7% - 5% (official A14B score; 88.2% was the 72B) @@ -38,9 +38,7 @@ MODEL_SCORE_THRESHOLDS = { "neuralmagic/Mistral-7B-Instruct-v0.3-FP8": 0.47, # 52.1% - 5% "neuralmagic/DeepSeek-Coder-V2-Lite-Instruct-FP8": 0.81, # 86.4% - 5% "zai-org/GLM-4.5-Air-FP8": 0.80, # ~85% - 5% - # GSM8K baseline for gemma-2-2b is ~40-45%; threshold set at 5% below. - # (Previously 0.50 based on MGSM-EN; tracked regression: https://github.com/sgl-project/sglang/issues/4324) - "neuralmagic/gemma-2-2b-it-FP8": 0.38, # ~43% - 5% + "neuralmagic/gemma-2-2b-it-FP8": 0.53, # 58.4% measured - 5% "neuralmagic/Meta-Llama-3.1-70B-Instruct-FP8": 0.89, # 94.1% - 5% "neuralmagic/Mixtral-8x7B-Instruct-v0.1-FP8": 0.69, # 74.4% - 5% "neuralmagic/Qwen2-72B-Instruct-FP8": 0.86, # 91.1% - 5%