test: fix gemma GSM8K thresholds in nightly text eval (#27342)
This commit is contained in:
@@ -30,7 +30,7 @@ MODEL_SCORE_THRESHOLDS = {
|
||||
"meta-llama/Llama-3.1-8B-Instruct": 0.80, # 84.5% - 5%
|
||||
"mistralai/Mistral-7B-Instruct-v0.3": 0.47, # 52.1% - 5%
|
||||
"deepseek-ai/DeepSeek-Coder-V2-Lite-Instruct": 0.81, # 86.4% - 5%
|
||||
"google/gemma-2-27b-it": 0.86, # 90.7% - 5%
|
||||
"google/gemma-2-27b-it": 0.81, # 85.5% measured - 5%
|
||||
"meta-llama/Llama-3.1-70B-Instruct": 0.89, # 94.1% - 5%
|
||||
"mistralai/Mixtral-8x7B-Instruct-v0.1": 0.69, # 74.4% - 5%
|
||||
"Qwen/Qwen2-57B-A14B-Instruct": 0.76, # 80.7% - 5% (official A14B score; 88.2% was the 72B)
|
||||
@@ -38,9 +38,7 @@ MODEL_SCORE_THRESHOLDS = {
|
||||
"neuralmagic/Mistral-7B-Instruct-v0.3-FP8": 0.47, # 52.1% - 5%
|
||||
"neuralmagic/DeepSeek-Coder-V2-Lite-Instruct-FP8": 0.81, # 86.4% - 5%
|
||||
"zai-org/GLM-4.5-Air-FP8": 0.80, # ~85% - 5%
|
||||
# GSM8K baseline for gemma-2-2b is ~40-45%; threshold set at 5% below.
|
||||
# (Previously 0.50 based on MGSM-EN; tracked regression: https://github.com/sgl-project/sglang/issues/4324)
|
||||
"neuralmagic/gemma-2-2b-it-FP8": 0.38, # ~43% - 5%
|
||||
"neuralmagic/gemma-2-2b-it-FP8": 0.53, # 58.4% measured - 5%
|
||||
"neuralmagic/Meta-Llama-3.1-70B-Instruct-FP8": 0.89, # 94.1% - 5%
|
||||
"neuralmagic/Mixtral-8x7B-Instruct-v0.1-FP8": 0.69, # 74.4% - 5%
|
||||
"neuralmagic/Qwen2-72B-Instruct-FP8": 0.86, # 91.1% - 5%
|
||||
|
||||
Reference in New Issue
Block a user