Reserve more memory for DeepSeekOCR model and adjust server start timeout for DeepGEMM to reduce flakiness (#15277)
This commit is contained in:
@@ -35,10 +35,11 @@ class TestFlashMLAAttnBackend(unittest.TestCase):
|
|||||||
"flashmla",
|
"flashmla",
|
||||||
]
|
]
|
||||||
)
|
)
|
||||||
|
# Use longer timeout for DeepGEMM JIT compilation which can take 10-20 minutes
|
||||||
cls.process = popen_launch_server(
|
cls.process = popen_launch_server(
|
||||||
cls.model,
|
cls.model,
|
||||||
cls.base_url,
|
cls.base_url,
|
||||||
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
|
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH * 2,
|
||||||
other_args=other_args,
|
other_args=other_args,
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -91,10 +92,11 @@ class TestFlashMLAMTP(CustomTestCase):
|
|||||||
"flashmla",
|
"flashmla",
|
||||||
]
|
]
|
||||||
)
|
)
|
||||||
|
# Use longer timeout for DeepGEMM JIT compilation which can take 10-20 minutes
|
||||||
cls.process = popen_launch_server(
|
cls.process = popen_launch_server(
|
||||||
cls.model,
|
cls.model,
|
||||||
cls.base_url,
|
cls.base_url,
|
||||||
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
|
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH * 2,
|
||||||
other_args=other_args,
|
other_args=other_args,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|||||||
@@ -162,6 +162,10 @@ class TestQwen2AudioServer(AudioOpenAITestMixin):
|
|||||||
class TestDeepseekOCRServer(TestOpenAIMLLMServerBase):
|
class TestDeepseekOCRServer(TestOpenAIMLLMServerBase):
|
||||||
model = "deepseek-ai/DeepSeek-OCR"
|
model = "deepseek-ai/DeepSeek-OCR"
|
||||||
trust_remote_code = False
|
trust_remote_code = False
|
||||||
|
extra_args = [
|
||||||
|
"--mem-fraction-static=0.70",
|
||||||
|
"--cuda-graph-max-bs=4",
|
||||||
|
]
|
||||||
|
|
||||||
def verify_single_image_response_for_ocr(self, response):
|
def verify_single_image_response_for_ocr(self, response):
|
||||||
"""Verify DeepSeek-OCR grounding output with coordinates"""
|
"""Verify DeepSeek-OCR grounding output with coordinates"""
|
||||||
|
|||||||
@@ -38,16 +38,23 @@ class TestOpenAIMLLMServerBase(CustomTestCase):
|
|||||||
def setUpClass(cls):
|
def setUpClass(cls):
|
||||||
cls.base_url = DEFAULT_URL_FOR_TEST
|
cls.base_url = DEFAULT_URL_FOR_TEST
|
||||||
cls.api_key = "sk-123456"
|
cls.api_key = "sk-123456"
|
||||||
|
|
||||||
|
# Build other_args: always include extra_args, conditionally include fixed_args
|
||||||
|
other_args = list(cls.extra_args)
|
||||||
|
if cls.trust_remote_code:
|
||||||
|
other_args.extend(cls.fixed_args)
|
||||||
|
else:
|
||||||
|
# Exclude --trust-remote-code but keep other fixed args like --enable-multimodal
|
||||||
|
other_args.extend(
|
||||||
|
arg for arg in cls.fixed_args if arg != "--trust-remote-code"
|
||||||
|
)
|
||||||
|
|
||||||
cls.process = popen_launch_server(
|
cls.process = popen_launch_server(
|
||||||
cls.model,
|
cls.model,
|
||||||
cls.base_url,
|
cls.base_url,
|
||||||
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
|
timeout=DEFAULT_TIMEOUT_FOR_SERVER_LAUNCH,
|
||||||
api_key=cls.api_key,
|
api_key=cls.api_key,
|
||||||
other_args=(
|
other_args=other_args,
|
||||||
cls.extra_args + cls.fixed_args + ["--trust-remote-code"]
|
|
||||||
if cls.trust_remote_code
|
|
||||||
else []
|
|
||||||
),
|
|
||||||
)
|
)
|
||||||
cls.base_url += "/v1"
|
cls.base_url += "/v1"
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user