[Config] Retire get_global_server_args, and clear the deprecated flags that have a replacement (#38375)
This commit is contained in:
@@ -70,7 +70,7 @@ runtime:
|
||||
# (no chunking). Largest buffer = max_total_num_tokens * 256 B, so the
|
||||
# ceiling is 16,777,216 tokens; MTP runs at 7,000,000 (the validated value,
|
||||
# well under the ceiling and below every MTP leg's natural pool).
|
||||
decode_extra_flags: "--max-total-tokens 7000000 --cuda-graph-bs 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 --prefill-round-robin-balance"
|
||||
decode_extra_flags: "--max-total-tokens 7000000 --cuda-graph-bs-decode 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128"
|
||||
prefill_extra_env:
|
||||
MORI_MAX_DISPATCH_TOKENS_PREFILL: 8192
|
||||
MORI_MAX_DISPATCH_TOKENS_DECODE: 256
|
||||
|
||||
@@ -69,7 +69,7 @@ runtime:
|
||||
# region over 4 GiB and mori registers each KV buffer as one region
|
||||
# (no chunking). Largest buffer = max_total_num_tokens * 256 B, so the
|
||||
# ceiling is 16,777,216 tokens; 16,000,000 leaves headroom.
|
||||
decode_extra_flags: "--max-total-tokens 16000000 --cuda-graph-bs 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 --prefill-round-robin-balance"
|
||||
decode_extra_flags: "--max-total-tokens 16000000 --cuda-graph-bs-decode 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128"
|
||||
prefill_extra_env:
|
||||
MORI_MAX_DISPATCH_TOKENS_PREFILL: 8192
|
||||
MORI_MAX_DISPATCH_TOKENS_DECODE: 256
|
||||
|
||||
@@ -70,7 +70,7 @@ runtime:
|
||||
# (no chunking). Largest buffer = max_total_num_tokens * 256 B, so the
|
||||
# ceiling is 16,777,216 tokens; MTP runs at 7,000,000 (the validated value,
|
||||
# well under the ceiling and below every MTP leg's natural pool).
|
||||
decode_extra_flags: "--max-total-tokens 7000000 --cuda-graph-bs 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 --prefill-round-robin-balance"
|
||||
decode_extra_flags: "--max-total-tokens 7000000 --cuda-graph-bs-decode 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128"
|
||||
prefill_extra_env:
|
||||
MORI_MAX_DISPATCH_TOKENS_PREFILL: 8192
|
||||
MORI_MAX_DISPATCH_TOKENS_DECODE: 256
|
||||
|
||||
@@ -69,7 +69,7 @@ runtime:
|
||||
# region over 4 GiB and mori registers each KV buffer as one region
|
||||
# (no chunking). Largest buffer = max_total_num_tokens * 256 B, so the
|
||||
# ceiling is 16,777,216 tokens; 16,000,000 leaves headroom.
|
||||
decode_extra_flags: "--max-total-tokens 16000000 --cuda-graph-bs 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 --prefill-round-robin-balance"
|
||||
decode_extra_flags: "--max-total-tokens 16000000 --cuda-graph-bs-decode 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128"
|
||||
prefill_extra_env:
|
||||
MORI_MAX_DISPATCH_TOKENS_PREFILL: 8192
|
||||
MORI_MAX_DISPATCH_TOKENS_DECODE: 256
|
||||
|
||||
@@ -84,7 +84,7 @@ runtime:
|
||||
decode_max_running_requests: 1024
|
||||
common_extra_flags: "--moe-dense-tp-size 1 --enable-dp-lm-head --decode-log-interval 100 --watchdog-timeout 3600 --load-balance-method round_robin"
|
||||
prefill_extra_flags: "--context-length 9217 --max-total-tokens 262144"
|
||||
decode_extra_flags: "--disable-cuda-graph --prefill-round-robin-balance"
|
||||
decode_extra_flags: "--disable-cuda-graph"
|
||||
prefill_extra_env:
|
||||
MORI_MAX_DISPATCH_TOKENS_PREFILL: 8192
|
||||
MORI_MAX_DISPATCH_TOKENS_DECODE: 256
|
||||
|
||||
@@ -84,7 +84,7 @@ runtime:
|
||||
decode_max_running_requests: 1024
|
||||
common_extra_flags: "--moe-dense-tp-size 1 --enable-dp-lm-head --decode-log-interval 100 --watchdog-timeout 3600 --load-balance-method round_robin"
|
||||
prefill_extra_flags: "--context-length 9217 --max-total-tokens 262144"
|
||||
decode_extra_flags: "--disable-cuda-graph --prefill-round-robin-balance"
|
||||
decode_extra_flags: "--disable-cuda-graph"
|
||||
prefill_extra_env:
|
||||
MORI_MAX_DISPATCH_TOKENS_PREFILL: 8192
|
||||
MORI_MAX_DISPATCH_TOKENS_DECODE: 256
|
||||
|
||||
Reference in New Issue
Block a user