[NPU] bugfix: resolve modelslim load weights bug (#19472)
This commit is contained in:
@@ -239,15 +239,6 @@ class ModelSlimConfig(QuantizationConfig):
|
|||||||
):
|
):
|
||||||
# adapted from vllm.model_executor.layers.quantization.utils.quant_utils.is_layer_skipped
|
# adapted from vllm.model_executor.layers.quantization.utils.quant_utils.is_layer_skipped
|
||||||
proj_name = prefix.split(".")[-1]
|
proj_name = prefix.split(".")[-1]
|
||||||
if not hasattr(self, "_quant_description_normalized"):
|
|
||||||
quant_description = {}
|
|
||||||
for prefix_, value in self.quant_description.items():
|
|
||||||
prefix_ = prefix_.replace("language_model.", "")
|
|
||||||
if "visual" in prefix_:
|
|
||||||
prefix_ = prefix_.replace("model.", "")
|
|
||||||
quant_description[prefix_] = value
|
|
||||||
self.quant_description = quant_description
|
|
||||||
self._quant_description_normalized = True
|
|
||||||
if proj_name in fused_mapping:
|
if proj_name in fused_mapping:
|
||||||
shard_prefixes = [
|
shard_prefixes = [
|
||||||
prefix.replace(proj_name, shard_proj_name)
|
prefix.replace(proj_name, shard_proj_name)
|
||||||
|
|||||||
Reference in New Issue
Block a user