[bugfix] fix the default value of llm_int8_threshold in BitsAndBytesC…

…onfig (vllm-project#10657) Signed-off-by: Andrew Feldman <[email protected]>
neuralmagic · Dec 2, 2024 · 429d17e · 429d17e
1 parent 0f196ac
commit 429d17e
Showing 1 changed file with 2 additions and 2 deletions.
diff --git a/vllm/model_executor/layers/quantization/bitsandbytes.py b/vllm/model_executor/layers/quantization/bitsandbytes.py
@@ -26,7 +26,7 @@ def __init__(
         llm_int8_enable_fp32_cpu_offload: bool = False,
         llm_int8_has_fp16_weight: bool = False,
         llm_int8_skip_modules: Optional[List[str]] = None,
-        llm_int8_threshold: float = 0.0,
+        llm_int8_threshold: float = 6.0,
     ) -> None:
 
         self.load_in_8bit = load_in_8bit
@@ -103,7 +103,7 @@ def get_safe_value(config, keys, default_value=None):
                                                ["llm_int8_skip_modules"],
                                                default_value=[])
         llm_int8_threshold = get_safe_value(config, ["llm_int8_threshold"],
-                                            default_value=0.0)
+                                            default_value=6.0)
 
         return cls(
             load_in_8bit=load_in_8bit,