inforce args.quantization for bnb load_froamt

This commit is contained in:
mohamednaji7
2025-01-21 14:01:32 +02:00
parent 7167985f23
commit a27f72a33a
+3
View File
@@ -146,6 +146,9 @@ def get_engine_args():
# Rename and match to vllm args
args = match_vllm_args(args)
if args.load_format=="bitsandbytes":
args.quantization = args.load_format
# Set tensor parallel size and max parallel loading workers if more than 1 GPU is available
num_gpus = device_count()