Merge pull request #147 from mohamednaji7/BitsAndBytes

completing the "bitsandbytes" option - based on  https://docs.vllm.ai/en/stable/quantization/bnb.html
This commit is contained in:
Marut Pandya
2025-04-21 11:39:53 -07:00
committed by GitHub
4 changed files with 9 additions and 4 deletions
+3
View File
@@ -147,6 +147,9 @@ def get_engine_args():
# Rename and match to vllm args
args = match_vllm_args(args)
if args.get("load_format") == "bitsandbytes":
args["quantization"] = args["load_format"]
# Set tensor parallel size and max parallel loading workers if more than 1 GPU is available
num_gpus = device_count()