Merge pull request #147 from mohamednaji7/BitsAndBytes

completing the "bitsandbytes" option - based on  https://docs.vllm.ai/en/stable/quantization/bnb.html
This commit is contained in:
Marut Pandya
2025-04-21 11:39:53 -07:00
committed by GitHub
4 changed files with 9 additions and 4 deletions
+1 -1
View File
@@ -125,7 +125,7 @@ Below is a summary of the available RunPod Worker images, categorized by image s
| `MAX_NUM_SEQS` | 256 | `int` | Maximum number of sequences per iteration. | | `MAX_NUM_SEQS` | 256 | `int` | Maximum number of sequences per iteration. |
| `MAX_LOGPROBS` | 20 | `int` | Max number of log probs to return when logprobs is specified in SamplingParams. | | `MAX_LOGPROBS` | 20 | `int` | Max number of log probs to return when logprobs is specified in SamplingParams. |
| `DISABLE_LOG_STATS` | False | `bool` | Disable logging statistics. | | `DISABLE_LOG_STATS` | False | `bool` | Disable logging statistics. |
| `QUANTIZATION` | None | ['awq', 'squeezellm', 'gptq'] | Method used to quantize the weights. | | `QUANTIZATION` | None | ['awq', 'squeezellm', 'gptq', 'bitsandbytes'] | Method used to quantize the weights. |
| `ROPE_SCALING` | None | `dict` | RoPE scaling configuration in JSON format. | | `ROPE_SCALING` | None | `dict` | RoPE scaling configuration in JSON format. |
| `ROPE_THETA` | None | `float` | RoPE theta. Use with rope_scaling. | | `ROPE_THETA` | None | `float` | RoPE theta. Use with rope_scaling. |
| `TOKENIZER_POOL_SIZE` | 0 | `int` | Size of tokenizer pool to use for asynchronous tokenization. | | `TOKENIZER_POOL_SIZE` | 0 | `int` | Size of tokenizer pool to use for asynchronous tokenization. |
+2 -1
View File
@@ -4,8 +4,9 @@ pyarrow
runpod~=1.7.7 runpod~=1.7.7
huggingface-hub huggingface-hub
packaging packaging
typing-extensions==4.7.1 typing-extensions>=4.8.0
pydantic pydantic
pydantic-settings pydantic-settings
hf-transfer hf-transfer
transformers transformers
bitsandbytes>=0.45.0
+3
View File
@@ -147,6 +147,9 @@ def get_engine_args():
# Rename and match to vllm args # Rename and match to vllm args
args = match_vllm_args(args) args = match_vllm_args(args)
if args.get("load_format") == "bitsandbytes":
args["quantization"] = args["load_format"]
# Set tensor parallel size and max parallel loading workers if more than 1 GPU is available # Set tensor parallel size and max parallel loading workers if more than 1 GPU is available
num_gpus = device_count() num_gpus = device_count()
+3 -2
View File
@@ -802,14 +802,15 @@
"env_var_name": "QUANTIZATION", "env_var_name": "QUANTIZATION",
"value": "", "value": "",
"title": "Quantization", "title": "Quantization",
"description": "Method used to quantize the weights.", "description": "Method used to quantize the weights.\nif the `Load Format` is 'bitsandbytes' then `Quantization` will be forced to 'bitsandbytes'",
"required": false, "required": false,
"type": "select", "type": "select",
"options": [ "options": [
{ "value": "None", "label": "None" }, { "value": "None", "label": "None" },
{ "value": "awq", "label": "AWQ" }, { "value": "awq", "label": "AWQ" },
{ "value": "squeezellm", "label": "SqueezeLLM" }, { "value": "squeezellm", "label": "SqueezeLLM" },
{ "value": "gptq", "label": "GPTQ" } { "value": "gptq", "label": "GPTQ" },
{ "value": "bitsandbytes", "label": "bitsandbytes" }
] ]
}, },
"ROPE_SCALING": { "ROPE_SCALING": {