Merge pull request #147 from mohamednaji7/BitsAndBytes
completing the "bitsandbytes" option - based on https://docs.vllm.ai/en/stable/quantization/bnb.html
This commit is contained in:
@@ -125,7 +125,7 @@ Below is a summary of the available RunPod Worker images, categorized by image s
|
|||||||
| `MAX_NUM_SEQS` | 256 | `int` | Maximum number of sequences per iteration. |
|
| `MAX_NUM_SEQS` | 256 | `int` | Maximum number of sequences per iteration. |
|
||||||
| `MAX_LOGPROBS` | 20 | `int` | Max number of log probs to return when logprobs is specified in SamplingParams. |
|
| `MAX_LOGPROBS` | 20 | `int` | Max number of log probs to return when logprobs is specified in SamplingParams. |
|
||||||
| `DISABLE_LOG_STATS` | False | `bool` | Disable logging statistics. |
|
| `DISABLE_LOG_STATS` | False | `bool` | Disable logging statistics. |
|
||||||
| `QUANTIZATION` | None | ['awq', 'squeezellm', 'gptq'] | Method used to quantize the weights. |
|
| `QUANTIZATION` | None | ['awq', 'squeezellm', 'gptq', 'bitsandbytes'] | Method used to quantize the weights. |
|
||||||
| `ROPE_SCALING` | None | `dict` | RoPE scaling configuration in JSON format. |
|
| `ROPE_SCALING` | None | `dict` | RoPE scaling configuration in JSON format. |
|
||||||
| `ROPE_THETA` | None | `float` | RoPE theta. Use with rope_scaling. |
|
| `ROPE_THETA` | None | `float` | RoPE theta. Use with rope_scaling. |
|
||||||
| `TOKENIZER_POOL_SIZE` | 0 | `int` | Size of tokenizer pool to use for asynchronous tokenization. |
|
| `TOKENIZER_POOL_SIZE` | 0 | `int` | Size of tokenizer pool to use for asynchronous tokenization. |
|
||||||
|
|||||||
@@ -4,8 +4,9 @@ pyarrow
|
|||||||
runpod~=1.7.7
|
runpod~=1.7.7
|
||||||
huggingface-hub
|
huggingface-hub
|
||||||
packaging
|
packaging
|
||||||
typing-extensions==4.7.1
|
typing-extensions>=4.8.0
|
||||||
pydantic
|
pydantic
|
||||||
pydantic-settings
|
pydantic-settings
|
||||||
hf-transfer
|
hf-transfer
|
||||||
transformers
|
transformers
|
||||||
|
bitsandbytes>=0.45.0
|
||||||
|
|||||||
@@ -148,6 +148,9 @@ def get_engine_args():
|
|||||||
# Rename and match to vllm args
|
# Rename and match to vllm args
|
||||||
args = match_vllm_args(args)
|
args = match_vllm_args(args)
|
||||||
|
|
||||||
|
if args.get("load_format") == "bitsandbytes":
|
||||||
|
args["quantization"] = args["load_format"]
|
||||||
|
|
||||||
# Set tensor parallel size and max parallel loading workers if more than 1 GPU is available
|
# Set tensor parallel size and max parallel loading workers if more than 1 GPU is available
|
||||||
num_gpus = device_count()
|
num_gpus = device_count()
|
||||||
if num_gpus > 1:
|
if num_gpus > 1:
|
||||||
|
|||||||
+3
-2
@@ -802,14 +802,15 @@
|
|||||||
"env_var_name": "QUANTIZATION",
|
"env_var_name": "QUANTIZATION",
|
||||||
"value": "",
|
"value": "",
|
||||||
"title": "Quantization",
|
"title": "Quantization",
|
||||||
"description": "Method used to quantize the weights.",
|
"description": "Method used to quantize the weights.\nif the `Load Format` is 'bitsandbytes' then `Quantization` will be forced to 'bitsandbytes'",
|
||||||
"required": false,
|
"required": false,
|
||||||
"type": "select",
|
"type": "select",
|
||||||
"options": [
|
"options": [
|
||||||
{ "value": "None", "label": "None" },
|
{ "value": "None", "label": "None" },
|
||||||
{ "value": "awq", "label": "AWQ" },
|
{ "value": "awq", "label": "AWQ" },
|
||||||
{ "value": "squeezellm", "label": "SqueezeLLM" },
|
{ "value": "squeezellm", "label": "SqueezeLLM" },
|
||||||
{ "value": "gptq", "label": "GPTQ" }
|
{ "value": "gptq", "label": "GPTQ" },
|
||||||
|
{ "value": "bitsandbytes", "label": "bitsandbytes" }
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
"ROPE_SCALING": {
|
"ROPE_SCALING": {
|
||||||
|
|||||||
Reference in New Issue
Block a user