From 7167985f230963a017a44cb21becd6a48db42ca0 Mon Sep 17 00:00:00 2001 From: mohamednaji7 Date: Tue, 21 Jan 2025 13:49:36 +0200 Subject: [PATCH 1/7] including bitsandbytes "src:https://docs.vllm.ai/en/stable/quantization/bnb.html" --- builder/requirements.txt | 1 + 1 file changed, 1 insertion(+) diff --git a/builder/requirements.txt b/builder/requirements.txt index 4a6b137..d3d1f61 100644 --- a/builder/requirements.txt +++ b/builder/requirements.txt @@ -9,3 +9,4 @@ pydantic pydantic-settings hf-transfer transformers +bitsandbytes>=0.45.0 From a27f72a33a7e66490adc0c36e6783f62359ef991 Mon Sep 17 00:00:00 2001 From: mohamednaji7 Date: Tue, 21 Jan 2025 14:01:32 +0200 Subject: [PATCH 2/7] inforce args.quantization for bnb load_froamt --- src/engine_args.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/src/engine_args.py b/src/engine_args.py index 45e50d1..66ed122 100644 --- a/src/engine_args.py +++ b/src/engine_args.py @@ -146,6 +146,9 @@ def get_engine_args(): # Rename and match to vllm args args = match_vllm_args(args) + + if args.load_format=="bitsandbytes": + args.quantization = args.load_format # Set tensor parallel size and max parallel loading workers if more than 1 GPU is available num_gpus = device_count() From 331bc3010171ace8f9721e99226f45911fd0030f Mon Sep 17 00:00:00 2001 From: mohamednaji7 Date: Tue, 21 Jan 2025 14:08:37 +0200 Subject: [PATCH 3/7] adding 'bitsandbytes' option --- worker-config.json | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/worker-config.json b/worker-config.json index 7845882..221da2c 100644 --- a/worker-config.json +++ b/worker-config.json @@ -558,14 +558,15 @@ "env_var_name": "QUANTIZATION", "value": "", "title": "Quantization", - "description": "Method used to quantize the weights.", + "description": "Method used to quantize the weights.\nif the `Load Format` is 'bitsandbytes' then `Quantization` will be forced to 'bitsandbytes'", "required": false, "type": "select", "options": [ { "value": "None", "label": "None" }, { "value": "awq", "label": "AWQ" }, { "value": "squeezellm", "label": "SqueezeLLM" }, - { "value": "gptq", "label": "GPTQ" } + { "value": "gptq", "label": "GPTQ" }, + { "value": "bitsandbytes", "label": "bitsandbytes" } ] }, "ROPE_SCALING": { From 8882f6d50b8ea4614b042553e304940eda96448b Mon Sep 17 00:00:00 2001 From: mohamednaji7 Date: Tue, 21 Jan 2025 14:46:40 +0200 Subject: [PATCH 4/7] adding 'bitsandbytes' to `QUANTIZATION` --- README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/README.md b/README.md index 82cc1a6..921e183 100644 --- a/README.md +++ b/README.md @@ -125,7 +125,7 @@ Below is a summary of the available RunPod Worker images, categorized by image s | `MAX_NUM_SEQS` | 256 | `int` | Maximum number of sequences per iteration. | | `MAX_LOGPROBS` | 20 | `int` | Max number of log probs to return when logprobs is specified in SamplingParams. | | `DISABLE_LOG_STATS` | False | `bool` | Disable logging statistics. | -| `QUANTIZATION` | None | ['awq', 'squeezellm', 'gptq'] | Method used to quantize the weights. | +| `QUANTIZATION` | None | ['awq', 'squeezellm', 'gptq'. 'bitsandbytes'] | Method used to quantize the weights. | | `ROPE_SCALING` | None | `dict` | RoPE scaling configuration in JSON format. | | `ROPE_THETA` | None | `float` | RoPE theta. Use with rope_scaling. | | `TOKENIZER_POOL_SIZE` | 0 | `int` | Size of tokenizer pool to use for asynchronous tokenization. | From 131c17569fb2fe9641b2f5f1e2d0ba1a6fd6daaa Mon Sep 17 00:00:00 2001 From: mohamednaji7 Date: Tue, 21 Jan 2025 22:33:28 +0200 Subject: [PATCH 5/7] correct access to "args" dictionary --- src/engine_args.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/engine_args.py b/src/engine_args.py index 66ed122..1d99005 100644 --- a/src/engine_args.py +++ b/src/engine_args.py @@ -147,8 +147,8 @@ def get_engine_args(): # Rename and match to vllm args args = match_vllm_args(args) - if args.load_format=="bitsandbytes": - args.quantization = args.load_format + if args.get("load_format") == "bitsandbytes": + args["quantization"] = args["load_format"] # Set tensor parallel size and max parallel loading workers if more than 1 GPU is available num_gpus = device_count() From 9299f43b8ba9f63b05f3ad2e2ddf24362e3f08ce Mon Sep 17 00:00:00 2001 From: Mohamed Nagy <69568400+mohamednaji7@users.noreply.github.com> Date: Wed, 22 Jan 2025 18:06:46 +0200 Subject: [PATCH 6/7] solving "typing_extensions" compatibility with "bitsandbytes" ``` 2025-01-22 18:04:01 [INFO] > [stage-0 5/8] RUN --mount=type=cache,target=/root/.cache/pip python3 -m pip install --upgrade pip && python3 -m pip install --upgrade -r /requirements.txt: 2025-01-22 18:04:01 [INFO] #13 8.904 2025-01-22 18:04:01 [INFO] #13 8.904 The conflict is caused by: 2025-01-22 18:04:01 [INFO] #13 8.904 The user requested typing-extensions==4.7.1 2025-01-22 18:04:01 [INFO] #13 8.904 bitsandbytes 0.45.0 depends on typing_extensions>=4.8.0 ``` --- builder/requirements.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/builder/requirements.txt b/builder/requirements.txt index d3d1f61..b0bfd94 100644 --- a/builder/requirements.txt +++ b/builder/requirements.txt @@ -4,7 +4,7 @@ pyarrow runpod~=1.7.0 huggingface-hub packaging -typing-extensions==4.7.1 +typing-extensions>=4.8.0 pydantic pydantic-settings hf-transfer From 04288240f679210c063f6ab2c14ce5322da71e5f Mon Sep 17 00:00:00 2001 From: Mohamed Nagy <69568400+mohamednaji7@users.noreply.github.com> Date: Sun, 2 Feb 2025 14:58:48 +0200 Subject: [PATCH 7/7] updating the `.` in ['awq', 'squeezellm', 'gptq'. 'bitsandbytes'] for the QUNATIZATION row --- README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/README.md b/README.md index 921e183..6466c25 100644 --- a/README.md +++ b/README.md @@ -125,7 +125,7 @@ Below is a summary of the available RunPod Worker images, categorized by image s | `MAX_NUM_SEQS` | 256 | `int` | Maximum number of sequences per iteration. | | `MAX_LOGPROBS` | 20 | `int` | Max number of log probs to return when logprobs is specified in SamplingParams. | | `DISABLE_LOG_STATS` | False | `bool` | Disable logging statistics. | -| `QUANTIZATION` | None | ['awq', 'squeezellm', 'gptq'. 'bitsandbytes'] | Method used to quantize the weights. | +| `QUANTIZATION` | None | ['awq', 'squeezellm', 'gptq', 'bitsandbytes'] | Method used to quantize the weights. | | `ROPE_SCALING` | None | `dict` | RoPE scaling configuration in JSON format. | | `ROPE_THETA` | None | `float` | RoPE theta. Use with rope_scaling. | | `TOKENIZER_POOL_SIZE` | 0 | `int` | Size of tokenizer pool to use for asynchronous tokenization. |