Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b7c6d4f9a2 | ||
|
|
d69cc021e8 | ||
|
|
61faa8f137 |
+3
-3
@@ -1,13 +1,13 @@
|
|||||||
FROM nvidia/cuda:12.8.0-base-ubuntu22.04
|
FROM nvidia/cuda:12.9.1-base-ubuntu22.04
|
||||||
|
|
||||||
RUN apt-get update -y \
|
RUN apt-get update -y \
|
||||||
&& apt-get install -y python3-pip
|
&& apt-get install -y python3-pip
|
||||||
|
|
||||||
RUN ldconfig /usr/local/cuda-12.8/compat/
|
RUN ldconfig /usr/local/cuda-12.9/compat/
|
||||||
|
|
||||||
# Install vLLM with FlashInfer - use CUDA 12.8 PyTorch wheels (compatible with vLLM 0.15.0)
|
# Install vLLM with FlashInfer - use CUDA 12.8 PyTorch wheels (compatible with vLLM 0.15.0)
|
||||||
RUN python3 -m pip install --upgrade pip && \
|
RUN python3 -m pip install --upgrade pip && \
|
||||||
python3 -m pip install "vllm[flashinfer]==0.15.0" --extra-index-url https://download.pytorch.org/whl/cu128
|
python3 -m pip install "vllm[flashinfer]==0.15.0" --extra-index-url https://download.pytorch.org/whl/cu129
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -80,6 +80,7 @@ To build an image with the model baked in, you must specify the following docker
|
|||||||
- `WORKER_CUDA_VERSION`: `12.1.0` (`12.1.0` is recommended for optimal performance).
|
- `WORKER_CUDA_VERSION`: `12.1.0` (`12.1.0` is recommended for optimal performance).
|
||||||
- `TOKENIZER_NAME`: Tokenizer repository if you would like to use a different tokenizer than the one that comes with the model. (default: `None`, which uses the model's tokenizer)
|
- `TOKENIZER_NAME`: Tokenizer repository if you would like to use a different tokenizer than the one that comes with the model. (default: `None`, which uses the model's tokenizer)
|
||||||
- `TOKENIZER_REVISION`: Tokenizer revision to load (default: `main`).
|
- `TOKENIZER_REVISION`: Tokenizer revision to load (default: `main`).
|
||||||
|
- `VLLM_NIGHTLY`: Set to `true` to replace the pinned vLLM release with the latest nightly build and the latest `transformers` from source. Useful for testing unreleased vLLM features. (default: `false`)
|
||||||
|
|
||||||
For the remaining settings, you may apply them as environment variables when running the container. Supported environment variables are listed in the [Environment Variables](#environment-variables) section.
|
For the remaining settings, you may apply them as environment variables when running the container. Supported environment variables are listed in the [Environment Variables](#environment-variables) section.
|
||||||
|
|
||||||
@@ -89,6 +90,20 @@ For the remaining settings, you may apply them as environment variables when run
|
|||||||
docker build -t username/image:tag --build-arg MODEL_NAME="openchat/openchat_3.5" --build-arg BASE_PATH="/models" .
|
docker build -t username/image:tag --build-arg MODEL_NAME="openchat/openchat_3.5" --build-arg BASE_PATH="/models" .
|
||||||
```
|
```
|
||||||
|
|
||||||
|
### Example: Building with vLLM Nightly
|
||||||
|
|
||||||
|
To use the latest unreleased vLLM build (installs from the nightly wheel index and `transformers` from source):
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker build -t username/image:tag --build-arg VLLM_NIGHTLY=true .
|
||||||
|
```
|
||||||
|
|
||||||
|
You can combine it with other arguments:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker build -t username/image:tag --build-arg VLLM_NIGHTLY=true --build-arg MODEL_NAME="meta-llama/Llama-3.1-8B-Instruct" --build-arg BASE_PATH="/models" .
|
||||||
|
```
|
||||||
|
|
||||||
### (Optional) Including Huggingface Token
|
### (Optional) Including Huggingface Token
|
||||||
|
|
||||||
If the model you would like to deploy is private or gated, you will need to include it during build time as a Docker secret, which will protect it from being exposed in the image and on DockerHub.
|
If the model you would like to deploy is private or gated, you will need to include it during build time as a Docker secret, which will protect it from being exposed in the image and on DockerHub.
|
||||||
|
|||||||
+11
-6
@@ -122,9 +122,14 @@ def get_speculative_config():
|
|||||||
# Option 2: Build config from individual environment variables
|
# Option 2: Build config from individual environment variables
|
||||||
spec_method = os.getenv('SPECULATIVE_METHOD')
|
spec_method = os.getenv('SPECULATIVE_METHOD')
|
||||||
spec_model = os.getenv('SPECULATIVE_MODEL')
|
spec_model = os.getenv('SPECULATIVE_MODEL')
|
||||||
num_spec_tokens = os.getenv('NUM_SPECULATIVE_TOKENS')
|
_num_spec_tokens = os.getenv('NUM_SPECULATIVE_TOKENS')
|
||||||
ngram_max = os.getenv('NGRAM_PROMPT_LOOKUP_MAX')
|
_ngram_max = os.getenv('NGRAM_PROMPT_LOOKUP_MAX')
|
||||||
ngram_min = os.getenv('NGRAM_PROMPT_LOOKUP_MIN')
|
_ngram_min = os.getenv('NGRAM_PROMPT_LOOKUP_MIN')
|
||||||
|
|
||||||
|
# Convert numeric vars to int so '0' (hub.json default) is treated as unset
|
||||||
|
num_spec_tokens = (int(_num_spec_tokens) or None) if _num_spec_tokens else None
|
||||||
|
ngram_max = (int(_ngram_max) or None) if _ngram_max else None
|
||||||
|
ngram_min = (int(_ngram_min) or None) if _ngram_min else None
|
||||||
|
|
||||||
if not any([spec_method, spec_model, ngram_max]):
|
if not any([spec_method, spec_model, ngram_max]):
|
||||||
return None
|
return None
|
||||||
@@ -150,11 +155,11 @@ def get_speculative_config():
|
|||||||
if spec_model:
|
if spec_model:
|
||||||
config['model'] = spec_model
|
config['model'] = spec_model
|
||||||
if num_spec_tokens:
|
if num_spec_tokens:
|
||||||
config['num_speculative_tokens'] = int(num_spec_tokens)
|
config['num_speculative_tokens'] = num_spec_tokens
|
||||||
if ngram_max:
|
if ngram_max:
|
||||||
config['prompt_lookup_max'] = int(ngram_max)
|
config['prompt_lookup_max'] = ngram_max
|
||||||
if ngram_min:
|
if ngram_min:
|
||||||
config['prompt_lookup_min'] = int(ngram_min)
|
config['prompt_lookup_min'] = ngram_min
|
||||||
|
|
||||||
draft_tp = os.getenv('SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE')
|
draft_tp = os.getenv('SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE')
|
||||||
if draft_tp:
|
if draft_tp:
|
||||||
|
|||||||
Reference in New Issue
Block a user