add changes for v0.15.0

This commit is contained in:
velaraptor-runpod
2026-02-05 17:24:16 -06:00
parent 6d6cbe7095
commit 8eb55b90c1
5 changed files with 143 additions and 42 deletions
+10 -3
View File
@@ -1,9 +1,9 @@
FROM nvidia/cuda:12.4.1-base-ubuntu22.04
FROM nvidia/cuda:12.9.0-base-ubuntu22.04
RUN apt-get update -y \
&& apt-get install -y python3-pip
RUN ldconfig /usr/local/cuda-12.4/compat/
RUN ldconfig /usr/local/cuda-12.9/compat/
# Install Python dependencies
COPY builder/requirements.txt /requirements.txt
@@ -12,7 +12,7 @@ RUN --mount=type=cache,target=/root/.cache/pip \
python3 -m pip install --upgrade -r /requirements.txt
# Install vLLM
RUN python3 -m pip install vllm==0.11.0
RUN python3 -m pip install vllm==0.15.0
# Setup for Option 2: Building the Image with the Model included
ARG MODEL_NAME=""
@@ -21,6 +21,7 @@ ARG BASE_PATH="/runpod-volume"
ARG QUANTIZATION=""
ARG MODEL_REVISION=""
ARG TOKENIZER_REVISION=""
ARG VLLM_NIGHTLY="true"
ENV MODEL_NAME=$MODEL_NAME \
MODEL_REVISION=$MODEL_REVISION \
@@ -35,6 +36,12 @@ ENV MODEL_NAME=$MODEL_NAME \
ENV PYTHONPATH="/:/vllm-workspace"
RUN if [ -n "${VLLM_NIGHTLY}" ]; then \
pip install -U vllm --pre --index-url https://pypi.org/simple --extra-index-url https://wheels.vllm.ai/nightly && \
apt-get update && apt-get install -y git && rm -rf /var/lib/apt/lists/* && \
pip install git+https://github.com/huggingface/transformers.git; \
fi
COPY src /src
RUN --mount=type=secret,id=HF_TOKEN,required=false \
+2 -2
View File
@@ -8,7 +8,7 @@ typing-extensions>=4.8.0
pydantic
pydantic-settings
hf-transfer
transformers>=4.57.0
transformers>=4.57.5
bitsandbytes>=0.45.0
kernels
torch==2.6.0
torch>=2.10.0
+8 -12
View File
@@ -9,10 +9,13 @@ import time
from vllm import AsyncLLMEngine
from vllm.entrypoints.logger import RequestLogger
from vllm.entrypoints.openai.serving_chat import OpenAIServingChat
from vllm.entrypoints.openai.serving_completion import OpenAIServingCompletion
from vllm.entrypoints.openai.protocol import ChatCompletionRequest, CompletionRequest, ErrorResponse
from vllm.entrypoints.openai.serving_models import BaseModelPath, LoRAModulePath, OpenAIServingModels
from vllm.entrypoints.openai.chat_completion.serving import OpenAIServingChat
from vllm.entrypoints.openai.chat_completion.protocol import ChatCompletionRequest
from vllm.entrypoints.openai.completion.serving import OpenAIServingCompletion
from vllm.entrypoints.openai.completion.protocol import CompletionRequest
from vllm.entrypoints.openai.engine.protocol import ErrorResponse
from vllm.entrypoints.openai.models.serving import OpenAIServingModels
from vllm.entrypoints.openai.models.protocol import BaseModelPath, LoRAModulePath
from utils import DummyRequest, JobInput, BatchSize, create_error_response
@@ -202,14 +205,12 @@ class OpenAIvLLMEngine(vLLMEngine):
return adapters
async def _initialize_engines(self):
self.model_config = await self.llm.get_model_config()
self.base_model_paths = [
BaseModelPath(name=self.engine_args.model, model_path=self.engine_args.model)
]
self.serving_models = OpenAIServingModels(
engine_client=self.llm,
model_config=self.model_config,
base_model_paths=self.base_model_paths,
lora_modules=self.lora_adapters,
)
@@ -222,25 +223,20 @@ class OpenAIvLLMEngine(vLLMEngine):
self.chat_engine = OpenAIServingChat(
engine_client=self.llm,
model_config=self.model_config,
models=self.serving_models,
response_role=self.response_role,
request_logger=None,
chat_template=chat_template,
chat_template_content_format="auto",
# enable_reasoning=os.getenv('ENABLE_REASONING', 'false').lower() == 'true',
reasoning_parser= os.getenv('REASONING_PARSER', "") or None,
# return_token_as_token_ids=False,
reasoning_parser=os.getenv('REASONING_PARSER', "") or None,
enable_auto_tools=os.getenv('ENABLE_AUTO_TOOL_CHOICE', 'false').lower() == 'true',
tool_parser=os.getenv('TOOL_CALL_PARSER', "") or None,
enable_prompt_tokens_details=False
)
self.completion_engine = OpenAIServingCompletion(
engine_client=self.llm,
model_config=self.model_config,
models=self.serving_models,
request_logger=None,
# return_token_as_token_ids=False,
)
async def generate(self, openai_request: JobInput):
+116 -17
View File
@@ -38,7 +38,6 @@ DEFAULT_ARGS = {
"block_size": int(os.getenv('BLOCK_SIZE', 16)),
"enable_prefix_caching": os.getenv('ENABLE_PREFIX_CACHING', 'False').lower() == 'true',
"disable_sliding_window": os.getenv('DISABLE_SLIDING_WINDOW', 'False').lower() == 'true',
"use_v2_block_manager": os.getenv('USE_V2_BLOCK_MANAGER', 'False').lower() == 'true',
"swap_space": int(os.getenv('SWAP_SPACE', 4)), # GiB
"cpu_offload_gb": int(os.getenv('CPU_OFFLOAD_GB', 0)), # GiB
"max_num_batched_tokens": int(os.getenv('MAX_NUM_BATCHED_TOKENS', 0)) or None,
@@ -71,29 +70,128 @@ DEFAULT_ARGS = {
"device": os.getenv('DEVICE', 'auto'),
"ray_workers_use_nsight": os.getenv('RAY_WORKERS_USE_NSIGHT', 'False').lower() == 'true',
"num_gpu_blocks_override": int(os.getenv('NUM_GPU_BLOCKS_OVERRIDE', 0)) or None,
"num_lookahead_slots": int(os.getenv('NUM_LOOKAHEAD_SLOTS', 0)),
"model_loader_extra_config": os.getenv('MODEL_LOADER_EXTRA_CONFIG', None),
"ignore_patterns": os.getenv('IGNORE_PATTERNS', None),
"preemption_mode": os.getenv('PREEMPTION_MODE', None),
"scheduler_delay_factor": float(os.getenv('SCHEDULER_DELAY_FACTOR', 0.0)),
"enable_chunked_prefill": os.getenv('ENABLE_CHUNKED_PREFILL', None),
"guided_decoding_backend": os.getenv('GUIDED_DECODING_BACKEND', 'outlines'),
"speculative_model": os.getenv('SPECULATIVE_MODEL', None),
"speculative_draft_tensor_parallel_size": int(os.getenv('SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE', 0)) or None,
"enable_expert_parallel": bool(os.getenv('ENABLE_EXPERT_PARALLEL', 'False').lower() == 'true'),
"num_speculative_tokens": int(os.getenv('NUM_SPECULATIVE_TOKENS', 0)) or None,
"speculative_max_model_len": int(os.getenv('SPECULATIVE_MAX_MODEL_LEN', 0)) or None,
"speculative_disable_by_batch_size": int(os.getenv('SPECULATIVE_DISABLE_BY_BATCH_SIZE', 0)) or None,
"ngram_prompt_lookup_max": int(os.getenv('NGRAM_PROMPT_LOOKUP_MAX', 0)) or None,
"ngram_prompt_lookup_min": int(os.getenv('NGRAM_PROMPT_LOOKUP_MIN', 0)) or None,
"spec_decoding_acceptance_method": os.getenv('SPEC_DECODING_ACCEPTANCE_METHOD', 'rejection_sampler'),
"typical_acceptance_sampler_posterior_threshold": float(os.getenv('TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_THRESHOLD', 0)) or None,
"typical_acceptance_sampler_posterior_alpha": float(os.getenv('TYPICAL_ACCEPTANCE_SAMPLER_POSTERIOR_ALPHA', 0)) or None,
"qlora_adapter_name_or_path": os.getenv('QLORA_ADAPTER_NAME_OR_PATH', None),
"disable_logprobs_during_spec_decoding": os.getenv('DISABLE_LOGPROBS_DURING_SPEC_DECODING', None),
"otlp_traces_endpoint": os.getenv('OTLP_TRACES_ENDPOINT', None),
"use_v2_block_manager": os.getenv('USE_V2_BLOCK_MANAGER', 'true'),
}
def get_speculative_config():
"""
Build speculative decoding configuration from environment variables.
Supports two modes:
1. Full JSON config via SPECULATIVE_CONFIG env var
2. Individual env vars for common settings
Speculative Methods:
- "draft_model": Use a smaller draft model for speculation
- "ngram": Use n-gram based prompt lookup (no additional model needed)
- "eagle" / "eagle3": Use EAGLE-based speculation
- "medusa": Use Medusa heads for speculation
- "mlp_speculator": Use MLP-based speculator
Returns:
dict | None: Speculative config dictionary or None if not configured
"""
# Option 1: Full JSON configuration
spec_config_json = os.getenv('SPECULATIVE_CONFIG')
if spec_config_json:
try:
config = json.loads(spec_config_json)
logging.info(f"Using speculative config from SPECULATIVE_CONFIG: {config}")
return config
except json.JSONDecodeError as e:
logging.error(f"Failed to parse SPECULATIVE_CONFIG JSON: {e}")
return None
# Option 2: Build config from individual environment variables
spec_method = os.getenv('SPECULATIVE_METHOD') # ngram, draft_model, eagle, eagle3, medusa, mlp_speculator
spec_model = os.getenv('SPECULATIVE_MODEL')
num_spec_tokens = os.getenv('NUM_SPECULATIVE_TOKENS')
# N-gram specific settings
ngram_max = os.getenv('NGRAM_PROMPT_LOOKUP_MAX')
ngram_min = os.getenv('NGRAM_PROMPT_LOOKUP_MIN')
# Check if any speculative decoding is configured
if not any([spec_method, spec_model, ngram_max]):
return None
config = {}
# Determine method
if spec_method:
config['method'] = spec_method
elif ngram_max and not spec_model:
config['method'] = 'ngram'
elif spec_model:
# Auto-detect method based on model name if not specified
model_lower = spec_model.lower()
if 'eagle3' in model_lower:
config['method'] = 'eagle3'
elif 'eagle' in model_lower:
config['method'] = 'eagle'
elif 'medusa' in model_lower:
config['method'] = 'medusa'
else:
config['method'] = 'draft_model'
# Model configuration
if spec_model:
config['model'] = spec_model
# Number of speculative tokens
if num_spec_tokens:
config['num_speculative_tokens'] = int(num_spec_tokens)
# N-gram settings
if ngram_max:
config['prompt_lookup_max'] = int(ngram_max)
if ngram_min:
config['prompt_lookup_min'] = int(ngram_min)
# Draft model tensor parallel size
draft_tp = os.getenv('SPECULATIVE_DRAFT_TENSOR_PARALLEL_SIZE')
if draft_tp:
config['draft_tensor_parallel_size'] = int(draft_tp)
# Max model length for draft
spec_max_len = os.getenv('SPECULATIVE_MAX_MODEL_LEN')
if spec_max_len:
config['max_model_len'] = int(spec_max_len)
# Disable by batch size
disable_batch = os.getenv('SPECULATIVE_DISABLE_BY_BATCH_SIZE')
if disable_batch:
config['disable_by_batch_size'] = int(disable_batch)
# Draft model quantization
spec_quant = os.getenv('SPECULATIVE_QUANTIZATION')
if spec_quant:
config['quantization'] = spec_quant
# Draft model revision
spec_revision = os.getenv('SPECULATIVE_MODEL_REVISION')
if spec_revision:
config['revision'] = spec_revision
# Enforce eager mode for draft model
spec_eager = os.getenv('SPECULATIVE_ENFORCE_EAGER')
if spec_eager:
config['enforce_eager'] = spec_eager.lower() == 'true'
if config:
logging.info(f"Built speculative config from env vars: {config}")
return config
return None
limit_mm_env = os.getenv('LIMIT_MM_PER_PROMPT')
if limit_mm_env is not None:
DEFAULT_ARGS["limit_mm_per_prompt"] = convert_limit_mm_per_prompt(limit_mm_env)
@@ -172,8 +270,9 @@ def get_engine_args():
args["max_seq_len_to_capture"] = int(os.getenv("MAX_CONTEXT_LEN_TO_CAPTURE"))
logging.warning("Using MAX_CONTEXT_LEN_TO_CAPTURE is deprecated. Please use MAX_SEQ_LEN_TO_CAPTURE instead.")
# if "gemma-2" in args.get("model", "").lower():
# os.environ["VLLM_ATTENTION_BACKEND"] = "FLASHINFER"
# logging.info("Using FLASHINFER for gemma-2 model.")
# Add speculative decoding configuration if present
speculative_config = get_speculative_config()
if speculative_config:
args["speculative_config"] = speculative_config
return AsyncEngineArgs(**args)
+1 -2
View File
@@ -3,11 +3,10 @@ import logging
from http import HTTPStatus
from functools import wraps
from time import time
from vllm.entrypoints.openai.protocol import RequestResponseMetadata
try:
from vllm.utils import random_uuid
from vllm.entrypoints.openai.protocol import ErrorResponse
from vllm.entrypoints.openai.engine.protocol import ErrorResponse, RequestResponseMetadata
from vllm import SamplingParams
except ImportError:
logging.warning("Error importing vllm, skipping related imports. This is ONLY expected when baking model into docker image from a machine without GPUs")