diff --git a/README.md b/README.md index 28fc0d3..5f4c8d5 100644 --- a/README.md +++ b/README.md @@ -71,10 +71,11 @@ Development Image: ```runpod/worker-vllm:dev``` Note that the more GPUs you split a model's weights accross, the slower it will be due to inter-GPU communication overhead. If you can fit the model on a single GPU, it is recommended to do so. - `TENSOR_PARALLEL_SIZE`: Number of GPUs to shard the model across (default: `1`). + - If you are having issues loading your model with Tensor Parallelism, try decreasing `VLLM_CPU_FRACTION` (default: `1`). - System Settings: - `GPU_MEMORY_UTILIZATION`: GPU VRAM utilization (default: `0.98`). - - `MAX_PARALLEL_LOADING_WORKERS`: Maximum number of parallel workers for loading models (default: `number of available CPU cores`). + - `MAX_PARALLEL_LOADING_WORKERS`: Maximum number of parallel workers for loading models (default: `number of available CPU cores` if `TENSOR_PARALLEL_SIZE` is `1`, otherwise `None`). - Serverless Settings: diff --git a/src/engine.py b/src/engine.py index 5b53e3f..356fee3 100644 --- a/src/engine.py +++ b/src/engine.py @@ -41,7 +41,7 @@ class vLLMEngine: load_dotenv() # For local development self.config = self._initialize_config() logging.info("vLLM config: %s", self.config) - self.tokenizer = Tokenizer(os.environ.get("TOKENIZER_NAME", os.environ.get("MODEL_NAME"))) + self.tokenizer = Tokenizer(os.getenv("TOKENIZER_NAME", os.getenv("MODEL_NAME"))) self.llm = self._initialize_llm() if engine is None else engine self.openai_engine = self._initialize_openai() self.max_concurrency = int(os.getenv("MAX_CONCURRENCY", DEFAULT_MAX_CONCURRENCY)) @@ -178,7 +178,7 @@ class vLLMEngine: "disable_log_requests": bool(int(os.getenv("DISABLE_LOG_REQUESTS", 1))), "trust_remote_code": bool(int(os.getenv("TRUST_REMOTE_CODE", 0))), "gpu_memory_utilization": float(os.getenv("GPU_MEMORY_UTILIZATION", 0.95)), - "max_parallel_loading_workers": int(os.getenv("MAX_PARALLEL_LOADING_WORKERS", count_physical_cores())), + "max_parallel_loading_workers": self._get_max_parallel_loading_workers(), "max_model_len": self._get_max_model_len(), "tensor_parallel_size": self._get_num_gpu_shard(), } @@ -196,6 +196,12 @@ class vLLMEngine: else: return None + def _get_max_parallel_loading_workers(self): + if int(os.getenv("TENSOR_PARALLEL_SIZE", 1)) > 1: + return None + else: + return int(os.getenv("MAX_PARALLEL_LOADING_WORKERS", count_physical_cores())) + def _get_model_name_and_path(self): if os.path.exists("/local_model_path.txt"): model, download_dir = open("/local_model_path.txt", "r").read().strip(), None @@ -213,7 +219,7 @@ class vLLMEngine: return num_gpu_shard def _get_max_model_len(self): - max_model_len = os.getenv("MAX_MODEL_LEN") + max_model_len = os.getenv("MAX_MODEL_LENGTH") return int(max_model_len) if max_model_len is not None else None def _get_n_current_jobs(self):