This commit is contained in:
Alpay Ariyak
2024-01-16 18:38:26 -05:00
committed by GitHub
parent a5dc8b53b7
commit ed48093792
+1 -1
View File
@@ -76,7 +76,7 @@ class vLLMEngine:
return total_sequences
def _get_quantization(self):
quantization = os.getenv("QUANTIZATION").lower()
quantization = os.getenv("QUANTIZATION", "").lower()
return quantization if quantization in ["awq", "squeezellm", "gptq"] else None
def concurrency_modifier(self, current_concurrency):