Compare commits
8
Commits
v0.0.1-alpha
...
v1.0.1
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
403d318ffc | ||
|
|
381ed9ffff | ||
|
|
c42b80ebbd | ||
|
|
40d4097799 | ||
|
|
36a8b32d1e | ||
|
|
5f6c099504 | ||
|
|
be3de61c52 | ||
|
|
558da755c3 |
+2
-2
@@ -14,10 +14,10 @@
|
|||||||
{
|
{
|
||||||
"key": "LLAMA_SERVER_CMD_ARGS",
|
"key": "LLAMA_SERVER_CMD_ARGS",
|
||||||
"input": {
|
"input": {
|
||||||
"name": "Model Name",
|
"name": "Command line arguments for llama-server",
|
||||||
"type": "string",
|
"type": "string",
|
||||||
"description": "Launch command line arguments (argv) for the llama-server binary. Do not define the port.",
|
"description": "Launch command line arguments (argv) for the llama-server binary. Do not define the port.",
|
||||||
"default": "-hf unsloth/Mistral-Small-3.2-24B-Instruct-2506-GGUF:Q4_K_M -ctx_size 4096",
|
"default": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096 -ngl 99",
|
||||||
"advanced": false
|
"advanced": false
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
|
|||||||
+1
-1
@@ -14,7 +14,7 @@
|
|||||||
"env": [
|
"env": [
|
||||||
{
|
{
|
||||||
"key": "LLAMA_SERVER_CMD_ARGS",
|
"key": "LLAMA_SERVER_CMD_ARGS",
|
||||||
"value": "-hf unsloth/Mistral-Small-3.2-24B-Instruct-2506-GGUF:Q4_K_M -ctx_size 4096"
|
"value": "-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096"
|
||||||
}
|
}
|
||||||
],
|
],
|
||||||
"allowedCudaVersions": [
|
"allowedCudaVersions": [
|
||||||
|
|||||||
+1
-1
@@ -33,7 +33,7 @@ WORKDIR /work
|
|||||||
ADD ./src /work
|
ADD ./src /work
|
||||||
|
|
||||||
# Install runpod and its dependencies
|
# Install runpod and its dependencies
|
||||||
RUN pip install -r requirements.txt && chmod +x /work/start.sh
|
RUN pip install -r ./requirements.txt && chmod +x /work/start.sh
|
||||||
|
|
||||||
# Set the entrypoint
|
# Set the entrypoint
|
||||||
ENTRYPOINT ["/bin/sh", "-c", "/work/start.sh"]
|
ENTRYPOINT ["/bin/sh", "-c", "/work/start.sh"]
|
||||||
|
|||||||
@@ -4,8 +4,6 @@
|
|||||||
|
|
||||||
# Serverless llama.cpp inference worker for RunPod
|
# Serverless llama.cpp inference worker for RunPod
|
||||||
|
|
||||||
[](https://console.runpod.io/hub/Jacob-ML/inference-worker)
|
|
||||||
|
|
||||||
This repository contains a serverless inference worker for running llama.cpp models on RunPod. It uses the `llama-server` image to provide an API for interacting with the models.
|
This repository contains a serverless inference worker for running llama.cpp models on RunPod. It uses the `llama-server` image to provide an API for interacting with the models.
|
||||||
The following OpenAI API endpoints are supported:
|
The following OpenAI API endpoints are supported:
|
||||||
|
|
||||||
@@ -24,9 +22,11 @@ Make sure your RunPod worker has access to the network volume, i.e. is located i
|
|||||||
|
|
||||||
The worker can be configured via environment variables set in the RunPod hub configuration:
|
The worker can be configured via environment variables set in the RunPod hub configuration:
|
||||||
|
|
||||||
- `LLAMA_SERVER_CMD_ARGS`: Command line arguments (argv) for the `llama-server` binary. Example: `-hf /path/to/model.gguf:Q4_K_M -ctx_size 4096`. **IMPORTANT**: Please do not define the port argument here, as the worker will always use port `3098` automatically.
|
- `LLAMA_SERVER_CMD_ARGS`: Command line arguments (argv) for the `llama-server` binary. Example: `-hf /path/to/model.gguf:Q4_K_M --ctx-size 4096`. **IMPORTANT**: Please do not define the port argument here, as the worker will always use port `3098` automatically.
|
||||||
- `MAX_CONCURRENCY`: Maximum number of concurrent requests the worker can handle. Default is `8`.
|
- `MAX_CONCURRENCY`: Maximum number of concurrent requests the worker can handle. Default is `8`.
|
||||||
|
|
||||||
## License
|
## License
|
||||||
|
|
||||||
Please see the [LICENSE](./LICENSE) file for more information.
|
Please see the [LICENSE](./LICENSE) file for more information.
|
||||||
|
|
||||||
|
[](https://console.runpod.io/hub/Jacob-ML/inference-worker)
|
||||||
|
|||||||
@@ -0,0 +1,6 @@
|
|||||||
|
"""
|
||||||
|
This is an empty file used to mark the repository as a Runpod-compatible
|
||||||
|
serverless endpoint because they won't stop pretending it's not.
|
||||||
|
|
||||||
|
I'm tired.
|
||||||
|
"""
|
||||||
@@ -21,7 +21,6 @@ Typical usage:
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
import json
|
import json
|
||||||
import os
|
|
||||||
|
|
||||||
from dotenv import load_dotenv
|
from dotenv import load_dotenv
|
||||||
from openai import OpenAI
|
from openai import OpenAI
|
||||||
|
|||||||
@@ -1,4 +1,3 @@
|
|||||||
runpod
|
runpod
|
||||||
python-dotenv
|
python-dotenv
|
||||||
openai
|
openai
|
||||||
orjson==3.10.14
|
|
||||||
|
|||||||
+35
-12
@@ -1,24 +1,33 @@
|
|||||||
#!/bin/bash
|
#!/bin/bash
|
||||||
|
|
||||||
|
# fail on error:
|
||||||
|
set -e -o pipefail
|
||||||
|
|
||||||
# This script starts the llama-server with the command line arguments
|
# This script starts the llama-server with the command line arguments
|
||||||
# specified in the environment variable LLAMA_SERVER_CMD_ARGS, ensuring
|
# specified in the environment variable LLAMA_SERVER_CMD_ARGS, ensuring
|
||||||
# that the server listens on port 3098. It also starts the handler.py
|
# that the server listens on port 3098. It also starts the handler.py
|
||||||
# script after the server is up and running.
|
# script after the server is up and running.
|
||||||
|
|
||||||
cleanup() {
|
cleanup() {
|
||||||
echo "Cleaning up..."
|
echo "start.sh: Cleaning up..."
|
||||||
pkill -P $$ # kill all child processes of the current script
|
pkill -P $$ # kill all child processes of the current script
|
||||||
exit 0
|
exit 0
|
||||||
}
|
}
|
||||||
|
|
||||||
# check if the substring /workspace is in LLAMA_SERVER_CMD_ARGS
|
# check if $LLAMA_SERVER_CMD_ARGS is set
|
||||||
if [[ "$LLAMA_SERVER_CMD_ARGS" != *"/workspace"* ]]; then
|
if [ -z "$LLAMA_SERVER_CMD_ARGS" ]; then
|
||||||
echo "Tip: For reduced downloads and faster startup times, consider using a model stored in a network volume mounted to /workspace."
|
echo "start.sh: Warning: LLAMA_SERVER_CMD_ARGS is not set. Defaulting to -hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096"
|
||||||
|
LLAMA_SERVER_CMD_ARGS="-hf unsloth/gemma-3-270m-it-GGUF:Q6_K --ctx-size 4096 -ngl 99"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# check if the substring -port is in LLAMA_SERVER_CMD_ARGS
|
# check if the substring /workspace is in LLAMA_SERVER_CMD_ARGS
|
||||||
if [[ "$LLAMA_SERVER_CMD_ARGS" != *"-port"* ]]; then
|
if [[ "$LLAMA_SERVER_CMD_ARGS" != *"/workspace"* ]]; then
|
||||||
echo "Error: You must not define -port in LLAMA_SERVER_CMD_ARGS, as port 3098 is required."
|
echo "start.sh: Tip: For reduced downloads and faster startup times, consider using a model stored in a network volume mounted to /workspace."
|
||||||
|
fi
|
||||||
|
|
||||||
|
# check if the substring --port is in LLAMA_SERVER_CMD_ARGS and if yes, raise an error:
|
||||||
|
if [[ "$LLAMA_SERVER_CMD_ARGS" == *"--port"* ]]; then
|
||||||
|
echo "start.sh: Error: You must not define --port in LLAMA_SERVER_CMD_ARGS, as port 3098 is required."
|
||||||
exit 1
|
exit 1
|
||||||
fi
|
fi
|
||||||
|
|
||||||
@@ -26,18 +35,27 @@ fi
|
|||||||
trap cleanup SIGINT SIGTERM
|
trap cleanup SIGINT SIGTERM
|
||||||
|
|
||||||
# kill any existing llama-server processes
|
# kill any existing llama-server processes
|
||||||
pgrep llama-server | xargs kill
|
echo "start.sh: Stopping existing llama-server instances (if any)..."
|
||||||
|
{
|
||||||
|
pkill llama-server 2>/dev/null
|
||||||
|
} || {
|
||||||
|
echo "start.sh: No llama-server running"
|
||||||
|
}
|
||||||
|
|
||||||
# we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS;
|
# we have a string with all the command line arguments in the env var LLAMA_SERVER_CMD_ARGS;
|
||||||
# it contains a.e. "-hf modelname -ctx_size 4096".
|
# it contains a.e. "-hf modelname --ctx-size 4096 -ngl 99".
|
||||||
|
|
||||||
|
echo "start.sh: Running llama-server $LLAMA_SERVER_CMD_ARGS --port 3098"
|
||||||
|
|
||||||
|
touch llama.server.log
|
||||||
|
|
||||||
# We need to pass these arguments to llama-server verbatim.
|
# We need to pass these arguments to llama-server verbatim.
|
||||||
llama-server $LLAMA_SERVER_CMD_ARGS -port 3098 2>&1 | tee llama.server.log &
|
LD_LIBRARY_PATH=/app /app/llama-server $LLAMA_SERVER_CMD_ARGS --port 3098 2>&1 | tee llama.server.log &
|
||||||
|
|
||||||
LLAMA_SERVER_PID=$! # store the process ID (PID) of the background command
|
LLAMA_SERVER_PID=$! # store the process ID (PID) of the background command
|
||||||
|
|
||||||
check_server_is_running() {
|
check_server_is_running() {
|
||||||
echo "Checking if llama-server is done initializing..."
|
echo "start.sh: Checking if llama-server is done initializing..."
|
||||||
|
|
||||||
if cat llama.server.log | grep -q "listening"; then
|
if cat llama.server.log | grep -q "listening"; then
|
||||||
return 0 # success
|
return 0 # success
|
||||||
@@ -46,9 +64,14 @@ check_server_is_running() {
|
|||||||
fi
|
fi
|
||||||
}
|
}
|
||||||
|
|
||||||
|
echo "start.sh: Waiting for llama-server to start..."
|
||||||
|
|
||||||
# wait for the server to start
|
# wait for the server to start
|
||||||
while ! check_server_is_running; do
|
while ! check_server_is_running; do
|
||||||
sleep 5
|
# we don't want to lose too much time, so we check very frequently
|
||||||
|
sleep 0.5
|
||||||
done
|
done
|
||||||
|
|
||||||
|
echo "start.sh: llama-server is up and running, delegating to the handler script."
|
||||||
|
|
||||||
python -u handler.py $1
|
python -u handler.py $1
|
||||||
|
|||||||
Reference in New Issue
Block a user