diff --git a/README.md b/README.md index e7bfdde..76798f0 100644 --- a/README.md +++ b/README.md @@ -20,6 +20,10 @@ See the [test_inputs](./test_inputs) directory for example test requests. Streaming for openai requests are fully working. +## Preload model into the docker image + +See the [embed_model](./embed_model/) directory for instructions. + ## Licence This project is licensed under the Creative Commons Attribution 4.0 International License. You are free to use, share, and adapt the material for any purpose, even commercially, under the following terms: diff --git a/embed_model/Dockerfile b/embed_model/Dockerfile new file mode 100644 index 0000000..0a2836a --- /dev/null +++ b/embed_model/Dockerfile @@ -0,0 +1,10 @@ +FROM svenbrnn/runpod-ollama:0.5.7 + +ARG MODEL_NAME +ENV MODEL_NAME=$MODEL_NAME +ADD preload_model.sh /preload_model.sh +RUN chmod +x /preload_model.sh && /preload_model.sh + +# Copy the model to the volume +FROM svenbrnn/runpod-ollama:0.5.7 +COPY --from=0 /runpod-volume /runpod-volume \ No newline at end of file diff --git a/embed_model/README.md b/embed_model/README.md new file mode 100644 index 0000000..3c9796d --- /dev/null +++ b/embed_model/README.md @@ -0,0 +1,3 @@ +# Embed model into runpod-worker-ollama + +Run ``docker build --build-arg MODEL_NAME="" -t /: .`` \ No newline at end of file diff --git a/embed_model/preload_model.sh b/embed_model/preload_model.sh new file mode 100644 index 0000000..d625056 --- /dev/null +++ b/embed_model/preload_model.sh @@ -0,0 +1,31 @@ +#!/bin/sh + +# Start the Ollama server in the background +echo "Starting Ollama server to preload model: $MODEL_NAME" +ollama serve & + +# Capture the PID of the Ollama server +OLLAMA_PID=$! + +# Wait for the server to be ready (adjust if necessary) +echo "Waiting for Ollama server to start..." +sleep 5 + +# Pull the specified model +echo "Pulling model: $MODEL_NAME" +if ollama pull "$MODEL_NAME"; then + echo "Successfully pulled model: $MODEL_NAME" +else + echo "Failed to pull model: $MODEL_NAME" + kill $OLLAMA_PID + exit 1 +fi + +# Stop the Ollama server +echo "Stopping Ollama server..." +kill $OLLAMA_PID + +# Wait for the server to terminate +wait $OLLAMA_PID + +echo "Model preloading complete."