Team Ai
Apppublic

matthoffner/llama-cpp-server

sourceHugging Faceupdated 3y agoView on Hugging Face
0likes
Dockerfile77 linesDownload Raw Back to root
1# Set arguments for versions2ARG UBUNTU_VERSION=22.043ARG CUDA_VERSION=11.7.14ARG BASE_CUDA_DEV_CONTAINER=nvidia/cuda:${CUDA_VERSION}-devel-ubuntu${UBUNTU_VERSION}5ARG BASE_CUDA_RUN_CONTAINER=nvidia/cuda:${CUDA_VERSION}-runtime-ubuntu${UBUNTU_VERSION}6 7# Build stage with CUDA development container8FROM ${BASE_CUDA_DEV_CONTAINER} as build9 10# Install build essentials and git11RUN apt-get update && \12    apt-get install -y build-essential git13 14# Install Python3 and pip15RUN apt-get install -y python3 python3-pip16 17# Set work directory to /app18WORKDIR /app19 20# Copy your application code to the container21COPY . .22 23# Create a non-root user 'user' in the build stage as well24RUN useradd -m -u 1000 user25 26# Switch to the non-root user for any further commands27USER user28 29# Set nvcc architecture and enable cuBLAS30ENV CUDA_DOCKER_ARCH=all \31    LLAMA_CUBLAS=132 33# Runtime stage with CUDA runtime container34FROM ${BASE_CUDA_RUN_CONTAINER} as runtime35 36# Re-create the non-root user 'user' in the runtime stage37RUN useradd -m -u 1000 user && \38    apt-get update && \39    apt-get install -y libopenblas-dev ninja-build build-essential pkg-config curl40 41# Switch to non-root user42USER user43 44# Set home and path for the user45ENV HOME=/home/user \46    PATH=/home/user/.local/bin:$PATH47 48# Set work directory to user's home directory49WORKDIR $HOME/app50 51# Install Python3 and pip for the runtime container52USER root53RUN apt-get install -y python3 python3-pip54 55# Switch back to the non-root user for installing Python packages56USER user57RUN pip install --no-cache-dir --upgrade pip setuptools wheel && \58    pip install --verbose llama-cpp-python[server]59 60# Download the model to the user's directory61RUN mkdir $HOME/model && \62    curl -L https://huggingface.co/matthoffner/Magicoder-S-DS-6.7B-GGUF/resolve/main/Magicoder-S-DS-6.7B_Q4_K_M.gguf -o $HOME/model/gguf-model.gguf63 64COPY --chown=user ./main.py $HOME/app/65 66# Set environment variables for the host67ENV HOST=0.0.0.0 \68    PORT=786069 70# Expose the server port71EXPOSE ${PORT}72 73RUN ls -la $HOME/model74 75# Run the server start script76CMD ["uvicorn", "main:app", "--host", "0.0.0.0", "--port", "7860", "--log-level", "debug"]77