codebyam/Llama-3.2-3B-Instruct-Q8_0-GGUF-DEMO2
0
1FROM ubuntu:22.042 3# Install system dependencies4RUN apt-get update && \5 apt-get install -y \6 build-essential \7 libssl-dev \8 zlib1g-dev \9 libboost-math-dev \10 libboost-python-dev \11 libboost-timer-dev \12 libboost-thread-dev \13 libboost-system-dev \14 libboost-filesystem-dev \15 libopenblas-dev \16 libomp-dev \17 cmake \18 pkg-config \19 git \20 python3-pip \21 curl \22 libcurl4-openssl-dev \23 wget && \24 rm -rf /var/lib/apt/lists/*25 26# Build llama.cpp with OpenBLAS27RUN git clone https://github.com/ggerganov/llama.cpp && \28 cd llama.cpp && \29 cmake -B build -S . \30 -DLLAMA_BUILD_SERVER=ON \31 -DLLAMA_BUILD_EXAMPLES=ON \32 -DGGML_BLAS=ON \33 -DGGML_BLAS_VENDOR=OpenBLAS \34 -DCMAKE_BUILD_TYPE=Release && \35 cmake --build build --config Release --target llama-server -j $(nproc)36 37RUN cd /llama.cpp/build && ./bin/llama-server --list-devices38 39# Download model40RUN mkdir -p /models && \41 wget -O /models/model.q8_0.gguf https://huggingface.co/unsloth/Llama-3.2-3B-Instruct-GGUF/resolve/main/Llama-3.2-3B-Instruct-Q6_K.gguf42 43 44RUN pip install fastapi uvicorn openai45 46# Copy app and startup script47COPY app.py /app.py48COPY start.sh /start.sh49RUN chmod +x /start.sh50 51# Expose ports52EXPOSE 7860 808053 54# Start services55CMD ["/start.sh"]