fsconvo / Dockerfile
hari7775's picture
perf: add OpenMP thread pinning, passive wait policy, and core utilization tuning
534bb06
Raw History Blame Contribute Delete
713 Bytes
FROM python:3.10-slim
WORKDIR /app
RUN apt-get update && apt-get install -y \
build-essential \
git \
tesseract-ocr \
&& rm -rf /var/lib/apt/lists/*
# Install pre-compiled CPU binary wheel for instant llama.cpp installation
RUN pip install --no-cache-dir llama-cpp-python --extra-index-url https://abetlen.github.io/llama-cpp-python/whl/cpu
COPY requirements.txt .
RUN pip install --no-cache-dir -r requirements.txt
COPY . .
RUN mkdir -p /data/users
ENV PYTHONUNBUFFERED=1
ENV HF_API_TOKEN=""
ENV OMP_NUM_THREADS=2
ENV OMP_PROC_BIND=CLOSE
ENV OMP_WAIT_POLICY=PASSIVE
ENV KMP_BLOCKTIME=0
ENV MKL_NUM_THREADS=2
EXPOSE 7860
CMD ["uvicorn", "main:app", "--host", "0.0.0.0", "--port", "7860"]