- Multi-stage Docker build (simplified to single-stage with pre-converted ONNX) - HTTP server with ONNX inference - API secret authentication - Uses pre-converted all-MiniLM-L6-v2 ONNX model from onnx-community - Image size: ~373 MB Generated by Mistral Vibe. Co-Authored-By: Mistral Vibe <vibe@mistral.ai>
27 lines
1.3 KiB
Docker
27 lines
1.3 KiB
Docker
# Minimal image: ~200-250MB
|
|
# Uses pre-converted ONNX model from onnx-community
|
|
|
|
FROM python:3.11-slim
|
|
WORKDIR /app
|
|
|
|
# Install runtime dependencies and download model
|
|
RUN apt-get update && \
|
|
apt-get install -y --no-install-recommends wget && \
|
|
pip install --no-cache-dir onnxruntime numpy && \
|
|
# Download pre-converted ONNX model and tokenizer files
|
|
wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/onnx/model.onnx -O /app/model.onnx && \
|
|
wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/onnx/model.onnx_data -O /app/model.onnx_data && \
|
|
wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/tokenizer.json -O /app/tokenizer.json && \
|
|
wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/tokenizer_config.json -O /app/tokenizer_config.json && \
|
|
wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/vocab.txt -O /app/vocab.txt && \
|
|
wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/config.json -O /app/config.json && \
|
|
# Clean up
|
|
apt-get remove -y wget 2>/dev/null || true && \
|
|
apt-get clean && \
|
|
rm -rf /var/lib/apt/lists/* /tmp/* /root/.cache /var/cache/apt/*
|
|
|
|
COPY server.py .
|
|
|
|
EXPOSE 8080
|
|
ENV PORT=8080
|
|
CMD ["python", "server.py"]
|