# Minimal image: ~200-250MB # Uses pre-converted ONNX model from onnx-community FROM python:3.11-slim WORKDIR /app # Install runtime dependencies and download model RUN apt-get update && \ apt-get install -y --no-install-recommends wget && \ pip install --no-cache-dir onnxruntime numpy && \ # Download pre-converted ONNX model and tokenizer files wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/onnx/model.onnx -O /app/model.onnx && \ wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/onnx/model.onnx_data -O /app/model.onnx_data && \ wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/tokenizer.json -O /app/tokenizer.json && \ wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/tokenizer_config.json -O /app/tokenizer_config.json && \ wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/vocab.txt -O /app/vocab.txt && \ wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/config.json -O /app/config.json && \ # Clean up apt-get remove -y wget 2>/dev/null || true && \ apt-get clean && \ rm -rf /var/lib/apt/lists/* /tmp/* /root/.cache /var/cache/apt/* COPY server.py . EXPOSE 8080 ENV PORT=8080 CMD ["python", "server.py"]