# Minimal image: ~200-250MB
# Uses pre-converted ONNX model from onnx-community

FROM python:3.11-slim
WORKDIR /app

# Install runtime dependencies and download model
RUN apt-get update && \
    apt-get install -y --no-install-recommends wget && \
    pip install --no-cache-dir onnxruntime numpy && \
    # Download pre-converted ONNX model and tokenizer files
    wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/onnx/model.onnx -O /app/model.onnx && \
    wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/onnx/model.onnx_data -O /app/model.onnx_data && \
    wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/tokenizer.json -O /app/tokenizer.json && \
    wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/tokenizer_config.json -O /app/tokenizer_config.json && \
    wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/vocab.txt -O /app/vocab.txt && \
    wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/config.json -O /app/config.json && \
    # Clean up
    apt-get remove -y wget 2>/dev/null || true && \
    apt-get clean && \
    rm -rf /var/lib/apt/lists/* /tmp/* /root/.cache /var/cache/apt/*

COPY server.py .

EXPOSE 8080
ENV PORT=8080
CMD ["python", "server.py"]
