Add Go implementation with ONNX Runtime

- Use golang:1.23-bookworm as builder
- Use debian:bookworm-slim as runtime
- Use github.com/yalue/onnxruntime_go for ONNX inference
- Use github.com/sugarme/tokenizer for tokenization
- Download ONNX Runtime v1.27.0 shared library
- Download model.onnx and model.onnx_data from onnx-community
- Support API_SECRET environment variable for authentication
- Final image size: ~266 MB

Generated by Mistral Vibe.
Co-Authored-By: Mistral Vibe <vibe@mistral.ai>
This commit is contained in:
Christoph Haas 2026-06-28 12:35:31 +02:00
parent 6101df3bfe
commit 21d467a284
4 changed files with 303 additions and 19 deletions

View file

@ -1,27 +1,59 @@
# Minimal image: ~200-250MB
# Uses pre-converted ONNX model from onnx-community
# Go implementation: Minimal image with ONNX Runtime
# Multi-stage build: build with Go + ONNX Runtime deps, runtime with minimal Debian
FROM python:3.11-slim
# Stage 1: Build Go binary
FROM golang:1.23-bookworm AS builder
WORKDIR /app
# Install runtime dependencies and download model
# Install build dependencies: git, g++, make, ca-certificates, curl, libc6-dev
RUN apt-get update && \
apt-get install -y --no-install-recommends wget && \
pip install --no-cache-dir onnxruntime numpy && \
# Download pre-converted ONNX model and tokenizer files
wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/onnx/model.onnx -O /app/model.onnx && \
wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/onnx/model.onnx_data -O /app/model.onnx_data && \
wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/tokenizer.json -O /app/tokenizer.json && \
wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/tokenizer_config.json -O /app/tokenizer_config.json && \
wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/vocab.txt -O /app/vocab.txt && \
wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/config.json -O /app/config.json && \
# Clean up
apt-get remove -y wget 2>/dev/null || true && \
apt-get clean && \
rm -rf /var/lib/apt/lists/* /tmp/* /root/.cache /var/cache/apt/*
apt-get install -y --no-install-recommends git g++ make ca-certificates curl libc6-dev && \
rm -rf /var/lib/apt/lists/*
COPY server.py .
# Configure git to avoid terminal prompt
ENV GIT_TERMINAL_PROMPT=0
# Copy Go files
COPY go.mod go.sum .
COPY main.go .
# Download Go dependencies
RUN go mod download 2>&1
# Download ONNX Runtime shared library for Linux x64
RUN curl -sL -o /tmp/onnx.tgz https://github.com/microsoft/onnxruntime/releases/download/v1.27.0/onnxruntime-linux-x64-1.27.0.tgz
RUN tar -xzf /tmp/onnx.tgz -C /tmp
RUN mkdir -p /usr/local/lib && cp /tmp/onnxruntime-linux-x64-1.27.0/lib/libonnxruntime.so* /usr/local/lib/
RUN rm -rf /tmp/onnxruntime-linux-x64-1.27.0 /tmp/onnx.tgz
# Build Go binary with CGO enabled
RUN CGO_ENABLED=1 GOOS=linux GOARCH=amd64 go build -o /app/vector-server main.go 2>&1
# Stage 2: Runtime
FROM debian:bookworm-slim
WORKDIR /app
# Install runtime dependencies: libstdc++, ca-certificates, curl
RUN apt-get update && \
apt-get install -y --no-install-recommends libstdc++6 ca-certificates curl && \
rm -rf /var/lib/apt/lists/*
# Download model and tokenizer files
RUN curl -sL -o /app/model.onnx https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/onnx/model.onnx && \
curl -sL -o /app/model.onnx_data https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/onnx/model.onnx_data && \
curl -sL -o /app/tokenizer.json https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/tokenizer.json && \
apt-get remove -y curl 2>/dev/null || true && \
apt-get clean && \
rm -rf /var/lib/apt/lists/* /tmp/* /var/tmp/*
# Copy binary and shared library from builder
COPY --from=builder /usr/local/lib/libonnxruntime.so* /usr/local/lib/
COPY --from=builder /app/vector-server .
# Set environment for ONNX Runtime
ENV LD_LIBRARY_PATH=/usr/local/lib:${LD_LIBRARY_PATH}
ENV ONNXRUNTIME_SHARED_LIBRARY_PATH=/usr/local/lib/libonnxruntime.so
EXPOSE 8080
ENV PORT=8080
CMD ["python", "server.py"]
CMD ["./vector-server"]