commit 6101df3bfe991a6116e9830754fcd751bc6ba573 Author: Christoph Haas Date: Sun Jun 28 12:04:00 2026 +0200 Initial Python implementation with ONNX Runtime - Multi-stage Docker build (simplified to single-stage with pre-converted ONNX) - HTTP server with ONNX inference - API secret authentication - Uses pre-converted all-MiniLM-L6-v2 ONNX model from onnx-community - Image size: ~373 MB Generated by Mistral Vibe. Co-Authored-By: Mistral Vibe diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..78218d1 --- /dev/null +++ b/Dockerfile @@ -0,0 +1,27 @@ +# Minimal image: ~200-250MB +# Uses pre-converted ONNX model from onnx-community + +FROM python:3.11-slim +WORKDIR /app + +# Install runtime dependencies and download model +RUN apt-get update && \ + apt-get install -y --no-install-recommends wget && \ + pip install --no-cache-dir onnxruntime numpy && \ + # Download pre-converted ONNX model and tokenizer files + wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/onnx/model.onnx -O /app/model.onnx && \ + wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/onnx/model.onnx_data -O /app/model.onnx_data && \ + wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/tokenizer.json -O /app/tokenizer.json && \ + wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/tokenizer_config.json -O /app/tokenizer_config.json && \ + wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/vocab.txt -O /app/vocab.txt && \ + wget -q https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX/resolve/main/config.json -O /app/config.json && \ + # Clean up + apt-get remove -y wget 2>/dev/null || true && \ + apt-get clean && \ + rm -rf /var/lib/apt/lists/* /tmp/* /root/.cache /var/cache/apt/* + +COPY server.py . + +EXPOSE 8080 +ENV PORT=8080 +CMD ["python", "server.py"] diff --git a/README.md b/README.md new file mode 100644 index 0000000..48e03a2 --- /dev/null +++ b/README.md @@ -0,0 +1,100 @@ +# Vector Service + +A minimal Docker container that provides text embedding vectors using the all-MiniLM-L6-v2 model. The service accepts POST requests with text and returns a 384-dimensional embedding vector. + +## Features + +- **Small image size**: ~373 MB (much smaller than typical Python-based solutions) +- **Fast inference**: Uses ONNX Runtime for efficient model execution +- **API authentication**: Optional API secret protection +- **Pre-converted ONNX**: Uses ready-to-use ONNX model from HuggingFace + +## Usage + +### Build the container + +```bash +podman build -t vector-service . +``` + +Or with Docker: + +```bash +docker build -t vector-service . +``` + +The build process will: +1. Download the pre-converted ONNX model from `onnx-community/all-MiniLM-L6-v2-ONNX` on HuggingFace +2. Install only the runtime dependencies (ONNX Runtime + NumPy) +3. Create a minimal image (~373 MB) + +### Run the service + +Without authentication: +```bash +podman run --rm -p 8080:8080 -d vector-service +``` + +With API secret authentication: +```bash +podman run --rm -p 8080:8080 -e API_SECRET=your-secret-key -d vector-service +``` + +### Query the service + +Send a POST request with JSON body: + +```bash +curl -X POST http://localhost:8080/vector \ + -H "Content-Type: application/json" \ + -H "Authorization: Bearer your-secret-key" \ + -d '{"text": "Hello World"}' +``` + +Response: +```json +{ + "vector": [0.1868536774709355, 0.8120285351760685, ...] +} +``` + +The vector has 384 dimensions. + +### Without authentication + +If you didn't set `API_SECRET`, you can query without the Authorization header: + +```bash +curl -X POST http://localhost:8080/vector \ + -H "Content-Type: application/json" \ + -d '{"text": "Hello World"}' +``` + +## Files + +- `Dockerfile`: Single-stage build with pre-downloaded ONNX model +- `server.py`: HTTP server with ONNX inference + +## Technical Details + +### Model +- **Model**: all-MiniLM-L6-v2 (80 MB on disk) +- **Source**: Pre-converted ONNX from [onnx-community/all-MiniLM-L6-v2-ONNX](https://huggingface.co/onnx-community/all-MiniLM-L6-v2-ONNX) +- **Dimensions**: 384 +- **Format**: ONNX (pre-converted) + +### Dependencies +- Runtime: Python 3.11, ONNX Runtime, NumPy +- No build-time dependencies needed (uses pre-converted model) + +### Image Size Breakdown +- Model files (ONNX + ONNX data + tokenizer + vocab): ~95 MB +- Python runtime and dependencies: ~278 MB +- Total: ~373 MB + +## Notes + +- The build will download the pre-converted ONNX model from HuggingFace (~95 MB total) +- Much faster builds since no PyTorch or model conversion is needed +- The ONNX model includes an external data file (`model.onnx_data`) which is normal for larger models +- For production use, consider adding rate limiting and HTTPS diff --git a/server.py b/server.py new file mode 100644 index 0000000..9f8e896 --- /dev/null +++ b/server.py @@ -0,0 +1,100 @@ +#!/usr/bin/env python3 +import json, os, sys, numpy as np +import onnxruntime as ort +from http.server import BaseHTTPRequestHandler, HTTPServer + +# Load model at startup +session = ort.InferenceSession("/app/model.onnx") + +# Load vocab from vocab.txt +vocab = {} +with open("/app/vocab.txt", "r", encoding="utf-8") as f: + for idx, token in enumerate(f): + token = token.strip() + vocab[token] = idx + +# Token IDs from tokenizer_config.json +with open("/app/tokenizer_config.json", "r") as f: + tokenizer_config = json.load(f) + +# Map token strings to IDs using vocab +cls_token_id = vocab.get(tokenizer_config["cls_token"], 0) +sep_token_id = vocab.get(tokenizer_config["sep_token"], 0) +pad_token_id = vocab.get(tokenizer_config["pad_token"], 0) +unk_token_id = vocab.get(tokenizer_config["unk_token"], 0) + +def wordpiece_tokenize(text): + text = text.lower() + tokens = [] + buffer = "" + for char in text: + if char.isspace(): + if buffer: + tokens.append(buffer if buffer in vocab else "[UNK]") + buffer = "" + else: + buffer += char + if buffer: + tokens.append(buffer if buffer in vocab else "[UNK]") + return tokens + +def encode(text, max_len=128): + token_ids = [vocab.get(t, unk_token_id) for t in wordpiece_tokenize(text)] + input_ids = [cls_token_id] + token_ids + [sep_token_id] + if len(input_ids) > max_len: + input_ids = input_ids[:max_len] + else: + input_ids = input_ids + [pad_token_id] * (max_len - len(input_ids)) + attention_mask = [1] * len(input_ids) + token_type_ids = [0] * len(input_ids) + return input_ids, attention_mask, token_type_ids + +print("Model loaded. Starting server on port 8080...") + +API_SECRET = os.getenv("API_SECRET") + +class VectorHandler(BaseHTTPRequestHandler): + def do_POST(self): + if self.path != "/vector": + self.send_error(404) + return + + # Check API secret + if API_SECRET: + auth_header = self.headers.get("Authorization", "") + if auth_header != f"Bearer {API_SECRET}": + self.send_error(401, "Unauthorized - Invalid or missing API key") + return + + content_length = int(self.headers.get("Content-Length", 0)) + body = self.rfile.read(content_length) + try: + req = json.loads(body) + text = req.get("text", "") + if not text: + self.send_error(400, "Text is required") + return + input_ids, attention_mask, token_type_ids = encode(text) + input_ids_np = np.array([input_ids], dtype=np.int64) + attention_mask_np = np.array([attention_mask], dtype=np.int64) + token_type_ids_np = np.array([token_type_ids], dtype=np.int64) + outputs = session.run( + ["last_hidden_state"], + {"input_ids": input_ids_np, "attention_mask": attention_mask_np, "token_type_ids": token_type_ids_np} + ) + token_embeddings = outputs[0][0] + input_mask_expanded = np.expand_dims(attention_mask_np[0], axis=-1) + sum_embeddings = np.sum(token_embeddings * input_mask_expanded, axis=0) + sum_mask = np.maximum(np.sum(input_mask_expanded, axis=0), 1e-9) + sentence_embedding = (sum_embeddings / sum_mask).tolist() + resp = json.dumps({"vector": sentence_embedding}) + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.end_headers() + self.wfile.write(resp.encode()) + except Exception as e: + self.send_error(500, str(e)) + def log_message(self, format, *args): + pass # Suppress logs + +HTTPServer(("0.0.0.0", int(os.getenv("PORT", 8080))), VectorHandler).serve_forever()