diff --git a/embedding_service/README.md b/embedding_service/README.md new file mode 100644 index 0000000..2a0ced3 --- /dev/null +++ b/embedding_service/README.md @@ -0,0 +1,104 @@ +# Embedding service + +A small FastAPI service that turns texts into 768-dimensional vectors with +[`intfloat/multilingual-e5-base`](https://huggingface.co/intfloat/multilingual-e5-base). +The portfolio assistant uses it to index the knowledge base and to embed +visitor questions. + +It runs as its own process so the model is loaded once. Inside Django, every +Gunicorn worker would load its own copy. + +## Setup + +The service has its own virtual environment, separate from the Django one: + +```bash +cd embedding_service +python -m venv .venv +``` + +Activate it. **Windows (PowerShell):** + +```powershell +.\.venv\Scripts\Activate.ps1 +``` + +**macOS / Linux:** + +```bash +source .venv/bin/activate +``` + +Then install the dependencies: + +```bash +python -m pip install --upgrade pip +python -m pip install -r requirements.txt +``` + +`requirements.txt` pulls the CPU-only PyTorch build from the PyTorch package +index. The default Linux wheel on PyPI would add several gigabytes of CUDA +libraries. + +## Running + +```bash +uvicorn main:app --host 127.0.0.1 --port 8002 +``` + +The first start downloads the model (about 1.1 GB) into the Hugging Face +cache. Uvicorn only accepts connections once the model is loaded. The process +uses about 1.8 GB of memory after the first request. + +## API + +Interactive documentation is available at `/docs` while the service runs. + +### `GET /health` + +```json +{"status": "ok", "model": "intfloat/multilingual-e5-base"} +``` + +### `POST /embed` + +```json +{"kind": "query", "texts": ["Wie deployt Benjamin seine Projekte?"]} +``` + +- `kind` is `query` for search questions and `passage` for knowledge base + sections. The service adds the prefix e5 was trained with. Without it the + model still returns vectors, but retrieval quality drops without any error. +- `texts` holds 1 to 64 non-blank strings. + +The response contains one vector per text, in input order: + +```json +{"model": "intfloat/multilingual-e5-base", "dimensions": 768, "vectors": [[0.026, 0.063, ...]]} +``` + +Vectors are normalized to length 1, so their dot product is the cosine +similarity. e5 scores cluster between roughly 0.7 and 1.0 even for unrelated +texts. Only their order is meaningful, not their absolute value. + +Invalid input returns `422`. So does any text longer than the model's limit +of 512 tokens, which the model would otherwise cut off silently. +`detail.indexes` lists the positions of the affected texts. + +## Smoke test + +With the service running: + +```bash +python smoke_test.py +``` + +It checks the vector size and that related passages, German and English, +score higher than an unrelated one. It exits with `0` on success and `1` on +failure, and needs nothing beyond the Python standard library. + +## Why this folder is not a Python package + +There is deliberately no `__init__.py`. Without it, Django's test runner and +coverage ignore this folder, which matters because FastAPI and PyTorch are not +installed in the Django environment or in CI. diff --git a/embedding_service/main.py b/embedding_service/main.py new file mode 100644 index 0000000..633a6a0 --- /dev/null +++ b/embedding_service/main.py @@ -0,0 +1,85 @@ +"""HTTP service that turns texts into multilingual-e5-base embeddings.""" + +from contextlib import asynccontextmanager +from typing import Annotated, Literal + +from fastapi import FastAPI, HTTPException +from pydantic import BaseModel, Field, StringConstraints +from sentence_transformers import SentenceTransformer + +MODEL_NAME = "intfloat/multilingual-e5-base" +MAX_TEXTS_PER_REQUEST = 64 + +ml_models = {} + + +@asynccontextmanager +async def lifespan(app: FastAPI): + """Load the model once at startup instead of once per request.""" + ml_models["embedder"] = SentenceTransformer(MODEL_NAME, device="cpu") + yield + ml_models.clear() + + +app = FastAPI(title="Embedding service", lifespan=lifespan) + +Text = Annotated[str, StringConstraints(strip_whitespace=True, min_length=1)] + + +class EmbedRequest(BaseModel): + """Texts to embed, marked as search queries or knowledge passages.""" + + kind: Literal["query", "passage"] + texts: Annotated[ + list[Text], Field(min_length=1, max_length=MAX_TEXTS_PER_REQUEST) + ] + + +class EmbedResponse(BaseModel): + """One normalized vector per input text, in input order.""" + + model: str + dimensions: int + vectors: list[list[float]] + + +@app.get("/health") +def health(): + """Report readiness. Uvicorn only answers once the model is loaded.""" + return {"status": "ok", "model": MODEL_NAME} + + +def _reject_truncated(embedder, texts): + """Refuse texts the model would silently cut off at its token limit.""" + token_counts = [ + len(ids) for ids in embedder.tokenizer(texts)["input_ids"] + ] + too_long = [ + index for index, count in enumerate(token_counts) + if count > embedder.max_seq_length + ] + if too_long: + raise HTTPException( + status_code=422, + detail={ + "error": "Text exceeds the model's token limit.", + "limit": embedder.max_seq_length, + "indexes": too_long, + }, + ) + + +@app.post("/embed") +def embed(request: EmbedRequest) -> EmbedResponse: + """Return one normalized embedding per text.""" + embedder = ml_models["embedder"] + # e5 was trained with these prefixes. Without them it still returns + # vectors, but retrieval quality drops without any error. + texts = [f"{request.kind}: {text}" for text in request.texts] + _reject_truncated(embedder, texts) + vectors = embedder.encode(texts, normalize_embeddings=True) + return EmbedResponse( + model=MODEL_NAME, + dimensions=vectors.shape[1], + vectors=vectors.tolist(), + ) diff --git a/embedding_service/requirements.txt b/embedding_service/requirements.txt new file mode 100644 index 0000000..331269c --- /dev/null +++ b/embedding_service/requirements.txt @@ -0,0 +1,53 @@ +# CPU-only PyTorch. The +cpu build only exists on this index; the default +# Linux wheel on PyPI pulls in several gigabytes of CUDA libraries. +--extra-index-url https://download.pytorch.org/whl/cpu + +annotated-doc==0.0.5 +annotated-types==0.8.0 +anyio==4.15.1 +certifi==2026.7.22 +click==8.5.0 +cloudpickle==3.1.2 +colorama==0.4.6 +fastapi==0.141.1 +filelock==3.32.3 +fsspec==2026.7.0 +h11==0.16.0 +hf-xet==1.6.0 +httpcore==1.0.9 +httpx==0.28.1 +huggingface_hub==1.33.0 +idna==3.20 +Jinja2==3.1.6 +joblib==1.6.0 +markdown-it-py==4.2.0 +MarkupSafe==3.0.3 +mdurl==0.1.2 +mpmath==1.3.0 +narwhals==2.26.0 +networkx==3.6.1 +numpy==2.5.3 +packaging==26.3 +pydantic==2.13.5 +pydantic_core==2.46.5 +Pygments==2.21.0 +PyYAML==6.0.3 +regex==2026.9.10 +rich==15.0.0 +safetensors==0.8.0 +scikit-learn==1.9.1 +scipy==1.18.1 +sentence-transformers==6.1.0 +setuptools==84.0.0 +shellingham==1.5.4 +starlette==1.7.0 +sympy==1.14.0 +threadpoolctl==3.7.0 +tokenizers==0.23.2 +torch==2.14.0+cpu +tqdm==4.70.1 +transformers==5.17.0 +typer==0.27.2 +typing-inspection==0.4.4 +typing_extensions==4.16.0 +uvicorn==0.53.0 diff --git a/embedding_service/smoke_test.py b/embedding_service/smoke_test.py new file mode 100644 index 0000000..29fc13f --- /dev/null +++ b/embedding_service/smoke_test.py @@ -0,0 +1,75 @@ +"""Smoke test for a running embedding service. + +Checks the vector size and that related passages, in German and in English, +score higher than an unrelated one. Uses only the standard library, so it +runs with any Python 3, including on the server after a deploy. +""" + +import json +import sys +from urllib.request import Request, urlopen + +BASE_URL = "http://127.0.0.1:8002" +EXPECTED_DIMENSIONS = 768 + +QUERY = "Wie bringt Benjamin seine Projekte auf den Server?" +PASSAGES = { + "related, German": ( + "Das Deployment läuft automatisch über GitHub Actions " + "auf einen eigenen VPS." + ), + "related, English": ( + "Deployment runs automatically through GitHub Actions " + "to a private VPS." + ), + "unrelated": ( + "Der Pokedex zeigt Pokémon mit ihren Typen und Werten " + "in einer Kartenansicht." + ), +} + + +def embed(kind, texts): + """Send texts to the service and return their vectors.""" + body = json.dumps({"kind": kind, "texts": texts}).encode("utf-8") + request = Request( + f"{BASE_URL}/embed", + data=body, + headers={"Content-Type": "application/json"}, + ) + with urlopen(request, timeout=30) as response: + return json.load(response)["vectors"] + + +def cosine(a, b): + """Cosine similarity of two normalized vectors, their dot product.""" + return sum(x * y for x, y in zip(a, b)) + + +def main(): + """Print the similarity scores and return 1 if a check fails.""" + [query] = embed("query", [QUERY]) + vectors = embed("passage", list(PASSAGES.values())) + scores = { + label: cosine(query, vector) + for label, vector in zip(PASSAGES, vectors) + } + + for label, score in scores.items(): + print(f"{score:.4f} {label}") + + failures = [] + if len(query) != EXPECTED_DIMENSIONS: + failures.append(f"expected {EXPECTED_DIMENSIONS} dimensions") + for label in ("related, German", "related, English"): + if scores[label] <= scores["unrelated"]: + failures.append(f"'{label}' does not beat 'unrelated'") + + for failure in failures: + print(f"FAIL: {failure}") + print("FAILED" if failures else "OK") + return 1 if failures else 0 + + +if __name__ == "__main__": + sys.exit(main())