|
| 1 | +"""Shared building blocks for the Modal embedding endpoints. |
| 2 | +
|
| 3 | +``modal_embeddings_en.py`` and ``modal_embeddings_multilang.py`` each define a tiny |
| 4 | +``EmbeddingModel`` class at module scope (Modal requires globally-defined classes |
| 5 | +with stacked ``@app.cls`` / ``@modal.concurrent`` decorators) that delegates to the |
| 6 | +helpers here. All the heavy lifting — the container image, model loading, pooling, |
| 7 | +and the embedding request handler — lives in this module so it is written once. |
| 8 | +
|
| 9 | +The endpoint contract (consumed by ``document_qa.custom_embeddings.ModalEmbeddings``): |
| 10 | +
|
| 11 | +- **Method**: ``POST`` |
| 12 | +- **Auth**: ``x-api-key`` header, compared against the ``API_KEY`` secret. |
| 13 | +- **Body**: form field ``text`` containing newline-separated strings. |
| 14 | +- **Response**: JSON list of L2-normalised embedding vectors, one per input line. |
| 15 | +""" |
| 16 | + |
| 17 | +import os |
| 18 | + |
| 19 | +import modal |
| 20 | +import torch |
| 21 | +import torch.nn.functional as F |
| 22 | +from fastapi import HTTPException, Request |
| 23 | +from torch import Tensor |
| 24 | + |
| 25 | +MINUTES = 60 # seconds |
| 26 | +N_GPU = 1 |
| 27 | + |
| 28 | +# Shared container image for every embedding model. |
| 29 | +image = ( |
| 30 | + modal.Image.debian_slim(python_version="3.11") |
| 31 | + .pip_install( |
| 32 | + "transformers", |
| 33 | + "huggingface_hub[hf_transfer]==0.26.2", |
| 34 | + "flashinfer-python==0.2.0.post2", # pinning, very unstable |
| 35 | + "fastapi[standard]", |
| 36 | + extra_index_url="https://flashinfer.ai/whl/cu124/torch2.5", |
| 37 | + ) |
| 38 | + .env({"HF_HUB_ENABLE_HF_TRANSFER": "1"}) # faster model transfers |
| 39 | + # Modal 1.0 no longer auto-mounts imported local modules; the wrapper scripts |
| 40 | + # import this module by name, so it must be added explicitly. Kept last so it |
| 41 | + # doesn't invalidate the (expensive) pip layer above on every code edit. |
| 42 | + .add_local_python_source("_embeddings_app") |
| 43 | +) |
| 44 | + |
| 45 | +hf_cache_vol = modal.Volume.from_name("huggingface-cache", create_if_missing=True) |
| 46 | +vllm_cache_vol = modal.Volume.from_name("vllm-cache", create_if_missing=True) |
| 47 | + |
| 48 | + |
| 49 | +def cls_kwargs() -> dict: |
| 50 | + """Common ``@app.cls`` configuration shared by every embedding endpoint.""" |
| 51 | + return dict( |
| 52 | + image=image, |
| 53 | + gpu=f"L40S:{N_GPU}", |
| 54 | + # how long should we stay up with no requests? |
| 55 | + scaledown_window=3 * MINUTES, |
| 56 | + volumes={ |
| 57 | + "/root/.cache/huggingface": hf_cache_vol, |
| 58 | + "/root/.cache/vllm": vllm_cache_vol, |
| 59 | + }, |
| 60 | + secrets=[modal.Secret.from_name("document-qa-embedding-key")], |
| 61 | + ) |
| 62 | + |
| 63 | + |
| 64 | +def average_pool(last_hidden_states: Tensor, attention_mask: Tensor) -> Tensor: |
| 65 | + """Mean-pool token embeddings, ignoring padding positions.""" |
| 66 | + last_hidden = last_hidden_states.masked_fill(~attention_mask[..., None].bool(), 0.0) |
| 67 | + return last_hidden.sum(dim=1) / attention_mask.sum(dim=1)[..., None] |
| 68 | + |
| 69 | + |
| 70 | +def load_embedding_model(model_name: str, model_revision: str): |
| 71 | + """Load a tokenizer + model onto the best available device, once per container. |
| 72 | +
|
| 73 | + Returns: |
| 74 | + tuple: ``(tokenizer, model, device)`` with ``model`` already in eval mode. |
| 75 | + """ |
| 76 | + # transformers is only available inside the Modal image, so import lazily. |
| 77 | + from transformers import AutoModel, AutoTokenizer |
| 78 | + |
| 79 | + device = torch.device("cuda" if torch.cuda.is_available() else "cpu") |
| 80 | + print(f"Loading {model_name} on {device}...") |
| 81 | + tokenizer = AutoTokenizer.from_pretrained(model_name, revision=model_revision) |
| 82 | + model = AutoModel.from_pretrained(model_name, revision=model_revision).to(device) |
| 83 | + model.eval() |
| 84 | + print("Model loaded successfully.") |
| 85 | + return tokenizer, model, device |
| 86 | + |
| 87 | + |
| 88 | +def run_embed(tokenizer, model, device, request: Request, text: str): |
| 89 | + """Authenticate, embed newline-separated ``text``, and return normalised vectors.""" |
| 90 | + api_key = request.headers.get("x-api-key") |
| 91 | + if api_key != os.environ["API_KEY"]: |
| 92 | + raise HTTPException(status_code=401, detail="Unauthorized") |
| 93 | + |
| 94 | + texts = [t for t in text.split("\n") if t.strip()] |
| 95 | + if not texts: |
| 96 | + return [] |
| 97 | + |
| 98 | + print(f"Start embedding {len(texts)} texts") |
| 99 | + try: |
| 100 | + with torch.no_grad(): |
| 101 | + batch_dict = tokenizer(texts, padding=True, truncation=True, return_tensors="pt") |
| 102 | + batch_dict = {k: v.to(device) for k, v in batch_dict.items()} |
| 103 | + |
| 104 | + outputs = model(**batch_dict) |
| 105 | + embeddings = average_pool(outputs.last_hidden_state, batch_dict["attention_mask"]) |
| 106 | + embeddings = F.normalize(embeddings, p=2, dim=1) |
| 107 | + embeddings = embeddings.cpu().numpy().tolist() |
| 108 | + |
| 109 | + print("Finished embedding texts.") |
| 110 | + return embeddings |
| 111 | + |
| 112 | + except RuntimeError as e: |
| 113 | + print(f"Error during embedding: {str(e)}") |
| 114 | + if "CUDA out of memory" in str(e): |
| 115 | + print("CUDA OOM. Try reducing batch size or using a smaller model.") |
| 116 | + raise |
0 commit comments