| """klaus-3 vision service: CLIP + all-MiniLM embeddings + LPIPS over HTTP. |
| |
| Runs on klaus-3 (has the GPU + torch). The Mac orchestrator calls these |
| endpoints so it never needs torch locally. |
| |
| Endpoints: |
| GET /health |
| POST /clip/similarity form: image=<file>, texts="a\nb\nc" -> {similarity:{text:cos}} |
| POST /clip/zero_shot form: image=<file>, labels="a\nb\nc" -> {scores:{label:prob}} |
| POST /clip/caption form: image=<file> -> {label: str} |
| POST /embed json: {texts:[...]} -> {embeddings:[[...]]} |
| POST /lpips form: a=<file>, b=<file> -> {lpips: float} |
| |
| Run: |
| pip install fastapi uvicorn python-multipart open_clip_torch sentence-transformers lpips torch pillow |
| uvicorn vision_service:app --host 0.0.0.0 --port 8090 |
| """ |
|
|
| from __future__ import annotations |
|
|
| import io |
|
|
| import numpy as np |
| import torch |
| from fastapi import FastAPI, File, Form, UploadFile |
| from PIL import Image |
|
|
| app = FastAPI(title="veil-pgd vision service") |
|
|
| _DEVICE = "cuda" if torch.cuda.is_available() else "cpu" |
|
|
| |
| _clip = {"model": None, "preprocess": None, "tokenizer": None} |
| _embed_model = None |
| _lpips_model = None |
|
|
| |
| _CAPTION_VOCAB = [ |
| "dog", "cat", "car", "person", "tree", "building", "food", "flower", |
| "bird", "boat", "bicycle", "mountain", "beach", "book", "phone", |
| ] |
|
|
|
|
| def _load_clip(): |
| if _clip["model"] is None: |
| import open_clip |
|
|
| model, _, preprocess = open_clip.create_model_and_transforms( |
| "ViT-B-32", pretrained="laion2b_s34b_b79k" |
| ) |
| model = model.to(_DEVICE).eval() |
| _clip["model"] = model |
| _clip["preprocess"] = preprocess |
| _clip["tokenizer"] = open_clip.get_tokenizer("ViT-B-32") |
| return _clip |
|
|
|
|
| def _load_embed(): |
| global _embed_model |
| if _embed_model is None: |
| from sentence_transformers import SentenceTransformer |
|
|
| _embed_model = SentenceTransformer("all-MiniLM-L6-v2", device=_DEVICE) |
| return _embed_model |
|
|
|
|
| def _load_lpips(): |
| global _lpips_model |
| if _lpips_model is None: |
| import lpips |
|
|
| _lpips_model = lpips.LPIPS(net="alex").to(_DEVICE).eval() |
| return _lpips_model |
|
|
|
|
| def _read_image(data: bytes) -> Image.Image: |
| return Image.open(io.BytesIO(data)).convert("RGB") |
|
|
|
|
| @torch.no_grad() |
| def _clip_features(image: Image.Image, texts: list[str]): |
| c = _load_clip() |
| img_t = c["preprocess"](image).unsqueeze(0).to(_DEVICE) |
| tok = c["tokenizer"](texts).to(_DEVICE) |
| img_f = c["model"].encode_image(img_t) |
| txt_f = c["model"].encode_text(tok) |
| img_f = img_f / img_f.norm(dim=-1, keepdim=True) |
| txt_f = txt_f / txt_f.norm(dim=-1, keepdim=True) |
| return img_f, txt_f |
|
|
|
|
| @app.get("/health") |
| def health(): |
| return {"ok": True, "device": _DEVICE} |
|
|
|
|
| @app.post("/clip/similarity") |
| async def clip_similarity(image: UploadFile = File(...), texts: str = Form(...)): |
| img = _read_image(await image.read()) |
| text_list = [t for t in texts.split("\n") if t.strip()] |
| img_f, txt_f = _clip_features(img, text_list) |
| sims = (img_f @ txt_f.T).squeeze(0).tolist() |
| return {"similarity": dict(zip(text_list, sims))} |
|
|
|
|
| @app.post("/clip/zero_shot") |
| async def clip_zero_shot(image: UploadFile = File(...), labels: str = Form(...)): |
| img = _read_image(await image.read()) |
| label_list = [t for t in labels.split("\n") if t.strip()] |
| img_f, txt_f = _clip_features(img, label_list) |
| logits = (100.0 * img_f @ txt_f.T).softmax(dim=-1).squeeze(0).tolist() |
| return {"scores": dict(zip(label_list, logits))} |
|
|
|
|
| @app.post("/clip/caption") |
| async def clip_caption(image: UploadFile = File(...)): |
| img = _read_image(await image.read()) |
| img_f, txt_f = _clip_features(img, _CAPTION_VOCAB) |
| idx = int((img_f @ txt_f.T).squeeze(0).argmax().item()) |
| return {"label": _CAPTION_VOCAB[idx]} |
|
|
|
|
| @app.post("/embed") |
| async def embed(payload: dict): |
| texts = payload.get("texts", []) |
| model = _load_embed() |
| vecs = model.encode(texts, normalize_embeddings=True) |
| return {"embeddings": np.asarray(vecs).tolist()} |
|
|
|
|
| @torch.no_grad() |
| @app.post("/lpips") |
| async def lpips_endpoint(a: UploadFile = File(...), b: UploadFile = File(...)): |
| ia = _read_image(await a.read()) |
| ib = _read_image(await b.read()).resize(ia.size) |
|
|
| def to_t(im: Image.Image) -> torch.Tensor: |
| arr = np.asarray(im, dtype=np.float32) / 127.5 - 1.0 |
| return torch.from_numpy(arr).permute(2, 0, 1).unsqueeze(0).to(_DEVICE) |
|
|
| model = _load_lpips() |
| d = model(to_t(ia), to_t(ib)).item() |
| return {"lpips": float(d)} |
|
|
|
|
| |
| |
| |
| |
|
|
| _encoders: dict = {} |
|
|
|
|
| class _OpenClipEncoder: |
| """OpenCLIP ViT-* wrapper with a differentiable image path.""" |
|
|
| def __init__(self, arch: str, pretrained: str): |
| import open_clip |
|
|
| self.model, _, self.preprocess = open_clip.create_model_and_transforms( |
| arch, pretrained=pretrained |
| ) |
| self.model = self.model.to(_DEVICE).eval() |
| self.tokenizer = open_clip.get_tokenizer(arch) |
| self.input_px = 224 |
|
|
| def image_tensor(self, image: Image.Image) -> torch.Tensor: |
| return self.preprocess(image).unsqueeze(0).to(_DEVICE) |
|
|
| def image_feat(self, t: torch.Tensor) -> torch.Tensor: |
| f = self.model.encode_image(t) |
| return f / f.norm(dim=-1, keepdim=True) |
|
|
| def text_feat(self, texts: list[str]) -> torch.Tensor: |
| tok = self.tokenizer(texts).to(_DEVICE) |
| f = self.model.encode_text(tok) |
| return f / f.norm(dim=-1, keepdim=True) |
|
|
|
|
| class _SiglipEncoder: |
| """HF SigLIP2 wrapper with a differentiable image path (fp32 for grads).""" |
|
|
| def __init__(self, model_id: str, dtype=torch.float16): |
| from transformers import AutoModel, AutoProcessor |
|
|
| self.dtype = dtype |
| self.model = AutoModel.from_pretrained(model_id, torch_dtype=dtype) |
| self.model = self.model.to(_DEVICE).eval() |
| self.processor = AutoProcessor.from_pretrained(model_id) |
| self.input_px = getattr( |
| self.processor.image_processor, "size", {} |
| ).get("height", 224) |
|
|
| def image_tensor(self, image: Image.Image) -> torch.Tensor: |
| px = self.processor(images=image, return_tensors="pt")["pixel_values"] |
| return px.to(_DEVICE, dtype=self.dtype) |
|
|
| @staticmethod |
| def _as_tensor(x): |
| if isinstance(x, torch.Tensor): |
| return x |
| for attr in ("image_embeds", "text_embeds", "pooler_output", "last_hidden_state"): |
| v = getattr(x, attr, None) |
| if isinstance(v, torch.Tensor): |
| return v.mean(dim=1) if v.dim() == 3 else v |
| raise TypeError(f"cannot extract tensor from {type(x)}") |
|
|
| def image_feat(self, t: torch.Tensor) -> torch.Tensor: |
| f = self._as_tensor(self.model.get_image_features(pixel_values=t)) |
| return f / f.norm(dim=-1, keepdim=True) |
|
|
| def text_feat(self, texts: list[str]) -> torch.Tensor: |
| tok = self.processor( |
| text=texts, padding="max_length", max_length=64, return_tensors="pt" |
| ).to(_DEVICE) |
| f = self._as_tensor(self.model.get_text_features(input_ids=tok["input_ids"])) |
| return f / f.norm(dim=-1, keepdim=True) |
|
|
|
|
| _ENCODER_SPECS = { |
| "openclip:ViT-B-32": ("openclip", "ViT-B-32", "laion2b_s34b_b79k"), |
| "openclip:ViT-L-14": ("openclip", "ViT-L-14", "laion2b_s32b_b82k"), |
| "siglip2-base": ("siglip", "google/siglip2-base-patch16-224", None), |
| "siglip-base": ("siglip", "google/siglip-base-patch16-224", None), |
| } |
|
|
|
|
| def _load_encoder(model_id: str): |
| if model_id not in _encoders: |
| if model_id not in _ENCODER_SPECS: |
| raise ValueError(f"unknown model_id {model_id}") |
| kind, a, b = _ENCODER_SPECS[model_id] |
| if kind == "openclip": |
| _encoders[model_id] = _OpenClipEncoder(a, b) |
| else: |
| _encoders[model_id] = _SiglipEncoder(a) |
| return _encoders[model_id] |
|
|
|
|
| @app.get("/vit/health") |
| def vit_health(): |
| mem = {} |
| if _DEVICE == "cuda": |
| mem = { |
| "allocated_mb": round(torch.cuda.memory_allocated() / 1e6, 1), |
| "reserved_mb": round(torch.cuda.memory_reserved() / 1e6, 1), |
| } |
| return {"ok": True, "device": _DEVICE, |
| "loaded": list(_encoders.keys()), |
| "available": list(_ENCODER_SPECS.keys()), "cuda_memory": mem} |
|
|
|
|
| @app.post("/vit/load") |
| async def vit_load(payload: dict): |
| mid = payload["model_id"] |
| enc = _load_encoder(mid) |
| return {"loaded": mid, "input_px": enc.input_px, "device": _DEVICE} |
|
|
|
|
| @torch.no_grad() |
| @app.post("/vit/score") |
| async def vit_score( |
| image: UploadFile = File(...), |
| truth: str = Form(...), |
| decoy: str = Form(...), |
| model_id: str = Form("openclip:ViT-B-32"), |
| clean: UploadFile | None = File(None), |
| ): |
| """Contrastive decoy-direction score for one candidate image. |
| |
| margin = cos(img, decoy) - cos(img, truth). If a clean image is supplied, |
| also returns the clean margin and delta (how much the overlay moved it) plus |
| away_from_clean = 1 - cos(img_candidate, img_clean). |
| """ |
| enc = _load_encoder(model_id) |
| img = _read_image(await image.read()) |
| it = enc.image_tensor(img) |
| ifeat = enc.image_feat(it) |
| tfeat = enc.text_feat([truth, decoy]) |
| sim_truth = float((ifeat @ tfeat[0:1].T).item()) |
| sim_decoy = float((ifeat @ tfeat[1:2].T).item()) |
| out = {"model_id": model_id, "sim_truth": sim_truth, "sim_decoy": sim_decoy, |
| "margin": sim_decoy - sim_truth} |
| if clean is not None: |
| cimg = _read_image(await clean.read()) |
| cfeat = enc.image_feat(enc.image_tensor(cimg)) |
| c_truth = float((cfeat @ tfeat[0:1].T).item()) |
| c_decoy = float((cfeat @ tfeat[1:2].T).item()) |
| out["clean_margin"] = c_decoy - c_truth |
| out["delta_margin"] = out["margin"] - out["clean_margin"] |
| out["away_from_clean"] = 1.0 - float((ifeat @ cfeat.T).item()) |
| return out |
|
|
|
|
| @torch.no_grad() |
| @app.post("/vit/zero_shot") |
| async def vit_zero_shot( |
| image: UploadFile = File(...), |
| labels: str = Form(...), |
| model_id: str = Form("openclip:ViT-B-32"), |
| ): |
| enc = _load_encoder(model_id) |
| img = _read_image(await image.read()) |
| label_list = [t for t in labels.split("\n") if t.strip()] |
| ifeat = enc.image_feat(enc.image_tensor(img)) |
| tfeat = enc.text_feat(label_list) |
| sims = (ifeat @ tfeat.T).squeeze(0).tolist() |
| return {"similarity": dict(zip(label_list, sims))} |
|
|
|
|
| @app.post("/vit/grad_region") |
| async def vit_grad_region( |
| image: UploadFile = File(...), |
| truth: str = Form(...), |
| decoy: str = Form(...), |
| model_id: str = Form("openclip:ViT-B-32"), |
| region_json: str = Form("null"), |
| step_eps: float = Form(0.01), |
| ): |
| """Single backward pass of margin loss wrt the input tensor. |
| |
| Proves the differentiable path works: returns grad L2 (whole + in region) and |
| whether one gradient-ascent step actually increases the decoy margin. |
| """ |
| import json as _json |
|
|
| enc = _load_encoder(model_id) |
| img = _read_image(await image.read()) |
| t = enc.image_tensor(img).clone().detach().requires_grad_(True) |
| tfeat = enc.text_feat([truth, decoy]).detach() |
| ifeat = enc.image_feat(t) |
| margin = (ifeat @ tfeat[1:2].T) - (ifeat @ tfeat[0:1].T) |
| margin0 = float(margin.item()) |
| enc.model.zero_grad(set_to_none=True) |
| margin.backward() |
| g = t.grad.detach() |
| grad_l2 = float(g.norm().item()) |
|
|
| region = _json.loads(region_json) |
| grad_l2_region = None |
| if region is not None: |
| |
| W, H = img.size |
| px = t.shape[-1] |
| sx, sy = px / W, px / H |
| x0, y0, x1, y1 = region |
| rx0, ry0 = max(0, int(x0 * sx)), max(0, int(y0 * sy)) |
| rx1, ry1 = min(px, int(x1 * sx)), min(px, int(y1 * sy)) |
| if rx1 > rx0 and ry1 > ry0: |
| grad_l2_region = float(g[..., ry0:ry1, rx0:rx1].norm().item()) |
|
|
| |
| with torch.no_grad(): |
| t2 = (t + step_eps * g.sign()).detach() |
| ifeat2 = enc.image_feat(t2) |
| margin1 = float(((ifeat2 @ tfeat[1:2].T) - (ifeat2 @ tfeat[0:1].T)).item()) |
|
|
| return {"model_id": model_id, "margin_before": margin0, "margin_after_step": margin1, |
| "margin_increased": margin1 > margin0, "grad_l2": grad_l2, |
| "grad_l2_region": grad_l2_region} |
|
|
|
|
| |
| _CLIP_MEAN = (0.48145466, 0.4578275, 0.40821073) |
| _CLIP_STD = (0.26862954, 0.26130258, 0.27577711) |
|
|
|
|
| def _clip_feat_from_pixels(enc, x: torch.Tensor) -> torch.Tensor: |
| """Differentiable CLIP image feature from a full-res [0,1] (1,3,H,W) tensor. |
| |
| Replicates resize+normalize (skips center crop) so gradients flow back to the |
| original-resolution pixels and the perturbation is meaningful at full res. |
| """ |
| import torch.nn.functional as F |
|
|
| x = F.interpolate(x, size=(224, 224), mode="bicubic", align_corners=False) |
| mean = torch.tensor(_CLIP_MEAN, device=x.device).view(1, 3, 1, 1) |
| std = torch.tensor(_CLIP_STD, device=x.device).view(1, 3, 1, 1) |
| x = (x.clamp(0, 1) - mean) / std |
| f = enc.model.encode_image(x) |
| return f / f.norm(dim=-1, keepdim=True) |
|
|
|
|
| @app.post("/vit/pgd_region") |
| async def vit_pgd_region( |
| image: UploadFile = File(...), |
| truth: str = Form(...), |
| decoy: str = Form(...), |
| model_id: str = Form("openclip:ViT-B-32"), |
| region_json: str = Form("null"), |
| mask: UploadFile | None = File(None), |
| eps: float = Form(0.0627), |
| steps: int = Form(60), |
| step_size: float = Form(0.0078), |
| return_image: bool = Form(True), |
| ): |
| """PGD that maximizes the decoy-truth margin, confined to a text box or an |
| arbitrary editable mask. This is the *ceiling probe*: how far can ANY in-region |
| perturbation move the encoder? mask (grayscale, >0 = editable) restricts edits |
| to the text glyphs (stealthy ceiling); region_json alone = full box (Nightshade |
| ceiling). Returns the optimized image so the caller can test caption flip / JPEG. |
| """ |
| import base64 |
| import io as _io |
| import json as _json |
|
|
| if not model_id.startswith("openclip"): |
| return {"error": "pgd_region currently supports openclip:* only"} |
| enc = _load_encoder(model_id) |
| img = _read_image(await image.read()) |
| W, H = img.size |
| x0 = (torch.from_numpy(np.asarray(img, dtype=np.float32) / 255.0) |
| .permute(2, 0, 1).unsqueeze(0).to(_DEVICE)) |
|
|
| |
| m = torch.zeros(1, 1, H, W, device=_DEVICE) |
| region = _json.loads(region_json) |
| if mask is not None: |
| mimg = Image.open(_io.BytesIO(await mask.read())).convert("L").resize((W, H)) |
| marr = torch.from_numpy(np.asarray(mimg, dtype=np.float32) / 255.0) |
| m[0, 0] = (marr > 0.05).float().to(_DEVICE) |
| elif region is not None: |
| rx0, ry0, rx1, ry1 = [int(v) for v in region] |
| m[0, 0, max(0, ry0):min(H, ry1), max(0, rx0):min(W, rx1)] = 1.0 |
| else: |
| m[:] = 1.0 |
| editable_px = int(m.sum().item()) |
|
|
| tfeat = enc.text_feat([truth, decoy]).detach() |
|
|
| def margin_of(x): |
| f = _clip_feat_from_pixels(enc, x) |
| return (f @ tfeat[1:2].T) - (f @ tfeat[0:1].T) |
|
|
| with torch.no_grad(): |
| margin0 = float(margin_of(x0).item()) |
|
|
| delta = torch.zeros_like(x0, requires_grad=True) |
| for _ in range(int(steps)): |
| x_adv = (x0 + delta * m).clamp(0, 1) |
| loss = margin_of(x_adv) |
| enc.model.zero_grad(set_to_none=True) |
| if delta.grad is not None: |
| delta.grad = None |
| loss.backward() |
| with torch.no_grad(): |
| delta += step_size * delta.grad.sign() |
| delta.clamp_(-eps, eps) |
| with torch.no_grad(): |
| x_final = (x0 + delta * m).clamp(0, 1) |
| margin1 = float(margin_of(x_final).item()) |
| linf = float((delta * m).abs().max().item()) |
|
|
| out = {"model_id": model_id, "margin_before": margin0, "margin_after": margin1, |
| "delta_margin": margin1 - margin0, "eps": eps, "steps": int(steps), |
| "editable_px": editable_px, "linf": linf, |
| "mode": "mask" if mask is not None else ("box" if region else "full")} |
| if return_image: |
| arr = (x_final.squeeze(0).permute(1, 2, 0).cpu().numpy() * 255.0) |
| arr = arr.round().astype("uint8") |
| buf = _io.BytesIO() |
| Image.fromarray(arr).save(buf, format="PNG") |
| out["image_png_b64"] = base64.b64encode(buf.getvalue()).decode() |
| return out |
|
|