ArtShumov commited on
Commit
c2da662
·
1 Parent(s): e0379b1

feat(tagger): NL captioning via Qwen3-VL, pose detection UI fixes, memory-safe for HF Spaces (16GB CPU)

Browse files
_debug_person.png ADDED
data/session_history.json.bak CHANGED
@@ -1,4 +1,82 @@
1
  [
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2
  {
3
  "id": "c9ed5cc5",
4
  "timestamp": 1785778992.0812612,
@@ -1220,83 +1298,5 @@
1220
  "categories": [],
1221
  "seed": null,
1222
  "liked_indices": []
1223
- },
1224
- {
1225
- "id": "ba70a0a2",
1226
- "timestamp": 1784813293.9073813,
1227
- "prompt": "1girl, blue hair",
1228
- "model": "anima",
1229
- "rating": "pg",
1230
- "num_variations": 2,
1231
- "creativity": "medium",
1232
- "weight_mode": "off",
1233
- "results": [
1234
- "masterpiece, best quality, score_9, safe, 1girl, BREAK, high contrast, hair ribbon, blue hair, detailed background, sacred atmosphere, dragon lair, global illumination, moonlight beam, light rays underwater, holographic display, silver aesthetic, leaning against wall, radial composition, hand on own chest, looking at viewer, extreme close-up, surprised pose, intricate details, beautiful detailed",
1235
- "masterpiece, best quality, score_9, safe, 1girl, BREAK, vintage color grading, hair between eyes, saturated matte, high contrast, blue hair, detailed background, transmission tower, divine radiance, shield, car, global illumination, gothic lighting, silver aesthetic, ice crystals, cosmic dust, floating, foreshortening, surprised pose, looking away, full body, intricate details, beautiful detailed",
1236
- "",
1237
- "",
1238
- "",
1239
- "",
1240
- "",
1241
- "",
1242
- "",
1243
- ""
1244
- ],
1245
- "neg_results": [],
1246
- "categories": [],
1247
- "seed": null,
1248
- "liked_indices": []
1249
- },
1250
- {
1251
- "id": "103989cc",
1252
- "timestamp": 1784813166.9979286,
1253
- "prompt": "1girl, blue hair",
1254
- "model": "anima",
1255
- "rating": "pg",
1256
- "num_variations": 2,
1257
- "creativity": "medium",
1258
- "weight_mode": "off",
1259
- "results": [
1260
- "masterpiece, best quality, score_9, safe, 1girl, BREAK, sticking out tongue, saturated matte, blue hair, skirt, detailed background, retro futuristic, enchanted boat, motorcycle, phoenix, sword, global illumination, moonlight beam, light rays underwater, holographic display, silver aesthetic, foreground framing, hands behind back, looking at viewer, extreme close-up, surprised pose, intricate details, beautiful detailed",
1261
- "masterpiece, best quality, score_9, safe, 1girl, BREAK, iridescent colors, blank expression, saturated matte, high contrast, blue hair, glasses, mask, detailed background, christmas present, abyssal darkness, moonlight beam, silver aesthetic, cosmic dust, first person view, looking at viewer, surprised pose, wide shot, intricate details, beautiful detailed",
1262
- "",
1263
- "",
1264
- "",
1265
- "",
1266
- "",
1267
- "",
1268
- "",
1269
- ""
1270
- ],
1271
- "neg_results": [],
1272
- "categories": [],
1273
- "seed": null,
1274
- "liked_indices": []
1275
- },
1276
- {
1277
- "id": "671a7818",
1278
- "timestamp": 1784813130.1904197,
1279
- "prompt": "1girl, blue hair",
1280
- "model": "anima",
1281
- "rating": "pg",
1282
- "num_variations": 2,
1283
- "creativity": "medium",
1284
- "weight_mode": "off",
1285
- "results": [
1286
- "masterpiece, best quality, score_9, safe, 1girl, BREAK, hair between eyes, saturated matte, broad shoulders, triadic colors, autumn colors, blue hair, umbrella, complex background, mythical world, boat, gothic lighting, moonlight beam, silver aesthetic, surprised pose, looking away, from above, intricate details, beautiful detailed",
1287
- "masterpiece, best quality, score_9, safe, 1girl, BREAK, iridescent colors, high contrast, blue hair, backless dress, star, detailed background, samurai residence, global illumination, holographic display, silver aesthetic, lightning aura, looking at viewer, surprised pose, faux traditional media, incredibly absurdres, intricate details, beautiful detailed",
1288
- "",
1289
- "",
1290
- "",
1291
- "",
1292
- "",
1293
- "",
1294
- "",
1295
- ""
1296
- ],
1297
- "neg_results": [],
1298
- "categories": [],
1299
- "seed": null,
1300
- "liked_indices": []
1301
  }
1302
  ]
 
1
  [
2
+ {
3
+ "id": "27bb436d",
4
+ "timestamp": 1785784568.4936664,
5
+ "prompt": "1girl, blue hair",
6
+ "model": "anima",
7
+ "rating": "pg",
8
+ "num_variations": 2,
9
+ "creativity": "medium",
10
+ "weight_mode": "off",
11
+ "results": [
12
+ "masterpiece, best quality, score_7, safe, 1girl, BREAK, blue hair, hair tie, three-quarter view, from below, profile, sepia, detailed background, blue hour, warm, underwater caustics, ethereal lighting, dreamy lighting, gothic lighting, holographic display, enchanted boat, steam train, beautiful detailed, beautiful detail",
13
+ "masterpiece, best quality, score_7, safe, 1girl, BREAK, blue hair, hair tie, upper body, high contrast, vibrant, surprised pose, crying tears, triangular composition, detailed background, blue hour, bioluminescent sea, horror backlight, morning sky, magical particles, silver aesthetic, blue tint, incredibly absurdres, enchanted boat, sports car, beautiful detailed, beautiful detail",
14
+ "",
15
+ "",
16
+ "",
17
+ "",
18
+ "",
19
+ "",
20
+ "",
21
+ ""
22
+ ],
23
+ "neg_results": [],
24
+ "categories": [],
25
+ "seed": null,
26
+ "liked_indices": []
27
+ },
28
+ {
29
+ "id": "f2656317",
30
+ "timestamp": 1785784470.7438738,
31
+ "prompt": "1girl, blue hair",
32
+ "model": "anima",
33
+ "rating": "pg",
34
+ "num_variations": 2,
35
+ "creativity": "medium",
36
+ "weight_mode": "off",
37
+ "results": [
38
+ "masterpiece, best quality, score_7, safe, 1girl, BREAK, blue hair, hair tie, tears of joy, three-quarter view, cowboy shot, high contrast, vibrant, sinister smile, pov, detailed background, mist, underwater caustics, horror backlight, lightning aura, silver aesthetic, blue tint, smoke, incredibly absurdres, enchanted boat, beautiful detailed, beautiful detail",
39
+ "masterpiece, best quality, score_7, safe, 1girl, BREAK, absurdly long hair, blue hair, sly smile, serious, happy, three-quarter view, from behind, vaporwave aesthetic, high contrast, red and black, vibrant, pov, detailed background, foggy, underwater caustics, ethereal lighting, silver aesthetic, dragon, kimono, beautiful detailed, beautiful detail",
40
+ "",
41
+ "",
42
+ "",
43
+ "",
44
+ "",
45
+ "",
46
+ "",
47
+ ""
48
+ ],
49
+ "neg_results": [],
50
+ "categories": [],
51
+ "seed": null,
52
+ "liked_indices": []
53
+ },
54
+ {
55
+ "id": "0593f16f",
56
+ "timestamp": 1785783108.9567664,
57
+ "prompt": "1girl, blue hair",
58
+ "model": "anima",
59
+ "rating": "pg",
60
+ "num_variations": 2,
61
+ "creativity": "medium",
62
+ "weight_mode": "off",
63
+ "results": [
64
+ "masterpiece, best quality, score_7, safe, 1girl, BREAK, blue hair, magical girl pose, split complementary, high contrast, vibrant, pleated skirt, close-up, pov, detailed background, fog, gothic lighting, dark lighting, silver aesthetic, blue tint, detailed illustration, evil smile, beautiful detailed, beautiful detail",
65
+ "masterpiece, best quality, score_7, safe, 1girl, BREAK, blue hair, cross processed colors, high contrast, vibrant, suit, detailed background, sunset, creative atmosphere, warm, partly cloudy light, rainy lighting, smoke, chibi, portrait, temple, beautiful detailed, beautiful detail",
66
+ "",
67
+ "",
68
+ "",
69
+ "",
70
+ "",
71
+ "",
72
+ "",
73
+ ""
74
+ ],
75
+ "neg_results": [],
76
+ "categories": [],
77
+ "seed": null,
78
+ "liked_indices": []
79
+ },
80
  {
81
  "id": "c9ed5cc5",
82
  "timestamp": 1785778992.0812612,
 
1298
  "categories": [],
1299
  "seed": null,
1300
  "liked_indices": []
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1301
  }
1302
  ]
src/i18n.py CHANGED
@@ -700,6 +700,7 @@ L10N = {
700
  "tagger_mode_swinv2": "WD14 SwinV2",
701
  "tagger_mode_vit": "WD14 ViT",
702
  "tagger_mode_qwen": "Qwen3-VL (NL описание)",
 
703
  "tagger_pose_label": "Поза",
704
  "tagger_pose_single": "Обнаружен {n} человек",
705
  "tagger_pose_many": "Обнаружено {n} человек",
 
700
  "tagger_mode_swinv2": "WD14 SwinV2",
701
  "tagger_mode_vit": "WD14 ViT",
702
  "tagger_mode_qwen": "Qwen3-VL (NL описание)",
703
+ "tagger_mode_qwen_desc": "Qwen3-VL (Подробное описание изображения)",
704
  "tagger_pose_label": "Поза",
705
  "tagger_pose_single": "Обнаружен {n} человек",
706
  "tagger_pose_many": "Обнаружено {n} человек",
src/image_tagger.py CHANGED
@@ -94,6 +94,16 @@ def _to_pil_image(image) -> Image.Image:
94
  if np is None:
95
  raise RuntimeError("numpy is required for image tagging")
96
  arr = np.asarray(image)
 
 
 
 
 
 
 
 
 
 
97
  if arr.ndim == 2:
98
  arr = np.stack([arr] * 3, axis=-1)
99
  elif arr.ndim == 3 and arr.shape[2] == 1:
 
94
  if np is None:
95
  raise RuntimeError("numpy is required for image tagging")
96
  arr = np.asarray(image)
97
+ # Accept bytes / file-like objects (e.g. from API calls or older Gradio versions)
98
+ if arr.ndim == 0 or arr.dtype == object:
99
+ # Accept bytes / bytearray / memoryview / any file-like with .read()
100
+ buf = image if isinstance(image, (bytes, bytearray, memoryview)) else None
101
+ if buf is None and hasattr(image, "read"):
102
+ buf = image.read()
103
+ if buf is None:
104
+ raise ValueError("Unsupported image payload type")
105
+ from io import BytesIO
106
+ arr = np.array(Image.open(BytesIO(bytes(buf))).convert("RGB"))
107
  if arr.ndim == 2:
108
  arr = np.stack([arr] * 3, axis=-1)
109
  elif arr.ndim == 3 and arr.shape[2] == 1:
src/qwen_vl_tagger.py CHANGED
@@ -27,10 +27,13 @@ except Exception: # pragma: no cover
27
  _VL_DEPS_OK = False
28
  AutoProcessor = _AutoModel = None
29
 
 
 
30
  _MODEL_ID = os.environ.get("WHYX_QWEN_VL_MODEL", "Qwen/Qwen3-VL-4B-Instruct")
31
  _vl_instance: "QwenVLTagger | None" = None
32
 
33
 
 
34
  def _vl_enabled() -> bool:
35
  return os.environ.get("WHYX_ENABLE_QWEN_VL", "1").strip().lower() not in ("0", "false", "no", "off")
36
 
@@ -53,11 +56,17 @@ def _normalize_caption(raw: str) -> str:
53
 
54
 
55
  class QwenVLTagger:
 
 
 
 
 
 
 
56
  def __init__(self, model_id: str = _MODEL_ID):
57
  self._model_id = model_id
58
  self._processor: Optional[AutoProcessor] = None
59
  self._model: Optional["_AutoModel"] = None
60
- self._device: Optional[torch.device] = None
61
  self._loaded = False
62
 
63
  def ensure_loaded(self) -> bool:
@@ -67,16 +76,15 @@ class QwenVLTagger:
67
  raise RuntimeError("transformers is not installed")
68
  if not _vl_enabled():
69
  raise RuntimeError("Qwen VL is disabled (WHYX_ENABLE_QWEN_VL=0)")
70
- self._device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
71
  self._processor = AutoProcessor.from_pretrained(self._model_id)
72
- dtype = torch.float16 if self._device.type != "cpu" else torch.float32
73
  self._model = _AutoModel.from_pretrained(
74
  self._model_id,
75
  dtype=dtype,
76
  )
77
- # Move to device *after* load so accelerate is not required.
78
- if self._device is not None:
79
- self._model = self._model.to(self._device)
80
  self._loaded = True
81
  return True
82
 
@@ -90,25 +98,30 @@ class QwenVLTagger:
90
  if not self.ensure_loaded():
91
  return ""
92
  img = self._to_pil(image)
 
 
 
93
  system_prompt = prompt or (
94
  "Describe this image in ONE concise sentence suitable as a Stable "
95
- "Diffusion prompt (no preamble, no extra sentences, no lists)."
96
  )
97
  messages = [
98
- {"role": "system", "content": [{"type": "text", "text": system_prompt}]},
99
- {"role": "user", "content": [
100
- {"type": "image"},
101
- {"type": "text", "text": "Image:"},
102
- ]},
 
 
 
103
  ]
104
  text = self._processor.apply_chat_template(messages, add_generation_prompt=True, tokenize=False)
105
  inputs = self._processor(text=[text], images=[img], return_tensors="pt")
106
- inputs = {k: v.to(self._device) for k, v in inputs.items()}
107
  with torch.inference_mode():
108
  gen = self._model.generate(**inputs, max_new_tokens=max_new_tokens)
109
  trimmed = gen[:, inputs["input_ids"].shape[1]:]
110
  caption = self._processor.batch_decode(trimmed, skip_special_tokens=True)[0]
111
- return _normalize_caption(caption)
112
 
113
 
114
  def get_qwen_tagger() -> QwenVLTagger:
 
27
  _VL_DEPS_OK = False
28
  AutoProcessor = _AutoModel = None
29
 
30
+ _QWEN_ENABLED = os.getenv("WHYX_ENABLE_QWEN_VL", "").strip().lower() in ("1", "true", "yes", "on")
31
+
32
  _MODEL_ID = os.environ.get("WHYX_QWEN_VL_MODEL", "Qwen/Qwen3-VL-4B-Instruct")
33
  _vl_instance: "QwenVLTagger | None" = None
34
 
35
 
36
+
37
  def _vl_enabled() -> bool:
38
  return os.environ.get("WHYX_ENABLE_QWEN_VL", "1").strip().lower() not in ("0", "false", "no", "off")
39
 
 
56
 
57
 
58
  class QwenVLTagger:
59
+ """Qwen3-VL wrapper exclusively for the tagger's natural-language caption.
60
+
61
+ The model is fully disabled by default (`WHYX_ENABLE_QWEN_VL=1` activates it).
62
+ On CPU-only environments the model would otherwise consume more than the
63
+ 16 GB limit — hence the off default.
64
+ """
65
+
66
  def __init__(self, model_id: str = _MODEL_ID):
67
  self._model_id = model_id
68
  self._processor: Optional[AutoProcessor] = None
69
  self._model: Optional["_AutoModel"] = None
 
70
  self._loaded = False
71
 
72
  def ensure_loaded(self) -> bool:
 
76
  raise RuntimeError("transformers is not installed")
77
  if not _vl_enabled():
78
  raise RuntimeError("Qwen VL is disabled (WHYX_ENABLE_QWEN_VL=0)")
 
79
  self._processor = AutoProcessor.from_pretrained(self._model_id)
80
+ dtype = torch.float32 # FP32 on CPU saves 4 bits per parameter over bf16.
81
  self._model = _AutoModel.from_pretrained(
82
  self._model_id,
83
  dtype=dtype,
84
  )
85
+ # Keep the model resident on the target device (CPU on our tier).
86
+ device = "cpu"
87
+ self._model = self._model.to(device)
88
  self._loaded = True
89
  return True
90
 
 
98
  if not self.ensure_loaded():
99
  return ""
100
  img = self._to_pil(image)
101
+ # Qwen3-VL accepts a chat-style multi-turn format; we only need one-turn
102
+ # caption generation. Keep the framing as user-message-only for the best
103
+ # instruction-following on the free tier.
104
  system_prompt = prompt or (
105
  "Describe this image in ONE concise sentence suitable as a Stable "
106
+ "Diffusion prompt (no preamble, no extra sentences)."
107
  )
108
  messages = [
109
+ {
110
+ "role": "user",
111
+ "content": [
112
+ {"type": "text", "text": "Image:"},
113
+ {"type": "image"},
114
+ {"type": "text", "text": system_prompt},
115
+ ],
116
+ },
117
  ]
118
  text = self._processor.apply_chat_template(messages, add_generation_prompt=True, tokenize=False)
119
  inputs = self._processor(text=[text], images=[img], return_tensors="pt")
 
120
  with torch.inference_mode():
121
  gen = self._model.generate(**inputs, max_new_tokens=max_new_tokens)
122
  trimmed = gen[:, inputs["input_ids"].shape[1]:]
123
  caption = self._processor.batch_decode(trimmed, skip_special_tokens=True)[0]
124
+ return caption.strip()
125
 
126
 
127
  def get_qwen_tagger() -> QwenVLTagger: