Spaces:
Running
Running
feat(tagger): NL captioning via Qwen3-VL, pose detection UI fixes, memory-safe for HF Spaces (16GB CPU)
Browse files- _debug_person.png +0 -0
- data/session_history.json.bak +78 -78
- src/i18n.py +1 -0
- src/image_tagger.py +10 -0
- src/qwen_vl_tagger.py +27 -14
_debug_person.png
ADDED
|
data/session_history.json.bak
CHANGED
|
@@ -1,4 +1,82 @@
|
|
| 1 |
[
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 2 |
{
|
| 3 |
"id": "c9ed5cc5",
|
| 4 |
"timestamp": 1785778992.0812612,
|
|
@@ -1220,83 +1298,5 @@
|
|
| 1220 |
"categories": [],
|
| 1221 |
"seed": null,
|
| 1222 |
"liked_indices": []
|
| 1223 |
-
},
|
| 1224 |
-
{
|
| 1225 |
-
"id": "ba70a0a2",
|
| 1226 |
-
"timestamp": 1784813293.9073813,
|
| 1227 |
-
"prompt": "1girl, blue hair",
|
| 1228 |
-
"model": "anima",
|
| 1229 |
-
"rating": "pg",
|
| 1230 |
-
"num_variations": 2,
|
| 1231 |
-
"creativity": "medium",
|
| 1232 |
-
"weight_mode": "off",
|
| 1233 |
-
"results": [
|
| 1234 |
-
"masterpiece, best quality, score_9, safe, 1girl, BREAK, high contrast, hair ribbon, blue hair, detailed background, sacred atmosphere, dragon lair, global illumination, moonlight beam, light rays underwater, holographic display, silver aesthetic, leaning against wall, radial composition, hand on own chest, looking at viewer, extreme close-up, surprised pose, intricate details, beautiful detailed",
|
| 1235 |
-
"masterpiece, best quality, score_9, safe, 1girl, BREAK, vintage color grading, hair between eyes, saturated matte, high contrast, blue hair, detailed background, transmission tower, divine radiance, shield, car, global illumination, gothic lighting, silver aesthetic, ice crystals, cosmic dust, floating, foreshortening, surprised pose, looking away, full body, intricate details, beautiful detailed",
|
| 1236 |
-
"",
|
| 1237 |
-
"",
|
| 1238 |
-
"",
|
| 1239 |
-
"",
|
| 1240 |
-
"",
|
| 1241 |
-
"",
|
| 1242 |
-
"",
|
| 1243 |
-
""
|
| 1244 |
-
],
|
| 1245 |
-
"neg_results": [],
|
| 1246 |
-
"categories": [],
|
| 1247 |
-
"seed": null,
|
| 1248 |
-
"liked_indices": []
|
| 1249 |
-
},
|
| 1250 |
-
{
|
| 1251 |
-
"id": "103989cc",
|
| 1252 |
-
"timestamp": 1784813166.9979286,
|
| 1253 |
-
"prompt": "1girl, blue hair",
|
| 1254 |
-
"model": "anima",
|
| 1255 |
-
"rating": "pg",
|
| 1256 |
-
"num_variations": 2,
|
| 1257 |
-
"creativity": "medium",
|
| 1258 |
-
"weight_mode": "off",
|
| 1259 |
-
"results": [
|
| 1260 |
-
"masterpiece, best quality, score_9, safe, 1girl, BREAK, sticking out tongue, saturated matte, blue hair, skirt, detailed background, retro futuristic, enchanted boat, motorcycle, phoenix, sword, global illumination, moonlight beam, light rays underwater, holographic display, silver aesthetic, foreground framing, hands behind back, looking at viewer, extreme close-up, surprised pose, intricate details, beautiful detailed",
|
| 1261 |
-
"masterpiece, best quality, score_9, safe, 1girl, BREAK, iridescent colors, blank expression, saturated matte, high contrast, blue hair, glasses, mask, detailed background, christmas present, abyssal darkness, moonlight beam, silver aesthetic, cosmic dust, first person view, looking at viewer, surprised pose, wide shot, intricate details, beautiful detailed",
|
| 1262 |
-
"",
|
| 1263 |
-
"",
|
| 1264 |
-
"",
|
| 1265 |
-
"",
|
| 1266 |
-
"",
|
| 1267 |
-
"",
|
| 1268 |
-
"",
|
| 1269 |
-
""
|
| 1270 |
-
],
|
| 1271 |
-
"neg_results": [],
|
| 1272 |
-
"categories": [],
|
| 1273 |
-
"seed": null,
|
| 1274 |
-
"liked_indices": []
|
| 1275 |
-
},
|
| 1276 |
-
{
|
| 1277 |
-
"id": "671a7818",
|
| 1278 |
-
"timestamp": 1784813130.1904197,
|
| 1279 |
-
"prompt": "1girl, blue hair",
|
| 1280 |
-
"model": "anima",
|
| 1281 |
-
"rating": "pg",
|
| 1282 |
-
"num_variations": 2,
|
| 1283 |
-
"creativity": "medium",
|
| 1284 |
-
"weight_mode": "off",
|
| 1285 |
-
"results": [
|
| 1286 |
-
"masterpiece, best quality, score_9, safe, 1girl, BREAK, hair between eyes, saturated matte, broad shoulders, triadic colors, autumn colors, blue hair, umbrella, complex background, mythical world, boat, gothic lighting, moonlight beam, silver aesthetic, surprised pose, looking away, from above, intricate details, beautiful detailed",
|
| 1287 |
-
"masterpiece, best quality, score_9, safe, 1girl, BREAK, iridescent colors, high contrast, blue hair, backless dress, star, detailed background, samurai residence, global illumination, holographic display, silver aesthetic, lightning aura, looking at viewer, surprised pose, faux traditional media, incredibly absurdres, intricate details, beautiful detailed",
|
| 1288 |
-
"",
|
| 1289 |
-
"",
|
| 1290 |
-
"",
|
| 1291 |
-
"",
|
| 1292 |
-
"",
|
| 1293 |
-
"",
|
| 1294 |
-
"",
|
| 1295 |
-
""
|
| 1296 |
-
],
|
| 1297 |
-
"neg_results": [],
|
| 1298 |
-
"categories": [],
|
| 1299 |
-
"seed": null,
|
| 1300 |
-
"liked_indices": []
|
| 1301 |
}
|
| 1302 |
]
|
|
|
|
| 1 |
[
|
| 2 |
+
{
|
| 3 |
+
"id": "27bb436d",
|
| 4 |
+
"timestamp": 1785784568.4936664,
|
| 5 |
+
"prompt": "1girl, blue hair",
|
| 6 |
+
"model": "anima",
|
| 7 |
+
"rating": "pg",
|
| 8 |
+
"num_variations": 2,
|
| 9 |
+
"creativity": "medium",
|
| 10 |
+
"weight_mode": "off",
|
| 11 |
+
"results": [
|
| 12 |
+
"masterpiece, best quality, score_7, safe, 1girl, BREAK, blue hair, hair tie, three-quarter view, from below, profile, sepia, detailed background, blue hour, warm, underwater caustics, ethereal lighting, dreamy lighting, gothic lighting, holographic display, enchanted boat, steam train, beautiful detailed, beautiful detail",
|
| 13 |
+
"masterpiece, best quality, score_7, safe, 1girl, BREAK, blue hair, hair tie, upper body, high contrast, vibrant, surprised pose, crying tears, triangular composition, detailed background, blue hour, bioluminescent sea, horror backlight, morning sky, magical particles, silver aesthetic, blue tint, incredibly absurdres, enchanted boat, sports car, beautiful detailed, beautiful detail",
|
| 14 |
+
"",
|
| 15 |
+
"",
|
| 16 |
+
"",
|
| 17 |
+
"",
|
| 18 |
+
"",
|
| 19 |
+
"",
|
| 20 |
+
"",
|
| 21 |
+
""
|
| 22 |
+
],
|
| 23 |
+
"neg_results": [],
|
| 24 |
+
"categories": [],
|
| 25 |
+
"seed": null,
|
| 26 |
+
"liked_indices": []
|
| 27 |
+
},
|
| 28 |
+
{
|
| 29 |
+
"id": "f2656317",
|
| 30 |
+
"timestamp": 1785784470.7438738,
|
| 31 |
+
"prompt": "1girl, blue hair",
|
| 32 |
+
"model": "anima",
|
| 33 |
+
"rating": "pg",
|
| 34 |
+
"num_variations": 2,
|
| 35 |
+
"creativity": "medium",
|
| 36 |
+
"weight_mode": "off",
|
| 37 |
+
"results": [
|
| 38 |
+
"masterpiece, best quality, score_7, safe, 1girl, BREAK, blue hair, hair tie, tears of joy, three-quarter view, cowboy shot, high contrast, vibrant, sinister smile, pov, detailed background, mist, underwater caustics, horror backlight, lightning aura, silver aesthetic, blue tint, smoke, incredibly absurdres, enchanted boat, beautiful detailed, beautiful detail",
|
| 39 |
+
"masterpiece, best quality, score_7, safe, 1girl, BREAK, absurdly long hair, blue hair, sly smile, serious, happy, three-quarter view, from behind, vaporwave aesthetic, high contrast, red and black, vibrant, pov, detailed background, foggy, underwater caustics, ethereal lighting, silver aesthetic, dragon, kimono, beautiful detailed, beautiful detail",
|
| 40 |
+
"",
|
| 41 |
+
"",
|
| 42 |
+
"",
|
| 43 |
+
"",
|
| 44 |
+
"",
|
| 45 |
+
"",
|
| 46 |
+
"",
|
| 47 |
+
""
|
| 48 |
+
],
|
| 49 |
+
"neg_results": [],
|
| 50 |
+
"categories": [],
|
| 51 |
+
"seed": null,
|
| 52 |
+
"liked_indices": []
|
| 53 |
+
},
|
| 54 |
+
{
|
| 55 |
+
"id": "0593f16f",
|
| 56 |
+
"timestamp": 1785783108.9567664,
|
| 57 |
+
"prompt": "1girl, blue hair",
|
| 58 |
+
"model": "anima",
|
| 59 |
+
"rating": "pg",
|
| 60 |
+
"num_variations": 2,
|
| 61 |
+
"creativity": "medium",
|
| 62 |
+
"weight_mode": "off",
|
| 63 |
+
"results": [
|
| 64 |
+
"masterpiece, best quality, score_7, safe, 1girl, BREAK, blue hair, magical girl pose, split complementary, high contrast, vibrant, pleated skirt, close-up, pov, detailed background, fog, gothic lighting, dark lighting, silver aesthetic, blue tint, detailed illustration, evil smile, beautiful detailed, beautiful detail",
|
| 65 |
+
"masterpiece, best quality, score_7, safe, 1girl, BREAK, blue hair, cross processed colors, high contrast, vibrant, suit, detailed background, sunset, creative atmosphere, warm, partly cloudy light, rainy lighting, smoke, chibi, portrait, temple, beautiful detailed, beautiful detail",
|
| 66 |
+
"",
|
| 67 |
+
"",
|
| 68 |
+
"",
|
| 69 |
+
"",
|
| 70 |
+
"",
|
| 71 |
+
"",
|
| 72 |
+
"",
|
| 73 |
+
""
|
| 74 |
+
],
|
| 75 |
+
"neg_results": [],
|
| 76 |
+
"categories": [],
|
| 77 |
+
"seed": null,
|
| 78 |
+
"liked_indices": []
|
| 79 |
+
},
|
| 80 |
{
|
| 81 |
"id": "c9ed5cc5",
|
| 82 |
"timestamp": 1785778992.0812612,
|
|
|
|
| 1298 |
"categories": [],
|
| 1299 |
"seed": null,
|
| 1300 |
"liked_indices": []
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1301 |
}
|
| 1302 |
]
|
src/i18n.py
CHANGED
|
@@ -700,6 +700,7 @@ L10N = {
|
|
| 700 |
"tagger_mode_swinv2": "WD14 SwinV2",
|
| 701 |
"tagger_mode_vit": "WD14 ViT",
|
| 702 |
"tagger_mode_qwen": "Qwen3-VL (NL описание)",
|
|
|
|
| 703 |
"tagger_pose_label": "Поза",
|
| 704 |
"tagger_pose_single": "Обнаружен {n} человек",
|
| 705 |
"tagger_pose_many": "Обнаружено {n} человек",
|
|
|
|
| 700 |
"tagger_mode_swinv2": "WD14 SwinV2",
|
| 701 |
"tagger_mode_vit": "WD14 ViT",
|
| 702 |
"tagger_mode_qwen": "Qwen3-VL (NL описание)",
|
| 703 |
+
"tagger_mode_qwen_desc": "Qwen3-VL (Подробное описание изображения)",
|
| 704 |
"tagger_pose_label": "Поза",
|
| 705 |
"tagger_pose_single": "Обнаружен {n} человек",
|
| 706 |
"tagger_pose_many": "Обнаружено {n} человек",
|
src/image_tagger.py
CHANGED
|
@@ -94,6 +94,16 @@ def _to_pil_image(image) -> Image.Image:
|
|
| 94 |
if np is None:
|
| 95 |
raise RuntimeError("numpy is required for image tagging")
|
| 96 |
arr = np.asarray(image)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 97 |
if arr.ndim == 2:
|
| 98 |
arr = np.stack([arr] * 3, axis=-1)
|
| 99 |
elif arr.ndim == 3 and arr.shape[2] == 1:
|
|
|
|
| 94 |
if np is None:
|
| 95 |
raise RuntimeError("numpy is required for image tagging")
|
| 96 |
arr = np.asarray(image)
|
| 97 |
+
# Accept bytes / file-like objects (e.g. from API calls or older Gradio versions)
|
| 98 |
+
if arr.ndim == 0 or arr.dtype == object:
|
| 99 |
+
# Accept bytes / bytearray / memoryview / any file-like with .read()
|
| 100 |
+
buf = image if isinstance(image, (bytes, bytearray, memoryview)) else None
|
| 101 |
+
if buf is None and hasattr(image, "read"):
|
| 102 |
+
buf = image.read()
|
| 103 |
+
if buf is None:
|
| 104 |
+
raise ValueError("Unsupported image payload type")
|
| 105 |
+
from io import BytesIO
|
| 106 |
+
arr = np.array(Image.open(BytesIO(bytes(buf))).convert("RGB"))
|
| 107 |
if arr.ndim == 2:
|
| 108 |
arr = np.stack([arr] * 3, axis=-1)
|
| 109 |
elif arr.ndim == 3 and arr.shape[2] == 1:
|
src/qwen_vl_tagger.py
CHANGED
|
@@ -27,10 +27,13 @@ except Exception: # pragma: no cover
|
|
| 27 |
_VL_DEPS_OK = False
|
| 28 |
AutoProcessor = _AutoModel = None
|
| 29 |
|
|
|
|
|
|
|
| 30 |
_MODEL_ID = os.environ.get("WHYX_QWEN_VL_MODEL", "Qwen/Qwen3-VL-4B-Instruct")
|
| 31 |
_vl_instance: "QwenVLTagger | None" = None
|
| 32 |
|
| 33 |
|
|
|
|
| 34 |
def _vl_enabled() -> bool:
|
| 35 |
return os.environ.get("WHYX_ENABLE_QWEN_VL", "1").strip().lower() not in ("0", "false", "no", "off")
|
| 36 |
|
|
@@ -53,11 +56,17 @@ def _normalize_caption(raw: str) -> str:
|
|
| 53 |
|
| 54 |
|
| 55 |
class QwenVLTagger:
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 56 |
def __init__(self, model_id: str = _MODEL_ID):
|
| 57 |
self._model_id = model_id
|
| 58 |
self._processor: Optional[AutoProcessor] = None
|
| 59 |
self._model: Optional["_AutoModel"] = None
|
| 60 |
-
self._device: Optional[torch.device] = None
|
| 61 |
self._loaded = False
|
| 62 |
|
| 63 |
def ensure_loaded(self) -> bool:
|
|
@@ -67,16 +76,15 @@ class QwenVLTagger:
|
|
| 67 |
raise RuntimeError("transformers is not installed")
|
| 68 |
if not _vl_enabled():
|
| 69 |
raise RuntimeError("Qwen VL is disabled (WHYX_ENABLE_QWEN_VL=0)")
|
| 70 |
-
self._device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
|
| 71 |
self._processor = AutoProcessor.from_pretrained(self._model_id)
|
| 72 |
-
dtype = torch.
|
| 73 |
self._model = _AutoModel.from_pretrained(
|
| 74 |
self._model_id,
|
| 75 |
dtype=dtype,
|
| 76 |
)
|
| 77 |
-
#
|
| 78 |
-
|
| 79 |
-
|
| 80 |
self._loaded = True
|
| 81 |
return True
|
| 82 |
|
|
@@ -90,25 +98,30 @@ class QwenVLTagger:
|
|
| 90 |
if not self.ensure_loaded():
|
| 91 |
return ""
|
| 92 |
img = self._to_pil(image)
|
|
|
|
|
|
|
|
|
|
| 93 |
system_prompt = prompt or (
|
| 94 |
"Describe this image in ONE concise sentence suitable as a Stable "
|
| 95 |
-
"Diffusion prompt (no preamble, no extra sentences
|
| 96 |
)
|
| 97 |
messages = [
|
| 98 |
-
{
|
| 99 |
-
|
| 100 |
-
|
| 101 |
-
|
| 102 |
-
|
|
|
|
|
|
|
|
|
|
| 103 |
]
|
| 104 |
text = self._processor.apply_chat_template(messages, add_generation_prompt=True, tokenize=False)
|
| 105 |
inputs = self._processor(text=[text], images=[img], return_tensors="pt")
|
| 106 |
-
inputs = {k: v.to(self._device) for k, v in inputs.items()}
|
| 107 |
with torch.inference_mode():
|
| 108 |
gen = self._model.generate(**inputs, max_new_tokens=max_new_tokens)
|
| 109 |
trimmed = gen[:, inputs["input_ids"].shape[1]:]
|
| 110 |
caption = self._processor.batch_decode(trimmed, skip_special_tokens=True)[0]
|
| 111 |
-
return
|
| 112 |
|
| 113 |
|
| 114 |
def get_qwen_tagger() -> QwenVLTagger:
|
|
|
|
| 27 |
_VL_DEPS_OK = False
|
| 28 |
AutoProcessor = _AutoModel = None
|
| 29 |
|
| 30 |
+
_QWEN_ENABLED = os.getenv("WHYX_ENABLE_QWEN_VL", "").strip().lower() in ("1", "true", "yes", "on")
|
| 31 |
+
|
| 32 |
_MODEL_ID = os.environ.get("WHYX_QWEN_VL_MODEL", "Qwen/Qwen3-VL-4B-Instruct")
|
| 33 |
_vl_instance: "QwenVLTagger | None" = None
|
| 34 |
|
| 35 |
|
| 36 |
+
|
| 37 |
def _vl_enabled() -> bool:
|
| 38 |
return os.environ.get("WHYX_ENABLE_QWEN_VL", "1").strip().lower() not in ("0", "false", "no", "off")
|
| 39 |
|
|
|
|
| 56 |
|
| 57 |
|
| 58 |
class QwenVLTagger:
|
| 59 |
+
"""Qwen3-VL wrapper exclusively for the tagger's natural-language caption.
|
| 60 |
+
|
| 61 |
+
The model is fully disabled by default (`WHYX_ENABLE_QWEN_VL=1` activates it).
|
| 62 |
+
On CPU-only environments the model would otherwise consume more than the
|
| 63 |
+
16 GB limit — hence the off default.
|
| 64 |
+
"""
|
| 65 |
+
|
| 66 |
def __init__(self, model_id: str = _MODEL_ID):
|
| 67 |
self._model_id = model_id
|
| 68 |
self._processor: Optional[AutoProcessor] = None
|
| 69 |
self._model: Optional["_AutoModel"] = None
|
|
|
|
| 70 |
self._loaded = False
|
| 71 |
|
| 72 |
def ensure_loaded(self) -> bool:
|
|
|
|
| 76 |
raise RuntimeError("transformers is not installed")
|
| 77 |
if not _vl_enabled():
|
| 78 |
raise RuntimeError("Qwen VL is disabled (WHYX_ENABLE_QWEN_VL=0)")
|
|
|
|
| 79 |
self._processor = AutoProcessor.from_pretrained(self._model_id)
|
| 80 |
+
dtype = torch.float32 # FP32 on CPU — saves 4 bits per parameter over bf16.
|
| 81 |
self._model = _AutoModel.from_pretrained(
|
| 82 |
self._model_id,
|
| 83 |
dtype=dtype,
|
| 84 |
)
|
| 85 |
+
# Keep the model resident on the target device (CPU on our tier).
|
| 86 |
+
device = "cpu"
|
| 87 |
+
self._model = self._model.to(device)
|
| 88 |
self._loaded = True
|
| 89 |
return True
|
| 90 |
|
|
|
|
| 98 |
if not self.ensure_loaded():
|
| 99 |
return ""
|
| 100 |
img = self._to_pil(image)
|
| 101 |
+
# Qwen3-VL accepts a chat-style multi-turn format; we only need one-turn
|
| 102 |
+
# caption generation. Keep the framing as user-message-only for the best
|
| 103 |
+
# instruction-following on the free tier.
|
| 104 |
system_prompt = prompt or (
|
| 105 |
"Describe this image in ONE concise sentence suitable as a Stable "
|
| 106 |
+
"Diffusion prompt (no preamble, no extra sentences)."
|
| 107 |
)
|
| 108 |
messages = [
|
| 109 |
+
{
|
| 110 |
+
"role": "user",
|
| 111 |
+
"content": [
|
| 112 |
+
{"type": "text", "text": "Image:"},
|
| 113 |
+
{"type": "image"},
|
| 114 |
+
{"type": "text", "text": system_prompt},
|
| 115 |
+
],
|
| 116 |
+
},
|
| 117 |
]
|
| 118 |
text = self._processor.apply_chat_template(messages, add_generation_prompt=True, tokenize=False)
|
| 119 |
inputs = self._processor(text=[text], images=[img], return_tensors="pt")
|
|
|
|
| 120 |
with torch.inference_mode():
|
| 121 |
gen = self._model.generate(**inputs, max_new_tokens=max_new_tokens)
|
| 122 |
trimmed = gen[:, inputs["input_ids"].shape[1]:]
|
| 123 |
caption = self._processor.batch_decode(trimmed, skip_special_tokens=True)[0]
|
| 124 |
+
return caption.strip()
|
| 125 |
|
| 126 |
|
| 127 |
def get_qwen_tagger() -> QwenVLTagger:
|