Spaces:
Running
Running
File size: 8,091 Bytes
e6404d0 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 | """Fetch the ~745 additional artists needed to reach 1000 in artist.json.
Reads the Danbooru tag list (category=1 = artists), ordered by post count,
and builds a full artist-pool entry for each one:
tag canonical artist tag from Danbooru (lowercased, underscored
surfaces converted to our internal spaced form)
popularity 0-100, linearly scaled from the Danbooru post count of the
maximum-count artist in the pool (so existing 255 keep their
curated popularity and new artists integrate smoothly)
style heuristic guess from the artist's top-5 co-occurring general
tags, pulled from the same cooccurrence graph the app uses
signature_tags top-5 co-occurring general tags (from tag_cooccurrence.json)
conflicts empty (we can't know hand-curated conflicts yet)
description_en Present only when the artist already existed (kept from the
curated list); new entries leave it empty so the UI shows the
tag list as fallback text instead of a placeholder sentence.
description_ru Same as description_en.
danbooru_url canonical posts page for that artist
The script reads/writes atomically with a .bak backup.
"""
import json
import os
import sys
import time
import urllib.parse
import urllib.request
DATA_DIR = os.path.join(os.path.dirname(os.path.dirname(__file__)), "data")
ARTIST_PATH = os.path.join(DATA_DIR, "tag_pools", "artist.json")
COOC_PATH = os.path.join(DATA_DIR, "tag_cooccurrence.json")
STYLE_HINTS = {
"vibrant": {"vibrant colors", "colorful", "bright colors", "flat color", "bold",
"psychedelic", "rainbow", "neon"},
"dark": {"dark colors", "gothic", "horror", "grimdark", "noir", "monochrome",
"sepia", "nocturne"},
"detailed":{"detailed", "intricate", "ornate", "highly detailed", "extremely detailed"},
"cute": {"cute", "kawaii", "chibi", "moe", "pastel colors", "blush", "animal ears"},
"elegant": {"elegant", "fantasy", "flowing", "refined", "royal", "hime"},
"dynamic": {"dynamic pose", "action", "motion blur", "running", "battle", "fighting"},
"moe": {"moe", "cute", "kawaii", "school uniform", "twintails", "maid"},
"sexy": {"suggestive", "cleavage", "lace", "lingerie", "reclining"},
"realistic": {"photorealistic", "realistic", "hyperrealism"},
"painterly": {"painterly", "watercolor", "oil painting", "impasto"},
"minimalist": {"minimalist", "flat design", "simple background"},
}
def _load_json(path: str):
with open(path, "r", encoding="utf-8") as fh:
return json.load(fh)
def _atomic_write_json(path: str, payload):
tmp = path + ".tmp"
bak = path + ".bak"
with open(tmp, "w", encoding="utf-8") as fh:
json.dump(payload, fh, ensure_ascii=False, indent=1)
if os.path.exists(bak):
os.remove(bak)
if os.path.exists(path):
os.replace(path, bak)
os.replace(tmp, path)
def _guess_style(tags: list[str]) -> str:
if not tags:
return "detailed"
scores: dict[str, int] = {}
tag_set = set(t.lower() for t in tags)
for style, hints in STYLE_HINTS.items():
scores[style] = len(tag_set & hints)
best = max(scores, key=scores.get)
return best if scores[best] else "detailed"
def _fetch_artists_page(page: int, limit: int = 1000) -> list[dict]:
# Danbooru API: /tags.json?search[category]=1&search[order]=count&limit=1000&page=N
params = urllib.parse.urlencode({
"search[category]": "1",
"search[order]": "count",
"limit": limit,
"page": page,
})
url = f"https://danbooru.donmai.us/tags.json?{params}"
req = urllib.request.Request(
url,
headers={"User-Agent": "Whyx-PROptimizer/1.0 (dataset sync)"},
)
with urllib.request.urlopen(req, timeout=30) as resp:
data = json.loads(resp.read().decode("utf-8"))
# The endpoint wraps results in {"tags": [...]} on some deployments,
# and returns a bare list on others. Handle both.
if isinstance(data, dict) and "tags" in data:
return data["tags"]
return data
def main(target_total: int = 1000, throttle_s: float = 1.0, dry_run: bool = False):
current = _load_json(ARTIST_PATH)
existing = {a["tag"].lower() for a in current["artists"]}
missing = max(0, target_total - len(current["artists"]))
if missing == 0:
print(f"already at {len(current['artists'])} artists — nothing to do")
return
cooc = _load_json(COOC_PATH).get("cooccurrence", {})
# Fetch pages until we have enough artists (or Danbooru runs out).
fetched: list[dict] = []
page = 1
while len(fetched) < missing:
batch = _fetch_artists_page(page=page)
if not batch:
break
fetched.extend(batch)
page += 1
time.sleep(throttle_s)
# Danbooru returns post counts under different key spellings across versions.
def _count(rec: dict) -> int:
for k in ("post_count", "tag_count", "postcount"):
if k in rec:
try:
return int(rec[k])
except (TypeError, ValueError):
pass
return 0
# Keep only artist-category rows (defensive: some pages may be off).
new_rows = [
r for r in fetched
if r.get("name") and r["name"].lower() not in existing
][:missing]
if not new_rows:
print("no additional artists found on Danbooru")
return
# Popularity anchors calibrated against the EXISTING 255 artists (which use
# curated values 60-99) and their Danbooru post_count at sync time:
# wlop 398 posts -> ~85 (existing: 97)
# mika_pikazo 1111 posts -> ~93 (existing: 98)
# kantoku 2463 posts -> ~98 (existing top artists are 95-99)
# Linear scale: pop = 70 + 30 * (count / 2500), clamped. Newly fetched
# artists are capped at 96 so the hand-curated top (97-99) keeps priority.
_POP_MIN, _POP_MAX, _POP_TOP_COUNT, _POP_NEW_CAP = 70, 99, 2500, 96
new_artists = []
for r in new_rows:
tag = r["name"].replace("_", " ").strip()
co = cooc.get(tag, [])[:5]
signature = [c["tag"].replace("_", " ") for c in co if c.get("tag")]
style = _guess_style(signature)
raw = _count(r)
pop = _POP_MIN + (_POP_MAX - _POP_MIN) * min(raw, _POP_TOP_COUNT) / _POP_TOP_COUNT
pop = min(_POP_NEW_CAP, int(round(pop)))
new_artists.append({
"popularity": pop,
"style": style,
"signature_tags": signature,
"tag": tag,
"conflicts": [],
"description_en": "",
"description_ru": "",
"danbooru_url": (
"https://danbooru.donmai.us/posts?tags=" +
urllib.parse.quote(r["name"])
),
})
out = {
"artists": current["artists"] + new_artists,
"tandems": current.get("tandems", []),
}
print(f"have {len(current['artists'])} + add {len(new_artists)} -> {len(out['artists'])}")
if dry_run:
for a in new_artists[:5]:
print(" sample:", a["tag"], f"pop={a['popularity']}", f"style={a['style']}")
print("(dry-run: file not written)")
return
_atomic_write_json(ARTIST_PATH, out)
print(f"wrote {ARTIST_PATH} (+ .bak)")
empty_sig = sum(1 for a in new_artists if not a["signature_tags"])
if empty_sig:
print(
f"NOTE: {empty_sig} artists have empty signature_tags (not in the\n"
f"cooccurrence anchor set). Run scripts/enrich_artist_signatures.py\n"
f"to fetch their top tags from Danbooru posts."
);
if __name__ == "__main__":
import argparse
ap = argparse.ArgumentParser()
ap.add_argument("--target", type=int, default=1000)
ap.add_argument("--throttle", type=float, default=1.0)
ap.add_argument("--dry-run", action="store_true")
args = ap.parse_args()
main(args.target, args.throttle, args.dry_run)
|