File size: 8,091 Bytes
e6404d0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
"""Fetch the ~745 additional artists needed to reach 1000 in artist.json.

Reads the Danbooru tag list (category=1 = artists), ordered by post count,
and builds a full artist-pool entry for each one:

  tag            canonical artist tag from Danbooru (lowercased, underscored
                 surfaces converted to our internal spaced form)
  popularity     0-100, linearly scaled from the Danbooru post count of the
                 maximum-count artist in the pool (so existing 255 keep their
                 curated popularity and new artists integrate smoothly)
  style          heuristic guess from the artist's top-5 co-occurring general
                 tags, pulled from the same cooccurrence graph the app uses
  signature_tags top-5 co-occurring general tags (from tag_cooccurrence.json)
  conflicts      empty (we can't know hand-curated conflicts yet)
  description_en Present only when the artist already existed (kept from the
                 curated list); new entries leave it empty so the UI shows the
                 tag list as fallback text instead of a placeholder sentence.
  description_ru Same as description_en.
  danbooru_url   canonical posts page for that artist

The script reads/writes atomically with a .bak backup.
"""

import json
import os
import sys
import time
import urllib.parse
import urllib.request

DATA_DIR = os.path.join(os.path.dirname(os.path.dirname(__file__)), "data")
ARTIST_PATH = os.path.join(DATA_DIR, "tag_pools", "artist.json")
COOC_PATH = os.path.join(DATA_DIR, "tag_cooccurrence.json")

STYLE_HINTS = {
    "vibrant": {"vibrant colors", "colorful", "bright colors", "flat color", "bold",
                "psychedelic", "rainbow", "neon"},
    "dark":    {"dark colors", "gothic", "horror", "grimdark", "noir", "monochrome",
                "sepia", "nocturne"},
    "detailed":{"detailed", "intricate", "ornate", "highly detailed", "extremely detailed"},
    "cute":    {"cute", "kawaii", "chibi", "moe", "pastel colors", "blush", "animal ears"},
    "elegant": {"elegant", "fantasy", "flowing", "refined", "royal", "hime"},
    "dynamic": {"dynamic pose", "action", "motion blur", "running", "battle", "fighting"},
    "moe":     {"moe", "cute", "kawaii", "school uniform", "twintails", "maid"},
    "sexy":    {"suggestive", "cleavage", "lace", "lingerie", "reclining"},
    "realistic": {"photorealistic", "realistic", "hyperrealism"},
    "painterly": {"painterly", "watercolor", "oil painting", "impasto"},
    "minimalist": {"minimalist", "flat design", "simple background"},
}


def _load_json(path: str):
    with open(path, "r", encoding="utf-8") as fh:
        return json.load(fh)


def _atomic_write_json(path: str, payload):
    tmp = path + ".tmp"
    bak = path + ".bak"
    with open(tmp, "w", encoding="utf-8") as fh:
        json.dump(payload, fh, ensure_ascii=False, indent=1)
    if os.path.exists(bak):
        os.remove(bak)
    if os.path.exists(path):
        os.replace(path, bak)
    os.replace(tmp, path)


def _guess_style(tags: list[str]) -> str:
    if not tags:
        return "detailed"
    scores: dict[str, int] = {}
    tag_set = set(t.lower() for t in tags)
    for style, hints in STYLE_HINTS.items():
        scores[style] = len(tag_set & hints)
    best = max(scores, key=scores.get)
    return best if scores[best] else "detailed"


def _fetch_artists_page(page: int, limit: int = 1000) -> list[dict]:
    # Danbooru API: /tags.json?search[category]=1&search[order]=count&limit=1000&page=N
    params = urllib.parse.urlencode({
        "search[category]": "1",
        "search[order]": "count",
        "limit": limit,
        "page": page,
    })
    url = f"https://danbooru.donmai.us/tags.json?{params}"
    req = urllib.request.Request(
        url,
        headers={"User-Agent": "Whyx-PROptimizer/1.0 (dataset sync)"},
    )
    with urllib.request.urlopen(req, timeout=30) as resp:
        data = json.loads(resp.read().decode("utf-8"))
    # The endpoint wraps results in {"tags": [...]} on some deployments,
    # and returns a bare list on others. Handle both.
    if isinstance(data, dict) and "tags" in data:
        return data["tags"]
    return data


def main(target_total: int = 1000, throttle_s: float = 1.0, dry_run: bool = False):
    current = _load_json(ARTIST_PATH)
    existing = {a["tag"].lower() for a in current["artists"]}
    missing = max(0, target_total - len(current["artists"]))
    if missing == 0:
        print(f"already at {len(current['artists'])} artists — nothing to do")
        return

    cooc = _load_json(COOC_PATH).get("cooccurrence", {})

    # Fetch pages until we have enough artists (or Danbooru runs out).
    fetched: list[dict] = []
    page = 1
    while len(fetched) < missing:
        batch = _fetch_artists_page(page=page)
        if not batch:
            break
        fetched.extend(batch)
        page += 1
        time.sleep(throttle_s)

    # Danbooru returns post counts under different key spellings across versions.
    def _count(rec: dict) -> int:
        for k in ("post_count", "tag_count", "postcount"):
            if k in rec:
                try:
                    return int(rec[k])
                except (TypeError, ValueError):
                    pass
        return 0

    # Keep only artist-category rows (defensive: some pages may be off).
    new_rows = [
        r for r in fetched
        if r.get("name") and r["name"].lower() not in existing
    ][:missing]
    if not new_rows:
        print("no additional artists found on Danbooru")
        return

    # Popularity anchors calibrated against the EXISTING 255 artists (which use
    # curated values 60-99) and their Danbooru post_count at sync time:
    #   wlop       398 posts  -> ~85   (existing: 97)
    #   mika_pikazo 1111 posts -> ~93  (existing: 98)
    #   kantoku     2463 posts -> ~98  (existing top artists are 95-99)
    # Linear scale: pop = 70 + 30 * (count / 2500), clamped. Newly fetched
    # artists are capped at 96 so the hand-curated top (97-99) keeps priority.
    _POP_MIN, _POP_MAX, _POP_TOP_COUNT, _POP_NEW_CAP = 70, 99, 2500, 96

    new_artists = []
    for r in new_rows:
        tag = r["name"].replace("_", " ").strip()
        co = cooc.get(tag, [])[:5]
        signature = [c["tag"].replace("_", " ") for c in co if c.get("tag")]
        style = _guess_style(signature)
        raw = _count(r)
        pop = _POP_MIN + (_POP_MAX - _POP_MIN) * min(raw, _POP_TOP_COUNT) / _POP_TOP_COUNT
        pop = min(_POP_NEW_CAP, int(round(pop)))
        new_artists.append({
            "popularity": pop,
            "style": style,
            "signature_tags": signature,
            "tag": tag,
            "conflicts": [],
            "description_en": "",
            "description_ru": "",
            "danbooru_url": (
                "https://danbooru.donmai.us/posts?tags=" +
                urllib.parse.quote(r["name"])
            ),
        })

    out = {
        "artists": current["artists"] + new_artists,
        "tandems": current.get("tandems", []),
    }
    print(f"have {len(current['artists'])} + add {len(new_artists)} -> {len(out['artists'])}")
    if dry_run:
        for a in new_artists[:5]:
            print("  sample:", a["tag"], f"pop={a['popularity']}", f"style={a['style']}")
        print("(dry-run: file not written)")
        return
    _atomic_write_json(ARTIST_PATH, out)
    print(f"wrote {ARTIST_PATH} (+ .bak)")
    empty_sig = sum(1 for a in new_artists if not a["signature_tags"])
    if empty_sig:
        print(
            f"NOTE: {empty_sig} artists have empty signature_tags (not in the\n"
            f"cooccurrence anchor set). Run scripts/enrich_artist_signatures.py\n"
            f"to fetch their top tags from Danbooru posts."
        );


if __name__ == "__main__":
    import argparse
    ap = argparse.ArgumentParser()
    ap.add_argument("--target", type=int, default=1000)
    ap.add_argument("--throttle", type=float, default=1.0)
    ap.add_argument("--dry-run", action="store_true")
    args = ap.parse_args()
    main(args.target, args.throttle, args.dry_run)