"""Evaluation data: public benchmark eval/test splits rewritten as typed decision questions. Each example is {"state", "question" (Laya/Jev-style typed question), "target" (distribution over the question's options in render order), "source"}. Only evaluation splits are loaded. Instruction wording and option descriptions are sampled per example with a fixed seed, so a run is reproducible. Label-set sources (banking77, MASSIVE) show up to 20 options per question: the gold label plus randomly sampled other labels. """ import json import random def _onehot(k, i): t = [0.0] * k t[i] = 1.0 return t def _pick(rng, xs): return xs[rng.randrange(len(xs))] def _choice_q(rng, instructions, labels, descs=None): crit = {lab: (descs[lab] if descs and rng.random() < 0.7 else None) for lab in labels} return {"type": "choice", "instructions": _pick(rng, instructions), "criteria": crit} def _subsample_labels(rng, all_labels, gold, max_opts): others = [l for l in all_labels if l != gold] rng.shuffle(others) labs = others[: max_opts - 1] + [gold] rng.shuffle(labs) return labs # --------------------------------------------------------------------------- # # Source converters: fn(row, rng) -> (state, question, target) # --------------------------------------------------------------------------- # EMOTIONS = ["sadness", "joy", "love", "anger", "fear", "surprise"] EMO_DESC = {"sadness": "the writer feels sad or down", "joy": "the writer feels happy", "love": "the writer feels affection or love", "anger": "the writer feels angry or annoyed", "fear": "the writer feels afraid or anxious", "surprise": "the writer feels surprised"} AG = ["World", "Sports", "Business", "Sci/Tech"] AG_DESC = {"World": "international news and politics", "Sports": "sport results and athletes", "Business": "companies, markets and the economy", "Sci/Tech": "science and technology"} NLI = ["entailment", "neutral", "contradiction"] NLI_DESC = {"entailment": "the hypothesis must be true given the premise", "neutral": "the hypothesis might or might not be true", "contradiction": "the hypothesis cannot be true given the premise"} STARS = ["very negative (1 star)", "negative (2 stars)", "mixed (3 stars)", "positive (4 stars)", "very positive (5 stars)"] def _emotion(r, rng): q = _choice_q(rng, ["Which emotion does the writer express?", "What is the main feeling in this message?", "Classify the emotion of this text."], EMOTIONS, EMO_DESC) return r["text"], q, _onehot(6, int(r["label"])) def _ag(r, rng): q = _choice_q(rng, ["Which news section does this article belong to?", "What is the topic of this article?", "Route this story to the right desk."], AG, AG_DESC) return r["text"], q, _onehot(4, int(r["label"])) def _intent(label_key, max_opts, instr): def fn(r, rng): labels = fn.labels # full label set, filled from the eval split by build() gold = str(r[label_key]) if labels is None or gold not in labels: return None labs = _subsample_labels(rng, labels, gold, max_opts) crit = {l: (l.replace("_", " ") if rng.random() < 0.5 else None) for l in labs} return r["text"], {"type": "choice", "instructions": _pick(rng, instr), "criteria": crit}, \ _onehot(len(labs), labs.index(gold)) fn.labels = None return fn def _boolq(r, rng): q = {"type": "noul", "instructions": r["question"].strip().rstrip("?") + "?"} if rng.random() < 0.5: q["criteria"] = {"false": "the passage says no", "true": "the passage says yes"} y = 1 if r["answer"] else 0 return {"passage": r["passage"]} if rng.random() < 0.5 else r["passage"], q, [1.0 - y, float(y)] def _sentiment(text_key): def fn(r, rng): y = int(r["label"]) if rng.random() < 0.5: q = {"type": "noul", "instructions": _pick(rng, ["Is the sentiment of this text positive?", "Is the reviewer happy with it?", "The writer is positive about the subject."])} return r[text_key], q, [1.0 - y, float(y)] q = _choice_q(rng, ["What is the sentiment of this text?", "How does the writer feel about it?"], ["negative", "positive"], {"negative": "unfavourable, critical", "positive": "favourable, pleased"}) return r[text_key], q, _onehot(2, y) return fn def _yelp(r, rng): q = {"type": "score", "instructions": _pick(rng, ["How positive is this review?", "Rate the customer's satisfaction.", "How many stars does this review deserve?"]), "criteria": STARS} return r["text"], q, _onehot(5, int(r["label"])) def _hate(r, rng): y = int(r["label"]) q = {"type": "noul", "instructions": _pick(rng, ["Does this text contain hate speech?", "This message attacks people for who they are."]), "criteria": {"false": "no hateful content", "true": "hateful towards a group"}} return r["text"], q, [1.0 - y, float(y)] def _nli(r, rng): if int(r["label"]) < 0: return None q = _choice_q(rng, ["How does the hypothesis relate to the premise?", "Given the premise, is the hypothesis entailed, neutral or contradicted?"], NLI, NLI_DESC) return {"premise": r["premise"], "hypothesis": r["hypothesis"]}, q, _onehot(3, int(r["label"])) SOURCES = { # name: (path, config, eval split, converter, label_key for label-set sources) "emotion": ("dair-ai/emotion", "split", "validation", _emotion, None), "ag_news": ("fancyzhx/ag_news", None, "test", _ag, None), "banking77": ("mteb/banking77", None, "test", _intent("label_text", 20, ["Which banking intent does the customer have?", "Route this customer message to the right intent."]), "label_text"), "massive": ("mteb/amazon_massive_intent", "en", "validation", _intent("label", 20, ["What does the user want the assistant to do?", "Which intent is this voice command?"]), "label"), "boolq": ("google/boolq", None, "validation", _boolq, None), "sst2": ("stanfordnlp/sst2", None, "validation", _sentiment("sentence"), None), "imdb": ("stanfordnlp/imdb", None, "test", _sentiment("text"), None), "yelp": ("Yelp/yelp_review_full", None, "test", _yelp, None), "hate": ("cardiffnlp/tweet_eval", "hate", "validation", _hate, None), "mnli": ("nyu-mll/multi_nli", None, "validation_matched", _nli, None), } def _eval_label_set(path, name, split, key): from datasets import load_dataset return sorted({str(x) for x in load_dataset(path, name, split=split)[key]}) def build(per_source: int = 200, seed: int = 0, sources=None): """Up to `per_source` examples from each source's evaluation split.""" from datasets import load_dataset rng = random.Random(seed + 7) out = [] for name in (sources or SOURCES): path, cfg, split, fn, label_key = SOURCES[name] if label_key and fn.labels is None: fn.labels = _eval_label_set(path, cfg, split, label_key) got = 0 for r in load_dataset(path, cfg, split=split, streaming=True).take(per_source * 2): ex = fn(r, rng) if ex is None: continue state, q, target = ex out.append({"state": state, "question": q, "target": target, "source": name}) got += 1 if got >= per_source: break print(f" [eval data] {name:10s}: {got}") rng.shuffle(out) return out def load_jsonl(paths): out = [] for p in paths: with open(p, encoding="utf-8") as fh: out.extend(json.loads(line) for line in fh if line.strip()) return out