| """Evaluation data: public benchmark eval/test splits rewritten as typed decision questions. |
| |
| Each example is {"state", "question" (Laya/Jev-style typed question), "target" (distribution over the |
| question's options in render order), "source"}. Only evaluation splits are loaded. Instruction wording |
| and option descriptions are sampled per example with a fixed seed, so a run is reproducible. |
| Label-set sources (banking77, MASSIVE) show up to 20 options per question: the gold label plus |
| randomly sampled other labels. |
| """ |
| import json |
| import random |
|
|
|
|
| def _onehot(k, i): |
| t = [0.0] * k |
| t[i] = 1.0 |
| return t |
|
|
|
|
| def _pick(rng, xs): |
| return xs[rng.randrange(len(xs))] |
|
|
|
|
| def _choice_q(rng, instructions, labels, descs=None): |
| crit = {lab: (descs[lab] if descs and rng.random() < 0.7 else None) for lab in labels} |
| return {"type": "choice", "instructions": _pick(rng, instructions), "criteria": crit} |
|
|
|
|
| def _subsample_labels(rng, all_labels, gold, max_opts): |
| others = [l for l in all_labels if l != gold] |
| rng.shuffle(others) |
| labs = others[: max_opts - 1] + [gold] |
| rng.shuffle(labs) |
| return labs |
|
|
|
|
| |
| |
| |
| EMOTIONS = ["sadness", "joy", "love", "anger", "fear", "surprise"] |
| EMO_DESC = {"sadness": "the writer feels sad or down", "joy": "the writer feels happy", |
| "love": "the writer feels affection or love", "anger": "the writer feels angry or annoyed", |
| "fear": "the writer feels afraid or anxious", "surprise": "the writer feels surprised"} |
| AG = ["World", "Sports", "Business", "Sci/Tech"] |
| AG_DESC = {"World": "international news and politics", "Sports": "sport results and athletes", |
| "Business": "companies, markets and the economy", "Sci/Tech": "science and technology"} |
| NLI = ["entailment", "neutral", "contradiction"] |
| NLI_DESC = {"entailment": "the hypothesis must be true given the premise", |
| "neutral": "the hypothesis might or might not be true", |
| "contradiction": "the hypothesis cannot be true given the premise"} |
| STARS = ["very negative (1 star)", "negative (2 stars)", "mixed (3 stars)", |
| "positive (4 stars)", "very positive (5 stars)"] |
|
|
|
|
| def _emotion(r, rng): |
| q = _choice_q(rng, ["Which emotion does the writer express?", "What is the main feeling in this message?", |
| "Classify the emotion of this text."], EMOTIONS, EMO_DESC) |
| return r["text"], q, _onehot(6, int(r["label"])) |
|
|
|
|
| def _ag(r, rng): |
| q = _choice_q(rng, ["Which news section does this article belong to?", "What is the topic of this article?", |
| "Route this story to the right desk."], AG, AG_DESC) |
| return r["text"], q, _onehot(4, int(r["label"])) |
|
|
|
|
| def _intent(label_key, max_opts, instr): |
| def fn(r, rng): |
| labels = fn.labels |
| gold = str(r[label_key]) |
| if labels is None or gold not in labels: |
| return None |
| labs = _subsample_labels(rng, labels, gold, max_opts) |
| crit = {l: (l.replace("_", " ") if rng.random() < 0.5 else None) for l in labs} |
| return r["text"], {"type": "choice", "instructions": _pick(rng, instr), "criteria": crit}, \ |
| _onehot(len(labs), labs.index(gold)) |
| fn.labels = None |
| return fn |
|
|
|
|
| def _boolq(r, rng): |
| q = {"type": "noul", "instructions": r["question"].strip().rstrip("?") + "?"} |
| if rng.random() < 0.5: |
| q["criteria"] = {"false": "the passage says no", "true": "the passage says yes"} |
| y = 1 if r["answer"] else 0 |
| return {"passage": r["passage"]} if rng.random() < 0.5 else r["passage"], q, [1.0 - y, float(y)] |
|
|
|
|
| def _sentiment(text_key): |
| def fn(r, rng): |
| y = int(r["label"]) |
| if rng.random() < 0.5: |
| q = {"type": "noul", "instructions": _pick(rng, ["Is the sentiment of this text positive?", |
| "Is the reviewer happy with it?", |
| "The writer is positive about the subject."])} |
| return r[text_key], q, [1.0 - y, float(y)] |
| q = _choice_q(rng, ["What is the sentiment of this text?", "How does the writer feel about it?"], |
| ["negative", "positive"], {"negative": "unfavourable, critical", "positive": "favourable, pleased"}) |
| return r[text_key], q, _onehot(2, y) |
| return fn |
|
|
|
|
| def _yelp(r, rng): |
| q = {"type": "score", "instructions": _pick(rng, ["How positive is this review?", "Rate the customer's satisfaction.", |
| "How many stars does this review deserve?"]), |
| "criteria": STARS} |
| return r["text"], q, _onehot(5, int(r["label"])) |
|
|
|
|
| def _hate(r, rng): |
| y = int(r["label"]) |
| q = {"type": "noul", "instructions": _pick(rng, ["Does this text contain hate speech?", |
| "This message attacks people for who they are."]), |
| "criteria": {"false": "no hateful content", "true": "hateful towards a group"}} |
| return r["text"], q, [1.0 - y, float(y)] |
|
|
|
|
| def _nli(r, rng): |
| if int(r["label"]) < 0: |
| return None |
| q = _choice_q(rng, ["How does the hypothesis relate to the premise?", |
| "Given the premise, is the hypothesis entailed, neutral or contradicted?"], NLI, NLI_DESC) |
| return {"premise": r["premise"], "hypothesis": r["hypothesis"]}, q, _onehot(3, int(r["label"])) |
|
|
|
|
| SOURCES = { |
| |
| "emotion": ("dair-ai/emotion", "split", "validation", _emotion, None), |
| "ag_news": ("fancyzhx/ag_news", None, "test", _ag, None), |
| "banking77": ("mteb/banking77", None, "test", |
| _intent("label_text", 20, ["Which banking intent does the customer have?", |
| "Route this customer message to the right intent."]), "label_text"), |
| "massive": ("mteb/amazon_massive_intent", "en", "validation", |
| _intent("label", 20, ["What does the user want the assistant to do?", |
| "Which intent is this voice command?"]), "label"), |
| "boolq": ("google/boolq", None, "validation", _boolq, None), |
| "sst2": ("stanfordnlp/sst2", None, "validation", _sentiment("sentence"), None), |
| "imdb": ("stanfordnlp/imdb", None, "test", _sentiment("text"), None), |
| "yelp": ("Yelp/yelp_review_full", None, "test", _yelp, None), |
| "hate": ("cardiffnlp/tweet_eval", "hate", "validation", _hate, None), |
| "mnli": ("nyu-mll/multi_nli", None, "validation_matched", _nli, None), |
| } |
|
|
|
|
| def _eval_label_set(path, name, split, key): |
| from datasets import load_dataset |
| return sorted({str(x) for x in load_dataset(path, name, split=split)[key]}) |
|
|
|
|
| def build(per_source: int = 200, seed: int = 0, sources=None): |
| """Up to `per_source` examples from each source's evaluation split.""" |
| from datasets import load_dataset |
| rng = random.Random(seed + 7) |
| out = [] |
| for name in (sources or SOURCES): |
| path, cfg, split, fn, label_key = SOURCES[name] |
| if label_key and fn.labels is None: |
| fn.labels = _eval_label_set(path, cfg, split, label_key) |
| got = 0 |
| for r in load_dataset(path, cfg, split=split, streaming=True).take(per_source * 2): |
| ex = fn(r, rng) |
| if ex is None: |
| continue |
| state, q, target = ex |
| out.append({"state": state, "question": q, "target": target, "source": name}) |
| got += 1 |
| if got >= per_source: |
| break |
| print(f" [eval data] {name:10s}: {got}") |
| rng.shuffle(out) |
| return out |
|
|
|
|
| def load_jsonl(paths): |
| out = [] |
| for p in paths: |
| with open(p, encoding="utf-8") as fh: |
| out.extend(json.loads(line) for line in fh if line.strip()) |
| return out |
|
|