Byrne-Jev-79M / eval_data.py
Quazim0t0's picture
Initial public release of Byrne-Jev-79M
d492e75 verified
Raw
History Blame Contribute Delete
8.11 kB
"""Evaluation data: public benchmark eval/test splits rewritten as typed decision questions.
Each example is {"state", "question" (Laya/Jev-style typed question), "target" (distribution over the
question's options in render order), "source"}. Only evaluation splits are loaded. Instruction wording
and option descriptions are sampled per example with a fixed seed, so a run is reproducible.
Label-set sources (banking77, MASSIVE) show up to 20 options per question: the gold label plus
randomly sampled other labels.
"""
import json
import random
def _onehot(k, i):
t = [0.0] * k
t[i] = 1.0
return t
def _pick(rng, xs):
return xs[rng.randrange(len(xs))]
def _choice_q(rng, instructions, labels, descs=None):
crit = {lab: (descs[lab] if descs and rng.random() < 0.7 else None) for lab in labels}
return {"type": "choice", "instructions": _pick(rng, instructions), "criteria": crit}
def _subsample_labels(rng, all_labels, gold, max_opts):
others = [l for l in all_labels if l != gold]
rng.shuffle(others)
labs = others[: max_opts - 1] + [gold]
rng.shuffle(labs)
return labs
# --------------------------------------------------------------------------- #
# Source converters: fn(row, rng) -> (state, question, target)
# --------------------------------------------------------------------------- #
EMOTIONS = ["sadness", "joy", "love", "anger", "fear", "surprise"]
EMO_DESC = {"sadness": "the writer feels sad or down", "joy": "the writer feels happy",
"love": "the writer feels affection or love", "anger": "the writer feels angry or annoyed",
"fear": "the writer feels afraid or anxious", "surprise": "the writer feels surprised"}
AG = ["World", "Sports", "Business", "Sci/Tech"]
AG_DESC = {"World": "international news and politics", "Sports": "sport results and athletes",
"Business": "companies, markets and the economy", "Sci/Tech": "science and technology"}
NLI = ["entailment", "neutral", "contradiction"]
NLI_DESC = {"entailment": "the hypothesis must be true given the premise",
"neutral": "the hypothesis might or might not be true",
"contradiction": "the hypothesis cannot be true given the premise"}
STARS = ["very negative (1 star)", "negative (2 stars)", "mixed (3 stars)",
"positive (4 stars)", "very positive (5 stars)"]
def _emotion(r, rng):
q = _choice_q(rng, ["Which emotion does the writer express?", "What is the main feeling in this message?",
"Classify the emotion of this text."], EMOTIONS, EMO_DESC)
return r["text"], q, _onehot(6, int(r["label"]))
def _ag(r, rng):
q = _choice_q(rng, ["Which news section does this article belong to?", "What is the topic of this article?",
"Route this story to the right desk."], AG, AG_DESC)
return r["text"], q, _onehot(4, int(r["label"]))
def _intent(label_key, max_opts, instr):
def fn(r, rng):
labels = fn.labels # full label set, filled from the eval split by build()
gold = str(r[label_key])
if labels is None or gold not in labels:
return None
labs = _subsample_labels(rng, labels, gold, max_opts)
crit = {l: (l.replace("_", " ") if rng.random() < 0.5 else None) for l in labs}
return r["text"], {"type": "choice", "instructions": _pick(rng, instr), "criteria": crit}, \
_onehot(len(labs), labs.index(gold))
fn.labels = None
return fn
def _boolq(r, rng):
q = {"type": "noul", "instructions": r["question"].strip().rstrip("?") + "?"}
if rng.random() < 0.5:
q["criteria"] = {"false": "the passage says no", "true": "the passage says yes"}
y = 1 if r["answer"] else 0
return {"passage": r["passage"]} if rng.random() < 0.5 else r["passage"], q, [1.0 - y, float(y)]
def _sentiment(text_key):
def fn(r, rng):
y = int(r["label"])
if rng.random() < 0.5:
q = {"type": "noul", "instructions": _pick(rng, ["Is the sentiment of this text positive?",
"Is the reviewer happy with it?",
"The writer is positive about the subject."])}
return r[text_key], q, [1.0 - y, float(y)]
q = _choice_q(rng, ["What is the sentiment of this text?", "How does the writer feel about it?"],
["negative", "positive"], {"negative": "unfavourable, critical", "positive": "favourable, pleased"})
return r[text_key], q, _onehot(2, y)
return fn
def _yelp(r, rng):
q = {"type": "score", "instructions": _pick(rng, ["How positive is this review?", "Rate the customer's satisfaction.",
"How many stars does this review deserve?"]),
"criteria": STARS}
return r["text"], q, _onehot(5, int(r["label"]))
def _hate(r, rng):
y = int(r["label"])
q = {"type": "noul", "instructions": _pick(rng, ["Does this text contain hate speech?",
"This message attacks people for who they are."]),
"criteria": {"false": "no hateful content", "true": "hateful towards a group"}}
return r["text"], q, [1.0 - y, float(y)]
def _nli(r, rng):
if int(r["label"]) < 0:
return None
q = _choice_q(rng, ["How does the hypothesis relate to the premise?",
"Given the premise, is the hypothesis entailed, neutral or contradicted?"], NLI, NLI_DESC)
return {"premise": r["premise"], "hypothesis": r["hypothesis"]}, q, _onehot(3, int(r["label"]))
SOURCES = {
# name: (path, config, eval split, converter, label_key for label-set sources)
"emotion": ("dair-ai/emotion", "split", "validation", _emotion, None),
"ag_news": ("fancyzhx/ag_news", None, "test", _ag, None),
"banking77": ("mteb/banking77", None, "test",
_intent("label_text", 20, ["Which banking intent does the customer have?",
"Route this customer message to the right intent."]), "label_text"),
"massive": ("mteb/amazon_massive_intent", "en", "validation",
_intent("label", 20, ["What does the user want the assistant to do?",
"Which intent is this voice command?"]), "label"),
"boolq": ("google/boolq", None, "validation", _boolq, None),
"sst2": ("stanfordnlp/sst2", None, "validation", _sentiment("sentence"), None),
"imdb": ("stanfordnlp/imdb", None, "test", _sentiment("text"), None),
"yelp": ("Yelp/yelp_review_full", None, "test", _yelp, None),
"hate": ("cardiffnlp/tweet_eval", "hate", "validation", _hate, None),
"mnli": ("nyu-mll/multi_nli", None, "validation_matched", _nli, None),
}
def _eval_label_set(path, name, split, key):
from datasets import load_dataset
return sorted({str(x) for x in load_dataset(path, name, split=split)[key]})
def build(per_source: int = 200, seed: int = 0, sources=None):
"""Up to `per_source` examples from each source's evaluation split."""
from datasets import load_dataset
rng = random.Random(seed + 7)
out = []
for name in (sources or SOURCES):
path, cfg, split, fn, label_key = SOURCES[name]
if label_key and fn.labels is None:
fn.labels = _eval_label_set(path, cfg, split, label_key)
got = 0
for r in load_dataset(path, cfg, split=split, streaming=True).take(per_source * 2):
ex = fn(r, rng)
if ex is None:
continue
state, q, target = ex
out.append({"state": state, "question": q, "target": target, "source": name})
got += 1
if got >= per_source:
break
print(f" [eval data] {name:10s}: {got}")
rng.shuffle(out)
return out
def load_jsonl(paths):
out = []
for p in paths:
with open(p, encoding="utf-8") as fh:
out.extend(json.loads(line) for line in fh if line.strip())
return out