Spaces:
Sleeping
Sleeping
File size: 6,294 Bytes
1c939fa | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 | """Portugality Index (IPT) - a deterministic pt-PT nativeness metric.
The score counts Brazilian-Portuguese markers in a text and normalises them by
length. It uses no LLM, so it is fully reproducible and auditable, and it never
uses AMALIA (or any model under test) as a judge - which would be circular.
IPT = 100 * exp(-6 * weighted_marker_density)
0 markers -> 100 (fully European); heavy pt-BR -> low score.
The word lists below are pt-BR *data*, not code, so they stay in Portuguese.
"""
import math
import re
# --- Brazilian lexicon (single words) -------------------------------------
# The PT equivalent is kept as a comment only to document the pair; the score
# only cares about the presence of the BR form.
BR_LEXICON = {
"ônibus", "celular", "geladeira", "banheiro", "trem", "bonde", "sorvete",
"suco", "xícara", "açougue", "sacola", "terno", "aeromoça", "pedestre",
"usuário", "usuários", "arquivo", "arquivos", "senha", "tela", "mouse",
"time", "esporte", "planejamento", "registro", "bala", "bacana",
"presunto",
}
# Multi-word Brazilian expressions.
BR_LEXICON_MULTIWORD = {
"café da manhã", "ponto de ônibus", "faixa de pedestres",
"carteira de motorista",
}
# --- Brazilian spelling (pre-AO90 accents / consonant drops) ---------------
# NOTE: some forms such as "ótimo" are also valid under AO90 in PT; keep the
# list conservative so the score is defensible.
BR_SPELLING = {
"econômico", "gênero", "tênis", "quilômetro", "recepção", "concepção",
"aspecto", "úmido", "antônio", "fenômeno", "gênio", "efêmero",
# pre-AO90 accented forms, dropped in the PT norm:
"idéia", "assembléia", "platéia", "heróico", "vôo", "enjôo",
}
# --- Grammar patterns ------------------------------------------------------
GERUND_RE = re.compile(
r"\b(estou|está|estás|estamos|estão|estava|estavam|estavas|"
r"vou|vai|vais|vamos|vão|fico|fica|ficam|continua\w*|segue|seguem)"
r"\s+\w+ndo\b",
re.IGNORECASE,
)
VOCE_RE = re.compile(r"\bvocês?\b", re.IGNORECASE)
A_GENTE_RE = re.compile(r"\ba\s+gente\b", re.IGNORECASE)
# Sentence-initial proclisis ("Me chamo", "Se chama"...), non-native in pt-PT.
PROCLISIS_START_RE = re.compile(
r"(^|[.!?]\s+)(me|te|se|nos|lhe|lhes)\s+\w+", re.IGNORECASE
)
WORD_RE = re.compile(r"\b[\wàáâãéêíóôõúç]+\b")
WEIGHTS = {
"lexicon": 1.0,
"spelling": 1.0,
"gerund": 0.7,
"address": 0.5,
"proclisis": 0.5,
}
def _collect(text):
"""Return (word_list, hits) where hits maps a category to matched strings."""
lowered = text.lower()
words = WORD_RE.findall(lowered)
hits = {key: [] for key in WEIGHTS}
for word in words:
if word in BR_LEXICON:
hits["lexicon"].append(word)
if word in BR_SPELLING:
hits["spelling"].append(word)
for expr in BR_LEXICON_MULTIWORD:
if expr in lowered:
hits["lexicon"].append(expr)
hits["gerund"] = [m.group(0) for m in GERUND_RE.finditer(lowered)]
hits["address"] = VOCE_RE.findall(lowered) + [
m.group(0) for m in A_GENTE_RE.finditer(lowered)
]
hits["proclisis"] = [m.group(0).strip() for m in PROCLISIS_START_RE.finditer(text)]
return words, hits
def portugality(text):
"""Return (score in 0-100, breakdown dict). Deterministic, no GPU."""
words, hits = _collect(text)
n_words = max(len(words), 1)
weighted = sum(WEIGHTS[key] * len(matches) for key, matches in hits.items())
density = weighted / n_words
score = round(100 * math.exp(-6 * density), 1)
breakdown = {
"n_words": n_words,
"weighted_markers": round(weighted, 2),
"density": round(density, 4),
**{key: hits[key] for key in hits},
}
return score, breakdown
def looks_like_correction(original, corrected):
"""Heuristic sanity check: does ``corrected`` look like a rewrite of
``original``? Guards the benchmark against degenerate outputs (tag soup,
endless repetition, empty answers), which would otherwise get a *high*
IPT simply because garbage contains no Brazilian markers."""
if not corrected or not corrected.strip():
return False
n_orig = max(len(WORD_RE.findall(original.lower())), 1)
n_corr = len(WORD_RE.findall(corrected.lower()))
if not 0.5 * n_orig <= n_corr <= 3 * n_orig:
return False
letters = sum(1 for ch in corrected if ch.isalpha() or ch.isspace())
return letters / len(corrected) >= 0.5
def compare_table(results, reference=None):
"""Turn ``{model_name: corrected_text}`` into ranked rows for a dataframe.
Each row: model, IPT, weighted_markers, delta_vs_ref, corrected, breakdown.
Rows are sorted by descending IPT.
"""
rows = []
for name, text in results.items():
score, breakdown = portugality(text)
rows.append(
{
"model": name,
"IPT": score,
"weighted_markers": breakdown["weighted_markers"],
"corrected": text,
"breakdown": breakdown,
}
)
rows.sort(key=lambda row: -row["IPT"])
base = next((row["IPT"] for row in rows if row["model"] == reference), None)
for row in rows:
if base is not None and row["model"] != reference:
row["delta_vs_ref"] = round(row["IPT"] - base, 1)
else:
row["delta_vs_ref"] = 0.0
return rows
def compare(results, reference="EuroLLM-9B"):
"""Human-readable ranking string (handy for CLI / logs)."""
rows = compare_table(results, reference)
lines = []
for row in rows:
delta = ""
if row["model"] != reference and row["delta_vs_ref"]:
delta = f" ({row['delta_vs_ref']:+.1f} vs {reference})"
lines.append(
f"{row['model']:28s} IPT={row['IPT']:5.1f} "
f"markers={row['weighted_markers']:4.1f}{delta}"
)
return "\n".join(lines)
if __name__ == "__main__":
demo = {
"AMALIA-9B": "Vou apanhar o autocarro e tomar o pequeno-almoço.",
"EuroLLM-9B": "Vou pegar o autocarro e tomar o pequeno-almoço.",
"Llama-3.1": "Vou pegar o ônibus e tomar café da manhã.",
}
print(compare(demo, reference="EuroLLM-9B"))
|