File size: 6,294 Bytes
1c939fa
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
"""Portugality Index (IPT) - a deterministic pt-PT nativeness metric.

The score counts Brazilian-Portuguese markers in a text and normalises them by
length. It uses no LLM, so it is fully reproducible and auditable, and it never
uses AMALIA (or any model under test) as a judge - which would be circular.

    IPT = 100 * exp(-6 * weighted_marker_density)

0 markers -> 100 (fully European); heavy pt-BR -> low score.

The word lists below are pt-BR *data*, not code, so they stay in Portuguese.
"""

import math
import re

# --- Brazilian lexicon (single words) -------------------------------------
# The PT equivalent is kept as a comment only to document the pair; the score
# only cares about the presence of the BR form.
BR_LEXICON = {
    "ônibus", "celular", "geladeira", "banheiro", "trem", "bonde", "sorvete",
    "suco", "xícara", "açougue", "sacola", "terno", "aeromoça", "pedestre",
    "usuário", "usuários", "arquivo", "arquivos", "senha", "tela", "mouse",
    "time", "esporte", "planejamento", "registro", "bala", "bacana",
    "presunto",
}

# Multi-word Brazilian expressions.
BR_LEXICON_MULTIWORD = {
    "café da manhã", "ponto de ônibus", "faixa de pedestres",
    "carteira de motorista",
}

# --- Brazilian spelling (pre-AO90 accents / consonant drops) ---------------
# NOTE: some forms such as "ótimo" are also valid under AO90 in PT; keep the
# list conservative so the score is defensible.
BR_SPELLING = {
    "econômico", "gênero", "tênis", "quilômetro", "recepção", "concepção",
    "aspecto", "úmido", "antônio", "fenômeno", "gênio", "efêmero",
    # pre-AO90 accented forms, dropped in the PT norm:
    "idéia", "assembléia", "platéia", "heróico", "vôo", "enjôo",
}

# --- Grammar patterns ------------------------------------------------------
GERUND_RE = re.compile(
    r"\b(estou|está|estás|estamos|estão|estava|estavam|estavas|"
    r"vou|vai|vais|vamos|vão|fico|fica|ficam|continua\w*|segue|seguem)"
    r"\s+\w+ndo\b",
    re.IGNORECASE,
)
VOCE_RE = re.compile(r"\bvocês?\b", re.IGNORECASE)
A_GENTE_RE = re.compile(r"\ba\s+gente\b", re.IGNORECASE)
# Sentence-initial proclisis ("Me chamo", "Se chama"...), non-native in pt-PT.
PROCLISIS_START_RE = re.compile(
    r"(^|[.!?]\s+)(me|te|se|nos|lhe|lhes)\s+\w+", re.IGNORECASE
)

WORD_RE = re.compile(r"\b[\wàáâãéêíóôõúç]+\b")

WEIGHTS = {
    "lexicon": 1.0,
    "spelling": 1.0,
    "gerund": 0.7,
    "address": 0.5,
    "proclisis": 0.5,
}


def _collect(text):
    """Return (word_list, hits) where hits maps a category to matched strings."""
    lowered = text.lower()
    words = WORD_RE.findall(lowered)
    hits = {key: [] for key in WEIGHTS}

    for word in words:
        if word in BR_LEXICON:
            hits["lexicon"].append(word)
        if word in BR_SPELLING:
            hits["spelling"].append(word)

    for expr in BR_LEXICON_MULTIWORD:
        if expr in lowered:
            hits["lexicon"].append(expr)

    hits["gerund"] = [m.group(0) for m in GERUND_RE.finditer(lowered)]
    hits["address"] = VOCE_RE.findall(lowered) + [
        m.group(0) for m in A_GENTE_RE.finditer(lowered)
    ]
    hits["proclisis"] = [m.group(0).strip() for m in PROCLISIS_START_RE.finditer(text)]
    return words, hits


def portugality(text):
    """Return (score in 0-100, breakdown dict). Deterministic, no GPU."""
    words, hits = _collect(text)
    n_words = max(len(words), 1)
    weighted = sum(WEIGHTS[key] * len(matches) for key, matches in hits.items())
    density = weighted / n_words
    score = round(100 * math.exp(-6 * density), 1)
    breakdown = {
        "n_words": n_words,
        "weighted_markers": round(weighted, 2),
        "density": round(density, 4),
        **{key: hits[key] for key in hits},
    }
    return score, breakdown


def looks_like_correction(original, corrected):
    """Heuristic sanity check: does ``corrected`` look like a rewrite of
    ``original``?  Guards the benchmark against degenerate outputs (tag soup,
    endless repetition, empty answers), which would otherwise get a *high*
    IPT simply because garbage contains no Brazilian markers."""
    if not corrected or not corrected.strip():
        return False
    n_orig = max(len(WORD_RE.findall(original.lower())), 1)
    n_corr = len(WORD_RE.findall(corrected.lower()))
    if not 0.5 * n_orig <= n_corr <= 3 * n_orig:
        return False
    letters = sum(1 for ch in corrected if ch.isalpha() or ch.isspace())
    return letters / len(corrected) >= 0.5


def compare_table(results, reference=None):
    """Turn ``{model_name: corrected_text}`` into ranked rows for a dataframe.

    Each row: model, IPT, weighted_markers, delta_vs_ref, corrected, breakdown.
    Rows are sorted by descending IPT.
    """
    rows = []
    for name, text in results.items():
        score, breakdown = portugality(text)
        rows.append(
            {
                "model": name,
                "IPT": score,
                "weighted_markers": breakdown["weighted_markers"],
                "corrected": text,
                "breakdown": breakdown,
            }
        )
    rows.sort(key=lambda row: -row["IPT"])

    base = next((row["IPT"] for row in rows if row["model"] == reference), None)
    for row in rows:
        if base is not None and row["model"] != reference:
            row["delta_vs_ref"] = round(row["IPT"] - base, 1)
        else:
            row["delta_vs_ref"] = 0.0
    return rows


def compare(results, reference="EuroLLM-9B"):
    """Human-readable ranking string (handy for CLI / logs)."""
    rows = compare_table(results, reference)
    lines = []
    for row in rows:
        delta = ""
        if row["model"] != reference and row["delta_vs_ref"]:
            delta = f"  ({row['delta_vs_ref']:+.1f} vs {reference})"
        lines.append(
            f"{row['model']:28s}  IPT={row['IPT']:5.1f}  "
            f"markers={row['weighted_markers']:4.1f}{delta}"
        )
    return "\n".join(lines)


if __name__ == "__main__":
    demo = {
        "AMALIA-9B": "Vou apanhar o autocarro e tomar o pequeno-almoço.",
        "EuroLLM-9B": "Vou pegar o autocarro e tomar o pequeno-almoço.",
        "Llama-3.1": "Vou pegar o ônibus e tomar café da manhã.",
    }
    print(compare(demo, reference="EuroLLM-9B"))