Fix: MFL averaged per lexeme, not per paradigm cell
Browse filesPrevious implementation averaged token count across all paradigm cells,
giving disproportionate weight to lexemes with more cells (e.g. through
syncretism). Now groups by lexeme first, then averages the per-lexeme
means, matching the "aggregated per lexeme" definition in the docstring
and README.
Numeric impact is small (±0.02-0.03 across all four tokenizers), no
conclusions in README change.
- metrics.py +11 -3
metrics.py
CHANGED
|
@@ -153,15 +153,23 @@ def main():
|
|
| 153 |
w.writeheader()
|
| 154 |
w.writerows(rows)
|
| 155 |
|
| 156 |
-
|
| 157 |
prof = []
|
| 158 |
for name in toks:
|
| 159 |
-
ns = [r[f"{name}__n"] for r in rows]
|
| 160 |
si = [r[f"{name}__SI"] for r in rows if r[f"{name}__SI"] != ""]
|
| 161 |
iss = [r[f"{name}__ISS"] for r in rows if r[f"{name}__ISS"] != ""]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 162 |
prof.append({
|
| 163 |
"tokenizer": name,
|
| 164 |
-
"MFL": round(sum(
|
| 165 |
"SI_pct": round(100 * si.count("1") / len(si), 1),
|
| 166 |
"ISS_pct": round(100 * iss.count("1") / len(iss), 1),
|
| 167 |
"n_forms_SI": len(si),
|
|
|
|
| 153 |
w.writeheader()
|
| 154 |
w.writerows(rows)
|
| 155 |
|
| 156 |
+
# profile
|
| 157 |
prof = []
|
| 158 |
for name in toks:
|
|
|
|
| 159 |
si = [r[f"{name}__SI"] for r in rows if r[f"{name}__SI"] != ""]
|
| 160 |
iss = [r[f"{name}__ISS"] for r in rows if r[f"{name}__ISS"] != ""]
|
| 161 |
+
|
| 162 |
+
# MFL is a mean of per-lexeme means, not a flat mean over all cells.
|
| 163 |
+
# A flat mean over-weights lexemes with more paradigm cells
|
| 164 |
+
# (e.g. through syncretism); grouping first removes that bias.
|
| 165 |
+
by_lexeme = defaultdict(list)
|
| 166 |
+
for r in rows:
|
| 167 |
+
by_lexeme[r["lexeme"]].append(r[f"{name}__n"])
|
| 168 |
+
lexeme_means = [sum(v) / len(v) for v in by_lexeme.values()]
|
| 169 |
+
|
| 170 |
prof.append({
|
| 171 |
"tokenizer": name,
|
| 172 |
+
"MFL": round(sum(lexeme_means) / len(lexeme_means), 2),
|
| 173 |
"SI_pct": round(100 * si.count("1") / len(si), 1),
|
| 174 |
"ISS_pct": round(100 * iss.count("1") / len(iss), 1),
|
| 175 |
"n_forms_SI": len(si),
|