lizzy-606 commited on
Commit
0caab31
·
unverified ·
1 Parent(s): d4d6921

Fix: MFL averaged per lexeme, not per paradigm cell

Browse files

Previous implementation averaged token count across all paradigm cells,
giving disproportionate weight to lexemes with more cells (e.g. through
syncretism). Now groups by lexeme first, then averages the per-lexeme
means, matching the "aggregated per lexeme" definition in the docstring
and README.

Numeric impact is small (±0.02-0.03 across all four tokenizers), no
conclusions in README change.

Files changed (1) hide show
  1. metrics.py +11 -3
metrics.py CHANGED
@@ -153,15 +153,23 @@ def main():
153
  w.writeheader()
154
  w.writerows(rows)
155
 
156
- # profile
157
  prof = []
158
  for name in toks:
159
- ns = [r[f"{name}__n"] for r in rows]
160
  si = [r[f"{name}__SI"] for r in rows if r[f"{name}__SI"] != ""]
161
  iss = [r[f"{name}__ISS"] for r in rows if r[f"{name}__ISS"] != ""]
 
 
 
 
 
 
 
 
 
162
  prof.append({
163
  "tokenizer": name,
164
- "MFL": round(sum(ns) / len(ns), 2),
165
  "SI_pct": round(100 * si.count("1") / len(si), 1),
166
  "ISS_pct": round(100 * iss.count("1") / len(iss), 1),
167
  "n_forms_SI": len(si),
 
153
  w.writeheader()
154
  w.writerows(rows)
155
 
156
+ # profile
157
  prof = []
158
  for name in toks:
 
159
  si = [r[f"{name}__SI"] for r in rows if r[f"{name}__SI"] != ""]
160
  iss = [r[f"{name}__ISS"] for r in rows if r[f"{name}__ISS"] != ""]
161
+
162
+ # MFL is a mean of per-lexeme means, not a flat mean over all cells.
163
+ # A flat mean over-weights lexemes with more paradigm cells
164
+ # (e.g. through syncretism); grouping first removes that bias.
165
+ by_lexeme = defaultdict(list)
166
+ for r in rows:
167
+ by_lexeme[r["lexeme"]].append(r[f"{name}__n"])
168
+ lexeme_means = [sum(v) / len(v) for v in by_lexeme.values()]
169
+
170
  prof.append({
171
  "tokenizer": name,
172
+ "MFL": round(sum(lexeme_means) / len(lexeme_means), 2),
173
  "SI_pct": round(100 * si.count("1") / len(si), 1),
174
  "ISS_pct": round(100 * iss.count("1") / len(iss), 1),
175
  "n_forms_SI": len(si),