File size: 79,624 Bytes
20e20c3
 
 
38c2e17
 
 
20e20c3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
38c2e17
 
 
 
20e20c3
 
 
 
 
 
 
38c2e17
20e20c3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
38c2e17
 
20e20c3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
38c2e17
 
 
 
 
 
20e20c3
 
 
 
 
 
 
 
 
38c2e17
 
 
 
20e20c3
 
38c2e17
 
20e20c3
 
 
 
 
 
 
 
 
 
 
38c2e17
 
 
20e20c3
 
 
 
 
 
 
 
 
 
 
 
 
38c2e17
 
 
 
20e20c3
c904ce8
20e20c3
 
 
 
 
 
 
 
38c2e17
 
 
20e20c3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
38c2e17
 
 
 
20e20c3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
38c2e17
20e20c3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
38c2e17
 
20e20c3
 
 
 
 
 
 
 
38c2e17
 
20e20c3
38c2e17
 
20e20c3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
38c2e17
 
 
20e20c3
 
 
 
 
 
 
38c2e17
 
 
20e20c3
 
38c2e17
 
 
 
20e20c3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
c904ce8
20e20c3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
38c2e17
 
 
20e20c3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
38c2e17
 
 
20e20c3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
1011
1012
1013
1014
1015
1016
1017
1018
1019
1020
1021
1022
1023
1024
1025
1026
1027
1028
1029
1030
1031
1032
1033
1034
1035
1036
1037
1038
1039
1040
1041
1042
1043
1044
1045
1046
1047
1048
1049
1050
1051
1052
1053
1054
1055
1056
1057
1058
1059
1060
1061
1062
1063
1064
1065
1066
1067
1068
1069
1070
1071
1072
1073
# Source-1 evaluation

Source-1 was compared with its teacher and 16 public quality scorers on three test sets. An independent proprietary
LLM grader scored every chunk with Source-1's 13-field rubric: its grades were never trained on, and on the held-out
set it was given the same instructions as Source-1's teacher. Each number says how closely a model's ranking agrees
with the grader's, so it measures agreement with this grader applying Source-1's own rubric, on home ground.

## Summary

Rank agreement (Spearman) with the grader's overall score. Each model is read on the chunks it scored.

| test set | chunks | Source-1 (307M) | propella-1 4B (4.0B) | FineWeb-Edu classifier (English only) | teacher (open-weight, 27B) |
|---|---|---|---|---|---|
| held-out set, 53 languages (main result) | 495 | 0.900 | 0.756 | 0.529 | 0.912 |
| English exam | 413 | 0.921 | 0.820 | 0.453 | 0.912 |
| 12-language exam | 352 | 0.895 | 0.637 | - | 0.880 |

propella-1 4B scored 493, 412 and 350 of these chunks. The FineWeb-Edu classifier is read on the 159 English held-out
chunks, where Source-1 scores 0.864. The English exam has 414 chunks; the teacher scored 413, and comparisons use those.

- Source-1 agrees with the grader more closely than each of the 16 public scorers, on every set where they were
  compared. All 39 of these comparisons ([how they are counted](#every-public-scorer)) have 95% intervals clear of
  zero.
- It is level with its teacher within noise, with about 1/88 of the teacher's parameters. On the held-out set it is
  0.012 lower (95% interval of the difference -0.03 to +0.004).
- The closest public scorer, propella-1 4B, trails by 0.144 on the held-out set. A more generous reading of it,
  chosen after the results were known, narrows the gap to 0.064-0.081 ([details](#how-propella-1-is-read)).
- At its shipped drop line (the default `keep` flag), Source-1 catches 43 of the 64 chunks the grader drops on the
  held-out set. The teacher, at its own keep flags, catches 46.

The held-out set is the main result. The exams count less, because they helped choose the teacher
([why](#independence-from-development)).

## How we measured

**The grader.** Its overall score and keep flag are computed from its 13 fields with the rubric's own formula, as for
Source-1. Its drops are the chunks that hit the rubric's hard filters (spam, boilerplate or toxic text). Its labels
were never trained on.

**Home ground.** The held-out documents come from the same kinds of sources as the training data. The public scorers
were built for their own definitions of quality, most for educational value, and are not wrong when they disagree with
this rubric. Even a scorer that matched the grader's own `educational_value` scores exactly would reach only 0.874 on
the held-out set, 0.900 on the English exam and 0.817 on the 12-language exam, so a small part of the gap to the
educational-value classifiers comes from what this measure asks for;
[Educational value alone](#educational-value-alone) is the fairer comparison for them.

**The three test sets.**

| test set | chunks | languages | grader drops | what it is |
|---|---|---|---|---|
| held-out set (main result) | 495, one per document | 53 | 64 | documents from Source-1's held-out test split, never trained or calibrated on |
| English exam | 414, from 332 documents (413 scored by the teacher) | English | 25 | 200 chunks drawn at random, plus 214 harder cases added on purpose |
| 12-language exam | 352, one per document | 12 | 46 | web text drawn from FineWeb-2's test split (blocks of 10 consecutive rows at random positions; Spanish from its first rows), about 30 chunks per language |

**95% intervals.** Each difference between two models comes with a 95% interval from a paired bootstrap over
documents. When the interval excludes zero, chance alone is an unlikely explanation for the difference.

**How the public scorers were run.** Each public scorer ran from the pinned revision in
[Appendix C](#appendix-c-public-scorer-repositories), following its model card and its own code or prompt where it
publishes one, except as noted here. propella-1 and EAI-Distill were decoded greedily, although the propella-1 4B and
EAI-Distill repositories default to sampling. propella-1 ran in plain transformers, without the SGLang server and JSON
grammar its card recommends; the serving engine mainly affects speed, so we claim no speed comparison with it.

<details>
<summary>How the public scorers were run, in full</summary>

### How the public scorers were run

- Each public scorer ran from the repository and revision listed in
  [Appendix C](#appendix-c-public-scorer-repositories), following its model card and its own code or prompt where it
  publishes one.
- The encoder classifiers ran in Hugging Face transformers as their model cards show, in the precision each card
  states (bfloat16 where it says so, float32 otherwise).
- Source-1 itself ran in bfloat16, its default. In float32 its rank agreement on the held-out set is unchanged to
  three decimal places.
- The two fastText models ran in the fasttext library. They read the whole text with newlines turned into spaces, as
  the DCLM code does.
- EAI-Distill ran in transformers in float32 with greedy decoding (its repository's default settings sample).
- propella-1 ran in plain transformers in bfloat16, with its repository's own prompt and chat template, greedy
  decoding and no JSON grammar. Its card recommends serving with SGLang and a JSON grammar, and the 4B's default
  settings sample at temperature 0.7. The serving engine mainly affects speed. Greedy decoding without a grammar gives
  the same tokens as grammar-constrained greedy decoding, up to the first token the grammar would forbid, and it
  avoids sampling noise. An answer that did not parse got one retry with more new tokens
  ([details](#propella-1s-answers)). We claim no speed comparison with propella-1.

</details>

### Independence from development

- The exam sets helped choose the teacher and its prompt setup, by agreement with the exam grader's labels, and
  Source-1's backbone was kept over two other multilingual encoders in a comparison that looked at these sets.
- Grades and reviews by proprietary LLMs, among them the grader's model family, informed the rubric, the drop-line
  candidates and floor, and the data filters.
- AI assistants helped draft the rubric text and write the project's code. The grader's model family includes one of
  the assistants that helped draft the rubric text.
- The held-out set was built after the teacher and its prompt were fixed. It was not used to choose the model. It was
  not hidden, though: the candidate drop lines were drafted after its grades had been seen.
- No grade or review by a proprietary LLM was ever used as a label, a training target or a training example.

<details>
<summary>Independence from development, in full</summary>

What grades from the grader's model family did and did not influence:

- **The exam sets helped choose the teacher.** The English and 12-language exams are the samples on which the teacher
  was vetted and its prompt setup was chosen, by agreement with the exam grader's labels. The prompt setup covers the
  answer format, reasoning on or off, and a rubric rule for pages stitched together from unrelated or scrambled text,
  added after reviewing disagreements on these samples. The backbone was also kept over two other multilingual
  encoders in a comparison that looked at these sets. So the exams are not independent of the grader: Source-1's
  teacher was in part selected to agree with it there. They are reported, but they count less than the held-out set.
- **The held-out set did not.** It was built after the teacher and its prompt had been fixed, and was not used to
  choose either. It is called held-out because Source-1 never trained or calibrated on it. But it was not hidden
  during development: Source-1's results on it were checked, and the candidate drop lines were drafted after its
  grades had been seen. The released model was not chosen by its score on this set: it is the model trained on the
  final, fully cleaned data.
- **The rubric and the drop-line rule.** The choice between rubric revisions, the list of candidate drop lines and the
  0.95 keep-agreement floor of the drop-line rule were decided using grades from proprietary LLM graders, among them
  models of the grader's family, on samples of documents, some of them training chunks. The drop line itself was then
  picked by that fixed rule on the teacher's validation labels. The grader's labels were not used to pick it.
- **Training.** Every training label comes from the open-weight teacher. The grader's labels, and every other grade or
  review by a proprietary LLM, were never used as labels, training targets or training examples. The learning rate
  and the natural language mix came from a short sweep (two learning rates, two language mixes, half an epoch each)
  on a smaller training set labeled by the same teacher, read on validation agreement with the teacher. The epoch
  count came from the same validation curves. The checkpoint is the final step, which had the best validation
  agreement with the teacher (Spearman 0.955). No graded test set was used for these choices.
- **Data filtering.** Samples reviewed by a proprietary LLM measured how often the license and table-of-contents
  filters missed or over-fired. This informed which collections were filtered out; the reviewed documents were then
  kept out of training. AI assistants also reviewed source terms and document notices across the training data,
  which informed the license rules. They also helped draft the rubric text and write the project's code.

</details>

<details>
<summary>The test sets in detail: composition, near-duplicate check, safety filter</summary>

### The test sets in detail

- **Held-out set** (the main result): 495 chunks in 53 languages, one per document. All come from Source-1's held-out
  test split, drawn from the four data stages (web 129, multilingual web 276, conversations/code/synthetic 48, open
  books 42). The grader drops 64 of them. 159 chunks are English and 29 Chinese; most other languages have 1 to 12
  chunks. Chinese was oversampled on purpose: 15 of its 29 chunks were added to the proportional sample, and they hold
  7 of the 64 grader drops. 26 of the 53 languages have 5 or fewer chunks. Source-1 never trained or calibrated on
  these documents. They come from the same kinds of sources as the training data. 42 of the 495 chunks (8.5%) come
  from sources that were later removed from training under the license rules (19 from DCLM-baseline, 16 raw Common
  Crawl pages, and 7 from collections with unreliable or gated license terms). The test split also keeps documents
  that the license filtering removed from training.
- **English exam**: 414 chunks from 332 whole documents split into chunks. 200 chunks from 196 documents were drawn at
  random from web, wiki, Common Pile, math and code sources (the **random-sample** chunks and documents). 214 chunks
  from 136 documents were added on purpose to cover harder cases (long documents 57 chunks, academic 50, math 28,
  code 27, spam 23, toxic 20, fiction 9). The grader drops 25 of the chunks (24 documents). The added chunks make the
  set unlike a random draw. They move each model's numbers, up for some models and down for others (Source-1: rank
  0.919 on the random-sample chunks alone and 0.921 on all 414; AUC 0.987 and 0.979). The teacher has no score for
  one of the 414 chunks (a random-sample chunk). So the comparisons with the teacher and the public scorers' English
  rows use the 413 chunks it scored (411 or 412 where a public scorer also lacks one). The 512-token comparison uses
  all 414.
- **12-language exam**: 352 chunks, one per document, all FineWeb-2 web text in 12 languages (ar, bn, de, es, hi,
  ja, ko, ru, sw, th, vi, zh; about 30 each), drawn from FineWeb-2's test split (blocks of 10 consecutive rows at
  random positions; Spanish from its first rows). It has no code or math: 351 of the 352 chunks are plain text by the
  grader's label. The grader drops 46.

Every exam document was checked against the training data (exact text, URL, title, long-line and shingle matching).
None has half or more of its text in a training document; the largest share of an exam text found in a training
document is 41%.

#### Near-duplicate check

The 495 held-out chunks were compared with every training and validation record. The check measures the share of a
chunk's distinctive 40-character pieces that a single record contains. No chunk is covered 80% or more by any record.
4 are covered 64% to 71%: three short files from code collections and one US government record, 98 to 345 tokens
each, all short templated texts. 9 in all are covered 20% or more. These chunks stay in the reported set. Without the
4, Source-1's rank agreement is 0.901 (491 chunks); without all 9, 0.900.

#### Safety filter

A safety filter, with a rule fixed before it was first run, removed a small number of documents from every evaluation
set and from Source-1's training, validation and test data. Documents that substantially copy removed text were
removed from its data as well. Every model is compared on the same filtered chunks.

</details>

<details>
<summary>The grader in detail, and how much its grades move when repeated</summary>

### The grader in detail

Strictly, these are two graders from one proprietary model family: one graded the exams, another the held-out set.
This file calls either "the grader". That model family also includes one of the assistants that helped draft the
rubric text. The exams were graded with an older revision of the rubric, from before the rubric settled how to score
ads.

The held-out grader was given the teacher's own instructions word for word: the 13-field rubric and the special rules
in [Appendix A](#appendix-a-rubric-anchors), for example that ads are scored `spam_seo` 3 and kept. The public scorers
follow none of these rules.

The grader returns the 13 fields only. Its reference overall score and keep flag are computed from its scores in the
same way as Source-1's: the overall formula of the rubric (see [README.md](README.md#what-you-get)), and keep =
false when the rubric's hard filters match (`toxicity >= 4`, `spam_seo >= 4` or `boilerplate >= 4.5`). "The grader's
drops" in this file are those chunks: spam, boilerplate or toxic text under the hard filters.

The same sets were scored by the teacher and by the 16 public quality scorers, open-weight models run as described
under [How the public scorers were run](#how-the-public-scorers-were-run).

#### Grader self-agreement

The exam grader also graded part of each exam a second time. The two gradings agree at rank 0.902 on 97 English
random-sample chunks and 0.945 on 58 chunks of the 12-language exam. On exactly these chunks Source-1 reaches 0.886
and 0.892 and the teacher 0.879 and 0.882, below the grader's agreement with itself (within noise for English).

</details>

### The one-line header

<details>
<summary>The one-line header that Source-1, the teacher and the grader saw</summary>


Source-1, the teacher and the grader saw each chunk after a one-line header. It gives the chunk's source type
("dataset record" for 478 of the 495 held-out chunks, and a code-file type such as "Python source file" for the other
17), its part number when it is one part of a longer document and, for 121 of the 495 held-out chunks, a title. The
public scorers read the text alone. On the 1,261 chunks of the held-out set and both exams, dropping the whole header
moved Source-1's overall score by 0.03 on average (at most 0.655). Dropping a title moved it by 0.05 on average on the
chunks that had one. Adding a URL, which no training input had, moved it by up to 0.7 and did not improve its rank
agreement with the grader. So `source1.py` does not show a URL to the model by default.

</details>

<details>
<summary>Metric definitions: rank, AUC, matched keep rate, intervals</summary>

### Metric definitions

Computed per chunk:

- **Rank**: Spearman correlation between a model's overall score and the grader's overall score.
- **AUC**: how well a model's score separates the grader's drops from its keeps (area under the ROC curve; a low score
  means drop; ties count one half).
- **Matched keep rate**: every model keeps the same share of chunks the grader keeps (87% on the held-out set), taking
  its highest-scored chunks. **Keep agreement** is the share of chunks where the model's keep/drop matches the
  grader's. **Drop recall** is the share of the grader's drops the model also drops. This puts scorers with different
  scales on the same operating point. Chunks tied at the cut are kept fractionally (the expected value over every
  order of the tied chunks), so a count of drops caught can be fractional; such counts are given as "about".
- For Source-1 and the teacher, AUC and the matched keep rate use the overall score before it is clipped to 0, so
  heavily penalized chunks are not tied at zero.
- Intervals at a drop line are Wilson 95% intervals.
- **Differences between two models** (Source-1 minus the other, on the chunks both scored) come with paired bootstrap
  95% intervals: documents resampled with replacement, 2,000 resamples, seed 0, the same resamples for both models,
  percentile intervals. Interval ends are given to two decimals (three when they are close to zero), because the
  third decimal moves with the random seed.

</details>

## Results

### Held-out set

The table shows Source-1, the teacher and the three multilingual public scorers that agree best with the grader. Keep
agreement and drop recall are read at a matched keep rate: each model keeps its top-scored 87% of chunks, the share
the grader keeps ([Metric definitions](#metric-definitions)).

| model | chunks | rank | AUC | keep agreement (87% kept) | drop recall (87% kept) |
|---|---|---|---|---|---|
| **Source-1** | 495 | 0.900 | 0.946 | 0.927 | 0.719 (46/64) |
| Teacher (open-weight 27B LLM) | 495 | 0.912 | 0.948 | 0.919 | 0.688 (44/64) |
| propella-1 4B | 493 | 0.756 | 0.862 | 0.886 | 0.562 (36/64) |
| propella-1 1.7B | 495 | 0.736 | 0.855 | 0.896 | 0.599 (about 38/64) |
| JQL-Edu (mean of 3 balanced heads) | 495 | 0.600 | 0.737 | 0.826 | 0.328 (21/64) |

- At this matched rate Source-1 catches 46 of the grader's 64 drops and the teacher 44, a difference within noise
  (without the 15 added Chinese chunks: about 39 and 39 of 57). At each model's own drop line (for the teacher, its
  own keep flags) they catch 43 and 46 ([The drop line](#the-drop-line)). Source-1 and the teacher are also level
  within noise on rank and AUC.
- The lead over propella-1 4B, the closest public scorer, is at least as large outside English: 0.916 against 0.765
  on the 335 non-English chunks.

<details>
<summary>Held-out set in detail: every interval, and results by data stage and for Chinese</summary>

### Held-out set in detail

- Every public scorer ranks the held-out set well below Source-1 on this rubric. The closest, propella-1 4B, reaches
  0.756 on the 493 chunks it scored (Source-1 minus propella-1 4B: +0.144, 95% interval +0.11 to +0.18). It is also
  further from the grader on the drops: AUC 0.862 against Source-1's 0.946 (+0.084, +0.05 to +0.12). At the matched
  rate it catches 36 of the 64 drops to Source-1's 46 (drop recall +0.156, +0.06 to +0.25). Most public scorers do
  not target spam, boilerplate or toxicity, which is what the grader drops. With propella-1's own ratings for them
  added, its AUC gap is +0.049 (+0.02 to +0.07; see [How propella-1 is read](#how-propella-1-is-read)).
- The lead over propella-1 4B is at least as large outside English: 0.916 against 0.765 on the 335 non-English chunks
  in 52 languages (+0.151, +0.11 to +0.20; 34 of them are in languages propella-1's card does not list). It is +0.117
  (+0.06 to +0.18) on the 158 English chunks it scored (of 159).
- Source-1 ranks 0.012 below its teacher (0.900 vs 0.912; 95% interval of the difference -0.03 to +0.004), also
  outside English (-0.012, -0.03 to +0.006). It is level with it on AUC (0.946 vs 0.948; -0.02 to +0.01). At the
  matched rate it catches 46 of the grader's 64 drops and the teacher 44, a difference within noise (drop recall
  +0.031, -0.04 to +0.10).

By data stage, and for Chinese (same chunks; drops caught at each model's own drop line):

| model | web (129) | multilingual web (276) | conversations, code, synthetic (48) | open books (42) | Chinese (29) |
|---|---|---|---|---|---|
| **Source-1**, rank | 0.891 | 0.905 | 0.728 | 0.677 | 0.934 |
| Teacher, rank | 0.902 | 0.917 | 0.821 | 0.741 | 0.938 |
| **Source-1**, drops caught | 14 of 17 | 28 of 44 | 1 of 2 | 0 of 1 | 11 of 14 |
| Teacher, drops caught | 14 of 17 | 29 of 44 | 2 of 2 | 1 of 1 | 10 of 14 |

The held-out set has 159 English chunks and 1 to 12 chunks for most other languages (29 for Chinese), so
per-language results are noisy. Groups with fewer than 50 chunks are indicative only.

</details>

### Every public scorer

Source-1 ranks the chunks closer to the grader than each of the 16 public scorers, on every set each was run on. All
39 of these rank-agreement leads have 95% intervals clear of zero, also after a Bonferroni adjustment, while the
teacher's intervals all include zero. The smallest lead in size is +0.101 (+0.07 to +0.14), over propella-1 4B on the
English exam.

<details>
<summary>All 17 public-scorer rows with intervals, and how the 39 comparisons are counted</summary>

We compared 16 public scorers. The tables have 17 public rows, because the Nemotron-CC 3-way ensemble is computed
from three of the 16 (the two NeMo Curator classifiers and DCLM fastText). That gives 39 differences with intervals:
17 on the held-out set, 17 on the English exam and 5 on the 12-language exam, where only the five multilingual
scorers were run. Each cell is a rank agreement with the grader, and the lead is Source-1 minus that scorer on exactly
the chunks it scored. The held-out set has 495 chunks unless noted; the English exam 411 to 413; the 12-language exam
352 unless noted.

Multilingual scorers, and the teacher:

| model | held-out: rank | held-out: Source-1 lead (95% interval) | English exam: rank | English exam: lead | 12 languages: rank | 12 languages: lead |
|---|---|---|---|---|---|---|
| **Source-1** | 0.900 | - | 0.921 | - | 0.895 | - |
| Teacher (open-weight 27B LLM) | 0.912 | -0.012 (-0.03 to +0.004) | 0.912 | +0.009 (-0.01 to +0.02) | 0.880 | +0.015 (-0.007 to +0.04) |
| propella-1 4B | 0.756 (493 chunks) | +0.144 (+0.11 to +0.18) | 0.820 | +0.101 (+0.07 to +0.14) | 0.637 (350 chunks) | +0.258 (+0.19 to +0.33) |
| propella-1 1.7B | 0.736 | +0.163 (+0.13 to +0.20) | 0.812 | +0.108 (+0.07 to +0.14) | 0.631 | +0.264 (+0.20 to +0.33) |
| JQL-Edu | 0.600 | +0.299 (+0.25 to +0.36) | 0.541 | +0.380 (+0.30 to +0.46) | 0.379 | +0.516 (+0.42 to +0.61) |
| FinePDFs-Edu | 0.476 | +0.424 (+0.36 to +0.49) | 0.787 | +0.134 (+0.10 to +0.18) | 0.332 | +0.563 (+0.47 to +0.66) |
| FineWeb2-HQ (21 of the 53 languages) | 0.449 (342 chunks) | +0.456 (+0.38 to +0.54) | 0.448 | +0.473 (+0.39 to +0.56) | 0.496 (205 chunks) | +0.415 (+0.31 to +0.53) |

English-only scorers (held-out set: its 159 English chunks):

| model | held-out: rank | held-out: Source-1 lead (95% interval) | English exam: rank | English exam: lead |
|---|---|---|---|---|
| **Source-1** | 0.864 | - | 0.921 | - |
| FineWeb-Edu classifier | 0.529 | +0.336 (+0.22 to +0.45) | 0.453 | +0.468 (+0.38 to +0.56) |
| DCLM fastText (OH+ELI5) | 0.299 | +0.565 (+0.42 to +0.71) | 0.193 | +0.728 (+0.63 to +0.83) |
| NeMo Curator edu (Nemotron-4 labels) | 0.412 | +0.452 (+0.32 to +0.59) | 0.260 | +0.661 (+0.55 to +0.78) |
| NeMo Curator edu (Mixtral labels) | 0.498 | +0.366 (+0.24 to +0.50) | 0.422 | +0.499 (+0.41 to +0.59) |
| Meta-rater reasoning | 0.580 | +0.284 (+0.19 to +0.39) | 0.710 | +0.211 (+0.15 to +0.27) |
| Meta-rater readability | 0.574 | +0.291 (+0.19 to +0.40) | 0.619 | +0.302 (+0.24 to +0.37) |
| Meta-rater cleanliness | 0.597 | +0.267 (+0.18 to +0.37) | 0.698 | +0.223 (+0.17 to +0.28) |
| Meta-rater professionalism | 0.538 | +0.327 (+0.23 to +0.43) | 0.688 | +0.233 (+0.18 to +0.30) |
| EAI-Distill 0.5B | 0.592 | +0.273 (+0.18 to +0.37) | 0.685 | +0.236 (+0.19 to +0.29) |
| NVIDIA quality classifier (DeBERTa) | 0.326 | +0.539 (+0.39 to +0.70) | 0.181 | +0.740 (+0.59 to +0.90) |
| Dolma 3 fastText quality | 0.341 | +0.524 (+0.39 to +0.66) | 0.386 | +0.535 (+0.46 to +0.62) |
| Nemotron-CC 3-way ensemble (derived) | 0.353 | +0.511 (+0.37 to +0.67) | 0.295 | +0.626 (+0.53 to +0.72) |

- Every one of the 39 public-scorer intervals excludes zero, and the leads stay clear of zero after a Bonferroni
  adjustment for the 39 comparisons. The teacher's intervals all include zero.
- This holds for rank agreement. The other measures are much noisier on the 159 English held-out chunks (see
  [AUC and noise](#auc-and-noise)).

</details>

<details>
<summary>AUC of every scorer, Source-1 on each scorer's chunks, and the noise behind the tables</summary>

### AUC and noise

AUC against the grader's drops, each scorer on the chunks it scored:

| model | held-out: AUC | English exam: AUC | 12 languages: AUC |
|---|---|---|---|
| **Source-1** | 0.946 | 0.979 | 0.961 |
| Teacher (open-weight 27B LLM) | 0.948 | 0.965 | 0.945 |
| propella-1 4B | 0.862 | 0.924 | 0.836 |
| propella-1 1.7B | 0.855 | 0.941 | 0.846 |
| JQL-Edu | 0.737 | 0.789 | 0.645 |
| FinePDFs-Edu | 0.731 | 0.855 | 0.659 |
| FineWeb2-HQ | 0.780 | 0.795 | 0.768 |
| FineWeb-Edu classifier | 0.812 | 0.754 | - |
| DCLM fastText (OH+ELI5) | 0.741 | 0.576 | - |
| NeMo Curator edu (Nemotron-4 labels) | 0.706 | 0.645 | - |
| NeMo Curator edu (Mixtral labels) | 0.788 | 0.775 | - |
| Meta-rater reasoning | 0.763 | 0.793 | - |
| Meta-rater readability | 0.817 | 0.832 | - |
| Meta-rater cleanliness | 0.873 | 0.880 | - |
| Meta-rater professionalism | 0.735 | 0.777 | - |
| EAI-Distill 0.5B | 0.786 | 0.835 | - |
| NVIDIA quality classifier (DeBERTa) | 0.794 | 0.706 | - |
| Dolma 3 fastText quality | 0.752 | 0.710 | - |
| Nemotron-CC 3-way ensemble | 0.750 | 0.700 | - |

Source-1's own rank / AUC on each scorer's chunks:

- held-out set: 0.900 / 0.946 (on all 495 chunks and on propella-1 4B's 493), 0.864 / 0.941 on the 159 English chunks,
  and 0.905 / 0.957 on FineWeb2-HQ's 342;
- English exam: 0.921 / 0.979 (on each set of 411 to 413 chunks);
- 12-language exam: 0.895 / 0.961 (on 352 chunks and on propella-1 4B's 350), and 0.910 / 0.962 on FineWeb2-HQ's 205.

Noise:

- None of the 2,000 resamples put any of the 39 differences at or below zero. Measured against its own noise, the
  smallest lead is 5.3 bootstrap standard deviations above zero (Meta-rater cleanliness on the 159 English held-out
  chunks). So the leads stay clear of zero after a Bonferroni adjustment for the 39 comparisons (normal
  approximation).
- On the 159 English held-out chunks, with 15 grader drops, the other measures are much noisier. The keep-agreement
  or drop-recall interval at the matched rate reaches zero for 7 of the 12 English-only rows. Two AUC leads are only
  just clear of zero: over the FineWeb-Edu classifier (+0.129, +0.004 to +0.275) and over Meta-rater cleanliness
  (+0.069, +0.008 to +0.133).
- The English-only scorers are compared on far fewer held-out chunks (159) than the multilingual ones. The exam sets
  are not independent of the grader ([Independence from development](#independence-from-development)).
- All of this is agreement with Source-1's own rubric. The public scorers were built for their own definitions of
  quality (most for educational value) and are not wrong when they disagree with it.

</details>

<details>
<summary>How each public scorer is read: inputs, main scores, EAI-Distill, the Nemotron-CC ensemble</summary>

### How each public scorer is read

Each public scorer is read through one main score, on the chunks it scored. The English-only scorers are read on
English chunks and FineWeb2-HQ on its 21 languages. propella-1, JQL-Edu and FinePDFs-Edu (with its fallback model for
five languages) are read on all 53, although propella-1's card does not list ten of them (az, fil, gu, kk, kn, ml, mr,
ms, ta, te) and JQL-Edu's backbone covers 52. Of the 16, 11 are English-only and one covers 21 of the 53 languages;
the other four were run on all 53. The public scorers read the text without Source-1's header.

| public scorer | languages it was run on | input it reads | main score used here |
|---|---|---|---|
| propella-1 4B | all 53 (its card lists 43 of them) | the whole chunk, up to 50,000 characters | a weighted mean of four of its quality ratings, defined by us (see [How propella-1 is read](#how-propella-1-is-read)) |
| propella-1 1.7B | all 53 (its card lists 43 of them) | the whole chunk, up to 50,000 characters | as propella-1 4B |
| JQL-Edu | all 53 (its backbone covers 52) | the first 8,192 tokens | the mean of its three balanced educational-value heads |
| FinePDFs-Edu | all 53 (a model per language; a fallback model for five) | about 2,000 tokens from the start, and from the end of long texts (the higher score counts) | its educational-value score |
| FineWeb2-HQ (the per-language classifiers in epfml/FineWeb-HQ-Classifiers; on English text, its FineWeb-HQ classifier) | 21 of the 53 | the first 512 tokens | its probability of high quality |
| FineWeb-Edu classifier | English | the first 512 tokens | its educational-value score |
| DCLM fastText (OH+ELI5) | English | the whole text | its probability of the high-quality label |
| NeMo Curator edu (Nemotron-4 labels) | English | the first 512 tokens | its educational-value score |
| NeMo Curator edu (Mixtral labels) | English | the first 512 tokens | its educational-value score |
| Meta-rater reasoning, readability, cleanliness, professionalism (four models) | English | the first 4,096 tokens | each model's expected rating |
| EAI-Distill 0.5B | English | the whole text up to 30,000 characters (beyond that, the start, a middle part and the end, as its card says) | a 0-5 score defined by us from four of its labels (see [How EAI-Distill and the Nemotron-CC ensemble are read](#how-eai-distill-and-the-nemotron-cc-ensemble-are-read)) |
| NVIDIA quality classifier (DeBERTa) | English | the first 1,024 tokens | its expected class (low 0, medium 1, high 2) |
| Dolma 3 fastText quality | English | the whole text | its probability of the high-quality label |
| Nemotron-CC 3-way ensemble (derived) | English | as its three parts (the two NeMo Curator classifiers and DCLM fastText) | the highest of the three parts' percentile buckets, with percentiles taken within each evaluation set |

#### How EAI-Distill and the Nemotron-CC ensemble are read

- **EAI-Distill**: reasoning depth and technical correctness are read as their position among the five ordered levels
  (0 to 4) divided by 4. Extraction artifacts and missing content are 1 when the model reports any and 0 when it
  reports none. Main score = 5 x mean(reasoning depth, technical correctness) - extraction artifacts - missing content,
  clipped to 0-5. An indeterminate or abstaining answer is left out of the mean (no score when both are), and a missing
  penalty counts 0. Its answer codes were read with the code tables of the Essential-AI/eai-taxonomy README at commit
  `e8a934d5ca77a05f8daddc73466aedf8a9eb7a6c`.
- **Nemotron-CC 3-way ensemble**: each part's score becomes a bucket, floor(20 x its percentile rank within the
  evaluation set) (0 to 19), and the ensemble's score is the highest of the three buckets.

</details>

<details>
<summary>Educational value alone: Source-1's educational_value field against every public scorer</summary>

### Educational value alone

Most public scorers were built to rate educational value. Read that way, Source-1's `educational_value` field ranks
the held-out set closer to the grader's `educational_value` (0.879) than every public scorer's main score does. All
17 intervals exclude zero, as they do on both exams. The closest is propella-1 4B's composite (0.823; Source-1 +0.056,
95% interval +0.03 to +0.08). Among the dedicated educational-value classifiers, JQL-Edu trails by +0.174 (+0.14 to
+0.22) and, on the 159 English chunks, the FineWeb-Edu classifier by +0.238 (+0.16 to +0.33). On the English exam,
Source-1's `educational_value` reaches 0.904 against 0.614 for the FineWeb-Edu classifier and 0.884 for propella-1 4B.
The margin over propella-1 4B is small there: +0.020 (+0.002 to +0.040). The grader's `educational_value` follows
Source-1's rubric anchors, not the annotation prompts these classifiers were trained on. propella-1's own
educational-value rating agrees less with it (0.776 on the held-out set) than its composite does.

</details>

### How propella-1 is read

<details>
<summary>The composite we read it through, a more generous reading, and propella-1's answers</summary>

propella-1 answers in words, not numbers. We map each of its ordered ratings to integers. We read it through a
weighted mean of four quality ratings (educational value, reasoning, content quality, information density), weighted
as in Source-1's quality formula. That leaves out its ratings for commercial bias, content ratio and integrity, and
content safety, which are close to the grader's drop rules. It was also run on all 53 languages, ten of which its card
does not list. As a check of a more generous reading, chosen after the results were known, we also restricted it to
its 43 listed languages and subtracted rubric-style penalties for those ratings. Each was rescaled to 0-5, then 0.5,
0.4 and 0.8 times the excess over 1 was subtracted for commercial bias, the larger of content ratio and integrity, and
content safety, as in Source-1's overall score. Held-out set:

| reading of propella-1 4B | chunks | propella-1 4B rank | Source-1 rank | Source-1 minus propella-1 4B: rank (95% interval) | AUC (95% interval) |
|---|---|---|---|---|---|
| main score, all 53 languages (the tables above) | 493 | 0.756 | 0.900 | +0.144 (+0.11 to +0.18) | +0.084 (+0.05 to +0.12) |
| main score, its 43 listed languages | 459 | 0.766 | 0.899 | +0.133 (+0.10 to +0.17) | +0.080 (+0.05 to +0.12) |
| with its own red-flag ratings as penalties, all 53 languages | 493 | 0.828 | 0.900 | +0.072 (+0.05 to +0.10) | +0.049 (+0.02 to +0.07) |
| with its own red-flag ratings as penalties, its 43 listed languages | 459 | 0.835 | 0.899 | +0.064 (+0.04 to +0.09) | +0.048 (+0.03 to +0.07) |

We tried four ways of weighting the penalties: the one above, boilerplate from content ratio alone, from the sum of
content ratio and integrity, and no rescaling. Across them, on all 53 languages and on its 43 listed ones, the
held-out rank gap ranges from 0.064 to 0.081. With the penalties as above, the exam gaps are +0.075 (+0.05 to +0.10)
in English and +0.155 (+0.10 to +0.21) in the 12 languages. The lead holds under every reading we tried, but it
roughly halves: the 0.144 in the main tables depends on reading propella-1 through its quality ratings alone.

The exact mapping: each rating word is mapped to its position in the rating's scale, from 0: educational value (none,
minimal, basic, moderate, high), reasoning (none, minimal, basic_reasoning, explanatory, analytical), content quality
(unacceptable, poor, adequate, good, excellent), information density (empty, thin, moderate, adequate, dense). Main
score = (0.30 x educational value + 0.20 x reasoning + 0.15 x content quality + 0.20 x information density) / 0.85, on
that 0-4 scale. The penalized reading also maps commercial bias (none, minimal, moderate, heavy, pure_marketing),
content ratio (complete_content, mostly_content, mixed_content, mostly_navigation, minimal_content) and content safety
(safe, mild_concerns, nsfw, harmful, illegal) to 0-4 and content integrity (complete, mostly_complete, fragment,
severely_degraded) to 0-3. It rescales each of them and the main score to 0-5, and subtracts 0.5 x max(0, commercial
bias - 1) + 0.4 x max(0, max(content ratio, content integrity) - 1) + 0.8 x max(0, content safety - 1), clipped to
0-5.

#### propella-1's answers

An answer that did not parse as JSON was run once more, with a limit of 1,536 new tokens instead of 512. On the
evaluation chunks, the 4B was retried on 1 held-out chunk, and the 1.7B on 4 held-out chunks and 1 English exam chunk.
After the retry every answer parsed, except that one 1.7B English exam answer: it ran to the token limit, and its
ratings were read from the raw text. Seven answers used an educational-value word outside its scale (the 4B: 2
held-out, 1 English exam and 2 12-language exam chunks; the 1.7B: 2 English exam chunks). Those chunks have no main
score. This is why propella-1 4B is compared on 493 of the 495 held-out chunks and 350 of the 352 12-language chunks.

</details>

### Exam sets

On both exams Source-1 and the teacher are level within noise (English +0.009, 95% interval -0.01 to +0.02;
12 languages +0.015, -0.007 to +0.04).

<details>
<summary>Exam sets in detail: rank, AUC, whole documents and the drop line per document</summary>

### Exam sets in detail

| model | English: rank (413 chunks) | English: AUC | English: whole documents (rank) | 12 languages: rank (352 chunks) | 12 languages: AUC |
|---|---|---|---|---|---|
| **Source-1** | 0.921 | 0.979 | 0.920 | 0.895 | 0.961 |
| Teacher | 0.912 | 0.965 | 0.901 | 0.880 | 0.945 |

Whole documents: the token-weighted aggregation over each document's chunks, on the English exam's random-sample
documents (196 for Source-1; 195 for the teacher, which has no score for one of them). Source-1 on the same 195 is also
0.920.

At the shipped drop line, counted per document (the teacher at its own keep flags):

| exam documents | grader drops | Source-1: drops caught | Source-1: keep agreement | teacher: drops caught | teacher: keep agreement |
|---|---|---|---|---|---|
| English, all 332 | 24 | 14 (0.583) | 96.7% | 14 (0.583) | 96.1% |
| English, the 196 random-sample documents | 13 | 9 (0.692) | 97.4% | 6 (0.462) | 94.9% |
| 12 languages, all 352 (none added on purpose) | 46 | 31 (0.674) | 94.3% | 28 (0.609) | 93.5% |

The exams were graded with an older revision of the rubric, before it settled how to score ads (`spam_seo` 3, kept).
So part of the gap is rubric drift that affects the teacher and Source-1 alike.

</details>

### More results

### The drop line

<details>
<summary>The drop line: how it was chosen, stricter lines, results on the held-out set</summary>


`keep` is false when the scores cross the drop line stored in `calibration.json`:

```
drop if  toxicity >= 4  or  spam_seo >= 3.5  or  boilerplate >= 4.5
```

The rubric's own hard filters use `spam_seo >= 4`. Pages scored `spam_seo` 3, the level for ads and promotional pages,
are kept by both lines. The line was chosen on the teacher's labels for the validation split (9,744 chunks) by a fixed
rule: among five candidate lines, take the highest drop recall whose keep agreement with the teacher stays at or above
0.95. On that split it agrees with the teacher's keep flags on 96.4% of chunks, catches 77.1% of the teacher's drops
(803 of 1,042) and drops 9.4% of chunks. On the held-out test split (9,553 chunks, not used for the choice) it agrees
on 96.7% and catches 80.2% (840 of 1,047). The candidate lines and the 0.95 floor were set with help from grades by
the grader's model family ([Independence from development](#independence-from-development)).

If you need to catch more low-quality text and can afford to lose more good text, pass a stricter line as
`drop_line`. The candidates trade keep agreement for recall (validation split, against the teacher's keep flags):

| drop line | keep agreement | teacher drops caught | wrong drops | share dropped |
|---|---|---|---|---|
| `toxicity >= 4 or spam_seo >= 4 or boilerplate >= 4.5` (rubric default) | 0.960 | 0.711 (741/1,042) | 89 | 8.5% |
| **`toxicity >= 4 or spam_seo >= 3.5 or boilerplate >= 4.5` (shipped)** | **0.964** | **0.771 (803/1,042)** | **114** | **9.4%** |
| `toxicity >= 4 or spam_seo >= 3 or boilerplate >= 4.5` | 0.929 | 0.830 (865/1,042) | 515 | 14.2% |
| `toxicity >= 4 or spam_seo >= 3.5 or boilerplate >= 4` | 0.945 | 0.880 (917/1,042) | 408 | 13.6% |
| `toxicity >= 4 or spam_seo >= 2.5 or boilerplate >= 4.5` | 0.856 | 0.872 (909/1,042) | 1,271 | 22.4% |

`calibration.json` also stores one offset per quality score (mean teacher label minus mean model score on the
validation split). They are small, between -0.016 and -0.003 points, and `source1.py` leaves them off unless you pass
`apply_offsets=True`. The results in this file use the scores without offsets, as `source1.py` returns them by
default, except the per-field bias table, which says where it applies them.

On the held-out set, against the grader:

| model | line | keep agreement | drop recall (caught / grader drops) | wrong drops | share dropped |
|---|---|---|---|---|---|
| **Source-1** | shipped line | 0.941 (0.917 to 0.959) | 0.672 (43/64; 0.550 to 0.774) | 8 | 10.3% |
| Source-1 | rubric default line (`spam_seo >= 4`) | 0.931 | 0.594 (38/64) | 8 | 9.3% |
| Teacher | its own keep flags | 0.943 | 0.719 (46/64) | 10 | 11.3% |

17 of Source-1's 21 misses at the shipped line are also missed by the teacher at its own keep flags, and 16 of the 21
are in multilingual web text. On the 29 Chinese chunks Source-1's line agrees with the grader on 89.7% and catches 11
of 14 drops. The candidate lines were drafted after the held-out set's grades had been seen once, so treat these
held-out drop-line numbers as slightly optimistic. The test-split numbers above do not have this problem.

</details>

<details>
<summary>Per-field agreement and bias on the exam sets, and the red flags on the held-out set</summary>

### Per-field agreement and bias

Agreement per field on the exam sets. For the five quality scores: quadratic-weighted kappa on levels rounded to the
nearest integer (English: the 200 random-sample chunks, 199 for the teacher; 12 languages: all 352 chunks). For the
labels: unweighted Cohen's kappa (English: all 414 chunks, 413 for the teacher; 12 languages: all 352 chunks). A dash
means the kappa is not meaningful. 351 of the 352 chunks in the 12-language exam are plain text by the grader's label,
so content type has almost no variation there (Source-1 matches the grader on 350 of them, and its kappa is about 0).

| field | English: Source-1 kappa | English: teacher kappa | 12 languages: Source-1 kappa | 12 languages: teacher kappa |
|---|---|---|---|---|
| educational_value | 0.831 | 0.816 | 0.788 | 0.803 |
| reasoning_depth | 0.798 | 0.781 | 0.719 | 0.730 |
| writing_quality | 0.769 | 0.811 | 0.785 | 0.786 |
| information_density | 0.861 | 0.828 | 0.812 | 0.800 |
| reliability | 0.739 | 0.790 | 0.735 | 0.743 |
| format | 0.748 | 0.742 | 0.745 | 0.745 |
| topic | 0.776 | 0.794 | 0.786 | 0.808 |
| content_type | 0.833 | 0.864 | - | - |

Bias is the mean of model minus grader on rounded levels, on the same chunks as the quality kappas. Source-1's values
here include its optional calibration offsets (`apply_offsets=True`). Without them, as `source1.py` returns scores by
default, they differ by at most 0.014 (English: `reasoning_depth` +0.390, `writing_quality` +0.325, `reliability`
+0.270). The teacher has no offsets.

| field | English: Source-1 bias | English: teacher bias | 12 languages: Source-1 bias | 12 languages: teacher bias |
|---|---|---|---|---|
| educational_value | +0.280 | +0.307 | +0.111 | +0.142 |
| reasoning_depth | +0.385 | +0.422 | +0.301 | +0.321 |
| writing_quality | +0.320 | +0.281 | +0.131 | +0.153 |
| information_density | -0.105 | -0.085 | -0.196 | -0.159 |
| reliability | +0.260 | +0.261 | +0.196 | +0.196 |

On the held-out set (quadratic-weighted kappa on rounded levels, all 495 chunks), the red flags reach 0.87 for
`spam_seo`, 0.81 for `boilerplate` and 0.70 for `toxicity` (the teacher: 0.88, 0.81 and 0.73). The gated scores can
only be compared on the few chunks where both the grader and the model give them: `math_quality` 0.64 on 8 chunks (the
teacher 0.90) and `code_quality` 0.67 on 23 (the teacher 0.82 on 22). These numbers are very noisy.

</details>

### Reading the whole chunk

<details>
<summary>Reading the whole chunk: what the text past 512 tokens adds</summary>


The same model was run with every input cut to its first 512 tokens at scoring time. It was trained on full chunks,
so this measures what the text past 512 tokens adds, not how a model trained for 512 tokens would do. On the held-out
set, 197 of the 495 chunks fit in 512 tokens and score identically both ways.

| chunks | full chunk: rank | first 512 tokens: rank | difference (95% interval) |
|---|---|---|---|
| held-out set, all (495; 1,650 tokens on average) | 0.900 | 0.848 | +0.052 (+0.02 to +0.08) |
| held-out set, 513 to 2,048 tokens (192) | 0.902 | 0.881 | +0.021 (-0.01 to +0.05) |
| held-out set, over 2,048 tokens (106) | 0.875 | 0.753 | +0.122 (+0.05 to +0.21) |
| English exam (414) | 0.921 | 0.845 | +0.075 (+0.05 to +0.11) |
| 12-language exam (352) | 0.895 | 0.853 | +0.042 (+0.02 to +0.08) |

The gain comes from the longer chunks. Finding the grader's drops barely changes (held-out AUC 0.946 against 0.943;
+0.003, -0.01 to +0.01). Only 74 held-out chunks are longer than 4,096 tokens, so this does not test the far end of
the 8,192-token window.

The 512-token limit of some public scorers does not explain their lower agreement on the held-out set. Cut to 512
tokens, Source-1 still ranks it closer to the grader than propella-1 4B reading the whole chunk (0.848 against 0.756
on its 493 chunks; +0.092, +0.05 to +0.14). It also ranks closer than each of the four scorers that read 512 tokens
(leads +0.30 to +0.42, every interval clear of zero). On the English exam the cut model is only level with
propella-1 4B (+0.026, -0.01 to +0.07).

</details>

<details>
<summary>Agreement with the teacher on the test split: the job Source-1 was trained for</summary>

### Agreement with the teacher on the test split

How closely Source-1 reproduces the teacher's labels on 9,553 test chunks (8,473 documents, 53 languages) it never
trained on: the job it was trained for. The grader results above instead measure agreement with an independent LLM
grader applying the same rubric.

| chunks | n | overall score rank agreement | keep/drop agreement (rubric default line) | label accuracy | quality score error (MAE, 0-5 scale) |
|---|---|---|---|---|---|
| all | 9,553 | 0.953 (0.950 to 0.955) | 0.963 | 0.907 | 0.254 |
| web | 2,677 | 0.952 | 0.964 | 0.904 | 0.250 |
| multilingual web | 5,344 | 0.955 | 0.960 | 0.912 | 0.245 |
| conversations, code, synthetic | 950 | 0.895 | 0.964 | 0.890 | 0.325 |
| open books | 582 | 0.844 | 0.985 | 0.901 | 0.248 |

The keep/drop column applies the rubric's default hard filters to Source-1's scores. With the shipped drop line,
agreement on this split is 0.967 (see [The drop line](#the-drop-line)). The interval on the overall rank agreement is
a 95% bootstrap interval over the 8,473 documents (2,000 resamples, seed 0).

Label accuracy is 0.877 for format, 0.853 for topic and 0.991 for content type. The gated scores agree least with the
teacher: kappa 0.58 for `code_quality` (555 chunks where it applies) and 0.50 for `math_quality` (257 chunks).
Agreement with the teacher is lowest for Gujarati (0.788), Georgian (0.867), Croatian (0.887), Malayalam (0.893),
Bengali and Marathi (0.897) and Serbian (0.899).

</details>

<details>
<summary>Speed, and how much scores move with precision and batching</summary>

### Speed

Measured with `source1.py` on one RTX 3090 in bf16 with the default batches, excluding load time, on the 495 held-out
chunks and the 766 exam chunks. Repeated runs on the same GPU differed by up to about 10%. No speed comparison with
the public scorers or the teacher was made under the same conditions, so none is claimed.

| set | chunks | mean input tokens per chunk | chunks/s | input tokens/s | peak VRAM |
|---|---|---|---|---|---|
| held-out | 495 | 1,650 | 32.2 | 53,198 | 4.10 GB |
| exam | 766 | 1,766 | 31.0 | 54,644 | 4.21 GB |

Precision and batching: computing in bfloat16, both weight files give the same scores. On these 1,261 chunks, scoring
each chunk alone instead of in the default batches moved `overall` by up to 0.04 and a single field by up to 0.10 (3
labels changed, no keep decision). float32 differs from bfloat16 by a similar amount (up to 0.03 on `overall` and 0.09
on a single field; 4 labels and 1 keep decision changed); computing in float32 with the default bfloat16 weights moved
`overall` by up to 0.05. In float32, scores do not depend on the batch.

</details>

## Limitations

- **Home ground.** It measures agreement with Source-1's rubric, as applied by graders from one proprietary model
  family whose grades also steered development. It does not show how Source-1 does on text from other sources, or
  that filtering with it trains better language models.
- **The numbers are agreement with one grader family, not accuracy.** Another grader applying the same rubric would
  give different values.
- **The exams helped choose the teacher**, so they are not independent of the grader.
- **Agreement is lower among good texts.** Among the chunks the grader keeps, Source-1's rank agreement is 0.87, and in
  the better half of those 0.73.
- **It misses about a third of the grader's drops** at the shipped line: it catches 43 of 64 on the held-out set, the
  teacher, at its own keep flags, 46.
- **Like its teacher, it rates some qualities higher than the grader does.** On the English exam, `educational_value`
  is 0.28 levels above the grader on average.
- **Weaker on books and on conversations, code and synthetic text** (held-out rank 0.677 and 0.728, against about 0.90
  for web text).
- **One chunk of up to 8,192 tokens at a time.** Nothing outside a chunk is visible to it.
- **It copies the teacher, biases included.** For example, it can score a thin affiliate page 3 (kept) where the
  rubric says 4 (dropped).
- **The gated scores are the least reliable fields, and `toxicity` is the weakest red flag.**
- **Not a fact checker or a safety tool.**
- **Less data for some languages.** The 16 smallest have 1,249 to 1,470 training chunks each.
- **License screening has limits.** Notices the patterns miss, and opt-outs outside the text, were not caught.
- **No reproduction kit.** The numbers cannot be recomputed from this repository alone
  ([what is included](#reproducing-the-evaluation)).

### Limitations in detail

<details>
<summary>The full text of each limitation</summary>

- **Home-ground evaluation.** The benchmark measures agreement with Source-1's rubric as applied by independent LLM
  graders from one proprietary model family (one graded the exams, another the held-out set). They are independent in
  that their grades were never trained on; the held-out grader was given the teacher's own instructions word for word
  ([The grader in detail](#the-grader-in-detail)). Grades from that family,
  among other proprietary LLM graders, also steered the rubric revisions, the choice of the teacher and its prompt
  setup, and the drop-line candidates and floor ([Independence from development](#independence-from-development)).
  The held-out documents come from the same kinds of sources as the training data. The public scorers were built for
  other definitions of quality, read the text without Source-1's header and are each read through one main score. A
  more generous reading of propella-1 halves its gap to Source-1 ([How propella-1 is read](#how-propella-1-is-read)).
  The comparison does not show how Source-1 does on text from other sources, or that filtering with Source-1 trains
  better language models; neither has been tested.
- **The numbers are agreement with one grader family, not accuracy.** Another grader applying the same rubric would
  give different values, and differences of a few hundredths near the top (Source-1 against its teacher) may reflect
  this grader's own habits.
- **The exams are not independent of the grader.** They are the samples on which the teacher and its prompt setup were
  chosen against the exam grader's labels. The held-out set, which played no part in that choice, is the main result.
- **Agreement is lower among good texts.** About 13% of the held-out chunks are spam, boilerplate or toxic, which are
  easy to tell apart. Among the chunks the grader keeps, Source-1's rank agreement is 0.87, and in the better half of
  those 0.73 (teacher 0.75, propella-1 4B 0.61). Every scorer drops like this on already-filtered text; if you rank
  filtered text, expect the lower figure.
- **It misses about a third of the chunks the grader drops at the shipped line.** On the held-out set the shipped line
  catches 67.2% of the grader's drops (43/64; interval 55.0% to 77.4%); the teacher catches 71.9% (46/64). 17 of
  Source-1's 21 misses are also missed by the teacher, so most of what Source-1 misses its teacher misses too. Most
  misses are in multilingual web text (28 of 44 caught there). If recall matters more than keeping good text, use a
  stricter line ([The drop line](#the-drop-line)) or rank on `overall` and cut lower.
- **Like its teacher, it rates some qualities higher than the grader does.** On the English exam's 200 random-sample
  chunks, Source-1's `educational_value` is on average 0.28 levels above the grader's (the teacher 0.31). Its
  `reasoning_depth`, `writing_quality` and `reliability` are 0.26 to 0.39 levels above (rounded levels, with or without
  the optional calibration offsets; the teacher: 0.26 to 0.42). For `writing_quality` its bias is larger than the
  teacher's (+0.32 to +0.325 against +0.28). In the 12 languages they are smaller (`reasoning_depth` +0.30,
  `reliability` +0.20, `writing_quality` +0.13 to +0.14, `educational_value` +0.11; the teacher +0.32, +0.20, +0.15
  and +0.14). See [Per-field agreement and bias](#per-field-agreement-and-bias).
- **Weaker on books and on conversations, code and synthetic text.** Held-out rank is 0.677 for open books and 0.728
  for conversations/code/synthetic (the teacher: 0.741 and 0.821; 42 and 48 chunks), against about 0.90 for web text.
  Against the teacher on the test split it is 0.844 and 0.895. Book labels are nearly constant (long, formal, almost
  always kept), which leaves little signal to learn from.
- **8,192 tokens per chunk.** Longer documents are split and each chunk is judged on its own. Nothing outside a chunk
  is visible to it, except the "Part i of n" header. Text in scripts that need many tokens per character fills the
  window sooner: in training, 10.5% of Bengali chunks, 7.2% of Georgian, 6.3% of Arabic and 5.0% of Korean chunks
  were longer than 8,192 tokens and were truncated. When you score with `source1.py`, a document longer than the
  window is split into chunks rather than cut, so all of its text is read (the `truncated` field reports the rare
  chunk that still had to be cut). `max_chunks` trades that for speed by scoring only some evenly spaced chunks.
- **It copies the teacher, biases included.** The rubric puts ads, company pages and product pages at `spam_seo` 3,
  which the shipped line keeps, and thin affiliate and doorway pages at 4, which it drops. The teacher does not always
  follow the second rule and sometimes scores such pages 3, and Source-1 learned from those labels.
- **The gated scores are the least reliable fields, and toxicity is the weakest red flag.** Against the teacher on
  the test split, kappa is 0.58 for `code_quality` and 0.50 for `math_quality`. Against the grader on the held-out set
  they can be compared only on 8 and 22 to 23 chunks ([Per-field agreement and bias](#per-field-agreement-and-bias)).
  `toxicity` reaches 0.70 against the grader (the teacher 0.73).
- **Not a fact checker or a safety tool.** `reliability` is a surface judgment of care and plausibility; the model does
  not verify claims. Toxic text is rare in the training data. `toxicity` is meant as a data-filtering red flag, not a
  moderation classifier.
- **Languages with little data.** The 16 languages with the fewest training chunks (az, et, fil, gu, ka, kk, kn, lv,
  ml, mr, ms, sq, sw, ta, te, ur) have 1,249 to 1,470 each, and Kannada had no books. Agreement with the teacher on the
  test split is lowest for Gujarati (0.788), Georgian (0.867), Croatian (0.887), Malayalam (0.893), Bengali and Marathi
  (0.897) and Serbian (0.899). The held-out set has 1 to 12 chunks for most non-English languages, so per-language
  results there are noisy.
- **License screening has limits.** Licenses come from each source's metadata. The training documents were also
  screened by pattern matching on their own text: books for NonCommercial, NoDerivatives and all-rights-reserved
  notices in their front and back matter, web pages for such terms and for the sites they come from, and every
  document for text-and-data-mining and AI-training reservations. This screening did not catch notices worded in ways
  the patterns miss, or reservations made outside the text itself (on the terms pages of sites the rules do not list,
  or in machine-readable opt-out signals such as robots.txt). If you find such a document, tell us (see the contact
  section of [README.md](README.md#contact)).
- **No reproduction kit.** See [Reproducing the evaluation](#reproducing-the-evaluation).

</details>

## Reproducing and appendices

<details>
<summary>Reproducing the evaluation: what this repository does and does not include</summary>

### Reproducing the evaluation

What this repository gives you:

- Source-1 itself: the weights, `source1.py` and `calibration.json`, which produce the Source-1 scores used here
  (computed in bfloat16 on one RTX 3090 with the default batches; see [Speed](#speed) for how much scores move with
  precision and batching).
- The metric definitions and bootstrap settings ([Metric definitions](#metric-definitions)).
- How each public scorer was run ([How the public scorers were run](#how-the-public-scorers-were-run)) and read
  ([How each public scorer is read](#how-each-public-scorer-is-read)), with the exact definitions of the scores we
  defined ourselves ([How propella-1 is read](#how-propella-1-is-read),
  [How EAI-Distill and the Nemotron-CC ensemble are read](#how-eai-distill-and-the-nemotron-cc-ensemble-are-read)).
- The Hugging Face repositories and revisions of the 16 public scorers
  ([Appendix C](#appendix-c-public-scorer-repositories)).

What it does not include: the ids and texts of the evaluation chunks, the grader's labels, any model's per-chunk
scores, the script that computes the metrics, and the teacher's prompt and decoding settings. The numbers in this file
cannot be recomputed from this repository alone.

</details>

### Appendix A: rubric anchors

<details>
<summary>Appendix A: rubric anchors, and the teacher's extra rules</summary>


| field | 0 | 1 | 2 | 3 | 4 | 5 |
|---|---|---|---|---|---|---|
| educational_value | Teaches nothing: spam, ads, navigation, gibberish | Almost nothing to learn: a few incidental facts in promotional, personal or trivial text | Some useful information, but superficial, fragmentary, or mixed with irrelevant material | Useful and coherent; real knowledge or skills, without much depth or completeness | Clearly educational; explains concepts or methods well enough to learn from, minor gaps | Outstanding teaching material, comparable to an excellent textbook or expert tutorial |
| reasoning_depth | No reasoning: fragments, lists, boilerplate | Bare assertions or opinions | Occasional explanation, mostly unsupported; steps skipped | Explains the why behind key points, some step-by-step structure | Consistent explicit reasoning: derivations, cause and effect, worked examples | Rigorous multi-step reasoning throughout: proofs, careful derivations, thorough analysis |
| writing_quality | Unreadable: garbled, broken encoding, keyword soup | Very poor: frequent errors, incoherent | Below average: understandable but disorganized or repetitive | Adequate: clear and coherent, minor issues | Good: well organized, fluent, precise | Excellent: publication quality |
| information_density | No real content | Mostly padding around a little content | Noticeable padding or digressions | Reasonable: mostly on point, some filler | Dense: most sentences carry information | Very dense yet readable |
| reliability | Fabricated, nonsensical or deceptive | Largely unreliable: many errors, sensational claims | Questionable: some errors or unsupported claims | Generally plausible and consistent, informal or unverifiable | Careful and accurate; shows its work or cites sources | Authoritative: expert-level accuracy, well sourced |
| spam_seo | None | Minor promotion: a call to action or a brief ad in otherwise genuine content | Noticeable promotion: repeated calls to action, affiliate links, marketing tone | Substantial: the text mainly exists to promote, sell or rank; visible keyword repetition | Mostly spam: keyword stuffing, clickbait, thin affiliate or doorway content | Pure spam: auto-generated SEO text, scams, keyword lists |
| boilerplate | None: all real content | A little: a stray header, footer or copyright line | Noticeable: roughly 10-25% navigation, cookie or legal text | Substantial: roughly 25-50% templates, menus, link lists | Mostly boilerplate: over half templated or navigational | Entirely boilerplate: auto-generated pages, link lists, cookie banners, error or index pages |
| toxicity | None | Mild: occasional profanity or rudeness | Moderate: insults, crude humor, mild sexual references | Significant: harassment, demeaning stereotypes, graphic violence, partly explicit | Severe: hate speech, targeted harassment, sexually explicit as the main content | Extreme: violent extremism, dehumanizing hate, incitement, sexual content involving minors |
| code_quality | Not usable code: garbled, minified, obfuscated | Very poor: likely non-functional fragments, no structure, or auto-generated boilerplate | Poor: may work but messy | Acceptable: readable, plausibly correct, minimal docs | Good: clean, idiomatic, documented | Excellent: exemplary, production quality, instructive |
| math_quality | Garbled math | Mostly wrong or incoherent | Some correct math, but errors or skipped steps | Generally correct, key steps shown | Correct, clean notation, complete steps | Rigorous and elegant, every step justified |

The rubric anchors and the head layout are in `source1.json`. The teacher's prompt, which the held-out grader also
received word for word, had a few special rules that are not in `source1.json`. Source-1 was trained on labels that
follow them, as far as the teacher did:

- Pages whose main purpose is to promote or sell a business, product or service (company "about us" pages, product
  and landing pages, shop listings, brochures) are ads: format `product_page` and `spam_seo` 3, even when cleanly
  written. Self-promotional press releases stay `news` with `spam_seo` 3. Selling alone is never a reason for
  `spam_seo` 4 or 5; those levels are for keyword-stuffed text, doorway or thin affiliate pages made to rank, and
  scams. Independent reviews, comparisons and news about products are not ads.
- Pages stitched together from unrelated or scrambled text (often a keyword title over copied or shuffled
  paragraphs) are `spam_seo` 4-5, with `reliability` and `writing_quality` 0-1.
- Tag, category, archive and search-result pages, feeds, link directories and other index pages that mostly list
  other pages are `boilerplate` 4, and 5 when the list is all they contain; empty auto-generated stub pages are 4-5.
- Sexually explicit material as the main content is `toxicity` 4 in any language (5 if it involves minors).
- General rules: judge only the text shown (a part of a longer document is not penalized for starting or ending
  mid-thought); ignore personal-data placeholders such as `<EMAIL>`; judge every language by its own standards;
  poor machine translation lowers `writing_quality`, and machine-translated filler written to rank is spam.

</details>

### Appendix B: training in detail

<details>
<summary>Appendix B: training in detail (model, data, filtering, labels, recipe)</summary>


#### Model

- Backbone: [mmBERT-base](https://huggingface.co/jhu-clsp/mmBERT-base) (ModernBERT architecture, 22 layers, hidden
  size 768, 8,192-token context; trained by its authors on 3T+ tokens across 1800+ languages). All backbone weights
  were fine-tuned.
- Heads: 13 linear heads on the mean-pooled final hidden states (about 68k parameters): one softmax head per label
  (10, 15 and 4 classes) and one six-level softmax head per 0-5 field.
- Total: 307M parameters. Trained with float32 weights in bfloat16 mixed precision; released in bfloat16 (default)
  and float32.

#### Data

220,346 chunks were labeled. After license filtering, the safety filter and the held-out splits, and after setting
aside a reserve that was never trained on (about 1% of training documents, picked by a hash of the document id and
held back for label-quality checks, plus the training and validation documents reviewed during development; see
[Independence from development](#independence-from-development)), 172,895 chunks (350M tokens, from 151,281 documents)
were used for training, 9,744 for validation and 9,553 for testing. No label-quality result from the reserve is
reported here. Documents were split 90/5/5 by a hash of the document id, so no document spans two splits.

| stage | what it is | training chunks |
|---|---|---|
| Web | A stratified sample of filtered and unfiltered web text, PDFs, wikis, math pages, permissively licensed code, Common Pile sources and toxicity datasets, in 53 languages; low-quality pages included on purpose | 46,389 |
| Multilingual web | A larger sample of the same kinds of sources, weighted toward languages other than English | 96,778 |
| Conversations, code, synthetic | Chat and instruction data, permissively licensed code and commits, synthetic and machine-generated text, comments and other short or noisy text, and domain text (law, parliament proceedings, science articles, historical and OCR text) | 17,614 |
| Open books | Books recorded as openly licensed or public domain, from Project Gutenberg, HAL, Wikibooks, Wikisource, OpenStax, the World Bank, EU and FAO publications, DOAB, OAPEN and others, after removing books whose own text states stricter terms (see below) | 12,114 |
| **Total** | 53 languages; English is 38.9% of training chunks | **172,895** |

Languages: ar, az, bg, bn, ca, cs, da, de, el, en, es, et, fa, fi, fil, fr, gu, he, hi, hr, hu, id, it, ja, ka, kk,
kn, ko, lt, lv, ml, mr, ms, nl, no, pl, pt, ro, ru, sk, sl, sq, sr, sv, sw, ta, te, th, tr, uk, ur, vi, zh. Every
non-English language has 1,249 to 3,437 training chunks; the 16 languages with the fewest are listed under
[Limitations in detail](#limitations-in-detail).

To replace documents that the license filtering below removed, 18,113 of the training and validation chunks were
drawn by fixed sampling rules (no model chose documents) from the same kinds of open sources: FineWeb-2 (including its
removed-documents part), the FineWeb-Edu annotations, C4, HPLT, FinePDFs, FineMath, permissively licensed code, Common
Pile and Common Corpus documents, and open books. Every rule below was applied to them too.

Filtering before training (training and validation splits; the held-out test split keeps every document the license
filtering removed, for evaluation only):

- License filtering removed 21,576 chunks from 19,720 documents:
  - sources whose terms do not clearly cover this use: DCLM-baseline (its dataset card states that it is intended
    for research use), raw Common Crawl WET pages (no dataset license), Stack v2 Edu (its upstream terms are gated),
    Common Pile's YouTube transcripts (licenses asserted by uploaders over broadcasts), two collections with
    unreliable license metadata, a corpus of third-party social media posts and a toxicity corpus whose texts are
    not covered by its stated license, and French public data under the Licence Ouverte;
  - code: copyleft licenses (GPL, AGPL, LGPL, MPL, EPL), code outside a permissive allow-list (MIT, Apache-2.0, BSD,
    ISC, CC0, Unlicense), and code files whose own header states copyleft, proprietary or NonCommercial terms;
  - books whose own front or back matter states stricter terms than the open license their platform recorded: a
    scan of all 4,170 book documents found 137 that state NonCommercial or NoDerivatives terms or reserve all rights
    with no open grant, and a wider pass over the same pages found 28 that state such terms in other wordings, forbid
    sale or contradict their license record; also books deposited in HAL whose own text states no open license (84)
    and library books from the Norwegian Colossal Corpus published after 1955 (14);
  - other documents whose own text carries such notices (NonCommercial, NoDerivatives or all-rights-reserved
    statements, publishers' copyright notices, text reprinted with permission): 141; web pages under NonCommercial or
    NoDerivatives terms, or from sites whose terms put all their content under such terms (302); pages from sites
    that re-host other people's documents, homework, shadow-library, pirated-novel, lyrics and subtitle sites (529);
    and a few smaller groups (pages offering software cracks, open-education pages with no stated license,
    GFDL-only pages, papers marked closed-access);
  - documents whose own text reserves text-and-data-mining or AI-training rights: a scan of all 192,905 input
    documents found 16.
- Separately, a pre-specified safety filter removed documents from every split, and a rule fixed in advance also
  removed documents that substantially copy text the safety filter removed (every split).
- Book pages that are mostly a table of contents were left out of training (99 chunks).
- Every exam document was checked against the training data (exact text, URL, title, long-line and shingle matching).
  None has half or more of its text in a training document; the largest share of an exam text found in a training
  document is 41%.

#### Labels

- Every training label comes from one teacher: an open-weight 27B LLM scoring each chunk against the 13-field rubric,
  self-hosted on our own and rented GPUs. No human labels were used.
- No output of a proprietary model was used as a label or a training target. The evaluation grader's outputs were
  never trained on. What grades from proprietary LLMs did inform is listed under
  [Independence from development](#independence-from-development).
- About 2% of training documents (3,387) come from public datasets of model-written text (synthetic textbooks, chat
  logs, machine-generated-text detection sets, machine translations). They are there so the scorer learns to judge
  such text; the teacher scored them like any other input.

#### Recipe

| setting | value |
|---|---|
| epochs | 2 (5,320 optimizer steps) |
| tokens per step | 131,072 (about 65 chunks) |
| max length | 8,192 tokens; 964 training chunks (0.6%) were longer and were truncated |
| optimizer | AdamW, betas 0.9 / 0.98, eps 1e-6, weight decay 0.01, gradient clipping 1.0 |
| learning rate | 5e-5 for the backbone, 10x for the heads; 5% warmup, cosine decay to 10% |
| other | mean pooling, dropout 0.1, bf16 mixed precision, runs of spaces and tabs collapsed to one space before tokenizing (newlines kept), natural language mix (no reweighting), seed 0 |
| checkpoint | the final step, which had the best validation overall Spearman against the teacher (0.955) |
| how the settings were chosen | learning rate and language mix: a half-epoch sweep (two learning rates, two language mixes) on a smaller training set labeled by the same teacher, read on validation agreement with the teacher; epochs: the same validation curves (a third epoch added little while validation loss rose) |
| hardware | one rented NVIDIA H100 NVL, about 1.9 hours |

</details>

<details>
<summary>Appendix C: Hugging Face repositories and revisions of the public scorers</summary>

### Appendix C: public scorer repositories

| public scorer | Hugging Face repository | revision |
|---|---|---|
| propella-1 4B | `ellamind/propella-1-4b` | `bf607e62b6afa3e0e8d71c4d08d1429d9a09c82f` |
| propella-1 1.7B | `ellamind/propella-1-1.7b` | `2cb58fd324fce70e1cb106df20bf4e1d79696021` |
| JQL-Edu | `JQL-AI/JQL-Edu-Heads` (heads) and the embedding model its card names as the backbone (Snowflake's arctic-embed-m, version 2.0) | `5cb4a2d26c7961950b0facd1d8a390374027b7e4` (heads) and `95c2741480856aa9666782eb4afe11959938017f` (backbone) |
| FinePDFs-Edu | `HuggingFaceFW/finepdfs_edu_classifier_<code>`, one model per language (list below) | per model |
| FineWeb2-HQ | `epfml/FineWeb-HQ-Classifiers` (heads) and `FacebookAI/xlm-roberta-base` (backbone) | `1940ba2308cf2b12e530690c1eef183985dfcf29` and `e73636d4f797dec63c3081bb6ed5c7b0bb3f2089` |
| FineWeb-Edu classifier | `HuggingFaceFW/fineweb-edu-classifier` | `284663cbb2dabf9bda30d8f8cc49601251ee1631` |
| DCLM fastText (OH+ELI5) | `mlfoundations/fasttext-oh-eli5` | `cd8b714a90f2dbcd3b02cf5fc972e5d7c7f4f107` |
| NeMo Curator edu (Nemotron-4 labels) | `nvidia/nemocurator-fineweb-nemotron-4-edu-classifier` | `842316292abe5bc78521758f5498d6a05adc0f8b` |
| NeMo Curator edu (Mixtral labels) | `nvidia/nemocurator-fineweb-mixtral-edu-classifier` | `768fe255b7e7fbe222014e84cc6576a565516523` |
| Meta-rater reasoning | `opendatalab/meta-rater-reasoning-rating` | `0072a9a83971eb4af6d689dfc64f8f203c45b398` |
| Meta-rater readability | `opendatalab/meta-rater-readability-rating` | `5bfbee1110869ddcbf23447354a7311374784952` |
| Meta-rater cleanliness | `opendatalab/meta-rater-cleanliness-rating` | `4403a9535d47cbc7cc99de26b25099335fe2d9b6` |
| Meta-rater professionalism | `opendatalab/meta-rater-professionalism-rating` | `fc91d4be35fc91de3c65654bb59655ec533a1f61` |
| EAI-Distill 0.5B | `EssentialAI/eai-distill-0.5b` | `39f51ea6e8f1e959961feea0403c69ecfcc8b342` |
| NVIDIA quality classifier (DeBERTa) | `nvidia/quality-classifier-deberta` | `401824e175e89d3243bc376dc4ba262516615d81` |
| Dolma 3 fastText quality | `allenai/dolma3-fasttext-quality-classifier` | `bb89085994fef638ca8dc2ca25169db328e314bb` |

FinePDFs-Edu models used on the evaluation sets (`unknown` is its fallback model, used for fil, kn, ml, sw and te):

| repository | revision |
|---|---|
| `HuggingFaceFW/finepdfs_edu_classifier_als_Latn` | `8f538c2701074964af4941048ec66077b1b6ca1f` |
| `HuggingFaceFW/finepdfs_edu_classifier_arb_Arab` | `78462a34a522fbda15ad583ffa1cd98781571749` |
| `HuggingFaceFW/finepdfs_edu_classifier_azj_Latn` | `6860d1ada3da270bce499e82a8177bfc895a87b0` |
| `HuggingFaceFW/finepdfs_edu_classifier_ben_Beng` | `ca2a231ad78dc1948926cc5aa497240d95eeab40` |
| `HuggingFaceFW/finepdfs_edu_classifier_bul_Cyrl` | `a23563de023ccabecf4c1e0d2210fe3588e1c381` |
| `HuggingFaceFW/finepdfs_edu_classifier_cat_Latn` | `95c70a102e3862dc8708fe7b9e6bde361ed0643a` |
| `HuggingFaceFW/finepdfs_edu_classifier_ces_Latn` | `43c57ff228771a55c4f496a1a680a1a7942463e7` |
| `HuggingFaceFW/finepdfs_edu_classifier_cmn_Hani` | `b1157788380a284bac35fa96fb19654219f4f9b8` |
| `HuggingFaceFW/finepdfs_edu_classifier_dan_Latn` | `c3746210c23a292dc10c74a338addadb80c11d3c` |
| `HuggingFaceFW/finepdfs_edu_classifier_deu_Latn` | `eb2176fc3386be57b525a99fdee295b0580d1307` |
| `HuggingFaceFW/finepdfs_edu_classifier_ekk_Latn` | `5b66ac177115e31f9b304c56408a18dbadde838c` |
| `HuggingFaceFW/finepdfs_edu_classifier_ell_Grek` | `248951027a7d5e7853969863ef9f7628d379271b` |
| `HuggingFaceFW/finepdfs_edu_classifier_fas_Arab` | `e3d91254e276f6fd6415c2aa19705441b064bf92` |
| `HuggingFaceFW/finepdfs_edu_classifier_fin_Latn` | `b6925f941773d6a0716d8b3da09fed3130af16d6` |
| `HuggingFaceFW/finepdfs_edu_classifier_fra_Latn` | `f5050a44f837329386ec89e9c8bb4380aa8765b4` |
| `HuggingFaceFW/finepdfs_edu_classifier_guj_Gujr` | `72c1af11acd9085893fb1a6ae83c33ca49eaedaf` |
| `HuggingFaceFW/finepdfs_edu_classifier_heb_Hebr` | `95d6c2d065d0f6943d607d5b7eb192ed7efb5bf3` |
| `HuggingFaceFW/finepdfs_edu_classifier_hin_Deva` | `dfae45d02aa92adf72842e156d78a107e4f8a82d` |
| `HuggingFaceFW/finepdfs_edu_classifier_hrv_Latn` | `ecd3cb19a72491f32f5ce628f1a3b8cf8b9c0be0` |
| `HuggingFaceFW/finepdfs_edu_classifier_hun_Latn` | `078f8963005ccb83d24a444c8a87b0cf72443e74` |
| `HuggingFaceFW/finepdfs_edu_classifier_ind_Latn` | `ee3e75ff7ddc4c20eea2bf524b6a77b1d224f786` |
| `HuggingFaceFW/finepdfs_edu_classifier_ita_Latn` | `b67b3258ab616e68f2c1b61167e2d9b0673e95b1` |
| `HuggingFaceFW/finepdfs_edu_classifier_jpn_Jpan` | `3478261183e28a6214b82b85bfa47adc2b4a503f` |
| `HuggingFaceFW/finepdfs_edu_classifier_kat_Geor` | `5bf4a56cfa9249479098ce4a250fabb008c1278e` |
| `HuggingFaceFW/finepdfs_edu_classifier_kaz_Cyrl` | `f88ddf006ec15795340263cdc1de07c4d8e1a7d6` |
| `HuggingFaceFW/finepdfs_edu_classifier_kor_Hang` | `2baea20aa8f6640bd61ed879ba528292335f34ba` |
| `HuggingFaceFW/finepdfs_edu_classifier_lit_Latn` | `d4b7281f2b258c2a8057e42c949ab7fbe196c242` |
| `HuggingFaceFW/finepdfs_edu_classifier_lvs_Latn` | `0e1693b8ea51c6dfe76f8116604fc29ccf2119ce` |
| `HuggingFaceFW/finepdfs_edu_classifier_mar_Deva` | `aae791cf53a4959b66ee67f26acc5479aa38d8e5` |
| `HuggingFaceFW/finepdfs_edu_classifier_nld_Latn` | `b307a64f31a3c409d56c450f4f928aad596818e9` |
| `HuggingFaceFW/finepdfs_edu_classifier_nob_Latn` | `34166e84a7fb08a905774c637b6ef2eb7b19cc1d` |
| `HuggingFaceFW/finepdfs_edu_classifier_pol_Latn` | `58ff15760fa995fb7bea2c33a0761af1f9ce66a9` |
| `HuggingFaceFW/finepdfs_edu_classifier_por_Latn` | `d11bd310217f4cdb27522eaadc36202d2df705d4` |
| `HuggingFaceFW/finepdfs_edu_classifier_ron_Latn` | `92abfcae91841b726cc3ab71c3122cbfc77eb7b9` |
| `HuggingFaceFW/finepdfs_edu_classifier_rus_Cyrl` | `23ba4c39b4565af85282c1f1d1bc8479fdaa482d` |
| `HuggingFaceFW/finepdfs_edu_classifier_slk_Latn` | `7c1f7ec820a2d3f6eed0ede492d0417973dedb03` |
| `HuggingFaceFW/finepdfs_edu_classifier_slv_Latn` | `8825bc094303a48239f3b0c40a023689ae5a6c14` |
| `HuggingFaceFW/finepdfs_edu_classifier_spa_Latn` | `60eb17b37f8ea80fff614b50424359e974a43736` |
| `HuggingFaceFW/finepdfs_edu_classifier_srp_Cyrl` | `393b63976a35e266b21b14d91aea89990b4cf8cc` |
| `HuggingFaceFW/finepdfs_edu_classifier_swe_Latn` | `3da33c8970f10076e9da02649f51b10d359b2200` |
| `HuggingFaceFW/finepdfs_edu_classifier_tam_Taml` | `68239bbdb85ab737aaed970d45d313af9f18f051` |
| `HuggingFaceFW/finepdfs_edu_classifier_tha_Thai` | `db02cb1431acfb6baa956e8e09379f20ffd95980` |
| `HuggingFaceFW/finepdfs_edu_classifier_tur_Latn` | `dcdccca95c802edf5f1454ad33620e7342261da2` |
| `HuggingFaceFW/finepdfs_edu_classifier_ukr_Cyrl` | `1353ea90e4f65f8b33dce0570402f8692c982768` |
| `HuggingFaceFW/finepdfs_edu_classifier_unknown` | `d61616d51ece5ce2159936d8ece8aa39a6ef68bc` |
| `HuggingFaceFW/finepdfs_edu_classifier_urd_Arab` | `c2019778c7413f5a299d2ab4978acfadcdbe2030` |
| `HuggingFaceFW/finepdfs_edu_classifier_v2_eng_Latn` | `90ddef285f67230389057c14b2f6bbfeb70d40ea` |
| `HuggingFaceFW/finepdfs_edu_classifier_vie_Latn` | `870370fb168cc1c76549938b13f9cff953def4b7` |
| `HuggingFaceFW/finepdfs_edu_classifier_zsm_Latn` | `5ad2ae90901c74585f0f921ab84fac0a52e3cbbe` |

</details>