{ "version": 1, "model": "Source-1", "fitted_on": "a held-out validation split of 9,744 chunks (documents never trained on), against the labels of the teacher, an open-weight 27B LLM scoring the 13-field rubric; nothing here was fitted on evaluation labels", "drop_line": { "line": "toxicity >= 4 or spam_seo >= 3.5 or boilerplate >= 4.5", "meaning": "keep is false when this expression is true on a chunk's or a document's scores. It is the schema's default hard filters (schema_default) with the spam_seo threshold lowered from 4 to 3.5. source1.py applies it by default (drop_line=\"default\" applies schema_default instead).", "schema_default": "toxicity >= 4 or spam_seo >= 4 or boilerplate >= 4.5", "rule": "the candidate with the highest drop recall among those whose keep agreement with the teacher's keep flags is at least 0.95", "reference": "the teacher's keep flags: schema_default applied to the teacher's labels of each validation chunk", "columns": { "keep_agreement": "share of chunks where the line and the teacher make the same keep decision", "drop_recall": "share of the teacher's drops the line also drops", "teacher_drops": "chunks the teacher drops", "caught": "of those, chunks the line drops too", "wrong_drops": "chunks the line drops but the teacher keeps", "line_drops": "chunks the line drops", "drop_share": "share of all chunks the line drops" }, "candidates": [ { "line": "toxicity >= 4 or spam_seo >= 4 or boilerplate >= 4.5", "keep_agreement": 0.96, "drop_recall": 0.7111, "teacher_drops": 1042, "caught": 741, "wrong_drops": 89, "line_drops": 830, "drop_share": 0.0852 }, { "line": "toxicity >= 4 or spam_seo >= 3.5 or boilerplate >= 4.5", "keep_agreement": 0.9638, "drop_recall": 0.7706, "teacher_drops": 1042, "caught": 803, "wrong_drops": 114, "line_drops": 917, "drop_share": 0.0941 }, { "line": "toxicity >= 4 or spam_seo >= 3 or boilerplate >= 4.5", "keep_agreement": 0.929, "drop_recall": 0.8301, "teacher_drops": 1042, "caught": 865, "wrong_drops": 515, "line_drops": 1380, "drop_share": 0.1416 }, { "line": "toxicity >= 4 or spam_seo >= 2.5 or boilerplate >= 4.5", "keep_agreement": 0.8559, "drop_recall": 0.8724, "teacher_drops": 1042, "caught": 909, "wrong_drops": 1271, "line_drops": 2180, "drop_share": 0.2237 }, { "line": "toxicity >= 4 or spam_seo >= 3.5 or boilerplate >= 4", "keep_agreement": 0.9453, "drop_recall": 0.88, "teacher_drops": 1042, "caught": 917, "wrong_drops": 408, "line_drops": 1325, "drop_share": 0.136 } ] }, "offsets_meaning": "per quality score: mean teacher label minus mean Source-1 score over the validation chunks; added to Source-1's score, an offset removes its average lean against the teacher. All are below 0.02 in absolute value, so source1.py reports the model's own scores unless apply_offsets=True.", "offsets": { "educational_value": { "offset": -0.00264963054187195, "chunks": 9744, "mean_teacher": 1.9859, "mean_source1": 1.9886 }, "reasoning_depth": { "offset": -0.005093390804597586, "chunks": 9744, "mean_teacher": 1.5808, "mean_source1": 1.5859 }, "writing_quality": { "offset": -0.009194376026272266, "chunks": 9744, "mean_teacher": 2.9367, "mean_source1": 2.9459 }, "information_density": { "offset": -0.005678571428571644, "chunks": 9744, "mean_teacher": 2.485, "mean_source1": 2.4907 }, "reliability": { "offset": -0.01612294745484366, "chunks": 9744, "mean_teacher": 3.1512, "mean_source1": 3.1673 } } }