Source-1 / calibration.json
msmth's picture
Source-1: weights, loader and model card
20e20c3 verified
Raw History Blame Contribute Delete
4.04 kB
{
"version": 1,
"model": "Source-1",
"fitted_on": "a held-out validation split of 9,744 chunks (documents never trained on), against the labels of the teacher, an open-weight 27B LLM scoring the 13-field rubric; nothing here was fitted on evaluation labels",
"drop_line": {
"line": "toxicity >= 4 or spam_seo >= 3.5 or boilerplate >= 4.5",
"meaning": "keep is false when this expression is true on a chunk's or a document's scores. It is the schema's default hard filters (schema_default) with the spam_seo threshold lowered from 4 to 3.5. source1.py applies it by default (drop_line=\"default\" applies schema_default instead).",
"schema_default": "toxicity >= 4 or spam_seo >= 4 or boilerplate >= 4.5",
"rule": "the candidate with the highest drop recall among those whose keep agreement with the teacher's keep flags is at least 0.95",
"reference": "the teacher's keep flags: schema_default applied to the teacher's labels of each validation chunk",
"columns": {
"keep_agreement": "share of chunks where the line and the teacher make the same keep decision",
"drop_recall": "share of the teacher's drops the line also drops",
"teacher_drops": "chunks the teacher drops",
"caught": "of those, chunks the line drops too",
"wrong_drops": "chunks the line drops but the teacher keeps",
"line_drops": "chunks the line drops",
"drop_share": "share of all chunks the line drops"
},
"candidates": [
{
"line": "toxicity >= 4 or spam_seo >= 4 or boilerplate >= 4.5",
"keep_agreement": 0.96,
"drop_recall": 0.7111,
"teacher_drops": 1042,
"caught": 741,
"wrong_drops": 89,
"line_drops": 830,
"drop_share": 0.0852
},
{
"line": "toxicity >= 4 or spam_seo >= 3.5 or boilerplate >= 4.5",
"keep_agreement": 0.9638,
"drop_recall": 0.7706,
"teacher_drops": 1042,
"caught": 803,
"wrong_drops": 114,
"line_drops": 917,
"drop_share": 0.0941
},
{
"line": "toxicity >= 4 or spam_seo >= 3 or boilerplate >= 4.5",
"keep_agreement": 0.929,
"drop_recall": 0.8301,
"teacher_drops": 1042,
"caught": 865,
"wrong_drops": 515,
"line_drops": 1380,
"drop_share": 0.1416
},
{
"line": "toxicity >= 4 or spam_seo >= 2.5 or boilerplate >= 4.5",
"keep_agreement": 0.8559,
"drop_recall": 0.8724,
"teacher_drops": 1042,
"caught": 909,
"wrong_drops": 1271,
"line_drops": 2180,
"drop_share": 0.2237
},
{
"line": "toxicity >= 4 or spam_seo >= 3.5 or boilerplate >= 4",
"keep_agreement": 0.9453,
"drop_recall": 0.88,
"teacher_drops": 1042,
"caught": 917,
"wrong_drops": 408,
"line_drops": 1325,
"drop_share": 0.136
}
]
},
"offsets_meaning": "per quality score: mean teacher label minus mean Source-1 score over the validation chunks; added to Source-1's score, an offset removes its average lean against the teacher. All are below 0.02 in absolute value, so source1.py reports the model's own scores unless apply_offsets=True.",
"offsets": {
"educational_value": {
"offset": -0.00264963054187195,
"chunks": 9744,
"mean_teacher": 1.9859,
"mean_source1": 1.9886
},
"reasoning_depth": {
"offset": -0.005093390804597586,
"chunks": 9744,
"mean_teacher": 1.5808,
"mean_source1": 1.5859
},
"writing_quality": {
"offset": -0.009194376026272266,
"chunks": 9744,
"mean_teacher": 2.9367,
"mean_source1": 2.9459
},
"information_density": {
"offset": -0.005678571428571644,
"chunks": 9744,
"mean_teacher": 2.485,
"mean_source1": 2.4907
},
"reliability": {
"offset": -0.01612294745484366,
"chunks": 9744,
"mean_teacher": 3.1512,
"mean_source1": 3.1673
}
}
}