Text Classification
Transformers
Safetensors
English
Ukrainian
qwen3_5
image-text-to-text
openjudgement
judgment
classification
structured-output
preview
custom-code
Instructions to use kitaniai/OpenJudgement-4B-Preview with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use kitaniai/OpenJudgement-4B-Preview with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-classification", model="kitaniai/OpenJudgement-4B-Preview")# pip install -U transformers accelerate # Load model directly from transformers import AutoProcessor, AutoModelForMultimodalLM processor = AutoProcessor.from_pretrained("kitaniai/OpenJudgement-4B-Preview") model = AutoModelForMultimodalLM.from_pretrained("kitaniai/OpenJudgement-4B-Preview", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Download reports/final_validation_selection.json from kitaniai/OpenJudgement-4B-Preview: direct link, hf CLI and curl.
- Browser
- Download file 4.65 kB
-
https://huggingface.co/kitaniai/OpenJudgement-4B-Preview/resolve/main/reports/final_validation_selection.json
- Command line
-
hf download hf://kitaniai/OpenJudgement-4B-Preview/reports/final_validation_selection.json
-
curl -L -o final_validation_selection.json https://huggingface.co/kitaniai/OpenJudgement-4B-Preview/resolve/main/reports/final_validation_selection.json
4.65 kB
| { | |
| "selected_step": 700, | |
| "validation_improved": true, | |
| "metric": "uncalibrated macro-source NLL; improvement must exceed 1e-6", | |
| "base": { | |
| "n": 786, | |
| "nll": 0.8177371901208593, | |
| "brier": 0.3928857406198933, | |
| "argmax_agreement": 0.6424936386768448, | |
| "soft_target_fraction": 0.10941475826972011, | |
| "by_source": { | |
| "KSE-RESEARCH-Group/UAReviews": { | |
| "n": 200, | |
| "nll": 0.3008418807003928, | |
| "brier": 0.12895151835801125, | |
| "argmax_agreement": 0.915, | |
| "soft_target_fraction": 0.0 | |
| }, | |
| "nvidia/Aegis-AI-Content-Safety-Dataset-2.0": { | |
| "n": 200, | |
| "nll": 0.5550802189421661, | |
| "brier": 0.3796225903166879, | |
| "argmax_agreement": 0.695, | |
| "soft_target_fraction": 0.0 | |
| }, | |
| "nvidia/HelpSteer3": { | |
| "n": 200, | |
| "nll": 1.377823997184949, | |
| "brier": 0.5669972128285794, | |
| "argmax_agreement": 0.43, | |
| "soft_target_fraction": 0.255 | |
| }, | |
| "openai/coval": { | |
| "n": 36, | |
| "nll": 1.3788071523285632, | |
| "brier": 0.17675586768508195, | |
| "argmax_agreement": 0.5, | |
| "soft_target_fraction": 0.9722222222222222 | |
| }, | |
| "rabuahmad/climatecheck": { | |
| "n": 150, | |
| "nll": 0.9757010305711111, | |
| "brier": 0.5822047772661176, | |
| "argmax_agreement": 0.5266666666666666, | |
| "soft_target_fraction": 0.0 | |
| } | |
| }, | |
| "by_kind": { | |
| "choice": { | |
| "n": 468, | |
| "nll": 0.7127335941312066, | |
| "brier": 0.3476115614992396, | |
| "argmax_agreement": 0.7115384615384616, | |
| "soft_target_fraction": 0.0811965811965812 | |
| }, | |
| "noul": { | |
| "n": 200, | |
| "nll": 0.5550802189421661, | |
| "brier": 0.3796225903166879, | |
| "argmax_agreement": 0.695, | |
| "soft_target_fraction": 0.0 | |
| }, | |
| "score": { | |
| "n": 118, | |
| "nll": 1.6793734372301579, | |
| "brier": 0.594927654934361, | |
| "argmax_agreement": 0.2796610169491525, | |
| "soft_target_fraction": 0.4067796610169492 | |
| } | |
| }, | |
| "macro_source_nll": 0.9176508559454364, | |
| "note": "Accuracy is argmax agreement with annotation targets; ties use first-index argmax. Soft targets represent finite human votes or subjective preferences, not objective truth. Brier is sum of squared candidate-probability errors; its scale varies with candidate count." | |
| }, | |
| "candidate": { | |
| "n": 786, | |
| "nll": 0.5571027568779557, | |
| "brier": 0.2618064378625637, | |
| "argmax_agreement": 0.7544529262086515, | |
| "soft_target_fraction": 0.10941475826972011, | |
| "by_source": { | |
| "KSE-RESEARCH-Group/UAReviews": { | |
| "n": 200, | |
| "nll": 0.16488824965745785, | |
| "brier": 0.08122107897078124, | |
| "argmax_agreement": 0.955, | |
| "soft_target_fraction": 0.0 | |
| }, | |
| "nvidia/Aegis-AI-Content-Safety-Dataset-2.0": { | |
| "n": 200, | |
| "nll": 0.32759895917618914, | |
| "brier": 0.2073110988938831, | |
| "argmax_agreement": 0.855, | |
| "soft_target_fraction": 0.0 | |
| }, | |
| "nvidia/HelpSteer3": { | |
| "n": 200, | |
| "nll": 1.0104743724896212, | |
| "brier": 0.42899340445147355, | |
| "argmax_agreement": 0.52, | |
| "soft_target_fraction": 0.255 | |
| }, | |
| "openai/coval": { | |
| "n": 36, | |
| "nll": 1.1635381012599184, | |
| "brier": 0.07591942035684947, | |
| "argmax_agreement": 0.6666666666666666, | |
| "soft_target_fraction": 0.9722222222222222 | |
| }, | |
| "rabuahmad/climatecheck": { | |
| "n": 150, | |
| "nll": 0.6360205266404155, | |
| "brier": 0.39694429709267093, | |
| "argmax_agreement": 0.6866666666666666, | |
| "soft_target_fraction": 0.0 | |
| } | |
| }, | |
| "by_kind": { | |
| "choice": { | |
| "n": 468, | |
| "nll": 0.450252531502061, | |
| "brier": 0.2214117972369437, | |
| "argmax_agreement": 0.8141025641025641, | |
| "soft_target_fraction": 0.0811965811965812 | |
| }, | |
| "noul": { | |
| "n": 200, | |
| "nll": 0.32759895917618914, | |
| "brier": 0.2073110988938831, | |
| "argmax_agreement": 0.855, | |
| "soft_target_fraction": 0.0 | |
| }, | |
| "score": { | |
| "n": 118, | |
| "nll": 1.3698711044734813, | |
| "brier": 0.5143806718161742, | |
| "argmax_agreement": 0.3474576271186441, | |
| "soft_target_fraction": 0.4067796610169492 | |
| } | |
| }, | |
| "macro_source_nll": 0.6605040418447203, | |
| "note": "Accuracy is argmax agreement with annotation targets; ties use first-index argmax. Soft targets represent finite human votes or subjective preferences, not objective truth. Brier is sum of squared candidate-probability errors; its scale varies with candidate count." | |
| }, | |
| "fallback": null | |
| } |