BR-Voice-Reasoner / evaluation_protocol.json
zombie1315's picture
Add evaluation protocol and code hashes
13e1e16 verified
Raw History Blame Contribute Delete
2.78 kB
{
"schema": "br-voice-reasoner-evaluation-protocol.v1",
"evaluation_date_utc": "2026-09-02",
"benchmark": {
"name": "VoiceBench",
"upstream_repository": "https://github.com/matthewcym/VoiceBench",
"snapshot_identity": "SHA256 manifest",
"git_revision": null,
"revision_note": "The evaluated local snapshot did not retain Git metadata; release-relevant files are identified by SHA256."
},
"subset_samples": {
"wildvoice": 1000,
"bbh": 1000,
"alpacaeval_full": 636,
"mmsu": 3074,
"openbookqa": 455,
"ifeval": 345,
"advbench": 520,
"commoneval": 200,
"sdqa_usa": 553,
"total": 7783
},
"generation": {
"reasoning_enabled": true,
"temperature": 0.6,
"top_p": 0.95,
"top_k": 20,
"maximum_tokens": 16384,
"seed_policy": "A deterministic per-sample primary seed is derived from the input and base seed 1234.",
"empty_final_policy": {
"primary_attempts": 3,
"maximum_consecutive_fallback_seeds": 8,
"selection": "First non-empty visible final answer; no ranking or best-of selection.",
"samples_entering_seed_fallback": 5,
"samples_recovered": 4,
"samples_exhausted": 1,
"exhausted_record": "No answer.",
"exhausted_scoring": "incorrect"
}
},
"model_judge": {
"requested_model": "openai/gpt-4o-mini",
"model_kind": "alias",
"accessed_utc": "2026-09-02",
"gateway": "OpenRouter",
"provider_only": "OpenAI",
"allow_provider_fallbacks": false,
"judgments_per_sample": 3,
"vote_mode": "three separate requests without a seed",
"temperature": 0.5,
"top_p": 0.95,
"maximum_tokens": 1024
},
"aggregation": "Open-ended ratings are normalized to 0-100. Overall is the arithmetic mean of nine 0-100 subset scores.",
"sha256": {
"main.py": "cb7c86d7149b76d50e3316d09a377fb7cfa6549855ad2922f86b7dd980416f3e",
"evaluate.py": "0b610247b1373f2c256a4f183c2dfcbe7c562b576823cf4ce1396bf036ba1f42",
"src/models/qwen3_omni_thinking.py": "f44bcbc3c9b866c13b4db33ff156026dcf136f7fcfbd223ea006e600986543d8",
"qwen3omni_thinking/run_voicebench.sh": "4f040e0f695a12ee6a9fe4e1164c6fe3e57582160d5afca99d975f6ce7a7226c",
"qwen3omni_thinking/audit_outputs.py": "3ae9335dff1d568d19dc395bdf2b352b6a81685a9128edccdeca3882bdee7fb0",
"qwen3omni_thinking/score_local.sh": "332d701640ee1d76b3d7566e7250c55055546b3aa8186cd443531886635511ac",
"qwen3omni_thinking/score_thinking_mcq.py": "505018a7ae166f71f1a1d19b99887f69adc7fc802f91ceca930dbf2349f858eb",
"qwen3omni_thinking/api_judge_openrouter.py": "5b721566c1095e5dcf83b909f6ea667d127aed2533315b73f0819d168071ec33",
"qwen3omni_thinking/repro_utils.py": "9b36582a40c858f9ae6f5c64f7230dd7cf7db2e868d4f2bd77c446e70ea1a046"
}
}