MyeongHoJeong commited on
Commit
0f14d00
·
verified ·
1 Parent(s): 50ad5a9

Add files using upload-large-folder tool

Browse files
Files changed (42) hide show
  1. .gitattributes +6 -0
  2. docs/assets/00-benchmark-card.png +3 -0
  3. docs/assets/02-latency-vs-qwen.png +3 -0
  4. docs/assets/05-gain-over-base.png +3 -0
  5. docs/assets/06-vs-jev.png +3 -0
  6. model-00001-of-00004.safetensors +3 -0
  7. model-00002-of-00004.safetensors +3 -0
  8. model-00003-of-00004.safetensors +3 -0
  9. model-00004-of-00004.safetensors +3 -0
  10. server/benchmarks/DECISION_BENCHMARK_SELECTION.md +77 -0
  11. server/benchmarks/METRIC_VALIDATION.md +71 -0
  12. server/benchmarks/PUBLIC_DATASETS.md +99 -0
  13. server/benchmarks/README.md +144 -0
  14. server/benchmarks/data/jevbench-easy/LICENSE +21 -0
  15. server/benchmarks/data/jevbench-easy/THIRD-PARTY.md +51 -0
  16. server/benchmarks/data/jevbench-easy/public.jsonl +48 -0
  17. server/benchmarks/data/jevbench-easy/public.manifest.json +95 -0
  18. server/benchmarks/data/jevbench-hard/LICENSE +21 -0
  19. server/benchmarks/data/jevbench-hard/THIRD-PARTY.md +51 -0
  20. server/benchmarks/data/jevbench-hard/public.jsonl +0 -0
  21. server/benchmarks/data/jevbench-hard/public.manifest.json +109 -0
  22. server/benchmarks/data/jevbench-original/LICENSE +21 -0
  23. server/benchmarks/data/jevbench-original/THIRD-PARTY.md +51 -0
  24. server/benchmarks/data/jevbench-original/public.manifest.json +101 -0
  25. server/benchmarks/run_matrix.py +277 -0
  26. server/benchmarks/verify_metrics.py +140 -0
  27. server/examples/request.json +13 -0
  28. server/examples/smoke.py +49 -0
  29. server/jev_adapter/benchmarks/__init__.py +1 -0
  30. server/jev_adapter/benchmarks/compare.py +72 -0
  31. server/jev_adapter/benchmarks/data.py +217 -0
  32. server/jev_adapter/benchmarks/jevbench.py +292 -0
  33. server/jev_adapter/benchmarks/metrics.py +326 -0
  34. server/jev_adapter/benchmarks/prepare.py +227 -0
  35. server/jev_adapter/benchmarks/run.py +441 -0
  36. server/tests/test_benchmark_data.py +220 -0
  37. server/tests/test_disconnect_cleanup.py +150 -0
  38. server/tests/test_main.py +368 -0
  39. server/tests/test_service.py +324 -0
  40. server/tests/test_sglang.py +409 -0
  41. tekken.json +3 -0
  42. tokenizer.json +3 -0
.gitattributes CHANGED
@@ -33,3 +33,9 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ tekken.json filter=lfs diff=lfs merge=lfs -text
37
+ tokenizer.json filter=lfs diff=lfs merge=lfs -text
38
+ docs/assets/05-gain-over-base.png filter=lfs diff=lfs merge=lfs -text
39
+ docs/assets/00-benchmark-card.png filter=lfs diff=lfs merge=lfs -text
40
+ docs/assets/06-vs-jev.png filter=lfs diff=lfs merge=lfs -text
41
+ docs/assets/02-latency-vs-qwen.png filter=lfs diff=lfs merge=lfs -text
docs/assets/00-benchmark-card.png ADDED

Git LFS Details

  • SHA256: 2300b40c53e596eacf93f152375c68bd78a9d77c03fb15ec17e5a76ca9f632bd
  • Pointer size: 131 Bytes
  • Size of remote file: 270 kB
docs/assets/02-latency-vs-qwen.png ADDED

Git LFS Details

  • SHA256: 7313f2b0645ee401d082572986dcee0c08d3daed1ae79fa650cfa685570ad7e2
  • Pointer size: 131 Bytes
  • Size of remote file: 115 kB
docs/assets/05-gain-over-base.png ADDED

Git LFS Details

  • SHA256: f3211f7d1cbc624b0f558e7f9e7edac3ea51a381287cb7409719a12d3f11edaa
  • Pointer size: 131 Bytes
  • Size of remote file: 144 kB
docs/assets/06-vs-jev.png ADDED

Git LFS Details

  • SHA256: 4d41f1b782502403109c4074e7705fc42f18dfb9fc865b5aca5771f4c4991264
  • Pointer size: 131 Bytes
  • Size of remote file: 223 kB
model-00001-of-00004.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:41c1d711d80f96f9cd5ca8eb8b0b4375addd3174b2021bd8ab72dc1a9c9c7500
3
+ size 4999724576
model-00002-of-00004.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:57d50fd02ebafbb491d35d445c8279dda89ad8ad5034edf7637971564857b48d
3
+ size 4999820896
model-00003-of-00004.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:03616016aafb162714f7265f9e1a192f72bdc109e4d80d079c593fd0756c1cba
3
+ size 4915917688
model-00004-of-00004.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a77ea6878393937e46cead80bdfda98e090020cd93b746b89c38a5a81fbf823a
3
+ size 2920659992
server/benchmarks/DECISION_BENCHMARK_SELECTION.md ADDED
@@ -0,0 +1,77 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # JEV 형태의 판단 성능을 위한 추가 벤치 선택
2
+
3
+ 확인일: 2026-09-21. **현재 비교의 중심은 KEV의 고정 평가 세트와 JevBench**다.
4
+ 추가 공개 데이터는 아래 두 개를 추천한다. 이 문서는 선정 근거이며,
5
+ CLINC150과 ContractNLI의 변환·실행·추가 지표는 아직 통합하지 않았다.
6
+ 모두 텍스트 입력에서 정해진 선택지의 확률만 읽어 평가할 수 있다.
7
+
8
+ | 우선순위 | 데이터 | 측정하려는 판단 | 공식 테스트 규모 | 라이선스 |
9
+ |---|---|---|---|---|
10
+ | 1 | CLINC150 / oos-eval | 요청 라우팅과 지원 범위 밖 요청 거절 | 지원 범위 4,500 + 범위 밖 1,000 = 5,500문항 | CC BY 3.0 |
11
+ | 2 | ContractNLI | 문서가 조건을 지지하는지, 반박하는지, 근거가 없는지 | 123개 계약 × 17개 명제 = 2,091판단 | CC BY 4.0 |
12
+
13
+ ## CLINC150: 라우팅과 범위 밖 요청
14
+
15
+ 150개 intent와 `oos` 한 개를 합쳐 **151개 선택지**를 사용한다.
16
+ 공식 `data/data_full.json`은 intent마다 train 100개·validation 20개·test 30개이며,
17
+ 별도 OOS 분할은 train 100개·validation 100개·test 1,000개다.
18
+ 따라서 검증 세트는 총 3,100개, 테스트는 5,500개다.
19
+ 영어 단일 intent 요청을 사람이 작성·패러프레이즈한 데이터다.
20
+ [공식 설명](https://github.com/clinc/oos-eval/tree/828f8093932c8fe6ca7936c3d2e52903b1c523de)
21
+ [라이선스](https://github.com/clinc/oos-eval/blob/828f8093932c8fe6ca7936c3d2e52903b1c523de/LICENSE)
22
+
23
+ 권장 매핑은 `state=사용자 발화`, `choice=150개 intent + oos`다.
24
+ 공식 intent 이름의 밑줄을 공백으로 바꾸는 정도의 결정적 설명을 쓰고,
25
+ 테스트 발화나 정답을 이용해 선택지 설명을 만들지 않는다.
26
+ 모델마다 동일한 후보 목록·순서를 사용하며, 후보 일부를 정답에 맞춰 제거하지 않는다.
27
+ 151개 선택지는 기존 어댑터의 255개 한도 내에 있다.
28
+
29
+ 통합할 때 추가할 지표:
30
+
31
+ - 지원 범위 150개에 대한 accuracy·macro F1, 전체 151개 macro F1.
32
+ - `p(oos)`를 점수, OOS를 양성으로 정한 AUROC·AUPRC.
33
+ - FPR@95% OOS TPR: OOS의 95%를 탐지할 때 정상 요청을 잘못 거절하는 비율.
34
+ - 고정 임계값에서 OOS 탐지율과 정상 요청 오거절률, 수락한 요청의 정확도·coverage.
35
+ - Brier·NLL·ECE와 요청별 p50/p95 지연. 후보 151개의 비용을 함께 명시.
36
+
37
+ 운영 임계값을 선택한다면 validation만 사용하고 test에는 고정해서 적용한다.
38
+ 가중치 학습 없이 비교하되, 임계값 조정 여부는 별도로 기록한다.
39
+ OOS는 **지원하지 않는 intent**라는 뜻이다. 입력 근거가 부족해서 답을 알 수 없는
40
+ KEV `unknowable`과는 다른 능력을 측정하므로 둘을 합산하지 않는다.
41
+
42
+ 직접 검증한 원본: 커밋 `828f8093932c8fe6ca7936c3d2e52903b1c523de`,
43
+ [`data/data_full.json`](https://github.com/clinc/oos-eval/blob/828f8093932c8fe6ca7936c3d2e52903b1c523de/data/data_full.json),
44
+ SHA256 `36923c3705a59e08fe9c3883d8bc2dd966ef93e22cb78ac41171782a698d56e0`.
45
+ 다른 `small`·`plus` 변형과 혼합하지 않는다.
46
+
47
+ ## ContractNLI: 명시된 문서 근거에 따른 판단
48
+
49
+ 607개의 NDA에 동일한 17개 명제를 사람이 주석했다.
50
+ 분할은 train 423개·development 61개·test 123개 계약이다.
51
+ 정답은 `Entailment`, `Contradiction`, `NotMentioned`의 세 종류다.
52
+ 예외에 의한 부정, 문서에 없는 내용, 긴 문서의 근거를 찾는 판단을 점검하기 좋다.
53
+ [공식 데이터·스키마·라이선스](https://stanfordnlp.github.io/contract-nli/)
54
+ [원 논문의 분할 표](https://aclanthology.org/2021.findings-emnlp.164.pdf)
55
+
56
+ `state=전체 계약 텍스트`, 각 명제를 `choice` 질문으로 바꾼다.
57
+ 출력은 세 선택지의 확률이며, 근거 문장 생성은 요구하지 않는다.
58
+ `NotMentioned`를 `false`로 합치면 반박과 근거 부재를 구분할 수 없으므로
59
+ 기본 비교에는 `noul` 대신 3-way `choice`를 권장한다.
60
+ 원본의 정답 evidence span은 입력에 넣지 않는다.
61
+
62
+ 권장 지표는 3개 레이블 macro F1·accuracy, 명제별 성능, Brier·NLL·ECE,
63
+ `NotMentioned` precision/recall이다. 계약 단위로 신뢰구간을 계산해 한 문서의
64
+ 17개 질문을 독립 표본으로 취급하지 않는다.
65
+ 명제당 지연과 계약 전체 17개 질문의 지연을 함께 보고한다.
66
+
67
+ 긴 계약을 임의로 잘라 평가하지 않는다. 각 모델의 tokenizer로 입력 길이를 먼저
68
+ 측정하고, 공통 context 한도 밖의 문서는 실패·제외 수를 명시한다.
69
+ 본문 전체 판단만 수행한 결과를 원 논문의 evidence identification까지 포함한
70
+ 전체 과제 결과와 동일하다고 표현하지 않는다.
71
+
72
+ ## 해석 범위
73
+
74
+ 두 데이터 모두 공개된 지 오래되어 최신 Qwen·Mistral의 사전학습 오염을 배제할 수 없다.
75
+ 학습하지 않은 모델을 평가한다는 것은 **이번 실험에서 추가 튜닝하지 않는다**는 뜻이다.
76
+ 공개 테스���를 처음 접했다는 보장은 아니다. 영어 중심 결과를 한국어 운영 성능으로
77
+ 일반화하지 않으며, 모델 비교에서는 동일 입력·동일 분할·동일 판단 형식을 유지한다.
server/benchmarks/METRIC_VALIDATION.md ADDED
@@ -0,0 +1,71 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # KEV metric parity validation
2
+
3
+ On 2026-09-21, the adapter's scalar quality metrics were compared numerically
4
+ with the original KEV implementations at commit
5
+ `4f8110a3f8620cc3a182ae9a708e4398492c4b1a`.
6
+ All common scalar metrics matched within `1e-12` absolute error. The largest
7
+ observed difference was `8.881784197001252e-16` (score MAE). Complete contrastive
8
+ pair metrics matched exactly.
9
+
10
+ This validates the metric calculations. It does not validate H200 execution,
11
+ inference probabilities, model accuracy, or serving latency.
12
+
13
+ ## Method
14
+
15
+ The optional [verify_metrics.py](verify_metrics.py) script reads these original
16
+ files from a local KEV checkout and verifies their full SHA256 before execution:
17
+
18
+ | File | Extracted functions | SHA256 |
19
+ |---|---|---|
20
+ | `kev/evaluate.py` | `ece` | `1b20e3f9edf417aa8dae924b1526e52f74b710cadf7213c5ec68334f6e7f8fe1` |
21
+ | `kev/benchmark.py` | `coverage_at_error`, `metrics` | `4192ec3b26b065452f84bde38a091e6854a28fe185d2e0f39a0c91b2efe69df7` |
22
+ | `kev/contrastive.py` | `paired_flip` | `cbb979aa5d40265ad0e64695f94b281d91751fa405811ddfa8212fede111edcf` |
23
+
24
+ Python AST extraction keeps only those function definitions. KEV's package,
25
+ PyTorch, Transformers and model-loading code are not imported. The audit uses
26
+ NumPy in its own optional environment; NumPy is not an adapter dependency.
27
+ The original successful audit used NumPy `2.3.5`.
28
+
29
+ Input generation uses `numpy.random.default_rng(483)` for 100 batches of 50
30
+ rows, totaling 5,000 rows. Rows mix choice, boolean and ordinal score questions,
31
+ with 2–10 options. Probabilities come from Dirichlet distributions. The first
32
+ 10 rows in each batch additionally cover binary probability endpoints and
33
+ decimal boundaries, including zero and one. Both implementations receive the
34
+ same rows in the same order, including confidence ties.
35
+
36
+ Compared metrics include accuracy, NLL, multiclass Brier score, ten-bin ECE,
37
+ mean confidence, confidence bias, confident errors, coverage/accuracy at 0.9,
38
+ coverage at 1%/5% empirical error, score MAE and ranked probability score.
39
+ Ten complete two-sibling pairs additionally check relevant and invariant pair
40
+ metrics. Five pairs have changed gold labels and five have unchanged labels.
41
+
42
+ ## Reproduce
43
+
44
+ Use a Python environment that already has NumPy installed:
45
+
46
+ ```bash
47
+ git clone https://github.com/jaredpalmer/kev.git /tmp/kev-reference
48
+ git -C /tmp/kev-reference checkout 4f8110a3f8620cc3a182ae9a708e4398492c4b1a
49
+ python benchmarks/verify_metrics.py --kev-root /tmp/kev-reference
50
+ ```
51
+
52
+ The JSON output includes the reference hashes, current adapter metric-code hash,
53
+ NumPy version, sample counts, per-metric maximum differences and pair results.
54
+ A source mismatch or numerical discrepancy exits unsuccessfully. Reference
55
+ files are checked even if the local checkout has uncommitted changes.
56
+
57
+ ## Scope differences
58
+
59
+ - Accuracy headlines use clean question rows, not all submitted records.
60
+ Source `unknowable` is excluded from accuracy and scored for confidence.
61
+ - Incomplete pairs are counted explicitly for smoke subsets; KEV's original
62
+ pair function rejects incomplete pairs.
63
+ - The adapter omits meaningless unknowable accuracy from grouped reports;
64
+ KEV's original code includes it in some diagnostic subreports.
65
+ - This audit compares common scalar metrics and complete pairs. It does not
66
+ certify every report field, input conversion, HTTP behavior, or SemIf's
67
+ separate family-balanced aggregation.
68
+
69
+ Original definitions: [KEV benchmark.py](https://github.com/jaredpalmer/kev/blob/4f8110a3f8620cc3a182ae9a708e4398492c4b1a/kev/benchmark.py),
70
+ [evaluate.py](https://github.com/jaredpalmer/kev/blob/4f8110a3f8620cc3a182ae9a708e4398492c4b1a/kev/evaluate.py),
71
+ [contrastive.py](https://github.com/jaredpalmer/kev/blob/4f8110a3f8620cc3a182ae9a708e4398492c4b1a/kev/contrastive.py).
server/benchmarks/PUBLIC_DATASETS.md ADDED
@@ -0,0 +1,99 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Public decision benchmarks beyond the KEV suites
2
+
3
+ Verified on 2026-09-21. JevBench's 231 public decisions are integrated into the preparation tool and default H200 matrix. The other entries below are research candidates, not automatically downloaded.
4
+
5
+ The primary evaluation combines KEV and **JevBench's 231 public decisions** for explicit rule following and probability quality. **Typed Decisions' 400 test cases / 2,000 questions** is a secondary shared-state diagnostic because its labels measure teacher agreement. Image support is not a selection requirement. Additional objective intent/OOS and evidence-grounded candidates are assessed in [DECISION_BENCHMARK_SELECTION.md](DECISION_BENCHMARK_SELECTION.md). Keep each suite's results separate.
6
+
7
+ ## Official TypeSafe data versus community benchmarks
8
+
9
+ No downloadable, labeled benchmark released by TypeSafe itself was located in its [official documentation](https://docs.typesafe.ai/introduction), [announcement](https://typesafe.ai/blog/introducing-system-one-models-and-jev), or [public GitHub organization](https://github.com/typesafe-ai). This is a search finding, not proof that no such data exists. TypeSafe publishes an [LLM comparison adapter](https://github.com/typesafe-ai/system-one-adapter-python), but an adapter is not an evaluation dataset.
10
+
11
+ All Jev-specific datasets below are independent community work. A repository name containing `Jev` or `TypeSafe` does not make it official. The community [TypeSafeAI playground](https://github.com/TypeSafeAI/typesafe-playground) explicitly says it is independent and its examples are not validated accuracy benchmarks.
12
+
13
+ KEV's current [4B model card](https://github.com/jaredpalmer/kev/blob/4f8110a3f8620cc3a182ae9a708e4398492c4b1a/docs/model-cards/kev-4b.md) already reports:
14
+
15
+ - `decision-v7`, `transfer-v4`, and newer `transfer-v9` evaluations;
16
+ - SemIf's 144 authored examples;
17
+ - scienthoon's 900 support tickets;
18
+ - ekzhang's 1,000-question MMLU-Pro sample.
19
+
20
+ SemIf, scienthoon, and that MMLU-Pro sample are therefore **external to KEV's original training suite, but not additional benchmarks that KEV has never run**. The referenced KEV suites are text based; they do not measure the vision encoder. “Never trained” in KEV's terminology concerns its own fine-tuning sources and is not evidence that a foundation model or Jev never saw the data.
21
+
22
+ ## 1. JevBench public subset — first addition
23
+
24
+ [JevBench](https://github.com/fstandhartinger/jevbench) covers routing, policy checks, intent, enum extraction, ordinal severity, answer judging, and harder reasoning over supplied state. It explicitly disclaims TypeSafe affiliation. The full published evaluation has 534 decisions, while the downloadable public subset has **231**. Use a separate public-subset result rather than claiming reproduction of the full leaderboard.
25
+
26
+ Pinned repository: [`fd51755eb0c0b546ca206d764faf3302feca913e`](https://github.com/fstandhartinger/jevbench/tree/fd51755eb0c0b546ca206d764faf3302feca913e). Counts below were checked by parsing the pinned JSONL bytes.
27
+
28
+ | Download | Decisions | Choice | Noul | Score | License |
29
+ |---|---:|---:|---:|---:|---|
30
+ | [original.jsonl](https://raw.githubusercontent.com/fstandhartinger/jevbench/fd51755eb0c0b546ca206d764faf3302feca913e/datasets/public/original.jsonl) | 72 | 36 | 24 | 12 | MIT |
31
+ | [easy.jsonl](https://raw.githubusercontent.com/fstandhartinger/jevbench/fd51755eb0c0b546ca206d764faf3302feca913e/datasets/public/easy.jsonl) | 48 | 36 | 12 | 0 | MIT |
32
+ | [hard.jsonl](https://raw.githubusercontent.com/fstandhartinger/jevbench/fd51755eb0c0b546ca206d764faf3302feca913e/datasets/public/hard.jsonl) | 111 | 67 | 38 | 6 | MIT |
33
+ | Total | 231 | 139 | 74 | 18 | |
34
+
35
+ SHA-256 checksums, in the same order:
36
+
37
+ ```text
38
+ 5c2414edb3006b8bfcb70fda433f0f9ca015759433849f8d3104328a1f7c4180
39
+ 231df3c2c8e88a1a8c137ebe85de96ba70fabd330849098ac7b3c52c70b7172b
40
+ 89e9e6becb33ed88c1de7d42dcc87531b2fb64cfaef4e1986faf7c37b3f80ebb
41
+ ```
42
+
43
+ Each row has `state`, one `question`, ordered `labels`, `expected`, `id`, `group`, and `provenance`. Send only `state` and `question`. Noul gold is `"no"`/`"yes"`; Score gold is an integer rubric index. The hard tier's `provenance` can contain the answer rationale and `gold_probs`: neither belongs in model input. Some hard cases have exact reference probabilities, unlike ordinary hard-label classification. The [hard-tier protocol](https://github.com/fstandhartinger/jevbench/blob/fd51755eb0c0b546ca206d764faf3302feca913e/datasets/HARD-TIER.md) describes synthetic authoring, cross-model review, and the freeze.
44
+
45
+ The 303 unavailable decisions comprise 157 deliberately private items and 146 imported items whose task text the maintainers do not redistribute. Hashes and aggregate results do not reconstruct those inputs or grant redistribution rights. See [third-party terms](https://github.com/fstandhartinger/jevbench/blob/fd51755eb0c0b546ca206d764faf3302feca913e/THIRD-PARTY.md) and the [license](https://github.com/fstandhartinger/jevbench/blob/fd51755eb0c0b546ca206d764faf3302feca913e/LICENSE).
46
+
47
+ Implementation notes: retain paraphrase `group` identifiers for grouped uncertainty estimates. Label our adapter's output as **token-logprob-derived**, since JevBench's published ordinary-LLM adapter uses verbalized probabilities. Its leaderboard also applies assumed latency adjustments to some endpoints; our H200 results should report actual local measurements without adopting those adjustments.
48
+
49
+ ## 2. LocalLLaMA/typed-decisions — shared-state and soft labels
50
+
51
+ The [dataset card](https://huggingface.co/datasets/LocalLLaMA/typed-decisions/blob/ea9306458d6e9563628369a3d1e72e362fb381d2/README.md) specifies four workflows: agent traces, customer service, invoices, and security incidents. Each has 300 train and 100 test cases, with five Choice/Noul/Score questions per case. Evaluate **400 test cases / 2,000 questions**. The `all` configuration repeats the four workflow configurations; do not concatenate both.
52
+
53
+ - Dataset revision: `ea9306458d6e9563628369a3d1e72e362fb381d2`.
54
+ - [Combined test Parquet](https://huggingface.co/datasets/LocalLLaMA/typed-decisions/resolve/ea9306458d6e9563628369a3d1e72e362fb381d2/all/test-00000-of-00001.parquet), SHA-256 `4f294f218ea1da27f3efef936359389c62ea4d3973a41457732990f1d31b647c`.
55
+ - License: Apache-2.0, as declared in the pinned card.
56
+ - Parse JSON-string columns `state`, `questions`, and `gold`. Only the first two are model input; `factors` reveals the generating variables.
57
+
58
+ Gold averages three teacher-model probability samples. Consequently, this measures **agreement with a synthetic teacher**, not independently established correctness. Report hard-label agreement separately from soft-target Brier/KL and Score MAE. The teacher can be wrong, so a stronger model can receive a lower score. Its published Jev result is measured on this test set; its specialist baselines were trained on the corresponding training workflows.
59
+
60
+ ## 3. NPC addressee benchmark — application behavior
61
+
62
+ [wondertwins/jev-benchmark](https://github.com/wondertwins/jev-benchmark/tree/1c2509ac7d5508b8a7ce00ae4df6d7652d05de8b) provides **79 hand-labeled utterances**, each with clean, punctuation-free STT, and misheard-name variants. That is 237 case variants, not 237 independent utterances. Its strict addressee metrics exclude four explicitly ambiguous originals. Questions combine per-NPC Noul, primary-addressee Choice, intent, and urgency Score.
63
+
64
+ - Pin: `1c2509ac7d5508b8a7ce00ae4df6d7652d05de8b`.
65
+ - Fixtures: [npcaddress/dataset.py](https://github.com/wondertwins/jev-benchmark/blob/1c2509ac7d5508b8a7ce00ae4df6d7652d05de8b/npcaddress/dataset.py).
66
+ - Request construction: [npcaddress/questions.py](https://github.com/wondertwins/jev-benchmark/blob/1c2509ac7d5508b8a7ce00ae4df6d7652d05de8b/npcaddress/questions.py).
67
+ - [MIT license](https://github.com/wondertwins/jev-benchmark/blob/1c2509ac7d5508b8a7ce00ae4df6d7652d05de8b/LICENSE).
68
+
69
+ The same repository contains 30 chess positions and 25 mate-in-one puzzles, with cached Stockfish references. Treat chess as a separate reasoning stress test. Keep raw-board, code-enriched, and filtered hybrid variants distinct. These inputs are text/structured state, not screenshots.
70
+
71
+ ## 4. Prompt-injection decisions — domain-specific addition
72
+
73
+ [jev-sec-bench](https://github.com/Gaurav-Gosain/jev-sec-bench/tree/fdb16b94d37535db9bad77f8ef0faa971bd7d69a) evaluated all **662** public `deepset/prompt-injections` messages. The original data has 546 train and 116 test rows. Reproducing the 662-row community experiment requires explicitly naming the combined split; it must not be reported as 662 held-out test examples. Its policy context matters: the source labels concern a news assistant, not an unrestricted chatbot.
74
+
75
+ [Dataset and Apache-2.0 declaration](https://huggingface.co/datasets/deepset/prompt-injections/tree/4f61ecb038e9c3fb77e21034b22511b523772cdd), pinned revision `4f61ecb038e9c3fb77e21034b22511b523772cdd`. Schema: `text`, binary `label`; map to Noul using the documented deployment context. Report F1, false positives/negatives, ROC-AUC, Brier, and latency. This is an application-specific classification benchmark, not proof of general guardrail security.
76
+
77
+ ## Lower-priority or overlapping sources
78
+
79
+ [AbdelStark/jev-benchmarks](https://github.com/AbdelStark/jev-benchmarks/tree/0d610cc53e79bcbec691312b0c4adb4a0e371642) has a reproducible 300-example BTZSC pilot: AG News, Banking77, and Emotion. Those source families already occur in KEV training/evaluation, so the new harness is useful, but the data source is not independent of KEV's source selection. It pins BTZSC to `fef2a2ac62b69c58670047dddf045c53d7c3cb5e`; its Banking configuration has 72 available labels and excludes rows with no positive candidate. The harness is Apache-2.0; underlying datasets retain their own terms.
80
+
81
+ Small scenario collections such as [souvikr/jev-test](https://github.com/souvikr/jev-test) are useful smoke tests, but 17 cases / 23 checks are too small and easy to serve as the main accuracy benchmark. Documentation-derived instruction corpora and unlabeled live Jev outputs are not independent gold evaluation data.
82
+
83
+ ## Future vision additions — general benchmarks, not Jev releases
84
+
85
+ ### POPE: object-presence Noul
86
+
87
+ The [original POPE project](https://github.com/AoiDragon/POPE/tree/08d957b917e5a378a2f99d35b6293c536a66298b) supplies image filenames, questions, and yes/no labels. At the pinned revision, each of [random](https://raw.githubusercontent.com/AoiDragon/POPE/08d957b917e5a378a2f99d35b6293c536a66298b/output/coco/coco_pope_random.json), [popular](https://raw.githubusercontent.com/AoiDragon/POPE/08d957b917e5a378a2f99d35b6293c536a66298b/output/coco/coco_pope_popular.json), and [adversarial](https://raw.githubusercontent.com/AoiDragon/POPE/08d957b917e5a378a2f99d35b6293c536a66298b/output/coco/coco_pope_adversarial.json) contains **3,000 questions over 500 images**; counts were checked directly. These are related sampling variants, not three independent image datasets. Schema: `question_id`, `image`, `text`, `label`.
88
+
89
+ Use Noul for “is this object present?” and report each variant separately. The [repository license](https://github.com/AoiDragon/POPE/blob/08d957b917e5a378a2f99d35b6293c536a66298b/LICENSE) is MIT; COCO image rights remain separate. Obtain the named COCO 2014 images according to the [COCO terms](https://cocodataset.org/#termsofuse). Start with a deterministic image-grouped subset before a full run.
90
+
91
+ ### VSR: spatial-relation Noul
92
+
93
+ The original [VSR benchmark](https://github.com/cambridgeltl/visual-spatial-reasoning/tree/b27a0af0ee1462d2b6b92c8c83e869d9254a241a) asks whether a caption describes the spatial relationship in an image. Recommend the [random test file](https://huggingface.co/datasets/cambridgeltl/vsr_random/resolve/b2053328fafdd018ff56cf1dfa9643caaa4e69b8/test.jsonl): **2,195** image/caption pairs, SHA-256 `8ade82a0b93ac9dc1e53f6cf1f11e9d5536776a3715102b6b4d27d4f81d551cc`. Its [official dataset card](https://huggingface.co/datasets/cambridgeltl/vsr_random/tree/b2053328fafdd018ff56cf1dfa9643caaa4e69b8) declares CC-BY-4.0; the implementation repository is Apache-2.0, and COCO image terms remain separate.
94
+
95
+ Schema includes `image`, `image_link`, `caption`, binary `label`, and `relation`. Fetch images using the original project's [image instructions](https://github.com/cambridgeltl/visual-spatial-reasoning/blob/b27a0af0ee1462d2b6b92c8c83e869d9254a241a/data/README.md); only image and caption belong in input. Score by relation and overall.
96
+
97
+ Version pitfall: the repository README's zero-shot table says 616 test examples, but the pinned GitHub and [HF zero-shot](https://huggingface.co/datasets/cambridgeltl/vsr_zeroshot/tree/148b3777ceef4a1bfe46377614980836db7d12f2) `test.jsonl` both contain **1,222** rows, SHA-256 `914d9156723b912d8f794b179248af865a68c5f81c9d8e3a21d007778cacaad4`. Do not copy the README count into a report for those files.
98
+
99
+ For vision timing, record image dimensions, processed image-token counts, resize policy, and cache condition. Measure the same fixed examples on every model. A no-image ablation tests image dependence but does not make the text-only task equally answerable. Keep prompt tuning, temperature calibration, and final test evaluation separate; public availability does not establish absence from any model's pretraining.
server/benchmarks/README.md ADDED
@@ -0,0 +1,144 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # H200 한 장에서 학습 전 판단 성능 비교
2
+
3
+ 추가 학습·LoRA·확률 보정 없이 공식 모델을 실행해 KEV의 공개 평가 문항을 평가합니다. 모델과 SGLang은 서버에서 실행하고, 이 저장소의 어댑터와 벤치 클라이언트는 별도 Python 환경을 사용합니다. 기본 엔진을 수정하지 않습니다.
4
+
5
+ 현재 상태: **H200 실측 완료.** 네 모델 각각 4,006개 요청·4,852개 질문을 오류 없이 처리했습니다. [결과 보고서](../../outputs/h200-baseline-benchmark-2026-09-21/RESULTS.md)에서 정확도와 HTTP 지연을 확인할 수 있습니다. 제공된 모델은 추가 학습 전 공식 지시학습 배포본이며, 사전학습 전용 `*-Base` 모델을 뜻하지 않습니다.
6
+
7
+ 2026-09-21 로컬 검증: pytest 97개와 하위 사례 74개, H200 실행 도구 CPU 테스트 12개 통과. KEV 5종과 JevBench 공개 3종의 총 4,006개 요청·4,852개 질문을 준비했습니다. 합산 숫자는 실행 작업량이며 세트 간 중복을 제거한 독립 평가 문항 수가 아닙니다.
8
+
9
+ ## 모델과 비교 조건
10
+
11
+ | 프로필 | 공식 모델 | 기본 정밀도 |
12
+ |---|---|---|
13
+ | `ministral3-3b-bf16` | `mistralai/Ministral-3-3B-Instruct-2512-BF16` | BF16 |
14
+ | `qwen35-4b-bf16` | `Qwen/Qwen3.5-4B` | BF16 |
15
+ | `qwen36-27b-bf16` | `Qwen/Qwen3.6-27B` | BF16 |
16
+ | `qwen36-35b-a3b-bf16` | `Qwen/Qwen3.6-35B-A3B` | BF16 |
17
+
18
+ 원래 다운로드했던 Ministral의 공식 FP8 배포본은 `ministral3-3b-fp8` 선택형 프로필입니다. 기본 비교는 BF16으로 맞추며, FP8 결과는 별도 행으로 비교합니다. 모델 리비전, 엔진 리비전과 실행 옵션은 [h200/models.json](h200/models.json)에 고정했습니다. H200 1장, TP=1, 컨텍스트 8,192, 입력 절단 없음, speculative decoding/MTP 없음, 비전 인코더 유지 조건입니다. 최대 동시 요청은 32이며 기본 클라이언트 동시성은 1입니다.
19
+
20
+ 공식 지시학습 모델의 chat template에 `enable_thinking=False`를 전달하고, 선택지 라벨의 다음 토큰 확률을 읽습니다. 문장을 생성하지 않으며 `output_tokens=0`을 검증합니다. 질문마다 별도 프리필을 실행하므로 여러 질문을 하나의 특수 pointer head로 처리하는 KEV와 실행 구조는 다릅니다. 문항과 지표를 맞춘 **학습 전 어댑터 기준선**이며 KEV의 학습 결과나 기존 기본 모델 probe의 프롬프트를 그대로 복제한 실험은 아닙니다.
21
+
22
+ Ministral은 SGLang 0.5.20의 native chat 문자열 재인코딩 오류를 피하기 위해
23
+ 어댑터에서 공식 토크나이저의 `tokenize=True` 결과를 전달합니다. 전체 4,852문항의
24
+ 토큰 경계와 최대 입력 길이(4,037토큰)를 CPU에서 확인했습니다. Qwen은 엔진의
25
+ HTTP 토큰화 경로를 사용합니다. 따라서 모델 간 HTTP 지연에는 토큰화 위치와
26
+ 호출 횟수 차이가 포함되며, 순수 GPU 프리필 시간 비교가 아닙니다.
27
+ 초기 Ministral HTTP 점검은 `invalidated.json`으로 무효 표시하며 비교에서 제외합니다.
28
+
29
+ ## 준비된 KEV 데이터
30
+
31
+ 원본 커밋은 [`4f8110a3f8620cc3a182ae9a708e4398492c4b1a`](https://github.com/jaredpalmer/kev/tree/4f8110a3f8620cc3a182ae9a708e4398492c4b1a)입니다. 원본 manifest와 JSONL의 SHA256을 검증하고, 정답·출처·변형 정보를 모델 요청에서 제거합니다. 모든 질문을 유지하며 Banking77의 78개 선택지 변형도 포함합니다.
32
+
33
+ | 개발 세트 | 요청 수 | 전체 질문 | 기본 정확도에 포함되는 질문 |
34
+ |---|---:|---:|---:|
35
+ | `decision-v7` | 1,204 | 1,468 | 1,264 |
36
+ | `transfer-v4` | 764 | 764 | 656 |
37
+ | `transfer-v9` | 1,264 | 1,264 | 1,046 |
38
+ | `semif-v1` — 선택형 | 252 | 252 | 144 |
39
+ | `scienthoon-v1` — 선택형 | 291 | 873 | 873, 작업별 구분 필요 |
40
+
41
+ `decision-v7`은 KEV 학습에 사용된 출처의 평가 문항, `transfer-v4`는 KEV 미세조정에서 사용하지 않은 출처·정책 구조입니다. 이는 원래 Qwen/Mistral 사전학습에서 보지 않았다는 보장은 아닙니다. `transfer-v9`는 v4에 MMLU-Pro·긴 문맥·근거 제거 문항을 더한 것이므로 두 결과를 독립 데이터처럼 합산하지 않습니다.
42
+
43
+ 모델카드의 정확도와 비교할 값은 `report.json`의 `suites.<suite>/development.clean.acc`입니다. 순서 변형과 none-of-the-above 문항은 `variants`, 조건 반전은 `paired_flip`, 선택지 순서 영향은 `permutation`에 기록합니다. v9의 근거 제거 110문항은 정확도에서 제외하고 `unknowable`의 과신 비율로 평가합니다.
44
+
45
+ SemIf는 원래 가족별 balanced accuracy를 사용하므로 여기의 clean micro accuracy를 원래 지표와 혼동하지 않습니다. Scienthoon의 원본 900개 질문 행은 KEV 변환 과정에서 291개 고유 상태·873개 질문이 됐습니다. priority 라벨은 입력에 없는 조직 규칙에 의존하므로 queue·angry와 분리해 `tasks`를 확인합니다. 두 외부 세트도 KEV가 이미 평가한 데이터���니다.
46
+
47
+ ## JevBench 공개 데이터 — 기본 실행에 추가
48
+
49
+ [JevBench](https://github.com/fstandhartinger/jevbench/tree/fd51755eb0c0b546ca206d764faf3302feca913e)는 JEV 계열 판단 모델용 커뮤니티 벤치입니다. TypeSafe 공식 벤치는 아닙니다. 전체 결과표의 534문항 중 실제 공개된 231문항을 커밋과 SHA256으로 고정했습니다.
50
+
51
+ | 공개 세트 | 문항 | 주요 평가 |
52
+ |---|---:|---|
53
+ | `jevbench-original` | 72 | 정책, 라우팅, 의도, 등급, 추출, 답변 적합성 |
54
+ | `jevbench-easy` | 48 | 명확한 사실·의도·도구 선택 |
55
+ | `jevbench-hard` | 111 | 복합 규칙, 시간·수량, 확률, 모호성, 함정, 답변 판단 |
56
+
57
+ 정답과 해설·생성 근거는 모델 입력에 넣지 않습니다. 데이터는 `public.jsonl`로 저장하며 테스트나 비공개 문항으로 부르지 않습니다. 확률을 정확히 계산할 수 있는 hard 10문항은 `reference_distribution`에서 확률 분포 오차를 별도로 측정합니다. 최고 확률 동률은 JevBench 원본처럼 라벨 사전순으로 처리합니다. 기존 KEV 지표와 공개 문항별 지표를 기록하며, JevBench의 전체 534문항 종합 점수나 네트워크 지연 보정 점수는 재현했다고 주장하지 않습니다. 확률 출력 방식도 `token-logprob-derived`로 구분합니다.
58
+
59
+ 평가 선정 기준은 **정책·분류·근거 부족·다중 질문 판단과 확률 품질**이며 비전 여부는 조건이 아닙니다. 기본 준비 검사는 텍스트만 사용하고, `--probe-vision`을 추가했을 때만 이미지도 점검합니다. Typed Decisions는 작은 교사 모델과의 일치도를 측정하므로 보조 후보입니다. 추가 후보와 선정 근거는 [PUBLIC_DATASETS.md](PUBLIC_DATASETS.md), [DECISION_BENCHMARK_SELECTION.md](DECISION_BENCHMARK_SELECTION.md)에 있습니다.
60
+
61
+ ## H200 서버에서 실행
62
+
63
+ 이 저장소를 서버로 옮긴 뒤 저장소 루트에서 실행합니다. 모델 가중치는 실행 시 Hugging Face에서 받으며, 네 모델을 동시에 GPU에 올리지 않습니다. 전체 BF16 체크포인트 캐시를 위해 충분한 로컬 디스크 공간을 확보합니다. 엔진 설치 조건과 CUDA 13 환경은 [h200/README.md](h200/README.md)를 따릅니다.
64
+
65
+ 어댑터 환경:
66
+
67
+ ```bash
68
+ python3.12 -m venv .venv
69
+ .venv/bin/python -m pip install -e '.[dev,native-tokenizer]'
70
+ ```
71
+
72
+ 평가 데이터만 다운로드합니다. 함께 전달된 `benchmarks/data`가 있다면 이 단계는 다시 실행해도 동일 파일인지 확인합니다.
73
+
74
+ ```bash
75
+ .venv/bin/python -m jev_adapter.benchmarks.prepare \
76
+ --output benchmarks/data \
77
+ --suite decision-v7 --suite transfer-v4 --suite transfer-v9 \
78
+ --suite jevbench-original --suite jevbench-easy --suite jevbench-hard \
79
+ --suite semif-v1 --suite scienthoon-v1
80
+ ```
81
+
82
+ 오프라인에서는 같은 커밋의 KEV 체크아웃을 `--source-root /path/to/kev`, JevBench 체크아웃을 `--jevbench-source-root /path/to/jevbench`로 지정할 수 있습니다. 학습 데이터는 다운로드하지 않습니다.
83
+
84
+ 먼저 네 모델의 실행 명령만 확인합니다. `--execute`가 없으면 모델을 받거나 실행하지 않습니다.
85
+
86
+ ```bash
87
+ .venv/bin/python benchmarks/run_matrix.py \
88
+ --engine-python .venv-sglang/bin/python \
89
+ --output benchmarks/results/preview --include-external
90
+ ```
91
+
92
+ 서버 통합 점검은 소량 문항으로 진행합니다. 이 결과는 정확도 비교용으로 쓰지 않습니다.
93
+
94
+ ```bash
95
+ .venv/bin/python benchmarks/run_matrix.py \
96
+ --engine-python .venv-sglang/bin/python \
97
+ --output benchmarks/results/smoke \
98
+ --limit 8 --warmup 2 --execute
99
+ ```
100
+
101
+ 사전 점검이 통과하면 네 모델을 순서대로 평가합니다. 각 모델의 텍스트 입력, 토큰 경계, 0토큰 출력, 모델 리비전과 캐시 설정을 먼저 검증하고, 모델마다 어댑터를 다시 시작합니다. 기존 프로세스가 포트를 사용하면 중단합니다. JevBench만 실행할 때는 `--suite jevbench-original --suite jevbench-easy --suite jevbench-hard`를 지정하세요.
102
+
103
+ ```bash
104
+ .venv/bin/python benchmarks/run_matrix.py \
105
+ --engine-python .venv-sglang/bin/python \
106
+ --output benchmarks/results/bf16-c1 \
107
+ --include-external --execute
108
+ ```
109
+
110
+ 동시성 8에서의 처리량도 보려면 별도 출력 폴더로 실행합니다. 동시성 1의 지연시간과 구분해서 해석합니다.
111
+
112
+ ```bash
113
+ .venv/bin/python benchmarks/run_matrix.py \
114
+ --engine-python .venv-sglang/bin/python \
115
+ --output benchmarks/results/bf16-c8 \
116
+ --concurrency 8 --include-external --execute
117
+ ```
118
+
119
+ `--profile qwen36-35b-a3b-bf16`처럼 모델 하나만 선택할 수 있습니다. `--repeats 3`은 지연 측정을 반복하고 정확도는 첫 반복만 사용합니다. 모든 결과 디렉터리는 새 경로여야 하며 기존 결과를 덮어쓰지 않습니다.
120
+
121
+ ## 결과와 측정 범위
122
+
123
+ 각 모델 아래 `engine/launch.json`, `engine/preflight.json`, `engine/engine.log`, `c1/manifest.json`, `c1/predictions.jsonl`, `c1/report.json`이 생성됩니다. 전체 실행이 성공하면 `comparison.csv`에 모델·세트별 결과를 모읍니다.
124
+
125
+ - 정확도: clean 질문별 micro accuracy, 출처·작업별 정확도, 작업별 macro accuracy.
126
+ - 확률 품질: multiclass Brier, NLL, 10-bin ECE, 최대 확률 0.9 이상 오답 비율, 경험적 오류 예산별 coverage. 어댑터의 entropy confidence 대신 **최대 선택지 확률**로 ECE를 계산합니다.
127
+ - 순서·정책 반전: 라벨 정렬 후 flip rate, 최소 대조쌍 양쪽 정답률. 부분 집합의 누락된 짝은 별도 집계합니다.
128
+ - 시간: 요청별 HTTP 왕복 p50/p95/p99, 어댑터 내부 벽시계 시간, 전체 요청·질문 처리량, 입력 토큰 수. 모델 로딩과 20회 준비 요청은 제외합니다.
129
+
130
+ 지연시간에는 HTTP, 토큰화, 엔진 스케줄링과 프리필이 포함됩니다. **순수 GPU 커널 시간이나 프리필 FLOPS 측정값이 아닙니다.** 질문 수와 입력 토큰 수도 함께 보세요. 어댑터는 질문마다 엔진 HTTP 호출 4회를 사용합니다. 정확도 평가를 위해 지정한 라벨 logprob를 반환받는 비용도 포함됩니다.
131
+
132
+ 기본 실행은 radix cache와 이미지 전처리·비전 특징 재사용을 끕니다. 반복 입력의 캐시 효과가 섞이지 않는 조건입니다. 반대로 실제 서비스의 캐시 효과를 측정하려면 별도 엔진 설정과 `--cache-mode server-default`의 개별 runner 실행으로 기록해야 합니다.
133
+
134
+ 오류·시간 초과·확률 누락·출력 토큰 발생 시 재시도나 문항 제외로 성공률을 높이지 않습니다. 실패한 실행은 `status=failed`로 저장하고 기본 정확도를 출력하지 않습니다. 데이터 SHA256·부분 집합 여부·코드 SHA256·실행 옵션·엔진/모델 정보를 기록합니다. 실패한 실행은 비교 CSV에 포함하지 않습니다.
135
+
136
+ 개발 세트로 조건을 정한 뒤 최종 확인에만 test를 사용합니다. test 준비와 개별 실행은 모두 `--allow-test`를 명시해야 합니다. 행렬 실행기는 KEV 개발 세트와 JevBench 공개 세트를 사용합니다. 학습·프롬프트 선택·온도 보정 없이 현재 고정 조건을 그대로 비교하는 것이 이 기준선의 목적입니다.
137
+
138
+ ```bash
139
+ .venv/bin/python -m pytest
140
+ .venv/bin/python -m unittest discover -s benchmarks/h200 -p 'test_*.py'
141
+ .venv/bin/ruff check .
142
+ ```
143
+
144
+ 지표 수식의 원본 대조 결과와 재현 방법은 [METRIC_VALIDATION.md](METRIC_VALIDATION.md)를 참고하세요. 공개 데이터마다 원래 라이선스가 다르며 변환 manifest가 해당 출처를 보존합니다. JEV 유료 API 호출이나 학습은 실행 과정에 포함되지 않습니다.
server/benchmarks/data/jevbench-easy/LICENSE ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Florian Standhartinger and contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
server/benchmarks/data/jevbench-easy/THIRD-PARTY.md ADDED
@@ -0,0 +1,51 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Third-party projects, data and services
2
+
3
+ MIT (see `LICENSE`) covers this harness and the 72 original public decisions in
4
+ `datasets/public/original.jsonl`. Nothing else in this list is ours to license.
5
+
6
+ ## Systems evaluated
7
+
8
+ Each was reached through the interface its author published. We used public
9
+ endpoints as ordinary clients, at one request at a time, and never sent a
10
+ provider's API key to anyone else's endpoint.
11
+
12
+ | System | Author | Source |
13
+ |---|---|---|
14
+ | Jev 1.13.0 | TypeSafe AI | <https://docs.typesafe.ai> |
15
+ | openjev-sglang | ekzhang | <https://github.com/ekzhang/openjev-sglang> |
16
+ | system-one-open | mithalouni | <https://github.com/mithalouni/system-one-open> |
17
+ | open-alternative-jev | IkerMoel | <https://github.com/ikermoel/open-alternative-jev>, <https://huggingface.co/spaces/IkerMoel/open-alternative-jev> |
18
+ | open-jev-deberta-v3-large | Kotoba Labs | <https://huggingface.co/com-kotobalabs/open-jev-deberta-v3-large>, <https://github.com/kotoba-lang/typed-decisions> |
19
+ | GPT-5.6 Luna | OpenAI | <https://platform.openai.com> |
20
+ | Gemini 3.1 Flash-Lite | Google | <https://ai.google.dev> |
21
+ | DeepSeek V4.1 Flash | DeepSeek | <https://api-docs.deepseek.com> |
22
+ | Qwen3.8 27B | Qwen, served by Chutes | <https://chutes.ai> |
23
+
24
+ Model weights, base models and each project's own code keep their own licences.
25
+ A permissive licence on a repository is not a licence for the base model it
26
+ fine-tunes, and we do not restate either.
27
+
28
+ ## Wire format
29
+
30
+ The typed-decision request shape (`state`, `questions`, the `noul` / `choice` /
31
+ `score` primitives) is TypeSafe's public HTTP API, documented at
32
+ <https://docs.typesafe.ai/api>. Several of the open rebuilds implement it
33
+ deliberately, which is why one adapter reaches more than one of them. JevBench is
34
+ not affiliated with or endorsed by TypeSafe AI.
35
+
36
+ ## Imported decisions
37
+
38
+ 146 of the 242 decisions come from our own earlier auto-router experiment
39
+ (<https://github.com/fstandhartinger/auto-model-router>): 78 routing requests and
40
+ 68 answer-adequacy judgements whose ground truth is a deterministic grader's
41
+ verdict on a saved answer. The upstream task text is **not** redistributed here,
42
+ because the datasets it was drawn from keep their own terms. `datasets/manifest.json`
43
+ pins the hashes; `scripts/import_router.py` shows exactly what was taken and what
44
+ was excluded.
45
+
46
+ ## Held-out decisions
47
+
48
+ 24 scenarios are written by us and deliberately unpublished, so the suite cannot
49
+ be trained on in full. Only their whole-split hash and aggregate results appear
50
+ here. They are sent to the services being evaluated, which is not the same thing
51
+ as being public - see the limits section of the README.
server/benchmarks/data/jevbench-easy/public.jsonl ADDED
@@ -0,0 +1,48 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"id": "easy-intent-00", "suite": "jevbench-easy", "split": "public", "source": "intent", "variant": "clean", "record": {"state": "Where is my package? I ordered it last week and it still hasn't arrived.", "questions": {"decision": {"type": "choice", "instructions": "Which intent does the user's message express?", "criteria": {"billing_question": "Asks about a charge, invoice or payment", "cancel_order": "Wants to cancel an order", "change_address": "Wants to change the delivery address", "report_damage": "Received an item that is broken or damaged", "track_order": "Wants to know where an order is or when it arrives"}}}}, "expected": {"decision": {"labels": ["track_order", "cancel_order", "change_address", "report_damage", "billing_question"], "target": [1.0, 0.0, 0.0, 0.0, 0.0], "label": 0, "type": "choice", "task": "intent"}}, "metadata": {"id": "easy-intent-00", "source": "intent", "variant": "clean", "group_id": "easy-intent-00", "upstream_group": null, "canonical_labels": ["track_order", "cancel_order", "change_address", "report_damage", "billing_question"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
2
+ {"id": "easy-intent-01", "suite": "jevbench-easy", "split": "public", "source": "intent", "variant": "clean", "record": {"state": "Please cancel my order #4471, I don't need it anymore.", "questions": {"decision": {"type": "choice", "instructions": "Which intent does the user's message express?", "criteria": {"billing_question": "Asks about a charge, invoice or payment", "cancel_order": "Wants to cancel an order", "change_address": "Wants to change the delivery address", "report_damage": "Received an item that is broken or damaged", "track_order": "Wants to know where an order is or when it arrives"}}}}, "expected": {"decision": {"labels": ["track_order", "cancel_order", "change_address", "report_damage", "billing_question"], "target": [0.0, 1.0, 0.0, 0.0, 0.0], "label": 1, "type": "choice", "task": "intent"}}, "metadata": {"id": "easy-intent-01", "source": "intent", "variant": "clean", "group_id": "easy-intent-01", "upstream_group": null, "canonical_labels": ["track_order", "cancel_order", "change_address", "report_damage", "billing_question"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
3
+ {"id": "easy-intent-02", "suite": "jevbench-easy", "split": "public", "source": "intent", "variant": "clean", "record": {"state": "I moved. Can you ship my order to 12 Elm Street instead of the old address?", "questions": {"decision": {"type": "choice", "instructions": "Which intent does the user's message express?", "criteria": {"billing_question": "Asks about a charge, invoice or payment", "cancel_order": "Wants to cancel an order", "change_address": "Wants to change the delivery address", "report_damage": "Received an item that is broken or damaged", "track_order": "Wants to know where an order is or when it arrives"}}}}, "expected": {"decision": {"labels": ["track_order", "cancel_order", "change_address", "report_damage", "billing_question"], "target": [0.0, 0.0, 1.0, 0.0, 0.0], "label": 2, "type": "choice", "task": "intent"}}, "metadata": {"id": "easy-intent-02", "source": "intent", "variant": "clean", "group_id": "easy-intent-02", "upstream_group": null, "canonical_labels": ["track_order", "cancel_order", "change_address", "report_damage", "billing_question"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
4
+ {"id": "easy-intent-03", "suite": "jevbench-easy", "split": "public", "source": "intent", "variant": "clean", "record": {"state": "The mug I received arrived smashed into pieces.", "questions": {"decision": {"type": "choice", "instructions": "Which intent does the user's message express?", "criteria": {"billing_question": "Asks about a charge, invoice or payment", "cancel_order": "Wants to cancel an order", "change_address": "Wants to change the delivery address", "report_damage": "Received an item that is broken or damaged", "track_order": "Wants to know where an order is or when it arrives"}}}}, "expected": {"decision": {"labels": ["track_order", "cancel_order", "change_address", "report_damage", "billing_question"], "target": [0.0, 0.0, 0.0, 1.0, 0.0], "label": 3, "type": "choice", "task": "intent"}}, "metadata": {"id": "easy-intent-03", "source": "intent", "variant": "clean", "group_id": "easy-intent-03", "upstream_group": null, "canonical_labels": ["track_order", "cancel_order", "change_address", "report_damage", "billing_question"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
5
+ {"id": "easy-intent-04", "suite": "jevbench-easy", "split": "public", "source": "intent", "variant": "clean", "record": {"state": "Why was I charged twice on my credit card statement for one order?", "questions": {"decision": {"type": "choice", "instructions": "Which intent does the user's message express?", "criteria": {"billing_question": "Asks about a charge, invoice or payment", "cancel_order": "Wants to cancel an order", "change_address": "Wants to change the delivery address", "report_damage": "Received an item that is broken or damaged", "track_order": "Wants to know where an order is or when it arrives"}}}}, "expected": {"decision": {"labels": ["track_order", "cancel_order", "change_address", "report_damage", "billing_question"], "target": [0.0, 0.0, 0.0, 0.0, 1.0], "label": 4, "type": "choice", "task": "intent"}}, "metadata": {"id": "easy-intent-04", "source": "intent", "variant": "clean", "group_id": "easy-intent-04", "upstream_group": null, "canonical_labels": ["track_order", "cancel_order", "change_address", "report_damage", "billing_question"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
6
+ {"id": "easy-intent-05", "suite": "jevbench-easy", "split": "public", "source": "intent", "variant": "clean", "record": {"state": "When will my order be delivered? The tracking page shows nothing.", "questions": {"decision": {"type": "choice", "instructions": "Which intent does the user's message express?", "criteria": {"billing_question": "Asks about a charge, invoice or payment", "cancel_order": "Wants to cancel an order", "change_address": "Wants to change the delivery address", "report_damage": "Received an item that is broken or damaged", "track_order": "Wants to know where an order is or when it arrives"}}}}, "expected": {"decision": {"labels": ["track_order", "cancel_order", "change_address", "report_damage", "billing_question"], "target": [1.0, 0.0, 0.0, 0.0, 0.0], "label": 0, "type": "choice", "task": "intent"}}, "metadata": {"id": "easy-intent-05", "source": "intent", "variant": "clean", "group_id": "easy-intent-05", "upstream_group": null, "canonical_labels": ["track_order", "cancel_order", "change_address", "report_damage", "billing_question"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
7
+ {"id": "easy-intent-06", "suite": "jevbench-easy", "split": "public", "source": "intent", "variant": "clean", "record": {"state": "Wake me up at 6:30 tomorrow morning.", "questions": {"decision": {"type": "choice", "instructions": "Which intent does the user's message express?", "criteria": {"play_music": "Play a song, album, artist or playlist", "send_message": "Send a text message to someone", "set_alarm": "Set an alarm or wake-up time", "turn_off_lights": "Switch lights off", "weather": "Ask about the weather or forecast"}}}}, "expected": {"decision": {"labels": ["set_alarm", "play_music", "weather", "send_message", "turn_off_lights"], "target": [1.0, 0.0, 0.0, 0.0, 0.0], "label": 0, "type": "choice", "task": "intent"}}, "metadata": {"id": "easy-intent-06", "source": "intent", "variant": "clean", "group_id": "easy-intent-06", "upstream_group": null, "canonical_labels": ["set_alarm", "play_music", "weather", "send_message", "turn_off_lights"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
8
+ {"id": "easy-intent-07", "suite": "jevbench-easy", "split": "public", "source": "intent", "variant": "clean", "record": {"state": "Play some Taylor Swift.", "questions": {"decision": {"type": "choice", "instructions": "Which intent does the user's message express?", "criteria": {"play_music": "Play a song, album, artist or playlist", "send_message": "Send a text message to someone", "set_alarm": "Set an alarm or wake-up time", "turn_off_lights": "Switch lights off", "weather": "Ask about the weather or forecast"}}}}, "expected": {"decision": {"labels": ["set_alarm", "play_music", "weather", "send_message", "turn_off_lights"], "target": [0.0, 1.0, 0.0, 0.0, 0.0], "label": 1, "type": "choice", "task": "intent"}}, "metadata": {"id": "easy-intent-07", "source": "intent", "variant": "clean", "group_id": "easy-intent-07", "upstream_group": null, "canonical_labels": ["set_alarm", "play_music", "weather", "send_message", "turn_off_lights"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
9
+ {"id": "easy-intent-08", "suite": "jevbench-easy", "split": "public", "source": "intent", "variant": "clean", "record": {"state": "Will it rain in Berlin tomorrow?", "questions": {"decision": {"type": "choice", "instructions": "Which intent does the user's message express?", "criteria": {"play_music": "Play a song, album, artist or playlist", "send_message": "Send a text message to someone", "set_alarm": "Set an alarm or wake-up time", "turn_off_lights": "Switch lights off", "weather": "Ask about the weather or forecast"}}}}, "expected": {"decision": {"labels": ["set_alarm", "play_music", "weather", "send_message", "turn_off_lights"], "target": [0.0, 0.0, 1.0, 0.0, 0.0], "label": 2, "type": "choice", "task": "intent"}}, "metadata": {"id": "easy-intent-08", "source": "intent", "variant": "clean", "group_id": "easy-intent-08", "upstream_group": null, "canonical_labels": ["set_alarm", "play_music", "weather", "send_message", "turn_off_lights"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
10
+ {"id": "easy-intent-09", "suite": "jevbench-easy", "split": "public", "source": "intent", "variant": "clean", "record": {"state": "Text Anna that I'll be ten minutes late.", "questions": {"decision": {"type": "choice", "instructions": "Which intent does the user's message express?", "criteria": {"play_music": "Play a song, album, artist or playlist", "send_message": "Send a text message to someone", "set_alarm": "Set an alarm or wake-up time", "turn_off_lights": "Switch lights off", "weather": "Ask about the weather or forecast"}}}}, "expected": {"decision": {"labels": ["set_alarm", "play_music", "weather", "send_message", "turn_off_lights"], "target": [0.0, 0.0, 0.0, 1.0, 0.0], "label": 3, "type": "choice", "task": "intent"}}, "metadata": {"id": "easy-intent-09", "source": "intent", "variant": "clean", "group_id": "easy-intent-09", "upstream_group": null, "canonical_labels": ["set_alarm", "play_music", "weather", "send_message", "turn_off_lights"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
11
+ {"id": "easy-intent-10", "suite": "jevbench-easy", "split": "public", "source": "intent", "variant": "clean", "record": {"state": "Turn off the lights in the bedroom.", "questions": {"decision": {"type": "choice", "instructions": "Which intent does the user's message express?", "criteria": {"play_music": "Play a song, album, artist or playlist", "send_message": "Send a text message to someone", "set_alarm": "Set an alarm or wake-up time", "turn_off_lights": "Switch lights off", "weather": "Ask about the weather or forecast"}}}}, "expected": {"decision": {"labels": ["set_alarm", "play_music", "weather", "send_message", "turn_off_lights"], "target": [0.0, 0.0, 0.0, 0.0, 1.0], "label": 4, "type": "choice", "task": "intent"}}, "metadata": {"id": "easy-intent-10", "source": "intent", "variant": "clean", "group_id": "easy-intent-10", "upstream_group": null, "canonical_labels": ["set_alarm", "play_music", "weather", "send_message", "turn_off_lights"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
12
+ {"id": "easy-intent-11", "suite": "jevbench-easy", "split": "public", "source": "intent", "variant": "clean", "record": {"state": "Set an alarm for 7 am.", "questions": {"decision": {"type": "choice", "instructions": "Which intent does the user's message express?", "criteria": {"play_music": "Play a song, album, artist or playlist", "send_message": "Send a text message to someone", "set_alarm": "Set an alarm or wake-up time", "turn_off_lights": "Switch lights off", "weather": "Ask about the weather or forecast"}}}}, "expected": {"decision": {"labels": ["set_alarm", "play_music", "weather", "send_message", "turn_off_lights"], "target": [1.0, 0.0, 0.0, 0.0, 0.0], "label": 0, "type": "choice", "task": "intent"}}, "metadata": {"id": "easy-intent-11", "source": "intent", "variant": "clean", "group_id": "easy-intent-11", "upstream_group": null, "canonical_labels": ["set_alarm", "play_music", "weather", "send_message", "turn_off_lights"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
13
+ {"id": "easy-fact-00", "suite": "jevbench-easy", "split": "public", "source": "fact", "variant": "clean", "record": {"state": "Order #1182. Status: shipped on 3 September. Carrier: DHL.", "questions": {"decision": {"type": "noul", "instructions": "Has the order been shipped? Answer strictly from the facts stated in the text.", "criteria": {"false": "The text states that this is not so", "true": "The text states that this is so"}}}}, "expected": {"decision": {"labels": ["false", "true"], "target": [0.0, 1.0], "label": 1, "type": "noul", "task": "fact"}}, "metadata": {"id": "easy-fact-00", "source": "fact", "variant": "clean", "group_id": "easy-fact-00", "upstream_group": null, "canonical_labels": ["no", "yes"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
14
+ {"id": "easy-fact-01", "suite": "jevbench-easy", "split": "public", "source": "fact", "variant": "clean", "record": {"state": "Order #1183. Status: not yet shipped, waiting for stock.", "questions": {"decision": {"type": "noul", "instructions": "Has the order been shipped? Answer strictly from the facts stated in the text.", "criteria": {"false": "The text states that this is not so", "true": "The text states that this is so"}}}}, "expected": {"decision": {"labels": ["false", "true"], "target": [1.0, 0.0], "label": 0, "type": "noul", "task": "fact"}}, "metadata": {"id": "easy-fact-01", "source": "fact", "variant": "clean", "group_id": "easy-fact-01", "upstream_group": null, "canonical_labels": ["no", "yes"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
15
+ {"id": "easy-fact-02", "suite": "jevbench-easy", "split": "public", "source": "fact", "variant": "clean", "record": {"state": "Invoice 2026-044. Amount: 120 EUR. Payment status: paid in full.", "questions": {"decision": {"type": "noul", "instructions": "Is the invoice paid? Answer strictly from the facts stated in the text.", "criteria": {"false": "The text states that this is not so", "true": "The text states that this is so"}}}}, "expected": {"decision": {"labels": ["false", "true"], "target": [0.0, 1.0], "label": 1, "type": "noul", "task": "fact"}}, "metadata": {"id": "easy-fact-02", "source": "fact", "variant": "clean", "group_id": "easy-fact-02", "upstream_group": null, "canonical_labels": ["no", "yes"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
16
+ {"id": "easy-fact-03", "suite": "jevbench-easy", "split": "public", "source": "fact", "variant": "clean", "record": {"state": "Invoice 2026-045. Amount: 80 EUR. Payment status: unpaid, overdue since 1 August.", "questions": {"decision": {"type": "noul", "instructions": "Is the invoice paid? Answer strictly from the facts stated in the text.", "criteria": {"false": "The text states that this is not so", "true": "The text states that this is so"}}}}, "expected": {"decision": {"labels": ["false", "true"], "target": [1.0, 0.0], "label": 0, "type": "noul", "task": "fact"}}, "metadata": {"id": "easy-fact-03", "source": "fact", "variant": "clean", "group_id": "easy-fact-03", "upstream_group": null, "canonical_labels": ["no", "yes"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
17
+ {"id": "easy-fact-04", "suite": "jevbench-easy", "split": "public", "source": "fact", "variant": "clean", "record": {"state": "Guest profile: Maria Lopez. Allergies: peanuts. Diet: vegetarian.", "questions": {"decision": {"type": "noul", "instructions": "Is the guest allergic to peanuts? Answer strictly from the facts stated in the text.", "criteria": {"false": "The text states that this is not so", "true": "The text states that this is so"}}}}, "expected": {"decision": {"labels": ["false", "true"], "target": [0.0, 1.0], "label": 1, "type": "noul", "task": "fact"}}, "metadata": {"id": "easy-fact-04", "source": "fact", "variant": "clean", "group_id": "easy-fact-04", "upstream_group": null, "canonical_labels": ["no", "yes"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
18
+ {"id": "easy-fact-05", "suite": "jevbench-easy", "split": "public", "source": "fact", "variant": "clean", "record": {"state": "Guest profile: Jonas Weber. Allergies: none. Diet: no restrictions.", "questions": {"decision": {"type": "noul", "instructions": "Is the guest allergic to peanuts? Answer strictly from the facts stated in the text.", "criteria": {"false": "The text states that this is not so", "true": "The text states that this is so"}}}}, "expected": {"decision": {"labels": ["false", "true"], "target": [1.0, 0.0], "label": 0, "type": "noul", "task": "fact"}}, "metadata": {"id": "easy-fact-05", "source": "fact", "variant": "clean", "group_id": "easy-fact-05", "upstream_group": null, "canonical_labels": ["no", "yes"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
19
+ {"id": "easy-fact-06", "suite": "jevbench-easy", "split": "public", "source": "fact", "variant": "clean", "record": {"state": "Meeting room B: booked from 14:00 to 15:00 by the sales team.", "questions": {"decision": {"type": "noul", "instructions": "Is meeting room B booked from 14:00 to 15:00? Answer strictly from the facts stated in the text.", "criteria": {"false": "The text states that this is not so", "true": "The text states that this is so"}}}}, "expected": {"decision": {"labels": ["false", "true"], "target": [0.0, 1.0], "label": 1, "type": "noul", "task": "fact"}}, "metadata": {"id": "easy-fact-06", "source": "fact", "variant": "clean", "group_id": "easy-fact-06", "upstream_group": null, "canonical_labels": ["no", "yes"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
20
+ {"id": "easy-fact-07", "suite": "jevbench-easy", "split": "public", "source": "fact", "variant": "clean", "record": {"state": "Meeting room C: free all afternoon, no bookings.", "questions": {"decision": {"type": "noul", "instructions": "Is meeting room C booked this afternoon? Answer strictly from the facts stated in the text.", "criteria": {"false": "The text states that this is not so", "true": "The text states that this is so"}}}}, "expected": {"decision": {"labels": ["false", "true"], "target": [1.0, 0.0], "label": 0, "type": "noul", "task": "fact"}}, "metadata": {"id": "easy-fact-07", "source": "fact", "variant": "clean", "group_id": "easy-fact-07", "upstream_group": null, "canonical_labels": ["no", "yes"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
21
+ {"id": "easy-fact-08", "suite": "jevbench-easy", "split": "public", "source": "fact", "variant": "clean", "record": {"state": "Account settings: two-factor authentication is enabled.", "questions": {"decision": {"type": "noul", "instructions": "Is two-factor authentication enabled? Answer strictly from the facts stated in the text.", "criteria": {"false": "The text states that this is not so", "true": "The text states that this is so"}}}}, "expected": {"decision": {"labels": ["false", "true"], "target": [0.0, 1.0], "label": 1, "type": "noul", "task": "fact"}}, "metadata": {"id": "easy-fact-08", "source": "fact", "variant": "clean", "group_id": "easy-fact-08", "upstream_group": null, "canonical_labels": ["no", "yes"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
22
+ {"id": "easy-fact-09", "suite": "jevbench-easy", "split": "public", "source": "fact", "variant": "clean", "record": {"state": "Account settings: two-factor authentication is disabled.", "questions": {"decision": {"type": "noul", "instructions": "Is two-factor authentication enabled? Answer strictly from the facts stated in the text.", "criteria": {"false": "The text states that this is not so", "true": "The text states that this is so"}}}}, "expected": {"decision": {"labels": ["false", "true"], "target": [1.0, 0.0], "label": 0, "type": "noul", "task": "fact"}}, "metadata": {"id": "easy-fact-09", "source": "fact", "variant": "clean", "group_id": "easy-fact-09", "upstream_group": null, "canonical_labels": ["no", "yes"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
23
+ {"id": "easy-fact-10", "suite": "jevbench-easy", "split": "public", "source": "fact", "variant": "clean", "record": {"state": "The user wrote: 'Yes, please subscribe me to the newsletter.'", "questions": {"decision": {"type": "noul", "instructions": "Did the user agree to subscribe to the newsletter? Answer strictly from the facts stated in the text.", "criteria": {"false": "The text states that this is not so", "true": "The text states that this is so"}}}}, "expected": {"decision": {"labels": ["false", "true"], "target": [0.0, 1.0], "label": 1, "type": "noul", "task": "fact"}}, "metadata": {"id": "easy-fact-10", "source": "fact", "variant": "clean", "group_id": "easy-fact-10", "upstream_group": null, "canonical_labels": ["no", "yes"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
24
+ {"id": "easy-fact-11", "suite": "jevbench-easy", "split": "public", "source": "fact", "variant": "clean", "record": {"state": "The user wrote: 'No thanks, I do not want the newsletter.'", "questions": {"decision": {"type": "noul", "instructions": "Did the user agree to subscribe to the newsletter? Answer strictly from the facts stated in the text.", "criteria": {"false": "The text states that this is not so", "true": "The text states that this is so"}}}}, "expected": {"decision": {"labels": ["false", "true"], "target": [1.0, 0.0], "label": 0, "type": "noul", "task": "fact"}}, "metadata": {"id": "easy-fact-11", "source": "fact", "variant": "clean", "group_id": "easy-fact-11", "upstream_group": null, "canonical_labels": ["no", "yes"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
25
+ {"id": "easy-extraction-00", "suite": "jevbench-easy", "split": "public", "source": "extraction", "variant": "clean", "record": {"state": "I paid with PayPal yesterday evening.", "questions": {"decision": {"type": "choice", "instructions": "Which payment method does the customer name?", "criteria": {"bank_transfer": "Bank transfer / wire", "cash": "Cash", "credit_card": "Paid or wants to pay by credit card", "paypal": "PayPal"}}}}, "expected": {"decision": {"labels": ["credit_card", "paypal", "bank_transfer", "cash"], "target": [0.0, 1.0, 0.0, 0.0], "label": 1, "type": "choice", "task": "extraction"}}, "metadata": {"id": "easy-extraction-00", "source": "extraction", "variant": "clean", "group_id": "easy-extraction-00", "upstream_group": null, "canonical_labels": ["credit_card", "paypal", "bank_transfer", "cash"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
26
+ {"id": "easy-extraction-01", "suite": "jevbench-easy", "split": "public", "source": "extraction", "variant": "clean", "record": {"state": "I'll pay cash when the courier arrives.", "questions": {"decision": {"type": "choice", "instructions": "Which payment method does the customer name?", "criteria": {"bank_transfer": "Bank transfer / wire", "cash": "Cash", "credit_card": "Paid or wants to pay by credit card", "paypal": "PayPal"}}}}, "expected": {"decision": {"labels": ["credit_card", "paypal", "bank_transfer", "cash"], "target": [0.0, 0.0, 0.0, 1.0], "label": 3, "type": "choice", "task": "extraction"}}, "metadata": {"id": "easy-extraction-01", "source": "extraction", "variant": "clean", "group_id": "easy-extraction-01", "upstream_group": null, "canonical_labels": ["credit_card", "paypal", "bank_transfer", "cash"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
27
+ {"id": "easy-extraction-02", "suite": "jevbench-easy", "split": "public", "source": "extraction", "variant": "clean", "record": {"state": "I sent the money by bank transfer on Monday.", "questions": {"decision": {"type": "choice", "instructions": "Which payment method does the customer name?", "criteria": {"bank_transfer": "Bank transfer / wire", "cash": "Cash", "credit_card": "Paid or wants to pay by credit card", "paypal": "PayPal"}}}}, "expected": {"decision": {"labels": ["credit_card", "paypal", "bank_transfer", "cash"], "target": [0.0, 0.0, 1.0, 0.0], "label": 2, "type": "choice", "task": "extraction"}}, "metadata": {"id": "easy-extraction-02", "source": "extraction", "variant": "clean", "group_id": "easy-extraction-02", "upstream_group": null, "canonical_labels": ["credit_card", "paypal", "bank_transfer", "cash"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
28
+ {"id": "easy-extraction-03", "suite": "jevbench-easy", "split": "public", "source": "extraction", "variant": "clean", "record": {"state": "Ticket #88 - Priority: HIGH - Printer on floor 2 is jammed.", "questions": {"decision": {"type": "choice", "instructions": "Which priority does the ticket state?", "criteria": {"high": "High priority", "low": "Low priority", "medium": "Medium priority", "urgent": "Urgent / critical"}}}}, "expected": {"decision": {"labels": ["low", "medium", "high", "urgent"], "target": [0.0, 0.0, 1.0, 0.0], "label": 2, "type": "choice", "task": "extraction"}}, "metadata": {"id": "easy-extraction-03", "source": "extraction", "variant": "clean", "group_id": "easy-extraction-03", "upstream_group": null, "canonical_labels": ["low", "medium", "high", "urgent"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
29
+ {"id": "easy-extraction-04", "suite": "jevbench-easy", "split": "public", "source": "extraction", "variant": "clean", "record": {"state": "Ticket #89 - Priority: low - Please update my desk phone label.", "questions": {"decision": {"type": "choice", "instructions": "Which priority does the ticket state?", "criteria": {"high": "High priority", "low": "Low priority", "medium": "Medium priority", "urgent": "Urgent / critical"}}}}, "expected": {"decision": {"labels": ["low", "medium", "high", "urgent"], "target": [1.0, 0.0, 0.0, 0.0], "label": 0, "type": "choice", "task": "extraction"}}, "metadata": {"id": "easy-extraction-04", "source": "extraction", "variant": "clean", "group_id": "easy-extraction-04", "upstream_group": null, "canonical_labels": ["low", "medium", "high", "urgent"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
30
+ {"id": "easy-extraction-05", "suite": "jevbench-easy", "split": "public", "source": "extraction", "variant": "clean", "record": {"state": "Ticket #90 - Priority: URGENT - Production database is down.", "questions": {"decision": {"type": "choice", "instructions": "Which priority does the ticket state?", "criteria": {"high": "High priority", "low": "Low priority", "medium": "Medium priority", "urgent": "Urgent / critical"}}}}, "expected": {"decision": {"labels": ["low", "medium", "high", "urgent"], "target": [0.0, 0.0, 0.0, 1.0], "label": 3, "type": "choice", "task": "extraction"}}, "metadata": {"id": "easy-extraction-05", "source": "extraction", "variant": "clean", "group_id": "easy-extraction-05", "upstream_group": null, "canonical_labels": ["low", "medium", "high", "urgent"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
31
+ {"id": "easy-extraction-06", "suite": "jevbench-easy", "split": "public", "source": "extraction", "variant": "clean", "record": {"state": "Could I get the blue shirt in size M, please?", "questions": {"decision": {"type": "choice", "instructions": "Which shirt size does the customer ask for?", "criteria": {"L": "Large", "M": "Medium", "S": "Small", "XL": "Extra large"}}}}, "expected": {"decision": {"labels": ["S", "M", "L", "XL"], "target": [0.0, 1.0, 0.0, 0.0], "label": 1, "type": "choice", "task": "extraction"}}, "metadata": {"id": "easy-extraction-06", "source": "extraction", "variant": "clean", "group_id": "easy-extraction-06", "upstream_group": null, "canonical_labels": ["S", "M", "L", "XL"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
32
+ {"id": "easy-extraction-07", "suite": "jevbench-easy", "split": "public", "source": "extraction", "variant": "clean", "record": {"state": "I need the jacket in extra large (XL).", "questions": {"decision": {"type": "choice", "instructions": "Which shirt size does the customer ask for?", "criteria": {"L": "Large", "M": "Medium", "S": "Small", "XL": "Extra large"}}}}, "expected": {"decision": {"labels": ["S", "M", "L", "XL"], "target": [0.0, 0.0, 0.0, 1.0], "label": 3, "type": "choice", "task": "extraction"}}, "metadata": {"id": "easy-extraction-07", "source": "extraction", "variant": "clean", "group_id": "easy-extraction-07", "upstream_group": null, "canonical_labels": ["S", "M", "L", "XL"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
33
+ {"id": "easy-extraction-08", "suite": "jevbench-easy", "split": "public", "source": "extraction", "variant": "clean", "record": {"state": "Let's meet on Thursday at 10.", "questions": {"decision": {"type": "choice", "instructions": "Which day does the person propose for the meeting?", "criteria": {"friday": "Friday", "monday": "Monday", "thursday": "Thursday", "tuesday": "Tuesday", "wednesday": "Wednesday"}}}}, "expected": {"decision": {"labels": ["monday", "tuesday", "wednesday", "thursday", "friday"], "target": [0.0, 0.0, 0.0, 1.0, 0.0], "label": 3, "type": "choice", "task": "extraction"}}, "metadata": {"id": "easy-extraction-08", "source": "extraction", "variant": "clean", "group_id": "easy-extraction-08", "upstream_group": null, "canonical_labels": ["monday", "tuesday", "wednesday", "thursday", "friday"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
34
+ {"id": "easy-extraction-09", "suite": "jevbench-easy", "split": "public", "source": "extraction", "variant": "clean", "record": {"state": "Does Monday work for you? I'm free all day.", "questions": {"decision": {"type": "choice", "instructions": "Which day does the person propose for the meeting?", "criteria": {"friday": "Friday", "monday": "Monday", "thursday": "Thursday", "tuesday": "Tuesday", "wednesday": "Wednesday"}}}}, "expected": {"decision": {"labels": ["monday", "tuesday", "wednesday", "thursday", "friday"], "target": [1.0, 0.0, 0.0, 0.0, 0.0], "label": 0, "type": "choice", "task": "extraction"}}, "metadata": {"id": "easy-extraction-09", "source": "extraction", "variant": "clean", "group_id": "easy-extraction-09", "upstream_group": null, "canonical_labels": ["monday", "tuesday", "wednesday", "thursday", "friday"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
35
+ {"id": "easy-extraction-10", "suite": "jevbench-easy", "split": "public", "source": "extraction", "variant": "clean", "record": {"state": "How about Friday afternoon for the review?", "questions": {"decision": {"type": "choice", "instructions": "Which day does the person propose for the meeting?", "criteria": {"friday": "Friday", "monday": "Monday", "thursday": "Thursday", "tuesday": "Tuesday", "wednesday": "Wednesday"}}}}, "expected": {"decision": {"labels": ["monday", "tuesday", "wednesday", "thursday", "friday"], "target": [0.0, 0.0, 0.0, 0.0, 1.0], "label": 4, "type": "choice", "task": "extraction"}}, "metadata": {"id": "easy-extraction-10", "source": "extraction", "variant": "clean", "group_id": "easy-extraction-10", "upstream_group": null, "canonical_labels": ["monday", "tuesday", "wednesday", "thursday", "friday"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
36
+ {"id": "easy-extraction-11", "suite": "jevbench-easy", "split": "public", "source": "extraction", "variant": "clean", "record": {"state": "Please charge my credit card ending in 4242.", "questions": {"decision": {"type": "choice", "instructions": "Which payment method does the customer name?", "criteria": {"bank_transfer": "Bank transfer / wire", "cash": "Cash", "credit_card": "Paid or wants to pay by credit card", "paypal": "PayPal"}}}}, "expected": {"decision": {"labels": ["credit_card", "paypal", "bank_transfer", "cash"], "target": [1.0, 0.0, 0.0, 0.0], "label": 0, "type": "choice", "task": "extraction"}}, "metadata": {"id": "easy-extraction-11", "source": "extraction", "variant": "clean", "group_id": "easy-extraction-11", "upstream_group": null, "canonical_labels": ["credit_card", "paypal", "bank_transfer", "cash"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
37
+ {"id": "easy-tool_selection-00", "suite": "jevbench-easy", "split": "public", "source": "tool_selection", "variant": "clean", "record": {"state": "What's the weather like in Paris right now?", "questions": {"decision": {"type": "choice", "instructions": "Which single tool should be called to handle this request?", "criteria": {"convert_currency": "Convert an amount from one currency to another", "create_calendar_event": "Put an appointment or meeting into the calendar", "get_weather": "Current weather or forecast for a place", "send_email": "Send an email to a recipient", "translate_text": "Translate text into another language"}}}}, "expected": {"decision": {"labels": ["get_weather", "send_email", "create_calendar_event", "convert_currency", "translate_text"], "target": [1.0, 0.0, 0.0, 0.0, 0.0], "label": 0, "type": "choice", "task": "tool_selection"}}, "metadata": {"id": "easy-tool_selection-00", "source": "tool_selection", "variant": "clean", "group_id": "easy-tool_selection-00", "upstream_group": null, "canonical_labels": ["get_weather", "send_email", "create_calendar_event", "convert_currency", "translate_text"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
38
+ {"id": "easy-tool_selection-01", "suite": "jevbench-easy", "split": "public", "source": "tool_selection", "variant": "clean", "record": {"state": "Email Sarah the quarterly report and say it's attached.", "questions": {"decision": {"type": "choice", "instructions": "Which single tool should be called to handle this request?", "criteria": {"convert_currency": "Convert an amount from one currency to another", "create_calendar_event": "Put an appointment or meeting into the calendar", "get_weather": "Current weather or forecast for a place", "send_email": "Send an email to a recipient", "translate_text": "Translate text into another language"}}}}, "expected": {"decision": {"labels": ["get_weather", "send_email", "create_calendar_event", "convert_currency", "translate_text"], "target": [0.0, 1.0, 0.0, 0.0, 0.0], "label": 1, "type": "choice", "task": "tool_selection"}}, "metadata": {"id": "easy-tool_selection-01", "source": "tool_selection", "variant": "clean", "group_id": "easy-tool_selection-01", "upstream_group": null, "canonical_labels": ["get_weather", "send_email", "create_calendar_event", "convert_currency", "translate_text"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
39
+ {"id": "easy-tool_selection-02", "suite": "jevbench-easy", "split": "public", "source": "tool_selection", "variant": "clean", "record": {"state": "Add a dentist appointment on Friday at 3 pm to my calendar.", "questions": {"decision": {"type": "choice", "instructions": "Which single tool should be called to handle this request?", "criteria": {"convert_currency": "Convert an amount from one currency to another", "create_calendar_event": "Put an appointment or meeting into the calendar", "get_weather": "Current weather or forecast for a place", "send_email": "Send an email to a recipient", "translate_text": "Translate text into another language"}}}}, "expected": {"decision": {"labels": ["get_weather", "send_email", "create_calendar_event", "convert_currency", "translate_text"], "target": [0.0, 0.0, 1.0, 0.0, 0.0], "label": 2, "type": "choice", "task": "tool_selection"}}, "metadata": {"id": "easy-tool_selection-02", "source": "tool_selection", "variant": "clean", "group_id": "easy-tool_selection-02", "upstream_group": null, "canonical_labels": ["get_weather", "send_email", "create_calendar_event", "convert_currency", "translate_text"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
40
+ {"id": "easy-tool_selection-03", "suite": "jevbench-easy", "split": "public", "source": "tool_selection", "variant": "clean", "record": {"state": "How much is 250 US dollars in euros?", "questions": {"decision": {"type": "choice", "instructions": "Which single tool should be called to handle this request?", "criteria": {"convert_currency": "Convert an amount from one currency to another", "create_calendar_event": "Put an appointment or meeting into the calendar", "get_weather": "Current weather or forecast for a place", "send_email": "Send an email to a recipient", "translate_text": "Translate text into another language"}}}}, "expected": {"decision": {"labels": ["get_weather", "send_email", "create_calendar_event", "convert_currency", "translate_text"], "target": [0.0, 0.0, 0.0, 1.0, 0.0], "label": 3, "type": "choice", "task": "tool_selection"}}, "metadata": {"id": "easy-tool_selection-03", "source": "tool_selection", "variant": "clean", "group_id": "easy-tool_selection-03", "upstream_group": null, "canonical_labels": ["get_weather", "send_email", "create_calendar_event", "convert_currency", "translate_text"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
41
+ {"id": "easy-tool_selection-04", "suite": "jevbench-easy", "split": "public", "source": "tool_selection", "variant": "clean", "record": {"state": "Translate 'good morning' into Japanese.", "questions": {"decision": {"type": "choice", "instructions": "Which single tool should be called to handle this request?", "criteria": {"convert_currency": "Convert an amount from one currency to another", "create_calendar_event": "Put an appointment or meeting into the calendar", "get_weather": "Current weather or forecast for a place", "send_email": "Send an email to a recipient", "translate_text": "Translate text into another language"}}}}, "expected": {"decision": {"labels": ["get_weather", "send_email", "create_calendar_event", "convert_currency", "translate_text"], "target": [0.0, 0.0, 0.0, 0.0, 1.0], "label": 4, "type": "choice", "task": "tool_selection"}}, "metadata": {"id": "easy-tool_selection-04", "source": "tool_selection", "variant": "clean", "group_id": "easy-tool_selection-04", "upstream_group": null, "canonical_labels": ["get_weather", "send_email", "create_calendar_event", "convert_currency", "translate_text"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
42
+ {"id": "easy-tool_selection-05", "suite": "jevbench-easy", "split": "public", "source": "tool_selection", "variant": "clean", "record": {"state": "Schedule a team meeting next Tuesday at 10 in my calendar.", "questions": {"decision": {"type": "choice", "instructions": "Which single tool should be called to handle this request?", "criteria": {"convert_currency": "Convert an amount from one currency to another", "create_calendar_event": "Put an appointment or meeting into the calendar", "get_weather": "Current weather or forecast for a place", "send_email": "Send an email to a recipient", "translate_text": "Translate text into another language"}}}}, "expected": {"decision": {"labels": ["get_weather", "send_email", "create_calendar_event", "convert_currency", "translate_text"], "target": [0.0, 0.0, 1.0, 0.0, 0.0], "label": 2, "type": "choice", "task": "tool_selection"}}, "metadata": {"id": "easy-tool_selection-05", "source": "tool_selection", "variant": "clean", "group_id": "easy-tool_selection-05", "upstream_group": null, "canonical_labels": ["get_weather", "send_email", "create_calendar_event", "convert_currency", "translate_text"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
43
+ {"id": "easy-tool_selection-06", "suite": "jevbench-easy", "split": "public", "source": "tool_selection", "variant": "clean", "record": {"state": "Find me flights from Munich to Lisbon on 12 October.", "questions": {"decision": {"type": "choice", "instructions": "Which single tool should be called to handle this request?", "criteria": {"book_restaurant": "Reserve a table at a restaurant", "call_taxi": "Order a taxi to a location", "get_stock_price": "Look up the current price of a stock", "search_flights": "Find flights between two cities", "set_reminder": "Remind the user of something at a given time"}}}}, "expected": {"decision": {"labels": ["search_flights", "book_restaurant", "set_reminder", "get_stock_price", "call_taxi"], "target": [1.0, 0.0, 0.0, 0.0, 0.0], "label": 0, "type": "choice", "task": "tool_selection"}}, "metadata": {"id": "easy-tool_selection-06", "source": "tool_selection", "variant": "clean", "group_id": "easy-tool_selection-06", "upstream_group": null, "canonical_labels": ["search_flights", "book_restaurant", "set_reminder", "get_stock_price", "call_taxi"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
44
+ {"id": "easy-tool_selection-07", "suite": "jevbench-easy", "split": "public", "source": "tool_selection", "variant": "clean", "record": {"state": "Book a table for four at Luigi's tonight at 8.", "questions": {"decision": {"type": "choice", "instructions": "Which single tool should be called to handle this request?", "criteria": {"book_restaurant": "Reserve a table at a restaurant", "call_taxi": "Order a taxi to a location", "get_stock_price": "Look up the current price of a stock", "search_flights": "Find flights between two cities", "set_reminder": "Remind the user of something at a given time"}}}}, "expected": {"decision": {"labels": ["search_flights", "book_restaurant", "set_reminder", "get_stock_price", "call_taxi"], "target": [0.0, 1.0, 0.0, 0.0, 0.0], "label": 1, "type": "choice", "task": "tool_selection"}}, "metadata": {"id": "easy-tool_selection-07", "source": "tool_selection", "variant": "clean", "group_id": "easy-tool_selection-07", "upstream_group": null, "canonical_labels": ["search_flights", "book_restaurant", "set_reminder", "get_stock_price", "call_taxi"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
45
+ {"id": "easy-tool_selection-08", "suite": "jevbench-easy", "split": "public", "source": "tool_selection", "variant": "clean", "record": {"state": "Remind me to call my mother at 6 pm.", "questions": {"decision": {"type": "choice", "instructions": "Which single tool should be called to handle this request?", "criteria": {"book_restaurant": "Reserve a table at a restaurant", "call_taxi": "Order a taxi to a location", "get_stock_price": "Look up the current price of a stock", "search_flights": "Find flights between two cities", "set_reminder": "Remind the user of something at a given time"}}}}, "expected": {"decision": {"labels": ["search_flights", "book_restaurant", "set_reminder", "get_stock_price", "call_taxi"], "target": [0.0, 0.0, 1.0, 0.0, 0.0], "label": 2, "type": "choice", "task": "tool_selection"}}, "metadata": {"id": "easy-tool_selection-08", "source": "tool_selection", "variant": "clean", "group_id": "easy-tool_selection-08", "upstream_group": null, "canonical_labels": ["search_flights", "book_restaurant", "set_reminder", "get_stock_price", "call_taxi"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
46
+ {"id": "easy-tool_selection-09", "suite": "jevbench-easy", "split": "public", "source": "tool_selection", "variant": "clean", "record": {"state": "What is Apple's stock price right now?", "questions": {"decision": {"type": "choice", "instructions": "Which single tool should be called to handle this request?", "criteria": {"book_restaurant": "Reserve a table at a restaurant", "call_taxi": "Order a taxi to a location", "get_stock_price": "Look up the current price of a stock", "search_flights": "Find flights between two cities", "set_reminder": "Remind the user of something at a given time"}}}}, "expected": {"decision": {"labels": ["search_flights", "book_restaurant", "set_reminder", "get_stock_price", "call_taxi"], "target": [0.0, 0.0, 0.0, 1.0, 0.0], "label": 3, "type": "choice", "task": "tool_selection"}}, "metadata": {"id": "easy-tool_selection-09", "source": "tool_selection", "variant": "clean", "group_id": "easy-tool_selection-09", "upstream_group": null, "canonical_labels": ["search_flights", "book_restaurant", "set_reminder", "get_stock_price", "call_taxi"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
47
+ {"id": "easy-tool_selection-10", "suite": "jevbench-easy", "split": "public", "source": "tool_selection", "variant": "clean", "record": {"state": "Get me a taxi to the main station.", "questions": {"decision": {"type": "choice", "instructions": "Which single tool should be called to handle this request?", "criteria": {"book_restaurant": "Reserve a table at a restaurant", "call_taxi": "Order a taxi to a location", "get_stock_price": "Look up the current price of a stock", "search_flights": "Find flights between two cities", "set_reminder": "Remind the user of something at a given time"}}}}, "expected": {"decision": {"labels": ["search_flights", "book_restaurant", "set_reminder", "get_stock_price", "call_taxi"], "target": [0.0, 0.0, 0.0, 0.0, 1.0], "label": 4, "type": "choice", "task": "tool_selection"}}, "metadata": {"id": "easy-tool_selection-10", "source": "tool_selection", "variant": "clean", "group_id": "easy-tool_selection-10", "upstream_group": null, "canonical_labels": ["search_flights", "book_restaurant", "set_reminder", "get_stock_price", "call_taxi"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
48
+ {"id": "easy-tool_selection-11", "suite": "jevbench-easy", "split": "public", "source": "tool_selection", "variant": "clean", "record": {"state": "Remind me tomorrow at 9 to water the plants.", "questions": {"decision": {"type": "choice", "instructions": "Which single tool should be called to handle this request?", "criteria": {"book_restaurant": "Reserve a table at a restaurant", "call_taxi": "Order a taxi to a location", "get_stock_price": "Look up the current price of a stock", "search_flights": "Find flights between two cities", "set_reminder": "Remind the user of something at a given time"}}}}, "expected": {"decision": {"labels": ["search_flights", "book_restaurant", "set_reminder", "get_stock_price", "call_taxi"], "target": [0.0, 0.0, 1.0, 0.0, 0.0], "label": 2, "type": "choice", "task": "tool_selection"}}, "metadata": {"id": "easy-tool_selection-11", "source": "tool_selection", "variant": "clean", "group_id": "easy-tool_selection-11", "upstream_group": null, "canonical_labels": ["search_flights", "book_restaurant", "set_reminder", "get_stock_price", "call_taxi"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
server/benchmarks/data/jevbench-easy/public.manifest.json ADDED
@@ -0,0 +1,95 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "schema_version": 1,
3
+ "suite": "jevbench-easy",
4
+ "split": "public",
5
+ "description": "Public easy tier: 48 explicit facts, intents and tool selections.",
6
+ "data_file": "public.jsonl",
7
+ "data_sha256": "e186c6dd90b3a7a177ef15f2fd3673d7bbfce95340bfdc6062dc15bfa0e01a6a",
8
+ "upstream": {
9
+ "repository": "https://github.com/fstandhartinger/jevbench",
10
+ "commit": "fd51755eb0c0b546ca206d764faf3302feca913e",
11
+ "path": "datasets/public/easy.jsonl",
12
+ "url": "https://raw.githubusercontent.com/fstandhartinger/jevbench/fd51755eb0c0b546ca206d764faf3302feca913e/datasets/public/easy.jsonl",
13
+ "sha256": "231df3c2c8e88a1a8c137ebe85de96ba70fabd330849098ac7b3c52c70b7172b",
14
+ "notice_sha256": {
15
+ "LICENSE": "3e5beed774bb0bcbfb2fcf24ba9554212c6ae112937d308040989465ab0c5784",
16
+ "THIRD-PARTY.md": "396422e29055bba4073f4a9e5a7163cc724f13e970a34ef58e8ea1fb439b9b38"
17
+ }
18
+ },
19
+ "full_partition": {
20
+ "records": 48,
21
+ "questions": 48,
22
+ "clean_records": 48,
23
+ "clean_questions": 48,
24
+ "headline_questions": 48,
25
+ "variants": {
26
+ "clean": 48
27
+ },
28
+ "clean_sources": {
29
+ "intent": 12,
30
+ "fact": 12,
31
+ "extraction": 12,
32
+ "tool_selection": 12
33
+ },
34
+ "question_types": {
35
+ "choice": 36,
36
+ "noul": 12
37
+ },
38
+ "maximum_options": 5
39
+ },
40
+ "selected": {
41
+ "records": 48,
42
+ "questions": 48,
43
+ "clean_records": 48,
44
+ "clean_questions": 48,
45
+ "headline_questions": 48,
46
+ "variants": {
47
+ "clean": 48
48
+ },
49
+ "clean_sources": {
50
+ "intent": 12,
51
+ "fact": 12,
52
+ "extraction": 12,
53
+ "tool_selection": 12
54
+ },
55
+ "question_types": {
56
+ "choice": 36,
57
+ "noul": 12
58
+ },
59
+ "maximum_options": 5
60
+ },
61
+ "selection": {
62
+ "method": "full",
63
+ "requested_limit": null,
64
+ "is_full_partition": true,
65
+ "note": "Full frozen public tier; not the complete published leaderboard."
66
+ },
67
+ "protocol": {
68
+ "calibration_applied": false,
69
+ "training_data_downloaded": false,
70
+ "gold_labels_sent_to_model": false,
71
+ "headline_variant": "clean",
72
+ "exclude_from_headline_sources": [],
73
+ "locked_test": false,
74
+ "notes": [
75
+ "Community benchmark; not an official TypeSafe dataset release.",
76
+ "Public original/easy/hard have 72/48/111 questions; report separately.",
77
+ "The 534-item leaderboard contains private/untracked data not downloaded here.",
78
+ "Choice request criteria retain their native order; scoring labels retain canonical order.",
79
+ "Native accuracy uses argmax with lexicographic label tie-break, including Score.",
80
+ "Score also supports expected-value MAE; original has 36 paraphrase groups.",
81
+ "One-hot labels and ten explicit mathematical probability references are separate targets.",
82
+ "Provenance, rationale and gold probabilities are never sent in requests."
83
+ ]
84
+ },
85
+ "provenance": {
86
+ "dataset_license": "MIT",
87
+ "license_url": "https://github.com/fstandhartinger/jevbench/blob/fd51755eb0c0b546ca206d764faf3302feca913e/LICENSE",
88
+ "notices": [
89
+ "LICENSE",
90
+ "THIRD-PARTY.md"
91
+ ],
92
+ "reference_probability_questions": 0,
93
+ "gold_policy": "Authored rubric labels reviewed before inference; hard items retain author and review metadata. Explicit mathematical distributions are not teacher model confidence or population frequency estimates."
94
+ }
95
+ }
server/benchmarks/data/jevbench-hard/LICENSE ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Florian Standhartinger and contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
server/benchmarks/data/jevbench-hard/THIRD-PARTY.md ADDED
@@ -0,0 +1,51 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Third-party projects, data and services
2
+
3
+ MIT (see `LICENSE`) covers this harness and the 72 original public decisions in
4
+ `datasets/public/original.jsonl`. Nothing else in this list is ours to license.
5
+
6
+ ## Systems evaluated
7
+
8
+ Each was reached through the interface its author published. We used public
9
+ endpoints as ordinary clients, at one request at a time, and never sent a
10
+ provider's API key to anyone else's endpoint.
11
+
12
+ | System | Author | Source |
13
+ |---|---|---|
14
+ | Jev 1.13.0 | TypeSafe AI | <https://docs.typesafe.ai> |
15
+ | openjev-sglang | ekzhang | <https://github.com/ekzhang/openjev-sglang> |
16
+ | system-one-open | mithalouni | <https://github.com/mithalouni/system-one-open> |
17
+ | open-alternative-jev | IkerMoel | <https://github.com/ikermoel/open-alternative-jev>, <https://huggingface.co/spaces/IkerMoel/open-alternative-jev> |
18
+ | open-jev-deberta-v3-large | Kotoba Labs | <https://huggingface.co/com-kotobalabs/open-jev-deberta-v3-large>, <https://github.com/kotoba-lang/typed-decisions> |
19
+ | GPT-5.6 Luna | OpenAI | <https://platform.openai.com> |
20
+ | Gemini 3.1 Flash-Lite | Google | <https://ai.google.dev> |
21
+ | DeepSeek V4.1 Flash | DeepSeek | <https://api-docs.deepseek.com> |
22
+ | Qwen3.8 27B | Qwen, served by Chutes | <https://chutes.ai> |
23
+
24
+ Model weights, base models and each project's own code keep their own licences.
25
+ A permissive licence on a repository is not a licence for the base model it
26
+ fine-tunes, and we do not restate either.
27
+
28
+ ## Wire format
29
+
30
+ The typed-decision request shape (`state`, `questions`, the `noul` / `choice` /
31
+ `score` primitives) is TypeSafe's public HTTP API, documented at
32
+ <https://docs.typesafe.ai/api>. Several of the open rebuilds implement it
33
+ deliberately, which is why one adapter reaches more than one of them. JevBench is
34
+ not affiliated with or endorsed by TypeSafe AI.
35
+
36
+ ## Imported decisions
37
+
38
+ 146 of the 242 decisions come from our own earlier auto-router experiment
39
+ (<https://github.com/fstandhartinger/auto-model-router>): 78 routing requests and
40
+ 68 answer-adequacy judgements whose ground truth is a deterministic grader's
41
+ verdict on a saved answer. The upstream task text is **not** redistributed here,
42
+ because the datasets it was drawn from keep their own terms. `datasets/manifest.json`
43
+ pins the hashes; `scripts/import_router.py` shows exactly what was taken and what
44
+ was excluded.
45
+
46
+ ## Held-out decisions
47
+
48
+ 24 scenarios are written by us and deliberately unpublished, so the suite cannot
49
+ be trained on in full. Only their whole-split hash and aggregate results appear
50
+ here. They are sent to the services being evaluated, which is not the same thing
51
+ as being public - see the limits section of the README.
server/benchmarks/data/jevbench-hard/public.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
server/benchmarks/data/jevbench-hard/public.manifest.json ADDED
@@ -0,0 +1,109 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "schema_version": 1,
3
+ "suite": "jevbench-hard",
4
+ "split": "public",
5
+ "description": "Public hard tier: 111 decisions; ten supply probability references.",
6
+ "data_file": "public.jsonl",
7
+ "data_sha256": "6c93c82edf10d08b70ee0c3addcf7c1ea8fd19957e94fb8021969ff4c0027f95",
8
+ "upstream": {
9
+ "repository": "https://github.com/fstandhartinger/jevbench",
10
+ "commit": "fd51755eb0c0b546ca206d764faf3302feca913e",
11
+ "path": "datasets/public/hard.jsonl",
12
+ "url": "https://raw.githubusercontent.com/fstandhartinger/jevbench/fd51755eb0c0b546ca206d764faf3302feca913e/datasets/public/hard.jsonl",
13
+ "sha256": "89e9e6becb33ed88c1de7d42dcc87531b2fb64cfaef4e1986faf7c37b3f80ebb",
14
+ "notice_sha256": {
15
+ "LICENSE": "3e5beed774bb0bcbfb2fcf24ba9554212c6ae112937d308040989465ab0c5784",
16
+ "THIRD-PARTY.md": "396422e29055bba4073f4a9e5a7163cc724f13e970a34ef58e8ea1fb439b9b38"
17
+ }
18
+ },
19
+ "full_partition": {
20
+ "records": 111,
21
+ "questions": 111,
22
+ "clean_records": 111,
23
+ "clean_questions": 111,
24
+ "headline_questions": 111,
25
+ "variants": {
26
+ "clean": 111
27
+ },
28
+ "clean_sources": {
29
+ "long_policy": 19,
30
+ "probability": 10,
31
+ "temporal_numeric": 15,
32
+ "ambiguous": 7,
33
+ "multi_hop": 18,
34
+ "tradeoff": 6,
35
+ "adversarial": 6,
36
+ "trap": 8,
37
+ "judge_hard": 17,
38
+ "routing_hard": 5
39
+ },
40
+ "question_types": {
41
+ "choice": 67,
42
+ "noul": 38,
43
+ "score": 6
44
+ },
45
+ "maximum_options": 6
46
+ },
47
+ "selected": {
48
+ "records": 111,
49
+ "questions": 111,
50
+ "clean_records": 111,
51
+ "clean_questions": 111,
52
+ "headline_questions": 111,
53
+ "variants": {
54
+ "clean": 111
55
+ },
56
+ "clean_sources": {
57
+ "long_policy": 19,
58
+ "probability": 10,
59
+ "temporal_numeric": 15,
60
+ "ambiguous": 7,
61
+ "multi_hop": 18,
62
+ "tradeoff": 6,
63
+ "adversarial": 6,
64
+ "trap": 8,
65
+ "judge_hard": 17,
66
+ "routing_hard": 5
67
+ },
68
+ "question_types": {
69
+ "choice": 67,
70
+ "noul": 38,
71
+ "score": 6
72
+ },
73
+ "maximum_options": 6
74
+ },
75
+ "selection": {
76
+ "method": "full",
77
+ "requested_limit": null,
78
+ "is_full_partition": true,
79
+ "note": "Full frozen public tier; not the complete published leaderboard."
80
+ },
81
+ "protocol": {
82
+ "calibration_applied": false,
83
+ "training_data_downloaded": false,
84
+ "gold_labels_sent_to_model": false,
85
+ "headline_variant": "clean",
86
+ "exclude_from_headline_sources": [],
87
+ "locked_test": false,
88
+ "notes": [
89
+ "Community benchmark; not an official TypeSafe dataset release.",
90
+ "Public original/easy/hard have 72/48/111 questions; report separately.",
91
+ "The 534-item leaderboard contains private/untracked data not downloaded here.",
92
+ "Choice request criteria retain their native order; scoring labels retain canonical order.",
93
+ "Native accuracy uses argmax with lexicographic label tie-break, including Score.",
94
+ "Score also supports expected-value MAE; original has 36 paraphrase groups.",
95
+ "One-hot labels and ten explicit mathematical probability references are separate targets.",
96
+ "Provenance, rationale and gold probabilities are never sent in requests."
97
+ ]
98
+ },
99
+ "provenance": {
100
+ "dataset_license": "MIT",
101
+ "license_url": "https://github.com/fstandhartinger/jevbench/blob/fd51755eb0c0b546ca206d764faf3302feca913e/LICENSE",
102
+ "notices": [
103
+ "LICENSE",
104
+ "THIRD-PARTY.md"
105
+ ],
106
+ "reference_probability_questions": 10,
107
+ "gold_policy": "Authored rubric labels reviewed before inference; hard items retain author and review metadata. Explicit mathematical distributions are not teacher model confidence or population frequency estimates."
108
+ }
109
+ }
server/benchmarks/data/jevbench-original/LICENSE ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Florian Standhartinger and contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
server/benchmarks/data/jevbench-original/THIRD-PARTY.md ADDED
@@ -0,0 +1,51 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Third-party projects, data and services
2
+
3
+ MIT (see `LICENSE`) covers this harness and the 72 original public decisions in
4
+ `datasets/public/original.jsonl`. Nothing else in this list is ours to license.
5
+
6
+ ## Systems evaluated
7
+
8
+ Each was reached through the interface its author published. We used public
9
+ endpoints as ordinary clients, at one request at a time, and never sent a
10
+ provider's API key to anyone else's endpoint.
11
+
12
+ | System | Author | Source |
13
+ |---|---|---|
14
+ | Jev 1.13.0 | TypeSafe AI | <https://docs.typesafe.ai> |
15
+ | openjev-sglang | ekzhang | <https://github.com/ekzhang/openjev-sglang> |
16
+ | system-one-open | mithalouni | <https://github.com/mithalouni/system-one-open> |
17
+ | open-alternative-jev | IkerMoel | <https://github.com/ikermoel/open-alternative-jev>, <https://huggingface.co/spaces/IkerMoel/open-alternative-jev> |
18
+ | open-jev-deberta-v3-large | Kotoba Labs | <https://huggingface.co/com-kotobalabs/open-jev-deberta-v3-large>, <https://github.com/kotoba-lang/typed-decisions> |
19
+ | GPT-5.6 Luna | OpenAI | <https://platform.openai.com> |
20
+ | Gemini 3.1 Flash-Lite | Google | <https://ai.google.dev> |
21
+ | DeepSeek V4.1 Flash | DeepSeek | <https://api-docs.deepseek.com> |
22
+ | Qwen3.8 27B | Qwen, served by Chutes | <https://chutes.ai> |
23
+
24
+ Model weights, base models and each project's own code keep their own licences.
25
+ A permissive licence on a repository is not a licence for the base model it
26
+ fine-tunes, and we do not restate either.
27
+
28
+ ## Wire format
29
+
30
+ The typed-decision request shape (`state`, `questions`, the `noul` / `choice` /
31
+ `score` primitives) is TypeSafe's public HTTP API, documented at
32
+ <https://docs.typesafe.ai/api>. Several of the open rebuilds implement it
33
+ deliberately, which is why one adapter reaches more than one of them. JevBench is
34
+ not affiliated with or endorsed by TypeSafe AI.
35
+
36
+ ## Imported decisions
37
+
38
+ 146 of the 242 decisions come from our own earlier auto-router experiment
39
+ (<https://github.com/fstandhartinger/auto-model-router>): 78 routing requests and
40
+ 68 answer-adequacy judgements whose ground truth is a deterministic grader's
41
+ verdict on a saved answer. The upstream task text is **not** redistributed here,
42
+ because the datasets it was drawn from keep their own terms. `datasets/manifest.json`
43
+ pins the hashes; `scripts/import_router.py` shows exactly what was taken and what
44
+ was excluded.
45
+
46
+ ## Held-out decisions
47
+
48
+ 24 scenarios are written by us and deliberately unpublished, so the suite cannot
49
+ be trained on in full. Only their whole-split hash and aggregate results appear
50
+ here. They are sent to the services being evaluated, which is not the same thing
51
+ as being public - see the limits section of the README.
server/benchmarks/data/jevbench-original/public.manifest.json ADDED
@@ -0,0 +1,101 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "schema_version": 1,
3
+ "suite": "jevbench-original",
4
+ "split": "public",
5
+ "description": "Public original tier: 72 decisions in 36 paraphrase pairs.",
6
+ "data_file": "public.jsonl",
7
+ "data_sha256": "ad453e78624b1c150d5eb3f0f47c5c14caa79542329b4f7e1bcc6199e2eb7007",
8
+ "upstream": {
9
+ "repository": "https://github.com/fstandhartinger/jevbench",
10
+ "commit": "fd51755eb0c0b546ca206d764faf3302feca913e",
11
+ "path": "datasets/public/original.jsonl",
12
+ "url": "https://raw.githubusercontent.com/fstandhartinger/jevbench/fd51755eb0c0b546ca206d764faf3302feca913e/datasets/public/original.jsonl",
13
+ "sha256": "5c2414edb3006b8bfcb70fda433f0f9ca015759433849f8d3104328a1f7c4180",
14
+ "notice_sha256": {
15
+ "LICENSE": "3e5beed774bb0bcbfb2fcf24ba9554212c6ae112937d308040989465ab0c5784",
16
+ "THIRD-PARTY.md": "396422e29055bba4073f4a9e5a7163cc724f13e970a34ef58e8ea1fb439b9b38"
17
+ }
18
+ },
19
+ "full_partition": {
20
+ "records": 72,
21
+ "questions": 72,
22
+ "clean_records": 72,
23
+ "clean_questions": 72,
24
+ "headline_questions": 72,
25
+ "variants": {
26
+ "clean": 72
27
+ },
28
+ "clean_sources": {
29
+ "policy": 12,
30
+ "intent": 12,
31
+ "ordinal": 12,
32
+ "extraction": 12,
33
+ "adequacy": 12,
34
+ "routing": 12
35
+ },
36
+ "question_types": {
37
+ "noul": 24,
38
+ "choice": 36,
39
+ "score": 12
40
+ },
41
+ "maximum_options": 6
42
+ },
43
+ "selected": {
44
+ "records": 72,
45
+ "questions": 72,
46
+ "clean_records": 72,
47
+ "clean_questions": 72,
48
+ "headline_questions": 72,
49
+ "variants": {
50
+ "clean": 72
51
+ },
52
+ "clean_sources": {
53
+ "policy": 12,
54
+ "intent": 12,
55
+ "ordinal": 12,
56
+ "extraction": 12,
57
+ "adequacy": 12,
58
+ "routing": 12
59
+ },
60
+ "question_types": {
61
+ "noul": 24,
62
+ "choice": 36,
63
+ "score": 12
64
+ },
65
+ "maximum_options": 6
66
+ },
67
+ "selection": {
68
+ "method": "full",
69
+ "requested_limit": null,
70
+ "is_full_partition": true,
71
+ "note": "Full frozen public tier; not the complete published leaderboard."
72
+ },
73
+ "protocol": {
74
+ "calibration_applied": false,
75
+ "training_data_downloaded": false,
76
+ "gold_labels_sent_to_model": false,
77
+ "headline_variant": "clean",
78
+ "exclude_from_headline_sources": [],
79
+ "locked_test": false,
80
+ "notes": [
81
+ "Community benchmark; not an official TypeSafe dataset release.",
82
+ "Public original/easy/hard have 72/48/111 questions; report separately.",
83
+ "The 534-item leaderboard contains private/untracked data not downloaded here.",
84
+ "Choice request criteria retain their native order; scoring labels retain canonical order.",
85
+ "Native accuracy uses argmax with lexicographic label tie-break, including Score.",
86
+ "Score also supports expected-value MAE; original has 36 paraphrase groups.",
87
+ "One-hot labels and ten explicit mathematical probability references are separate targets.",
88
+ "Provenance, rationale and gold probabilities are never sent in requests."
89
+ ]
90
+ },
91
+ "provenance": {
92
+ "dataset_license": "MIT",
93
+ "license_url": "https://github.com/fstandhartinger/jevbench/blob/fd51755eb0c0b546ca206d764faf3302feca913e/LICENSE",
94
+ "notices": [
95
+ "LICENSE",
96
+ "THIRD-PARTY.md"
97
+ ],
98
+ "reference_probability_questions": 0,
99
+ "gold_policy": "Authored rubric labels reviewed before inference; hard items retain author and review metadata. Explicit mathematical distributions are not teacher model confidence or population frequency estimates."
100
+ }
101
+ }
server/benchmarks/run_matrix.py ADDED
@@ -0,0 +1,277 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """Run one pinned model at a time on one H200. Default is a command preview."""
3
+
4
+ import argparse
5
+ import json
6
+ import os
7
+ import shlex
8
+ import signal
9
+ import socket
10
+ import subprocess
11
+ import sys
12
+ import time
13
+ from pathlib import Path
14
+
15
+ ROOT = Path(__file__).resolve().parents[1]
16
+ RUNTIME = ROOT / "benchmarks/h200/runtime.py"
17
+
18
+
19
+ def commands(args, manifest, profile):
20
+ root = args.output.resolve() / profile
21
+ engine = [
22
+ sys.executable,
23
+ str(RUNTIME),
24
+ "launch",
25
+ profile,
26
+ "--engine-python",
27
+ # Resolving a venv Python symlink loses pyvenv.cfg and its installed engine.
28
+ str(args.engine_python.absolute()),
29
+ "--gpu",
30
+ args.gpu,
31
+ "--output",
32
+ str(root / "engine"),
33
+ "--timeout",
34
+ str(args.startup_timeout),
35
+ "--execute",
36
+ ]
37
+ if args.probe_vision:
38
+ engine.append("--probe-vision")
39
+ adapter = [
40
+ sys.executable,
41
+ "-m",
42
+ "jev_adapter",
43
+ "--engine-url",
44
+ "http://127.0.0.1:30000",
45
+ "--model",
46
+ "decision-model",
47
+ "--port",
48
+ "30120",
49
+ "--max-concurrency",
50
+ "32",
51
+ ]
52
+ model = manifest["models"][profile]
53
+ if model.get("tokenization") == "native_official_text":
54
+ adapter += [
55
+ "--tokenizer-model",
56
+ model["repo_id"],
57
+ "--tokenizer-revision",
58
+ model["revision"],
59
+ ]
60
+ suites = [
61
+ "decision-v7",
62
+ "transfer-v4",
63
+ "transfer-v9",
64
+ "jevbench-original",
65
+ "jevbench-easy",
66
+ "jevbench-hard",
67
+ ]
68
+ if args.suite:
69
+ suites = args.suite
70
+ if args.include_external:
71
+ suites += [
72
+ suite for suite in ["semif-v1", "scienthoon-v1"] if suite not in suites
73
+ ]
74
+ evaluations = []
75
+ for concurrency in args.concurrency:
76
+ run = [
77
+ sys.executable,
78
+ "-m",
79
+ "jev_adapter.benchmarks.run",
80
+ "--output",
81
+ str(root / f"c{concurrency}"),
82
+ "--engine-manifest",
83
+ str(root / "engine/launch.json"),
84
+ "--concurrency",
85
+ str(concurrency),
86
+ "--warmup",
87
+ str(args.warmup),
88
+ "--repeats",
89
+ str(args.repeats),
90
+ ]
91
+ for suite in suites:
92
+ split = "public" if suite.startswith("jevbench-") else "development"
93
+ run += [
94
+ "--data",
95
+ str(args.data_root.resolve() / suite / f"{split}.jsonl"),
96
+ ]
97
+ if args.limit is not None:
98
+ run += ["--limit", str(args.limit)]
99
+ evaluations.append(run)
100
+ return {
101
+ "profile": profile,
102
+ "model": manifest["models"][profile],
103
+ "engine": engine,
104
+ "adapter": adapter,
105
+ "evaluations": evaluations,
106
+ }
107
+
108
+
109
+ def stop_owned(process):
110
+ if process is not None and process.poll() is None:
111
+ # SIGINT can be inherited as ignored under nohup; the runtime handles TERM.
112
+ os.killpg(process.pid, signal.SIGTERM)
113
+ try:
114
+ process.wait(timeout=40)
115
+ except subprocess.TimeoutExpired as exc:
116
+ raise RuntimeError(
117
+ f"Owned process {process.pid} did not stop; check its log."
118
+ ) from exc
119
+
120
+
121
+ def wait_engine(process, ready_path, timeout):
122
+ deadline = time.monotonic() + timeout
123
+ while not ready_path.exists():
124
+ if process.poll() is not None:
125
+ raise RuntimeError(
126
+ "Engine launcher exited; see launcher.log and engine/engine.log"
127
+ )
128
+ if time.monotonic() > deadline:
129
+ raise TimeoutError("Engine startup deadline exceeded")
130
+ time.sleep(1)
131
+
132
+
133
+ def wait_adapter(process, timeout=60):
134
+ import httpx
135
+
136
+ deadline = time.monotonic() + timeout
137
+ headers = {}
138
+ if key := os.environ.get("JEV_API_KEY"):
139
+ headers["Authorization"] = f"Bearer {key}"
140
+ with httpx.Client(timeout=2) as client:
141
+ while True:
142
+ if process.poll() is not None:
143
+ raise RuntimeError("Adapter startup failed; see adapter.log")
144
+ try:
145
+ response = client.get(
146
+ "http://127.0.0.1:30120/v1/models", headers=headers
147
+ )
148
+ if response.is_success:
149
+ return
150
+ except httpx.HTTPError:
151
+ pass
152
+ if time.monotonic() > deadline:
153
+ raise TimeoutError("Adapter startup deadline exceeded")
154
+ time.sleep(1)
155
+
156
+
157
+ def main():
158
+ manifest = json.loads((RUNTIME.parent / "models.json").read_text())
159
+ parser = argparse.ArgumentParser(description=__doc__)
160
+ parser.add_argument("--output", type=Path, required=True)
161
+ parser.add_argument(
162
+ "--engine-python", type=Path, default=ROOT / ".venv-sglang/bin/python"
163
+ )
164
+ parser.add_argument("--data-root", type=Path, default=ROOT / "benchmarks/data")
165
+ parser.add_argument("--profile", action="append", choices=list(manifest["models"]))
166
+ parser.add_argument("--concurrency", type=int, action="append")
167
+ parser.add_argument("--warmup", type=int, default=20)
168
+ parser.add_argument("--repeats", type=int, default=1)
169
+ parser.add_argument("--limit", type=int)
170
+ parser.add_argument("--startup-timeout", type=int, default=3600)
171
+ parser.add_argument("--gpu", default="0")
172
+ parser.add_argument("--include-external", action="store_true")
173
+ parser.add_argument(
174
+ "--suite",
175
+ action="append",
176
+ choices=[
177
+ "decision-v7",
178
+ "transfer-v4",
179
+ "transfer-v9",
180
+ "jevbench-original",
181
+ "jevbench-easy",
182
+ "jevbench-hard",
183
+ "semif-v1",
184
+ "scienthoon-v1",
185
+ ],
186
+ )
187
+ parser.add_argument(
188
+ "--probe-vision", action="store_true", help="optional image readiness probe"
189
+ )
190
+ parser.add_argument("--execute", action="store_true")
191
+ args = parser.parse_args()
192
+ args.concurrency = args.concurrency or [1]
193
+ if (
194
+ any(c < 1 or c > 32 for c in args.concurrency)
195
+ or len(set(args.concurrency)) != len(args.concurrency)
196
+ or args.warmup < 0
197
+ or args.repeats < 1
198
+ or args.startup_timeout < 1
199
+ or (args.limit is not None and args.limit < 1)
200
+ ):
201
+ parser.error("invalid counts; concurrency must be unique values from 1 to 32")
202
+ profiles = args.profile or manifest["default_matrix"]
203
+ if len(set(profiles)) != len(profiles):
204
+ parser.error("profiles must be unique")
205
+ plans = [commands(args, manifest, profile) for profile in profiles]
206
+ for plan in plans:
207
+ print(f"\n{plan['profile']}")
208
+ for command in [plan["engine"], plan["adapter"], *plan["evaluations"]]:
209
+ print(shlex.join(command))
210
+ if not args.execute:
211
+ print("\nPreview only. Add --execute on the prepared H200 server.")
212
+ return
213
+ if not args.engine_python.is_file():
214
+ parser.error("engine Python is missing; follow benchmarks/h200/README.md")
215
+ for plan in plans:
216
+ for command in plan["evaluations"]:
217
+ for i, value in enumerate(command):
218
+ if value == "--data" and not Path(command[i + 1]).is_file():
219
+ parser.error(f"missing prepared dataset: {command[i + 1]}")
220
+ args.output.mkdir(parents=True, exist_ok=False)
221
+ (args.output / "matrix.json").write_text(json.dumps(plans, indent=2) + "\n")
222
+ for plan in plans:
223
+ # Never attach to or stop unrelated services using the benchmark ports.
224
+ for port in (30000, 30120):
225
+ with socket.socket() as listener:
226
+ # Match server bind semantics: ignore TIME_WAIT, reject live listeners.
227
+ listener.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1)
228
+ listener.bind(("127.0.0.1", port))
229
+ root = args.output.resolve() / plan["profile"]
230
+ root.mkdir()
231
+ engine, adapter = None, None
232
+ with (
233
+ (root / "launcher.log").open("x") as engine_log,
234
+ (root / "adapter.log").open("x") as adapter_log,
235
+ ):
236
+ try:
237
+ engine = subprocess.Popen(
238
+ plan["engine"],
239
+ cwd=ROOT,
240
+ stdout=engine_log,
241
+ stderr=subprocess.STDOUT,
242
+ start_new_session=True,
243
+ )
244
+ wait_engine(
245
+ engine, root / "engine/ready.json", args.startup_timeout + 90
246
+ )
247
+ adapter = subprocess.Popen(
248
+ plan["adapter"],
249
+ cwd=ROOT,
250
+ stdout=adapter_log,
251
+ stderr=subprocess.STDOUT,
252
+ start_new_session=True,
253
+ )
254
+ wait_adapter(adapter)
255
+ for command in plan["evaluations"]:
256
+ subprocess.run(command, cwd=ROOT, check=True)
257
+ finally:
258
+ try:
259
+ stop_owned(adapter)
260
+ finally:
261
+ stop_owned(engine)
262
+ subprocess.run(
263
+ [
264
+ sys.executable,
265
+ "-m",
266
+ "jev_adapter.benchmarks.compare",
267
+ str(args.output.resolve()),
268
+ "--output",
269
+ str(args.output.resolve() / "comparison.csv"),
270
+ ],
271
+ cwd=ROOT,
272
+ check=True,
273
+ )
274
+
275
+
276
+ if __name__ == "__main__":
277
+ main()
server/benchmarks/verify_metrics.py ADDED
@@ -0,0 +1,140 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Audit metric parity against verified KEV source, without importing KEV or torch.
2
+
3
+ Requires NumPy only in the optional audit environment, not in jev-adapter.
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ import argparse
9
+ import ast
10
+ import hashlib
11
+ import json
12
+ import math
13
+ import sys
14
+ from pathlib import Path
15
+
16
+ sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
17
+
18
+ from jev_adapter.benchmarks import metrics as adapter_metrics
19
+ from jev_adapter.benchmarks.data import KEV_COMMIT
20
+
21
+ REFERENCE_FILES = {
22
+ "kev/evaluate.py": (
23
+ "1b20e3f9edf417aa8dae924b1526e52f74b710cadf7213c5ec68334f6e7f8fe1",
24
+ {"ece"},
25
+ ),
26
+ "kev/benchmark.py": (
27
+ "4192ec3b26b065452f84bde38a091e6854a28fe185d2e0f39a0c91b2efe69df7",
28
+ {"coverage_at_error", "metrics"},
29
+ ),
30
+ "kev/contrastive.py": (
31
+ "cbb979aa5d40265ad0e64695f94b281d91751fa405811ddfa8212fede111edcf",
32
+ {"paired_flip"},
33
+ ),
34
+ }
35
+
36
+
37
+ def extract_reference(root, numpy):
38
+ scope = {"np": numpy, "math": math, "EPSILON": 1e-9}
39
+ for relative, (expected_hash, names) in REFERENCE_FILES.items():
40
+ payload = (root / relative).read_bytes()
41
+ if hashlib.sha256(payload).hexdigest() != expected_hash:
42
+ raise ValueError(
43
+ f"reference file differs from KEV {KEV_COMMIT}: {relative}"
44
+ )
45
+ tree = ast.parse(payload, filename=relative)
46
+ tree.body = [
47
+ node
48
+ for node in tree.body
49
+ if isinstance(node, ast.FunctionDef) and node.name in names
50
+ ]
51
+ if {node.name for node in tree.body} != names:
52
+ raise ValueError(f"missing reference functions in {relative}")
53
+ # Execute only the hash-verified function definitions. Top-level imports,
54
+ # model loading, decorators elsewhere and benchmark entry points are absent.
55
+ exec(compile(tree, relative, "exec"), scope) # noqa: S102
56
+ return scope
57
+
58
+
59
+ def audit(root, numpy):
60
+ reference = extract_reference(root, numpy)
61
+ rng = numpy.random.default_rng(483)
62
+ maximum_delta = {}
63
+ tolerance = 1e-12
64
+ for batch in range(100):
65
+ rows = []
66
+ for index in range(50):
67
+ kind = ["choice", "noul", "score"][index % 3]
68
+ options = 2 if kind == "noul" else int(rng.integers(2, 11))
69
+ probabilities = rng.dirichlet(numpy.ones(options)).tolist()
70
+ label = int(rng.integers(options))
71
+ if index < 10:
72
+ probabilities = [index / 10, 1 - index / 10]
73
+ label %= 2
74
+ kind = "noul"
75
+ rows.append({"p": probabilities, "label": label, "type": kind})
76
+ expected = reference["metrics"](rows)
77
+ actual = adapter_metrics.metrics(rows)
78
+ for key, value in actual.items():
79
+ if value is None or not isinstance(value, (int, float)):
80
+ continue
81
+ delta = abs(value - expected[key])
82
+ maximum_delta[key] = max(maximum_delta.get(key, 0), delta)
83
+ if delta > tolerance:
84
+ raise ValueError(
85
+ f"batch {batch}, {key}: adapter={value}, KEV={expected[key]}"
86
+ )
87
+ rows = []
88
+ for index in range(10):
89
+ for sibling, label in (("a", 0), ("b", index % 2)):
90
+ rows.append(
91
+ {
92
+ "pair_id": str(index),
93
+ "question": "q",
94
+ "keys": ["x", "y"],
95
+ "label": label,
96
+ "p": [0.7, 0.3] if index % 3 else [0.4, 0.6],
97
+ "sibling": sibling,
98
+ }
99
+ )
100
+ expected = reference["paired_flip"](rows)
101
+ actual = adapter_metrics.paired_flip(rows)
102
+ for key, value in expected.items():
103
+ if actual[key] != value:
104
+ raise ValueError(f"pair metric differs: {key}: {actual[key]} != {value}")
105
+ return {
106
+ "status": "passed",
107
+ "kev_commit": KEV_COMMIT,
108
+ "reference_sha256": {path: entry[0] for path, entry in REFERENCE_FILES.items()},
109
+ "adapter_metrics_sha256": hashlib.sha256(
110
+ Path(adapter_metrics.__file__).read_bytes()
111
+ ).hexdigest(),
112
+ "numpy_version": numpy.__version__,
113
+ "seed": 483,
114
+ "batches": 100,
115
+ "rows_per_batch": 50,
116
+ "total_metric_rows": 5000,
117
+ "absolute_tolerance": tolerance,
118
+ "maximum_absolute_delta": maximum_delta,
119
+ "complete_pair_metrics_exact": expected,
120
+ "limitations": [
121
+ "Numerical metric parity, not GPU/model inference validation.",
122
+ "Only common returned scalar metrics and complete pair metrics compared.",
123
+ "Partial-pair handling and unknowable reporting intentionally differ.",
124
+ ],
125
+ }
126
+
127
+
128
+ def main():
129
+ parser = argparse.ArgumentParser(description=__doc__)
130
+ parser.add_argument("--kev-root", required=True, type=Path)
131
+ args = parser.parse_args()
132
+ try:
133
+ import numpy
134
+ except ImportError:
135
+ parser.error("NumPy is required in this optional audit environment")
136
+ print(json.dumps(audit(args.kev_root, numpy), indent=2, allow_nan=False))
137
+
138
+
139
+ if __name__ == "__main__":
140
+ main()
server/examples/request.json ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": "jev-latest",
3
+ "state": "결제가 두 번 되었습니다. 중복 결제 건을 환불해 주세요.",
4
+ "questions": {
5
+ "route": {
6
+ "type": "choice",
7
+ "instructions": "어느 부서로 보내야 하나?",
8
+ "criteria": {"billing": "결제 및 환불", "technical": "기술 지원", "sales": "상품 문의"}
9
+ },
10
+ "refund": {"type": "noul", "instructions": "환불을 요청했나?"},
11
+ "urgency": {"type": "score", "instructions": "대응 긴급도", "criteria": ["낮음", "보통", "높음"]}
12
+ }
13
+ }
server/examples/smoke.py ADDED
@@ -0,0 +1,49 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Smoke-test a running adapter, optionally including actual image bytes."""
2
+
3
+ import argparse
4
+ import base64
5
+ import json
6
+ import mimetypes
7
+ import os
8
+ import time
9
+ from pathlib import Path
10
+ from urllib.request import Request, urlopen
11
+
12
+
13
+ def main():
14
+ parser = argparse.ArgumentParser(description=__doc__)
15
+ parser.add_argument("--base-url", default="http://127.0.0.1:30120")
16
+ parser.add_argument(
17
+ "--request", type=Path, default=Path(__file__).with_name("request.json")
18
+ )
19
+ parser.add_argument("--image", type=Path, action="append", default=[])
20
+ args = parser.parse_args()
21
+ body = json.loads(args.request.read_text())
22
+ for path in args.image:
23
+ mime = mimetypes.guess_type(path.name)[0]
24
+ if not mime or not mime.startswith("image/"):
25
+ parser.error(f"Cannot identify image type: {path.name}")
26
+ encoded = base64.b64encode(path.read_bytes()).decode("ascii")
27
+ body.setdefault("images", []).append(f"data:{mime};base64,{encoded}")
28
+ headers = {"Content-Type": "application/json"}
29
+ if key := os.environ.get("JEV_API_KEY"):
30
+ headers["Authorization"] = f"Bearer {key}"
31
+ request = Request(
32
+ args.base_url.rstrip("/") + "/v1/systemone",
33
+ data=json.dumps(body).encode(),
34
+ headers=headers,
35
+ )
36
+ started = time.perf_counter()
37
+ with urlopen(request, timeout=120) as response:
38
+ result = json.load(response)
39
+ elapsed = (time.perf_counter() - started) * 1000
40
+ if result.get("usage", {}).get("output_tokens") != 0:
41
+ raise RuntimeError("Expected zero generated tokens")
42
+ if set(result.get("answers", {})) != set(body["questions"]):
43
+ raise RuntimeError("Missing decision answers")
44
+ print(json.dumps(result, indent=2, ensure_ascii=False))
45
+ print(f"End-to-end HTTP latency: {elapsed:.1f} ms (not a GPU benchmark)")
46
+
47
+
48
+ if __name__ == "__main__":
49
+ main()
server/jev_adapter/benchmarks/__init__.py ADDED
@@ -0,0 +1 @@
 
 
1
+ """Reproducible, training-free decision benchmarks over the adapter HTTP API."""
server/jev_adapter/benchmarks/compare.py ADDED
@@ -0,0 +1,72 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Collect complete benchmark reports into a comparison table, one row per suite."""
2
+
3
+ import argparse
4
+ import csv
5
+ import json
6
+ from pathlib import Path
7
+
8
+
9
+ def collect(root):
10
+ rows = []
11
+ for path in sorted(root.rglob("report.json")):
12
+ if path.with_name("invalidated.json").exists():
13
+ continue
14
+ report = json.loads(path.read_text())
15
+ if report.get("status") != "complete" or not report.get("suites"):
16
+ continue
17
+ manifest = json.loads(path.with_name("manifest.json").read_text())
18
+ launch = manifest.get("launch_manifest", {})
19
+ engine_model = (
20
+ manifest.get("engine", {}).get("model_info", {}).get("model_path")
21
+ )
22
+ engine_revision = (
23
+ manifest.get("engine", {}).get("server_info", {}).get("revision")
24
+ )
25
+ for suite, result in report["suites"].items():
26
+ clean = result.get("clean")
27
+ if clean is None:
28
+ continue
29
+ latency = result["latency_ms"]
30
+ rows.append(
31
+ {
32
+ "profile": launch.get("profile", manifest["requested_model"]),
33
+ "model_repo": launch.get("model", {}).get("repo_id", engine_model),
34
+ "model_revision": launch.get("model", {}).get(
35
+ "revision", engine_revision
36
+ ),
37
+ "suite": suite,
38
+ "concurrency": manifest["concurrency"],
39
+ "partial_dataset": report["partial_dataset"],
40
+ "clean_questions": clean["n"],
41
+ "accuracy": clean["acc"],
42
+ "brier": clean["brier"],
43
+ "ece": clean["ece"],
44
+ "confident_error_rate": clean["confident_error_rate"],
45
+ "p50_ms": latency["p50"],
46
+ "p95_ms": latency["p95"],
47
+ "p99_ms": latency["p99"],
48
+ "cache_mode": manifest["cache_mode"],
49
+ "report": str(path.resolve()),
50
+ }
51
+ )
52
+ return rows
53
+
54
+
55
+ def main():
56
+ parser = argparse.ArgumentParser(description=__doc__)
57
+ parser.add_argument("root", type=Path)
58
+ parser.add_argument("--output", type=Path, required=True)
59
+ args = parser.parse_args()
60
+ rows = collect(args.root)
61
+ if not rows:
62
+ parser.error("no complete benchmark reports found; failed runs are not ranked")
63
+ args.output.parent.mkdir(parents=True, exist_ok=True)
64
+ with args.output.open("x", newline="") as handle:
65
+ writer = csv.DictWriter(handle, fieldnames=list(rows[0]))
66
+ writer.writeheader()
67
+ writer.writerows(rows)
68
+ print(f"Wrote {len(rows)} suite rows to {args.output}; latency is HTTP wall time.")
69
+
70
+
71
+ if __name__ == "__main__":
72
+ main()
server/jev_adapter/benchmarks/data.py ADDED
@@ -0,0 +1,217 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Normalize pinned, public KEV evaluation records without changing their tasks.
2
+
3
+ Only the request is sent to a model. Labels and source metadata remain outside
4
+ that request, in a separate expected/metadata envelope used by the evaluator.
5
+ This module is an independent format conversion, not imported KEV model code.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import copy
11
+ import hashlib
12
+ import json
13
+ from collections import Counter
14
+ from dataclasses import dataclass
15
+ from typing import Any
16
+
17
+ KEV_COMMIT = "4f8110a3f8620cc3a182ae9a708e4398492c4b1a"
18
+ KEV_REPOSITORY = "https://github.com/jaredpalmer/kev"
19
+ KEV_RAW = f"https://raw.githubusercontent.com/jaredpalmer/kev/{KEV_COMMIT}"
20
+ DEFAULT_SUITES = ("decision-v7", "transfer-v4", "transfer-v9")
21
+
22
+
23
+ @dataclass(frozen=True)
24
+ class SuiteSpec:
25
+ path: str
26
+ manifest_sha256: str
27
+ description: str
28
+ notes: tuple[str, ...] = ()
29
+
30
+
31
+ SUITES = {
32
+ "decision-v7": SuiteSpec(
33
+ "evals/v7/decision-v7",
34
+ "a8f50e481b7d90b97da049e0ff6a01cee2f1ed204aed61a8265af0edbb5514d2",
35
+ "Ten public sources and generated policies; KEV trained-source evaluation.",
36
+ ("Includes up to 78 choices with none-of-the-above variants.",),
37
+ ),
38
+ "transfer-v4": SuiteSpec(
39
+ "evals/v4/transfer-v4",
40
+ "31677c2256b406222e7d94ffdc0a02a70ce05746b9efe307876024c4e77291d1",
41
+ "Six sources unseen in KEV fine-tuning and held-out policy structures.",
42
+ ("Unseen means unseen in KEV fine-tuning, not in base-model pretraining.",),
43
+ ),
44
+ "transfer-v9": SuiteSpec(
45
+ "evals/v9/transfer-v9",
46
+ "3c4f0be94509a3612678bfd3a30fd99a8d0ca3c47ddfe7318075d95b2fa365e4",
47
+ "Transfer-v4 plus MMLU-Pro, buried evidence and unknowable/control pairs.",
48
+ (
49
+ "Contains transfer-v4 records; do not pool both suites as independent data.",
50
+ "Source 'unknowable' is evaluated for confidence, not accuracy.",
51
+ ),
52
+ ),
53
+ "semif-v1": SuiteSpec(
54
+ "evals/external/semif-v1",
55
+ "0de05eac16b0ddeeb2719c50a94a9148d6ae195f66303aec74aa103a3845ad11",
56
+ "SemIf's 144 authored choices plus 108 perturbations, frozen by KEV.",
57
+ (
58
+ "Already evaluated by KEV; this is not an additional independent suite.",
59
+ "SemIf's own headline is mean family balanced accuracy.",
60
+ ),
61
+ ),
62
+ "scienthoon-v1": SuiteSpec(
63
+ "evals/external/scienthoon-v1",
64
+ "ef31183425bf9d3c2d8ac5d245a14d30fa44d1e531c945e4405edcb1f75ae0d5",
65
+ "KEV's frozen conversion of scienthoon's synthetic support tickets.",
66
+ (
67
+ "Already evaluated by KEV; this is not an additional independent suite.",
68
+ "Original 900 question rows become 291 unique states / 873 questions in this conversion.",
69
+ "Priority labels depend on an organizational rule absent from the state; report separately.",
70
+ ),
71
+ ),
72
+ }
73
+
74
+
75
+ def sha256(data: bytes) -> str:
76
+ return hashlib.sha256(data).hexdigest()
77
+
78
+
79
+ def require_sha256(data: bytes, expected: str, name: str) -> None:
80
+ actual = sha256(data)
81
+ if actual != expected:
82
+ raise ValueError(
83
+ f"SHA256 mismatch for {name}: expected {expected}, got {actual}"
84
+ )
85
+
86
+
87
+ def normalize_record(raw: dict[str, Any], suite: str, split: str) -> dict[str, Any]:
88
+ """Retain option order, all sibling questions and KEV's original provenance."""
89
+ meta = copy.deepcopy(raw["_meta"])
90
+ if not isinstance(meta.get("id"), str) or not meta["id"]:
91
+ raise ValueError("record requires a nonempty _meta.id")
92
+ questions, expected = {}, {}
93
+ for qid, question in raw["questions"].items():
94
+ kind = question["type"]
95
+ if kind == "choice":
96
+ labels = list(question["criteria"])
97
+ try:
98
+ label = labels.index(question["label"])
99
+ except ValueError as exc:
100
+ raise ValueError(f"{meta['id']}/{qid}: label is not an option") from exc
101
+ elif kind == "noul":
102
+ if type(question["label"]) is not bool:
103
+ raise ValueError(f"{meta['id']}/{qid}: noul label must be a boolean")
104
+ labels, label = ["false", "true"], int(question["label"])
105
+ elif kind == "score":
106
+ labels = [str(i) for i in range(len(question["criteria"]))]
107
+ label = question["label"]
108
+ if type(label) is not int or not 0 <= label < len(labels):
109
+ raise ValueError(f"{meta['id']}/{qid}: score label is out of range")
110
+ else:
111
+ raise ValueError(f"unsupported question type: {kind}")
112
+ if len(labels) < 2 or len(set(labels)) != len(labels):
113
+ raise ValueError(f"{meta['id']}/{qid}: invalid option labels")
114
+ questions[qid] = {
115
+ key: copy.deepcopy(question[key])
116
+ for key in ("type", "instructions", "criteria")
117
+ if key in question
118
+ }
119
+ expected[qid] = {
120
+ "labels": labels,
121
+ "target": [float(i == label) for i in range(len(labels))],
122
+ "label": label,
123
+ "type": kind,
124
+ "task": question.get("src", meta["source"]),
125
+ }
126
+ if not questions:
127
+ raise ValueError(f"{meta['id']}: no questions")
128
+ record = {"state": copy.deepcopy(raw["state"]), "questions": questions}
129
+ for key in ("images", "options"):
130
+ if key in raw:
131
+ record[key] = copy.deepcopy(raw[key])
132
+ return {
133
+ "id": meta["id"],
134
+ "suite": suite,
135
+ "split": split,
136
+ "source": meta["source"],
137
+ "variant": meta.get("variant", "clean"),
138
+ "record": record,
139
+ "expected": expected,
140
+ "metadata": meta,
141
+ }
142
+
143
+
144
+ def parse_partition(data: bytes, suite: str, split: str) -> list[dict[str, Any]]:
145
+ records, seen = [], set()
146
+ for number, line in enumerate(data.decode("utf-8").splitlines(), 1):
147
+ if not line.strip():
148
+ raise ValueError(f"{suite}/{split}:{number}: blank JSONL record")
149
+ row = normalize_record(json.loads(line), suite, split)
150
+ if row["id"] in seen:
151
+ raise ValueError(f"duplicate record id: {row['id']}")
152
+ seen.add(row["id"])
153
+ records.append(row)
154
+ return records
155
+
156
+
157
+ def population_counts(records: list[dict[str, Any]]) -> dict[str, Any]:
158
+ clean = [row for row in records if row["variant"] == "clean"]
159
+ return {
160
+ "records": len(records),
161
+ "questions": sum(len(row["expected"]) for row in records),
162
+ "clean_records": len(clean),
163
+ "clean_questions": sum(len(row["expected"]) for row in clean),
164
+ "headline_questions": sum(
165
+ len(row["expected"]) for row in clean if row["source"] != "unknowable"
166
+ ),
167
+ "variants": dict(Counter(row["variant"] for row in records)),
168
+ "clean_sources": dict(Counter(row["source"] for row in clean)),
169
+ "question_types": dict(
170
+ Counter(q["type"] for row in records for q in row["expected"].values())
171
+ ),
172
+ "maximum_options": max(
173
+ (len(q["labels"]) for row in records for q in row["expected"].values()),
174
+ default=0,
175
+ ),
176
+ }
177
+
178
+
179
+ def source_provenance(
180
+ records: list[dict[str, Any]], upstream_manifest: dict[str, Any]
181
+ ) -> dict[str, Any]:
182
+ """Point to underlying dataset licenses; do not relicense mixed source data."""
183
+ datasets = {}
184
+ for row in records:
185
+ meta = row["metadata"]
186
+ repo = meta.get("repo")
187
+ if not repo:
188
+ continue
189
+ revision = meta.get("revision")
190
+ key = (repo, revision)
191
+ external = upstream_manifest.get("external", {})
192
+ is_external = external.get("repo", "").endswith("/" + repo)
193
+ datasets[key] = {
194
+ "repository": repo,
195
+ "revision": revision,
196
+ "source_url": (
197
+ external["repo"]
198
+ if is_external
199
+ else f"https://huggingface.co/datasets/{repo}"
200
+ ),
201
+ "license": external.get("license")
202
+ if is_external
203
+ else "see upstream dataset",
204
+ }
205
+ return {
206
+ "kev_repository_license": "Apache-2.0",
207
+ "kev_license_url": f"{KEV_REPOSITORY}/blob/{KEV_COMMIT}/LICENSE",
208
+ "dataset_notice": (
209
+ "Public and downloadable does not mean all source datasets share Apache-2.0. "
210
+ "Their individual licenses and attribution terms continue to apply."
211
+ ),
212
+ "datasets": list(datasets.values()),
213
+ "external": upstream_manifest.get("external"),
214
+ "dataset_revisions_from_manifest": upstream_manifest.get(
215
+ "dataset_revisions", {}
216
+ ),
217
+ }
server/jev_adapter/benchmarks/jevbench.py ADDED
@@ -0,0 +1,292 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Prepare JevBench's three pinned public tiers without exposing answer metadata.
2
+
3
+ This is an independent format conversion. The public 231 questions are not the
4
+ 534-question leaderboard, whose remaining questions are private or untracked.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import copy
10
+ import json
11
+ import math
12
+ from dataclasses import dataclass
13
+ from pathlib import Path
14
+ from typing import Any
15
+
16
+ import httpx
17
+
18
+ from jev_adapter.protocol import SystemOneRequest
19
+
20
+ from .data import normalize_record, population_counts, require_sha256, sha256
21
+
22
+ JEVBENCH_COMMIT = "fd51755eb0c0b546ca206d764faf3302feca913e"
23
+ JEVBENCH_REPOSITORY = "https://github.com/fstandhartinger/jevbench"
24
+ JEVBENCH_RAW = (
25
+ f"https://raw.githubusercontent.com/fstandhartinger/jevbench/{JEVBENCH_COMMIT}"
26
+ )
27
+ JEVBENCH_NOTICES = {
28
+ "LICENSE": "3e5beed774bb0bcbfb2fcf24ba9554212c6ae112937d308040989465ab0c5784",
29
+ "THIRD-PARTY.md": "396422e29055bba4073f4a9e5a7163cc724f13e970a34ef58e8ea1fb439b9b38",
30
+ }
31
+
32
+
33
+ @dataclass(frozen=True)
34
+ class JevBenchSpec:
35
+ path: str
36
+ sha256: str
37
+ records: int
38
+ description: str
39
+
40
+
41
+ JEVBENCH_SUITES = {
42
+ "jevbench-original": JevBenchSpec(
43
+ "datasets/public/original.jsonl",
44
+ "5c2414edb3006b8bfcb70fda433f0f9ca015759433849f8d3104328a1f7c4180",
45
+ 72,
46
+ "Public original tier: 72 decisions in 36 paraphrase pairs.",
47
+ ),
48
+ "jevbench-easy": JevBenchSpec(
49
+ "datasets/public/easy.jsonl",
50
+ "231df3c2c8e88a1a8c137ebe85de96ba70fabd330849098ac7b3c52c70b7172b",
51
+ 48,
52
+ "Public easy tier: 48 explicit facts, intents and tool selections.",
53
+ ),
54
+ "jevbench-hard": JevBenchSpec(
55
+ "datasets/public/hard.jsonl",
56
+ "89e9e6becb33ed88c1de7d42dcc87531b2fb64cfaef4e1986faf7c37b3f80ebb",
57
+ 111,
58
+ "Public hard tier: 111 decisions; ten supply probability references.",
59
+ ),
60
+ }
61
+
62
+
63
+ def normalize_jevbench_record(raw: dict[str, Any], suite: str) -> dict[str, Any]:
64
+ """Keep native question/option order and canonical scoring label order."""
65
+ if raw.get("split") != "public":
66
+ raise ValueError("JevBench converter accepts only the public subset")
67
+ if not isinstance(raw.get("id"), str) or not raw["id"]:
68
+ raise ValueError("JevBench record requires a nonempty id")
69
+ labels = raw["labels"]
70
+ if (
71
+ not isinstance(labels, list)
72
+ or len(labels) < 2
73
+ or any(not isinstance(label, str) or not label for label in labels)
74
+ or len(set(labels)) != len(labels)
75
+ ):
76
+ raise ValueError(f"{raw['id']}: invalid canonical labels")
77
+ question = raw["question"]
78
+ kind, gold = question["type"], raw["expected"]
79
+ if kind == "choice":
80
+ if not isinstance(question["criteria"], dict) or set(labels) != set(
81
+ question["criteria"]
82
+ ):
83
+ raise ValueError(f"{raw['id']}: choice criteria/labels mismatch")
84
+ if not isinstance(gold, str) or gold not in labels:
85
+ raise ValueError(f"{raw['id']}: invalid choice expected label")
86
+ canonical_labels, canonical_gold = list(labels), gold
87
+ elif kind == "noul":
88
+ if labels != ["no", "yes"] or gold not in ("no", "yes"):
89
+ raise ValueError(f"{raw['id']}: noul requires no/yes canonical labels")
90
+ canonical_labels = ["false", "true"]
91
+ canonical_gold, gold = ("true" if gold == "yes" else "false"), gold == "yes"
92
+ elif kind == "score":
93
+ if not isinstance(question["criteria"], list) or labels != [
94
+ str(i) for i in range(len(question["criteria"]))
95
+ ]:
96
+ raise ValueError(f"{raw['id']}: score criteria/labels mismatch")
97
+ if type(gold) is not int or not 0 <= gold < len(labels):
98
+ raise ValueError(f"{raw['id']}: invalid score expected label")
99
+ canonical_labels, canonical_gold = list(labels), str(gold)
100
+ else:
101
+ raise ValueError(f"unsupported JevBench question type: {kind}")
102
+ provenance = copy.deepcopy(raw["provenance"])
103
+ reference = provenance.get("gold_probs")
104
+ if provenance.get("exclude_reason"):
105
+ raise ValueError(f"{raw['id']}: excluded upstream item is not evaluable")
106
+ family = raw["family"]
107
+ if not isinstance(family, str) or not family:
108
+ raise ValueError(f"{raw['id']}: missing task family")
109
+ # Only these three native question fields are allowed into the request.
110
+ clean_question = {
111
+ key: copy.deepcopy(question[key])
112
+ for key in ("type", "instructions", "criteria")
113
+ if key in question
114
+ }
115
+ clean_question["label"] = gold
116
+ row = normalize_record(
117
+ {
118
+ "state": raw["state"],
119
+ "questions": {"decision": clean_question},
120
+ "_meta": {
121
+ "id": raw["id"],
122
+ "source": family,
123
+ "variant": "clean",
124
+ "group_id": raw.get("group") or raw["id"],
125
+ "upstream_group": raw.get("group"),
126
+ "canonical_labels": list(labels),
127
+ "provenance": provenance,
128
+ "gold_policy": {
129
+ "hard_label": "authored_reviewed_rubric",
130
+ "reference_probs": (
131
+ "countable_mathematical_probability"
132
+ if reference is not None
133
+ else None
134
+ ),
135
+ "argmax_tie_break": "lexicographic_label",
136
+ },
137
+ },
138
+ },
139
+ suite,
140
+ "public",
141
+ )
142
+ expected = row["expected"]["decision"]
143
+ expected["labels"] = canonical_labels
144
+ expected["label"] = canonical_labels.index(canonical_gold)
145
+ expected["target"] = [float(label == canonical_gold) for label in canonical_labels]
146
+ if reference is not None:
147
+ if not isinstance(reference, dict) or set(reference) != set(labels):
148
+ raise ValueError(f"{raw['id']}: reference probability labels mismatch")
149
+ probabilities = [reference[label] for label in labels]
150
+ if any(
151
+ isinstance(value, bool)
152
+ or not isinstance(value, (float, int))
153
+ or not math.isfinite(value)
154
+ or not 0 <= value <= 1
155
+ for value in probabilities
156
+ ) or not math.isclose(sum(probabilities), 1.0, rel_tol=0, abs_tol=1e-9):
157
+ raise ValueError(f"{raw['id']}: invalid reference probabilities")
158
+ expected["reference_probs"] = [float(value) for value in probabilities]
159
+ SystemOneRequest.model_validate({"model": "validation-only", **row["record"]})
160
+ return row
161
+
162
+
163
+ def parse_jevbench(data: bytes, suite: str) -> list[dict[str, Any]]:
164
+ records, seen = [], set()
165
+ for number, line in enumerate(data.decode("utf-8").splitlines(), 1):
166
+ if not line.strip():
167
+ raise ValueError(f"{suite}/public:{number}: blank JSONL record")
168
+ row = normalize_jevbench_record(json.loads(line), suite)
169
+ if row["id"] in seen:
170
+ raise ValueError(f"duplicate record id: {row['id']}")
171
+ seen.add(row["id"])
172
+ records.append(row)
173
+ return records
174
+
175
+
176
+ def prepare_jevbench(
177
+ suite: str,
178
+ output: Path,
179
+ *,
180
+ client: httpx.Client | None = None,
181
+ source_root: Path | None = None,
182
+ limit: int | None = None,
183
+ ) -> dict[str, Any]:
184
+ """Verify all source bytes before writing an immutable prepared public tier."""
185
+ if suite not in JEVBENCH_SUITES:
186
+ raise ValueError(f"unknown JevBench suite: {suite}")
187
+ if limit is not None and (type(limit) is not int or limit < 1):
188
+ raise ValueError("--limit must be positive")
189
+
190
+ def read_source(path: str) -> bytes:
191
+ if source_root is not None:
192
+ return (source_root / path).read_bytes()
193
+ if client is None:
194
+ raise ValueError("an HTTP client or --source-root is required")
195
+ response = client.get(f"{JEVBENCH_RAW}/{path}")
196
+ response.raise_for_status()
197
+ return response.content
198
+
199
+ spec = JEVBENCH_SUITES[suite]
200
+ raw_bytes = read_source(spec.path)
201
+ require_sha256(raw_bytes, spec.sha256, spec.path)
202
+ notices = {}
203
+ for path, digest in JEVBENCH_NOTICES.items():
204
+ content = read_source(path)
205
+ require_sha256(content, digest, path)
206
+ notices[path] = content
207
+ records = parse_jevbench(raw_bytes, suite)
208
+ counts = population_counts(records)
209
+ if counts["records"] != spec.records or counts["questions"] != spec.records:
210
+ raise ValueError(f"upstream record/question count mismatch: {suite}/public")
211
+ selected = records if limit is None else records[:limit]
212
+ serialized = "".join(
213
+ json.dumps(row, ensure_ascii=False, allow_nan=False) + "\n" for row in selected
214
+ ).encode("utf-8")
215
+ full = len(selected) == len(records)
216
+ manifest = {
217
+ "schema_version": 1,
218
+ "suite": suite,
219
+ "split": "public",
220
+ "description": spec.description,
221
+ "data_file": "public.jsonl",
222
+ "data_sha256": sha256(serialized),
223
+ "upstream": {
224
+ "repository": JEVBENCH_REPOSITORY,
225
+ "commit": JEVBENCH_COMMIT,
226
+ "path": spec.path,
227
+ "url": f"{JEVBENCH_RAW}/{spec.path}",
228
+ "sha256": spec.sha256,
229
+ "notice_sha256": dict(JEVBENCH_NOTICES),
230
+ },
231
+ "full_partition": counts,
232
+ "selected": population_counts(selected),
233
+ "selection": {
234
+ "method": "full" if full else "prefix",
235
+ "requested_limit": limit,
236
+ "is_full_partition": full,
237
+ "note": (
238
+ "Full frozen public tier; not the complete published leaderboard."
239
+ if full
240
+ else "Smoke subset only; not a full-tier result. Pairs may be incomplete."
241
+ ),
242
+ },
243
+ "protocol": {
244
+ "calibration_applied": False,
245
+ "training_data_downloaded": False,
246
+ "gold_labels_sent_to_model": False,
247
+ "headline_variant": "clean",
248
+ "exclude_from_headline_sources": [],
249
+ "locked_test": False,
250
+ "notes": [
251
+ "Community benchmark; not an official TypeSafe dataset release.",
252
+ "Public original/easy/hard have 72/48/111 questions; report separately.",
253
+ "The 534-item leaderboard contains private/untracked data not downloaded here.",
254
+ "Choice request criteria retain their native order; scoring labels retain canonical order.",
255
+ "Native accuracy uses argmax with lexicographic label tie-break, including Score.",
256
+ "Score also supports expected-value MAE; original has 36 paraphrase groups.",
257
+ "One-hot labels and ten explicit mathematical probability references are separate targets.",
258
+ "Provenance, rationale and gold probabilities are never sent in requests.",
259
+ ],
260
+ },
261
+ "provenance": {
262
+ "dataset_license": "MIT",
263
+ "license_url": f"{JEVBENCH_REPOSITORY}/blob/{JEVBENCH_COMMIT}/LICENSE",
264
+ "notices": list(JEVBENCH_NOTICES),
265
+ "reference_probability_questions": sum(
266
+ "reference_probs" in row["expected"]["decision"] for row in records
267
+ ),
268
+ "gold_policy": (
269
+ "Authored rubric labels reviewed before inference; hard items retain author "
270
+ "and review metadata. Explicit mathematical distributions are not teacher "
271
+ "model confidence or population frequency estimates."
272
+ ),
273
+ },
274
+ }
275
+ artifacts = {
276
+ "public.jsonl": serialized,
277
+ "public.manifest.json": (
278
+ json.dumps(manifest, ensure_ascii=False, indent=2, allow_nan=False) + "\n"
279
+ ).encode("utf-8"),
280
+ **notices,
281
+ }
282
+ directory = output / suite
283
+ for name, content in artifacts.items():
284
+ path = directory / name
285
+ if path.exists() and path.read_bytes() != content:
286
+ raise FileExistsError(
287
+ f"refusing to overwrite a different prepared artifact: {path}"
288
+ )
289
+ directory.mkdir(parents=True, exist_ok=True)
290
+ for name, content in artifacts.items():
291
+ (directory / name).write_bytes(content)
292
+ return manifest
server/jev_adapter/benchmarks/metrics.py ADDED
@@ -0,0 +1,326 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Dependency-free metrics following KEV's frozen benchmark definitions.
2
+
3
+ Reimplemented against jaredpalmer/kev commit
4
+ 4f8110a3f8620cc3a182ae9a708e4398492c4b1a, kev/benchmark.py,
5
+ kev/evaluate.py and kev/contrastive.py (Apache-2.0). Differences are explicit:
6
+ partial pairs are counted, and unknowable items never receive accuracy metrics.
7
+ """
8
+
9
+ import math
10
+ from collections import defaultdict
11
+ from statistics import fmean
12
+
13
+
14
+ def argmax(values):
15
+ return max(range(len(values)), key=values.__getitem__)
16
+
17
+
18
+ def predicted_index(row):
19
+ if row.get("argmax_tie_break") == "lexicographic_label":
20
+ return min(range(len(row["p"])), key=lambda i: (-row["p"][i], row["keys"][i]))
21
+ return argmax(row["p"])
22
+
23
+
24
+ def distribution(raw, labels):
25
+ if not isinstance(raw, dict) or set(raw) != set(labels):
26
+ raise ValueError("probability keys differ from requested labels")
27
+ p = [raw[key] for key in labels]
28
+ if any(
29
+ type(value) not in (int, float)
30
+ or not math.isfinite(value)
31
+ or not 0 <= value <= 1
32
+ for value in p
33
+ ):
34
+ raise ValueError("invalid probability value")
35
+ total = math.fsum(p)
36
+ if total <= 0 or abs(total - 1) > max(1e-5, len(labels) * 0.005 + 1e-8):
37
+ raise ValueError("invalid probability sum")
38
+ return [value / total for value in p], total
39
+
40
+
41
+ def prediction_rows(item, response):
42
+ expected = item["expected"]
43
+ answers = response.get("answers")
44
+ if not isinstance(answers, dict) or set(answers) != set(expected):
45
+ raise ValueError("response question IDs differ from request")
46
+ rows = []
47
+ meta = item["metadata"]
48
+ for qid, target in expected.items():
49
+ answer = answers[qid]
50
+ kind = target["type"]
51
+ if not isinstance(answer, dict) or answer.get("type") != kind:
52
+ raise ValueError("response answer type differs from request")
53
+ if kind == "noul":
54
+ value = answer.get("noul")
55
+ if type(value) not in (int, float):
56
+ raise ValueError("invalid noul probability")
57
+ raw = {"true": value, "false": 1 - value}
58
+ else:
59
+ raw = answer.get("probabilities")
60
+ labels = target["labels"]
61
+ p, total = distribution(raw, labels)
62
+ rows.append(
63
+ {
64
+ "id": item["id"],
65
+ "suite": item["suite"],
66
+ "split": item["split"],
67
+ "question": qid,
68
+ "source": item["source"],
69
+ "task": target["task"],
70
+ "type": kind,
71
+ "variant": item["variant"],
72
+ "keys": labels,
73
+ "label": target["label"],
74
+ "p": p,
75
+ "group": meta.get("group_id", item["id"]),
76
+ "pair_id": meta.get("pair_id"),
77
+ "sibling": meta.get("sibling"),
78
+ "control_id": meta.get("control_id"),
79
+ "parent": meta.get("parent_id")
80
+ or (item["id"] if item["variant"] == "clean" else meta.get("group_id")),
81
+ "raw_probability_sum": total,
82
+ "zero_count": sum(value == 0 for value in p),
83
+ "reference_probs": target.get("reference_probs"),
84
+ "argmax_tie_break": meta.get("gold_policy", {}).get("argmax_tie_break"),
85
+ }
86
+ )
87
+ return rows
88
+
89
+
90
+ def coverage_at_error(confidence, correct, budget):
91
+ # Stable ties deliberately match KEV. This empirical, retrospective cutoff
92
+ # is not a risk guarantee and must not be deployed as a fitted threshold.
93
+ order = sorted(range(len(confidence)), key=lambda i: -confidence[i])
94
+ wrong, accepted = 0, 0
95
+ for rank, index in enumerate(order, 1):
96
+ wrong += not correct[index]
97
+ if wrong <= budget * rank:
98
+ accepted = rank
99
+ return accepted / len(order)
100
+
101
+
102
+ def metrics(rows):
103
+ if not rows:
104
+ return None
105
+ nll, correct, confidence, brier, mae, rps = [], [], [], [], [], []
106
+ for row in rows:
107
+ p, y = row["p"], row["label"]
108
+ nll.append(-math.log(max(p[y], 1e-9)))
109
+ correct.append(predicted_index(row) == y)
110
+ confidence.append(max(p))
111
+ brier.append(math.fsum((value - (i == y)) ** 2 for i, value in enumerate(p)))
112
+ if row["type"] == "score":
113
+ mae.append(abs(math.fsum(i * value for i, value in enumerate(p)) - y))
114
+ cumulative = 0.0
115
+ errors = []
116
+ for i, value in enumerate(p[:-1]):
117
+ cumulative += value
118
+ errors.append((cumulative - (i >= y)) ** 2)
119
+ rps.append(fmean(errors))
120
+ ece = 0.0
121
+ for index in range(10):
122
+ # Match numpy.linspace(0, 1, 11), including floating-point bin edges.
123
+ lo, hi = index * 0.1, (index + 1) * 0.1
124
+ bucket = [
125
+ i
126
+ for i, c in enumerate(confidence)
127
+ if lo <= c and (c < hi if index < 9 else c <= 1)
128
+ ]
129
+ if bucket:
130
+ ece += (
131
+ len(bucket)
132
+ / len(rows)
133
+ * abs(
134
+ fmean(correct[i] for i in bucket)
135
+ - fmean(confidence[i] for i in bucket)
136
+ )
137
+ )
138
+ high = [i for i, value in enumerate(confidence) if value >= 0.9]
139
+ result = {
140
+ "n": len(rows),
141
+ "acc": fmean(correct),
142
+ "nll": fmean(nll),
143
+ "brier": fmean(brier),
144
+ "ece": ece,
145
+ "mean_conf": fmean(confidence),
146
+ "confidence_bias": fmean(confidence) - fmean(correct),
147
+ "confident_error_rate": sum(not correct[i] for i in high) / len(rows),
148
+ "coverage_at_0_9": len(high) / len(rows),
149
+ "accuracy_at_0_9": fmean(correct[i] for i in high) if high else None,
150
+ "coverage_at_5pct_error": coverage_at_error(confidence, correct, 0.05),
151
+ "coverage_at_1pct_error": coverage_at_error(confidence, correct, 0.01),
152
+ }
153
+ if mae:
154
+ result.update(score_mae=fmean(mae), ranked_probability_score=fmean(rps))
155
+ return result
156
+
157
+
158
+ def grouped_metrics(rows, key):
159
+ grouped = defaultdict(list)
160
+ for row in rows:
161
+ if row["source"] != "unknowable":
162
+ grouped[row[key]].append(row)
163
+ return {name: metrics(group) for name, group in sorted(grouped.items())}
164
+
165
+
166
+ def paired_flip(rows):
167
+ pairs = defaultdict(dict)
168
+ for row in rows:
169
+ if row.get("pair_id"):
170
+ pair = pairs[row["pair_id"], row["question"]]
171
+ if row["sibling"] in pair:
172
+ raise ValueError("duplicate contrastive sibling")
173
+ pair[row["sibling"]] = row
174
+ complete = [pair for pair in pairs.values() if set(pair) == {"a", "b"}]
175
+
176
+ def truth(row):
177
+ return row["keys"][row["label"]]
178
+
179
+ def prediction(row):
180
+ return row["keys"][predicted_index(row)]
181
+
182
+ relevant = [p for p in complete if truth(p["a"]) != truth(p["b"])]
183
+ invariant = [p for p in complete if truth(p["a"]) == truth(p["b"])]
184
+
185
+ def both(group):
186
+ return (
187
+ fmean(all(prediction(r) == truth(r) for r in p.values()) for p in group)
188
+ if group
189
+ else None
190
+ )
191
+
192
+ return {
193
+ "pairs": len(relevant),
194
+ "incomplete_pairs": len(pairs) - len(complete),
195
+ "flip_rate": fmean(prediction(p["a"]) != prediction(p["b"]) for p in relevant)
196
+ if relevant
197
+ else None,
198
+ "both_correct_rate": both(relevant),
199
+ "invariant_pairs": len(invariant),
200
+ "invariant_both_correct_rate": both(invariant),
201
+ "invariance_rate": fmean(
202
+ prediction(p["a"]) == prediction(p["b"]) for p in invariant
203
+ )
204
+ if invariant
205
+ else None,
206
+ }
207
+
208
+
209
+ def summarize(rows):
210
+ clean = [r for r in rows if r["variant"] == "clean"]
211
+ knowable = [r for r in clean if r["source"] != "unknowable"]
212
+ controls = {r["id"]: r for r in clean if r["source"] == "unknowable_control"}
213
+ unknown = [r for r in clean if r["source"] == "unknowable"]
214
+ unknowable = None
215
+ if unknown:
216
+ paired = [
217
+ (max(r["p"]), max(controls[r["control_id"]]["p"]))
218
+ for r in unknown
219
+ if r.get("control_id") in controls
220
+ ]
221
+ unknowable = {
222
+ "n": len(unknown),
223
+ "mean_max_p": fmean(max(r["p"]) for r in unknown),
224
+ "share_at_0_9": fmean(max(r["p"]) >= 0.9 for r in unknown),
225
+ "control_acc": (
226
+ fmean(predicted_index(r) == r["label"] for r in controls.values())
227
+ if controls
228
+ else None
229
+ ),
230
+ "control_mean_max_p": (
231
+ fmean(max(r["p"]) for r in controls.values()) if controls else None
232
+ ),
233
+ "control_share_at_0_9": (
234
+ fmean(max(r["p"]) >= 0.9 for r in controls.values())
235
+ if controls
236
+ else None
237
+ ),
238
+ "paired_confidence_drop": fmean(c - u for u, c in paired)
239
+ if paired
240
+ else None,
241
+ "share_less_confident_than_control": (
242
+ fmean(u < c for u, c in paired) if paired else None
243
+ ),
244
+ }
245
+ originals = {(r["id"], r["question"]): r for r in clean}
246
+ deltas, flips, missing = [], [], 0
247
+ for row in rows:
248
+ if row["variant"] != "permuted" or row["type"] != "choice":
249
+ continue
250
+ original = originals.get((row["parent"], row["question"]))
251
+ if original is None:
252
+ missing += 1
253
+ continue
254
+ aligned = [row["p"][row["keys"].index(key)] for key in original["keys"]]
255
+ deltas.append(max(abs(a - b) for a, b in zip(aligned, original["p"])))
256
+ flips.append(argmax(aligned) != argmax(original["p"]))
257
+ tasks = grouped_metrics(clean, "task")
258
+ return {
259
+ "clean": metrics(knowable),
260
+ "tasks": tasks,
261
+ "sources": grouped_metrics(clean, "source"),
262
+ "variants": grouped_metrics(rows, "variant"),
263
+ "macro_task_acc": fmean(t["acc"] for t in tasks.values()) if tasks else None,
264
+ "objective": -fmean(t["nll"] for t in tasks.values()) if tasks else None,
265
+ "paired_flip": paired_flip(clean),
266
+ "reference_distribution": reference_metrics(clean),
267
+ "unknowable": unknowable,
268
+ "permutation": {
269
+ "n": len(flips),
270
+ "missing_parents": missing,
271
+ "flip_rate": fmean(flips) if flips else None,
272
+ "mean_max_delta": fmean(deltas) if deltas else None,
273
+ },
274
+ "metric_policy": {
275
+ "nll_floor": 1e-9,
276
+ "ece_bins": 10,
277
+ "confidence_for_calibration": "maximum label probability",
278
+ "calibration_applied": False,
279
+ "unknown_accuracy_excluded": True,
280
+ "coverage_error_budget_is_retrospective": True,
281
+ "raw_sums_outside_1e_5": sum(
282
+ abs(r["raw_probability_sum"] - 1) > 1e-5 for r in rows
283
+ ),
284
+ "returned_zeros": sum(r["zero_count"] for r in rows),
285
+ },
286
+ }
287
+
288
+
289
+ def reference_metrics(rows):
290
+ selected = [row for row in rows if row.get("reference_probs") is not None]
291
+ if not selected:
292
+ return None
293
+ squared, variation, kl = [], [], []
294
+ for row in selected:
295
+ p, q = row["p"], row["reference_probs"]
296
+ squared.append(math.fsum((a - b) ** 2 for a, b in zip(p, q)))
297
+ variation.append(0.5 * math.fsum(abs(a - b) for a, b in zip(p, q)))
298
+ kl.append(
299
+ math.fsum(b * math.log(b / max(a, 1e-9)) for a, b in zip(p, q) if b > 0)
300
+ )
301
+ return {
302
+ "n": len(selected),
303
+ "squared_l2": fmean(squared),
304
+ "total_variation": fmean(variation),
305
+ "kl_reference_to_model": fmean(kl),
306
+ "note": "Exact reference-distribution fidelity, separate from hard-label Brier/ECE.",
307
+ }
308
+
309
+
310
+ def quantile(values, q):
311
+ if not values:
312
+ return None
313
+ ordered = sorted(values)
314
+ position = (len(ordered) - 1) * q
315
+ lo, hi = math.floor(position), math.ceil(position)
316
+ return ordered[lo] + (ordered[hi] - ordered[lo]) * (position - lo)
317
+
318
+
319
+ def latency_summary(values):
320
+ return {
321
+ "n": len(values),
322
+ "mean": fmean(values) if values else None,
323
+ "p50": quantile(values, 0.5),
324
+ "p95": quantile(values, 0.95),
325
+ "p99": quantile(values, 0.99),
326
+ }
server/jev_adapter/benchmarks/prepare.py ADDED
@@ -0,0 +1,227 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Download pinned KEV and JevBench evaluation data, without training data."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import json
7
+ from pathlib import Path
8
+
9
+ import httpx
10
+
11
+ from jev_adapter.protocol import SystemOneRequest
12
+
13
+ from .data import (
14
+ DEFAULT_SUITES,
15
+ KEV_COMMIT,
16
+ KEV_RAW,
17
+ KEV_REPOSITORY,
18
+ SUITES,
19
+ parse_partition,
20
+ population_counts,
21
+ require_sha256,
22
+ sha256,
23
+ source_provenance,
24
+ )
25
+ from .jevbench import JEVBENCH_SUITES, prepare_jevbench
26
+
27
+
28
+ def read_source(
29
+ path: str, client: httpx.Client | None, source_root: Path | None
30
+ ) -> bytes:
31
+ if source_root is not None:
32
+ return (source_root / path).read_bytes()
33
+ if client is None:
34
+ raise ValueError("an HTTP client or --source-root is required")
35
+ response = client.get(f"{KEV_RAW}/{path}")
36
+ response.raise_for_status()
37
+ return response.content
38
+
39
+
40
+ def prepare_suite(
41
+ suite: str,
42
+ split: str,
43
+ output: Path,
44
+ *,
45
+ client: httpx.Client | None = None,
46
+ source_root: Path | None = None,
47
+ allow_test: bool = False,
48
+ limit: int | None = None,
49
+ ) -> dict:
50
+ if suite not in SUITES:
51
+ raise ValueError(f"unknown suite: {suite}")
52
+ if split not in ("development", "test"):
53
+ raise ValueError("only development and test are evaluation partitions")
54
+ if split == "test" and not allow_test:
55
+ raise ValueError(
56
+ "locked test requires --allow-test; use development for iteration"
57
+ )
58
+ if limit is not None and limit < 1:
59
+ raise ValueError("--limit must be positive")
60
+ spec = SUITES[suite]
61
+ manifest_bytes = read_source(f"{spec.path}/manifest.json", client, source_root)
62
+ require_sha256(manifest_bytes, spec.manifest_sha256, f"{suite}/manifest.json")
63
+ upstream = json.loads(manifest_bytes)
64
+ expected_file = upstream["files"][f"{split}.jsonl"]
65
+ raw_bytes = read_source(f"{spec.path}/{split}.jsonl", client, source_root)
66
+ require_sha256(raw_bytes, expected_file["sha256"], f"{suite}/{split}.jsonl")
67
+ records = parse_partition(raw_bytes, suite, split)
68
+ counts = population_counts(records)
69
+ if counts["records"] != expected_file["records"]:
70
+ raise ValueError(f"upstream record count mismatch: {suite}/{split}")
71
+ if (
72
+ "questions" in expected_file
73
+ and counts["questions"] != expected_file["questions"]
74
+ ):
75
+ raise ValueError(f"upstream question count mismatch: {suite}/{split}")
76
+ if not records:
77
+ raise ValueError(
78
+ f"{suite}/{split} is empty; this suite may be development-only"
79
+ )
80
+ selected = records if limit is None else records[:limit]
81
+ for row in selected:
82
+ SystemOneRequest.model_validate({"model": "validation-only", **row["record"]})
83
+ serialized = (
84
+ "".join(
85
+ json.dumps(row, ensure_ascii=False, allow_nan=False) + "\n"
86
+ for row in selected
87
+ )
88
+ ).encode("utf-8")
89
+ directory = output / suite
90
+ data_path = directory / f"{split}.jsonl"
91
+ manifest_path = directory / f"{split}.manifest.json"
92
+ upstream_path = directory / "upstream.manifest.json"
93
+ manifest = {
94
+ "schema_version": 1,
95
+ "suite": suite,
96
+ "split": split,
97
+ "description": spec.description,
98
+ "data_file": data_path.name,
99
+ "data_sha256": sha256(serialized),
100
+ "upstream": {
101
+ "repository": KEV_REPOSITORY,
102
+ "commit": KEV_COMMIT,
103
+ "path": f"{spec.path}/{split}.jsonl",
104
+ "url": f"{KEV_RAW}/{spec.path}/{split}.jsonl",
105
+ "sha256": sha256(raw_bytes),
106
+ "manifest_sha256": spec.manifest_sha256,
107
+ },
108
+ "full_partition": counts,
109
+ "selected": population_counts(selected),
110
+ "selection": {
111
+ "method": "full" if len(selected) == len(records) else "prefix",
112
+ "requested_limit": limit,
113
+ "is_full_partition": len(selected) == len(records),
114
+ "note": (
115
+ "Full frozen partition."
116
+ if len(selected) == len(records)
117
+ else "Smoke subset only; not a full-suite result. Pairs may be incomplete."
118
+ ),
119
+ },
120
+ "protocol": {
121
+ "calibration_applied": False,
122
+ "training_data_downloaded": False,
123
+ "gold_labels_sent_to_model": False,
124
+ "headline_variant": "clean",
125
+ "exclude_from_headline_sources": ["unknowable"],
126
+ "locked_test": split == "test",
127
+ "notes": list(spec.notes),
128
+ "upstream_context_policy": upstream.get("context"),
129
+ },
130
+ "provenance": source_provenance(records, upstream),
131
+ }
132
+ manifest_payload = (
133
+ json.dumps(manifest, ensure_ascii=False, indent=2) + "\n"
134
+ ).encode()
135
+ # Idempotent preparation is safe; refuse to replace any changed frozen artifact.
136
+ for path, content in (
137
+ (data_path, serialized),
138
+ (manifest_path, manifest_payload),
139
+ (upstream_path, manifest_bytes),
140
+ ):
141
+ if path.exists() and path.read_bytes() != content:
142
+ raise FileExistsError(
143
+ f"refusing to overwrite a different prepared artifact: {path}"
144
+ )
145
+ directory.mkdir(parents=True, exist_ok=True)
146
+ data_path.write_bytes(serialized)
147
+ manifest_path.write_bytes(manifest_payload)
148
+ upstream_path.write_bytes(manifest_bytes)
149
+ return manifest
150
+
151
+
152
+ def main(argv: list[str] | None = None) -> None:
153
+ parser = argparse.ArgumentParser(description=__doc__)
154
+ parser.add_argument("--output", type=Path, required=True)
155
+ parser.add_argument(
156
+ "--suite", action="append", choices=tuple(SUITES) + tuple(JEVBENCH_SUITES)
157
+ )
158
+ parser.add_argument(
159
+ "--split", choices=("development", "test"), default="development"
160
+ )
161
+ parser.add_argument("--allow-test", action="store_true")
162
+ parser.add_argument(
163
+ "--limit", type=int, help="deterministic prefix smoke subset per suite"
164
+ )
165
+ parser.add_argument(
166
+ "--source-root", type=Path, help="offline KEV checkout; same SHA256 checks"
167
+ )
168
+ parser.add_argument(
169
+ "--jevbench-source-root", type=Path, help="offline pinned JevBench checkout"
170
+ )
171
+ args = parser.parse_args(argv)
172
+ if args.split == "test" and not args.allow_test:
173
+ parser.error("--split test requires --allow-test")
174
+ if args.limit is not None and args.limit < 1:
175
+ parser.error("--limit must be positive")
176
+ suites = args.suite or (
177
+ DEFAULT_SUITES if args.split == "test" else (*DEFAULT_SUITES, *JEVBENCH_SUITES)
178
+ )
179
+ if args.split == "test" and any(suite in JEVBENCH_SUITES for suite in suites):
180
+ parser.error("JevBench supplies a public subset, not a locked test split")
181
+ with httpx.Client(timeout=60, follow_redirects=True) as client:
182
+ for suite in dict.fromkeys(suites):
183
+ if suite in JEVBENCH_SUITES:
184
+ manifest = prepare_jevbench(
185
+ suite,
186
+ args.output,
187
+ client=client,
188
+ source_root=args.jevbench_source_root,
189
+ limit=args.limit,
190
+ )
191
+ print(
192
+ json.dumps(
193
+ {
194
+ "suite": suite,
195
+ "split": "public",
196
+ "selected": manifest["selected"],
197
+ "is_full_partition": manifest["selection"][
198
+ "is_full_partition"
199
+ ],
200
+ }
201
+ )
202
+ )
203
+ continue
204
+ manifest = prepare_suite(
205
+ suite,
206
+ args.split,
207
+ args.output,
208
+ client=client,
209
+ source_root=args.source_root,
210
+ allow_test=args.allow_test,
211
+ limit=args.limit,
212
+ )
213
+ print(
214
+ json.dumps(
215
+ {
216
+ "suite": suite,
217
+ "split": args.split,
218
+ "data": str(args.output / suite / f"{args.split}.jsonl"),
219
+ "selected": manifest["selected"],
220
+ "is_full_partition": manifest["selection"]["is_full_partition"],
221
+ }
222
+ )
223
+ )
224
+
225
+
226
+ if __name__ == "__main__":
227
+ main()
server/jev_adapter/benchmarks/run.py ADDED
@@ -0,0 +1,441 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Evaluate frozen decisions against a running, untrained-model adapter.
2
+
3
+ No retries, generation, calibration, truncation, or training. Latency includes
4
+ adapter HTTP, tokenization, engine scheduling and forward work; it is NOT CUDA
5
+ kernel time. Model load and warmup are excluded. Failed runs have no headline.
6
+ """
7
+
8
+ import argparse
9
+ import asyncio
10
+ import hashlib
11
+ import json
12
+ import math
13
+ import os
14
+ import platform
15
+ import random
16
+ import sys
17
+ import time
18
+ from collections import defaultdict
19
+ from datetime import UTC, datetime
20
+ from pathlib import Path
21
+
22
+ import httpx
23
+
24
+ from jev_adapter.protocol import SystemOneRequest
25
+
26
+ from .metrics import latency_summary, prediction_rows, summarize
27
+
28
+
29
+ def digest(data):
30
+ return hashlib.sha256(data).hexdigest()
31
+
32
+
33
+ def write_json(path, value):
34
+ path.write_text(
35
+ json.dumps(value, ensure_ascii=False, indent=2, allow_nan=False) + "\n"
36
+ )
37
+
38
+
39
+ def load_data(paths, model, allow_test=False, limit=None):
40
+ items, seen, sources = [], set(), []
41
+ for path in paths:
42
+ raw = path.read_bytes()
43
+ loaded = [json.loads(line) for line in raw.splitlines() if line.strip()]
44
+ source = {
45
+ "path": str(path.resolve()),
46
+ "sha256": digest(raw),
47
+ "records": len(loaded),
48
+ }
49
+ manifest_path = path.with_suffix(".manifest.json")
50
+ if manifest_path.exists():
51
+ prepared = json.loads(manifest_path.read_text())
52
+ if prepared.get("data_sha256") != digest(raw):
53
+ raise ValueError(f"prepared dataset checksum mismatch: {path}")
54
+ if prepared.get("selected", {}).get("records") != len(loaded):
55
+ raise ValueError(f"prepared dataset count mismatch: {path}")
56
+ source["prepared_manifest"] = prepared
57
+ sources.append(source)
58
+ for item in loaded:
59
+ key = item["suite"], item["split"], item["id"]
60
+ if key in seen:
61
+ raise ValueError(f"duplicate input record: {key}")
62
+ seen.add(key)
63
+ if item["split"] == "test" and not allow_test:
64
+ raise ValueError(
65
+ "test data requires --allow-test; use development first"
66
+ )
67
+ request = SystemOneRequest.model_validate(
68
+ {**item["record"], "model": model}
69
+ )
70
+ if set(request.questions) != set(item["expected"]):
71
+ raise ValueError(f"question/label mismatch: {key}")
72
+ for qid, expected in item["expected"].items():
73
+ labels, target, label = (
74
+ expected["labels"],
75
+ expected["target"],
76
+ expected["label"],
77
+ )
78
+ if (
79
+ len(labels) < 2
80
+ or len(labels) != len(set(labels))
81
+ or type(label) is not int
82
+ or not 0 <= label < len(labels)
83
+ or target != [int(i == label) for i in range(len(labels))]
84
+ ):
85
+ raise ValueError(
86
+ "this runner requires valid one-hot objective labels"
87
+ )
88
+ question = request.questions[qid]
89
+ actual = (
90
+ list(question.criteria)
91
+ if question.type == "choice"
92
+ else ["false", "true"]
93
+ if question.type == "noul"
94
+ else [str(i) for i in range(len(question.criteria))]
95
+ )
96
+ if set(labels) != set(actual) or expected["type"] != question.type:
97
+ raise ValueError(
98
+ f"expected labels/type differ from request: {key}/{qid}"
99
+ )
100
+ reference = expected.get("reference_probs")
101
+ if reference is not None and (
102
+ not isinstance(reference, list)
103
+ or len(reference) != len(labels)
104
+ or any(
105
+ type(p) not in (int, float)
106
+ or not math.isfinite(p)
107
+ or not 0 <= p <= 1
108
+ for p in reference
109
+ )
110
+ or abs(math.fsum(reference) - 1) > 1e-8
111
+ ):
112
+ raise ValueError("invalid reference probability distribution")
113
+ items.append(item)
114
+ if limit is not None:
115
+ items = items[:limit]
116
+ if not items:
117
+ raise ValueError("no input records")
118
+ return items, sources
119
+
120
+
121
+ async def engine_snapshot(client, base_url, cache_mode):
122
+ headers = {}
123
+ if key := os.environ.get("SGLANG_API_KEY"):
124
+ headers["Authorization"] = f"Bearer {key}"
125
+ results = {}
126
+ allowed = {
127
+ "model_path",
128
+ "model_type",
129
+ "architectures",
130
+ "served_model_name",
131
+ "revision",
132
+ "tokenizer_revision",
133
+ "dtype",
134
+ "quantization",
135
+ "tp_size",
136
+ "tp",
137
+ "version",
138
+ "context_length",
139
+ "max_total_tokens",
140
+ "max_running_requests",
141
+ "chunked_prefill_size",
142
+ "mem_fraction_static",
143
+ "disable_radix_cache",
144
+ "mm_preprocess_cache_size_mb",
145
+ "enable_prefix_mm_cache",
146
+ "enable_mm_global_cache",
147
+ "speculative_algorithm",
148
+ "is_generation",
149
+ "has_image_understanding",
150
+ "attention_backend",
151
+ "moe_runner_backend",
152
+ }
153
+ for name in ("model_info", "server_info"):
154
+ response = await client.get(base_url.rstrip("/") + "/" + name, headers=headers)
155
+ if response.status_code == 404:
156
+ response = await client.get(
157
+ base_url.rstrip("/") + "/get_" + name, headers=headers
158
+ )
159
+ response.raise_for_status()
160
+ value = response.json()
161
+ if not isinstance(value, dict):
162
+ raise TypeError(f"invalid engine {name}")
163
+ results[name] = {k: v for k, v in value.items() if k in allowed}
164
+ config = results["server_info"]
165
+ if config.get("speculative_algorithm") is not None:
166
+ raise ValueError("disable speculative decoding for this benchmark")
167
+ if cache_mode == "full-prefill":
168
+ if config.get("disable_radix_cache") is not True:
169
+ raise ValueError("full-prefill requires engine --disable-radix-cache")
170
+ if config.get("mm_preprocess_cache_size_mb") != 0:
171
+ raise ValueError("full-prefill requires --mm-preprocess-cache-size-mb 0")
172
+ if config.get("enable_prefix_mm_cache") or config.get("enable_mm_global_cache"):
173
+ raise ValueError("full-prefill requires multimodal feature caches disabled")
174
+ return results
175
+
176
+
177
+ def checked_usage(response):
178
+ usage, meta = response.get("usage", {}), response.get("metadata", {})
179
+ if type(usage.get("output_tokens")) is not int or usage["output_tokens"] != 0:
180
+ raise ValueError("engine must return exactly zero output tokens")
181
+ if type(usage.get("input_tokens")) is not int or usage["input_tokens"] < 1:
182
+ raise ValueError("missing or invalid input-token accounting")
183
+ if type(meta.get("evaluations")) is not int or meta["evaluations"] < 1:
184
+ raise ValueError("missing forward evaluation count")
185
+ elapsed = meta.get("adapter_elapsed_ms")
186
+ if type(elapsed) not in (int, float) or not math.isfinite(elapsed) or elapsed < 0:
187
+ raise ValueError("missing or invalid adapter elapsed time")
188
+ if not isinstance(response.get("model"), str) or not response["model"]:
189
+ raise ValueError("missing served model identity")
190
+ return usage, meta
191
+
192
+
193
+ def validate_launch(snapshot, launch):
194
+ model, engine = launch["model"], launch["engine"]
195
+ server, info = snapshot["server_info"], snapshot["model_info"]
196
+ for field, expected in {
197
+ "model_path": model["repo_id"],
198
+ "revision": model["revision"],
199
+ "dtype": model["dtype"],
200
+ "quantization": model["quantization"],
201
+ }.items():
202
+ if server.get(field) != expected:
203
+ raise ValueError(f"launch manifest {field} differs from running engine")
204
+ if info.get("model_path") != model["repo_id"]:
205
+ raise ValueError("launch manifest model differs from running engine model_info")
206
+ if not engine.get("revision") or not launch.get("profile"):
207
+ raise ValueError("launch manifest lacks pinned engine/profile identity")
208
+
209
+
210
+ async def evaluate(args, client=None):
211
+ items, inputs = load_data(args.data, args.model, args.allow_test, args.limit)
212
+ args.output.mkdir(parents=True, exist_ok=False)
213
+ manifest = {
214
+ "created_at": datetime.now(UTC).isoformat(),
215
+ "status": "started",
216
+ "python": sys.version,
217
+ "platform": platform.platform(),
218
+ "requested_model": args.model,
219
+ "base_url": args.base_url,
220
+ "engine_url": args.engine_url,
221
+ "data": inputs,
222
+ "limit": args.limit,
223
+ "concurrency": args.concurrency,
224
+ "warmup_requests": args.warmup,
225
+ "repeats": args.repeats,
226
+ "seed": args.seed,
227
+ "cache_mode": args.cache_mode,
228
+ "options": {"temperature": 1.0, "permutations": 1, "enable_thinking": False},
229
+ "latency_scope": "client HTTP wall time, excluding queue before dispatch; NOT GPU kernel time",
230
+ "quality_repeats": "first measured repetition only",
231
+ "assistant_prefix": args.assistant_prefix,
232
+ "code_sha256": {
233
+ str(p.relative_to(Path(__file__).parents[1])): digest(p.read_bytes())
234
+ for p in Path(__file__).parents[1].rglob("*.py")
235
+ },
236
+ }
237
+ write_json(args.output / "manifest.json", manifest)
238
+ owned = client is None
239
+ if owned:
240
+ client = httpx.AsyncClient(
241
+ timeout=args.timeout, limits=httpx.Limits(max_connections=args.concurrency)
242
+ )
243
+ headers = {}
244
+ if key := os.environ.get("JEV_API_KEY"):
245
+ headers["Authorization"] = f"Bearer {key}"
246
+ results, failures = [], []
247
+ try:
248
+ manifest["engine"] = await engine_snapshot(
249
+ client, args.engine_url, args.cache_mode
250
+ )
251
+ expected_served = manifest["engine"]["model_info"].get("served_model_name")
252
+ if not expected_served:
253
+ raise ValueError("engine did not identify its served model")
254
+ if args.engine_manifest:
255
+ manifest["launch_manifest"] = json.loads(args.engine_manifest.read_text())
256
+ validate_launch(manifest["engine"], manifest["launch_manifest"])
257
+ write_json(args.output / "manifest.json", manifest)
258
+
259
+ async def call(item):
260
+ request = {
261
+ **item["record"],
262
+ "model": args.model,
263
+ "options": {
264
+ "temperature": 1.0,
265
+ "permutations": 1,
266
+ "return_logprobs": True,
267
+ },
268
+ }
269
+ if args.assistant_prefix is not None:
270
+ request["options"]["assistant_prefix"] = args.assistant_prefix
271
+ started = time.perf_counter()
272
+ response = await client.post(
273
+ args.base_url.rstrip("/") + "/v1/systemone",
274
+ headers=headers,
275
+ json=request,
276
+ )
277
+ response.raise_for_status()
278
+ body = response.json()
279
+ elapsed = (time.perf_counter() - started) * 1000
280
+ usage, meta = checked_usage(body)
281
+ if body["model"] != expected_served:
282
+ raise ValueError("adapter response model differs from engine snapshot")
283
+ if meta["evaluations"] != len(item["record"]["questions"]):
284
+ raise ValueError("expected one forward evaluation per question")
285
+ rows = prediction_rows(item, body)
286
+ return {
287
+ "request_sha256": digest(json.dumps(request, sort_keys=True).encode()),
288
+ "id": item["id"],
289
+ "suite": item["suite"],
290
+ "split": item["split"],
291
+ "latency_ms": elapsed,
292
+ "adapter_ms": meta["adapter_elapsed_ms"],
293
+ "input_tokens": usage["input_tokens"],
294
+ "evaluations": meta["evaluations"],
295
+ "served_model": body["model"],
296
+ "response": body,
297
+ "rows": rows,
298
+ }
299
+
300
+ for i in range(args.warmup):
301
+ # Span the full input so warmup includes more than the first task.
302
+ item = items[(i * len(items) // max(args.warmup, 1)) % len(items)]
303
+ print(
304
+ f"Warmup {i + 1}/{args.warmup}: {item['suite']} {item['id']}",
305
+ flush=True,
306
+ )
307
+ await call(item)
308
+ jobs = [
309
+ (i, repeat) for repeat in range(args.repeats) for i in range(len(items))
310
+ ]
311
+ random.Random(args.seed).shuffle(jobs)
312
+ semaphore = asyncio.Semaphore(args.concurrency)
313
+ started = time.perf_counter()
314
+ with (args.output / "predictions.jsonl").open("w") as handle:
315
+
316
+ async def job(index, repeat):
317
+ async with semaphore:
318
+ try:
319
+ result = await call(items[index])
320
+ result.update(index=index, repeat=repeat)
321
+ results.append(result)
322
+ handle.write(
323
+ json.dumps(result, ensure_ascii=False, allow_nan=False)
324
+ + "\n"
325
+ )
326
+ handle.flush()
327
+ except (httpx.HTTPError, ValueError, KeyError, TypeError) as error:
328
+ failures.append(
329
+ {
330
+ "index": index,
331
+ "repeat": repeat,
332
+ "id": items[index]["id"],
333
+ "suite": items[index]["suite"],
334
+ "error_type": type(error).__name__,
335
+ "error": str(error),
336
+ }
337
+ )
338
+ done = len(results) + len(failures)
339
+ if done % 50 == 0 or done == len(jobs):
340
+ print(
341
+ f"{done}/{len(jobs)} requests; errors={len(failures)}",
342
+ flush=True,
343
+ )
344
+
345
+ await asyncio.gather(*(job(index, repeat) for index, repeat in jobs))
346
+ elapsed = time.perf_counter() - started
347
+ served = {r["served_model"] for r in results}
348
+ if len(served) > 1:
349
+ failures.append({"error": "served model changed during run"})
350
+ report = {
351
+ "status": "failed" if failures else "complete",
352
+ "requested_records": len(items),
353
+ "requested_questions": sum(len(i["expected"]) for i in items),
354
+ "attempted_requests": len(jobs),
355
+ "successful_requests": len(results),
356
+ "errors": len(failures),
357
+ "served_models": sorted(served),
358
+ "elapsed_measurement_s": elapsed,
359
+ "requests_per_second": len(results) / elapsed,
360
+ "questions_per_second": sum(r["evaluations"] for r in results) / elapsed,
361
+ "latency_ms": latency_summary([r["latency_ms"] for r in results]),
362
+ "adapter_ms": latency_summary([r["adapter_ms"] for r in results]),
363
+ "input_tokens_per_request": latency_summary(
364
+ [r["input_tokens"] for r in results]
365
+ ),
366
+ "suites": {},
367
+ "latency_scope": manifest["latency_scope"],
368
+ "partial_dataset": args.limit is not None
369
+ or any(
370
+ not source.get("prepared_manifest", {})
371
+ .get("selection", {})
372
+ .get("is_full_partition", False)
373
+ for source in inputs
374
+ ),
375
+ }
376
+ if not failures:
377
+ grouped, times = defaultdict(list), defaultdict(list)
378
+ for result in sorted(results, key=lambda r: (r["index"], r["repeat"])):
379
+ key = result["suite"] + "/" + result["split"]
380
+ times[key].append(result["latency_ms"])
381
+ if result["repeat"] == 0:
382
+ grouped[key].extend(result["rows"])
383
+ report["suites"] = {
384
+ key: {**summarize(rows), "latency_ms": latency_summary(times[key])}
385
+ for key, rows in grouped.items()
386
+ }
387
+ write_json(args.output / "report.json", report)
388
+ if failures:
389
+ write_json(args.output / "failures.json", failures)
390
+ manifest.update(status=report["status"], served_models=sorted(served))
391
+ write_json(args.output / "manifest.json", manifest)
392
+ return report
393
+ except BaseException as error:
394
+ manifest.update(
395
+ status="failed", failure_type=type(error).__name__, failure=str(error)
396
+ )
397
+ write_json(args.output / "manifest.json", manifest)
398
+ raise
399
+ finally:
400
+ if owned:
401
+ await client.aclose()
402
+
403
+
404
+ def main():
405
+ parser = argparse.ArgumentParser(description=__doc__)
406
+ parser.add_argument("--data", type=Path, action="append", required=True)
407
+ parser.add_argument("--output", type=Path, required=True)
408
+ parser.add_argument("--base-url", default="http://127.0.0.1:30120")
409
+ parser.add_argument("--engine-url", default="http://127.0.0.1:30000")
410
+ parser.add_argument("--engine-manifest", type=Path)
411
+ parser.add_argument("--model", default="decision-model")
412
+ parser.add_argument("--concurrency", type=int, default=1)
413
+ parser.add_argument("--warmup", type=int, default=20)
414
+ parser.add_argument("--repeats", type=int, default=1)
415
+ parser.add_argument("--seed", type=int, default=42)
416
+ parser.add_argument("--limit", type=int)
417
+ parser.add_argument("--timeout", type=float, default=120)
418
+ parser.add_argument("--assistant-prefix")
419
+ parser.add_argument("--allow-test", action="store_true")
420
+ parser.add_argument(
421
+ "--cache-mode",
422
+ choices=["full-prefill", "server-default"],
423
+ default="full-prefill",
424
+ )
425
+ args = parser.parse_args()
426
+ if (
427
+ args.concurrency < 1
428
+ or args.repeats < 1
429
+ or args.warmup < 0
430
+ or (args.limit is not None and args.limit < 1)
431
+ or not math.isfinite(args.timeout)
432
+ or args.timeout <= 0
433
+ ):
434
+ parser.error("counts and timeout must be positive; warmup may be zero")
435
+ report = asyncio.run(evaluate(args))
436
+ print(json.dumps(report, ensure_ascii=False, indent=2))
437
+ raise SystemExit(0 if report["status"] == "complete" else 1)
438
+
439
+
440
+ if __name__ == "__main__":
441
+ main()
server/tests/test_benchmark_data.py ADDED
@@ -0,0 +1,220 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Check benchmark integrity and prevent labels from entering inference requests."""
2
+
3
+ import copy
4
+ import json
5
+ import tempfile
6
+ import unittest
7
+ from pathlib import Path
8
+ from unittest.mock import patch
9
+
10
+ import httpx
11
+
12
+ from jev_adapter.benchmarks.data import (
13
+ SuiteSpec,
14
+ normalize_record,
15
+ parse_partition,
16
+ population_counts,
17
+ sha256,
18
+ )
19
+ from jev_adapter.benchmarks.prepare import prepare_suite
20
+
21
+
22
+ def fixture_record(name="example/1"):
23
+ return {
24
+ "state": {"message": "Needs a refund", "count": 4},
25
+ "questions": {
26
+ "queue": {
27
+ "type": "choice",
28
+ "instructions": "Choose the queue.",
29
+ "criteria": {"other": None, "billing": "Refunds"},
30
+ "label": "billing",
31
+ "src": "routing",
32
+ },
33
+ "urgent": {
34
+ "type": "noul",
35
+ "instructions": "Is the message urgent?",
36
+ "label": False,
37
+ "src": "urgency",
38
+ },
39
+ "priority": {
40
+ "type": "score",
41
+ "instructions": "Select priority.",
42
+ "criteria": ["low", "medium", "high"],
43
+ "label": 2,
44
+ "src": "priority",
45
+ },
46
+ },
47
+ "_meta": {
48
+ "id": name,
49
+ "source": "example",
50
+ "variant": "clean",
51
+ "group_id": "pair-1",
52
+ "pair_id": "pair-1",
53
+ "sibling": "a",
54
+ },
55
+ }
56
+
57
+
58
+ def fake_source(records):
59
+ payload = ("".join(json.dumps(r) + "\n" for r in records)).encode()
60
+ manifest = json.dumps(
61
+ {
62
+ "files": {
63
+ name: {
64
+ "sha256": sha256(payload),
65
+ "records": len(records),
66
+ "questions": sum(len(r["questions"]) for r in records),
67
+ }
68
+ for name in ("development.jsonl", "test.jsonl")
69
+ },
70
+ "dataset_revisions": {},
71
+ }
72
+ ).encode()
73
+ spec = SuiteSpec("evals/example", sha256(manifest), "Test fixture")
74
+ return spec, manifest, payload
75
+
76
+
77
+ class TestBenchmarkNormalization(unittest.TestCase):
78
+ def test_all_question_types_preserve_order_and_hide_gold(self):
79
+ original = fixture_record()
80
+ before = copy.deepcopy(original)
81
+ row = normalize_record(original, "example", "development")
82
+ self.assertEqual(original, before)
83
+ self.assertEqual(
84
+ list(row["record"]["questions"]), ["queue", "urgent", "priority"]
85
+ )
86
+ self.assertNotIn("model", row["record"])
87
+ self.assertNotIn("_meta", row["record"])
88
+ for q in row["record"]["questions"].values():
89
+ self.assertNotIn("label", q)
90
+ self.assertNotIn("src", q)
91
+ self.assertEqual(row["expected"]["queue"]["labels"], ["other", "billing"])
92
+ self.assertEqual(row["expected"]["queue"]["target"], [0, 1])
93
+ self.assertEqual(row["expected"]["urgent"]["labels"], ["false", "true"])
94
+ self.assertEqual(row["expected"]["urgent"]["target"], [1, 0])
95
+ self.assertEqual(row["expected"]["priority"]["target"], [0, 0, 1])
96
+ self.assertIsNone(row["record"]["questions"]["queue"]["criteria"]["other"])
97
+ self.assertEqual(row["metadata"]["pair_id"], "pair-1")
98
+
99
+ def test_large_choice_space_is_not_truncated(self):
100
+ raw = fixture_record()
101
+ raw["questions"] = {
102
+ "intent": {
103
+ "type": "choice",
104
+ "instructions": "Which intent?",
105
+ "criteria": {f"intent_{i}": None for i in range(78)},
106
+ "label": "intent_77",
107
+ }
108
+ }
109
+ row = normalize_record(raw, "example", "development")
110
+ self.assertEqual(len(row["expected"]["intent"]["labels"]), 78)
111
+ self.assertEqual(row["expected"]["intent"]["label"], 77)
112
+
113
+ def test_duplicate_ids_and_malformed_targets_fail(self):
114
+ raw = fixture_record()
115
+ payload = (json.dumps(raw) + "\n") * 2
116
+ with self.assertRaisesRegex(ValueError, "duplicate record"):
117
+ parse_partition(payload.encode(), "example", "development")
118
+ for question, label in (
119
+ ("queue", "absent"),
120
+ ("urgent", "false"),
121
+ ("priority", 9),
122
+ ):
123
+ modified = copy.deepcopy(raw)
124
+ modified["questions"][question]["label"] = label
125
+ with self.subTest(question=question), self.assertRaises(ValueError):
126
+ normalize_record(modified, "example", "development")
127
+
128
+ def test_population_separates_perturbations_and_unknowable(self):
129
+ raws = [fixture_record(str(i)) for i in range(3)]
130
+ raws[1]["_meta"]["variant"] = "permuted"
131
+ raws[2]["_meta"]["source"] = "unknowable"
132
+ counts = population_counts(
133
+ [normalize_record(raw, "example", "development") for raw in raws]
134
+ )
135
+ self.assertEqual(counts["records"], 3)
136
+ self.assertEqual(counts["questions"], 9)
137
+ self.assertEqual(counts["clean_questions"], 6)
138
+ self.assertEqual(counts["headline_questions"], 3)
139
+
140
+
141
+ class TestBenchmarkPreparation(unittest.TestCase):
142
+ def test_verified_download_subset_and_idempotency(self):
143
+ spec, manifest, payload = fake_source(
144
+ [fixture_record("one"), fixture_record("two")]
145
+ )
146
+ fetched = []
147
+
148
+ def transport(request):
149
+ fetched.append(request.url.path)
150
+ return httpx.Response(
151
+ 200,
152
+ content=manifest
153
+ if request.url.path.endswith("manifest.json")
154
+ else payload,
155
+ )
156
+
157
+ with (
158
+ tempfile.TemporaryDirectory() as tmp,
159
+ patch.dict("jev_adapter.benchmarks.prepare.SUITES", {"example": spec}),
160
+ httpx.Client(transport=httpx.MockTransport(transport)) as client,
161
+ ):
162
+ out = Path(tmp)
163
+ result = prepare_suite(
164
+ "example", "development", out, client=client, limit=1
165
+ )
166
+ again = prepare_suite("example", "development", out, client=client, limit=1)
167
+ self.assertEqual(result, again)
168
+ self.assertFalse(result["selection"]["is_full_partition"])
169
+ self.assertEqual(result["full_partition"]["questions"], 6)
170
+ self.assertEqual(result["selected"]["questions"], 3)
171
+ self.assertEqual(
172
+ result["data_sha256"],
173
+ sha256((out / "example/development.jsonl").read_bytes()),
174
+ )
175
+ self.assertTrue(all("train" not in path for path in fetched))
176
+ with self.assertRaises(FileExistsError):
177
+ prepare_suite("example", "development", out, client=client)
178
+
179
+ def test_tampered_partition_rejected_without_writing(self):
180
+ spec, manifest, payload = fake_source([fixture_record()])
181
+
182
+ def transport(request):
183
+ return httpx.Response(
184
+ 200,
185
+ content=manifest
186
+ if request.url.path.endswith("manifest.json")
187
+ else payload + b" ",
188
+ )
189
+
190
+ with (
191
+ tempfile.TemporaryDirectory() as tmp,
192
+ patch.dict("jev_adapter.benchmarks.prepare.SUITES", {"example": spec}),
193
+ httpx.Client(transport=httpx.MockTransport(transport)) as client,
194
+ ):
195
+ out = Path(tmp)
196
+ with self.assertRaisesRegex(ValueError, "SHA256 mismatch"):
197
+ prepare_suite("example", "development", out, client=client)
198
+ self.assertFalse((out / "example").exists())
199
+
200
+ def test_offline_source_has_same_integrity_and_test_gate(self):
201
+ spec, manifest, payload = fake_source([fixture_record()])
202
+ with (
203
+ tempfile.TemporaryDirectory() as tmp,
204
+ patch.dict("jev_adapter.benchmarks.prepare.SUITES", {"example": spec}),
205
+ ):
206
+ root = Path(tmp) / "source"
207
+ directory = root / "evals/example"
208
+ directory.mkdir(parents=True)
209
+ (directory / "manifest.json").write_bytes(manifest)
210
+ (directory / "test.jsonl").write_bytes(payload)
211
+ out = Path(tmp) / "out"
212
+ with self.assertRaisesRegex(ValueError, "locked test"):
213
+ prepare_suite("example", "test", out, source_root=root)
214
+ result = prepare_suite(
215
+ "example", "test", out, source_root=root, allow_test=True
216
+ )
217
+ self.assertTrue(result["protocol"]["locked_test"])
218
+ (directory / "manifest.json").write_bytes(manifest + b" ")
219
+ with self.assertRaisesRegex(ValueError, "SHA256 mismatch"):
220
+ prepare_suite("example", "test", out, source_root=root, allow_test=True)
server/tests/test_disconnect_cleanup.py ADDED
@@ -0,0 +1,150 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """A completed score must not wait forever for a cancelled disconnect poll."""
2
+
3
+ import asyncio
4
+
5
+ import pytest
6
+
7
+ from jev_adapter.backend import AdapterError, ScoringResult
8
+ from jev_adapter.protocol import SystemOneRequest
9
+ from jev_adapter.server import create_app
10
+
11
+
12
+ class ControlledBackend:
13
+ model = "test-model"
14
+
15
+ def __init__(self):
16
+ self.release = asyncio.Event()
17
+ self.entered = asyncio.Event()
18
+ self.cancelled = asyncio.Event()
19
+
20
+ def labels(self, count):
21
+ return tuple("AB"[:count]), tuple(range(count))
22
+
23
+ async def evaluate(self, prompt, images, labels, token_ids, assistant_prefix):
24
+ self.entered.set()
25
+ try:
26
+ await self.release.wait()
27
+ except asyncio.CancelledError:
28
+ self.cancelled.set()
29
+ raise
30
+ return ScoringResult((-0.1, -2.0), 10)
31
+
32
+
33
+ class CancellationSwallowingRequest:
34
+ """Reproduce a request probe whose internal cancellation scope wins a race.
35
+
36
+ CancelledError can be suppressed within a probe, leaving its caller alive.
37
+ A real Request.is_disconnected uses a self-cancelling AnyIO CancelScope;
38
+ deliberately control that boundary here rather than rely on scheduler luck.
39
+ """
40
+
41
+ def __init__(self):
42
+ self.entered = asyncio.Event()
43
+ self.swallowed = asyncio.Event()
44
+ self.probe_task = None
45
+ self.calls = 0
46
+
47
+ async def is_disconnected(self):
48
+ self.probe_task = asyncio.current_task()
49
+ self.calls += 1
50
+ self.entered.set()
51
+ try:
52
+ await asyncio.Future()
53
+ except asyncio.CancelledError:
54
+ self.swallowed.set()
55
+ return False
56
+
57
+
58
+ def endpoint_and_body(backend):
59
+ app = create_app(backend)
60
+ endpoint = next(
61
+ route.endpoint
62
+ for route in app.routes
63
+ if getattr(route, "path", None) == "/v1/systemone"
64
+ )
65
+ body = SystemOneRequest.model_validate(
66
+ {
67
+ "model": backend.model,
68
+ "state": "A billing question.",
69
+ "questions": {
70
+ "route": {
71
+ "type": "choice",
72
+ "instructions": "Choose the team.",
73
+ "criteria": {"billing": None, "technical": None},
74
+ }
75
+ },
76
+ }
77
+ )
78
+ return endpoint, body
79
+
80
+
81
+ @pytest.mark.asyncio
82
+ async def test_completed_score_returns_when_disconnect_probe_swallows_cancel():
83
+ backend = ControlledBackend()
84
+ endpoint, body = endpoint_and_body(backend)
85
+ request = CancellationSwallowingRequest()
86
+ response = asyncio.create_task(endpoint(body, request))
87
+ try:
88
+ await asyncio.wait_for(request.entered.wait(), timeout=1)
89
+ await asyncio.wait_for(backend.entered.wait(), timeout=1)
90
+ backend.release.set()
91
+ await asyncio.wait_for(request.swallowed.wait(), timeout=1)
92
+ done, _ = await asyncio.wait({response}, timeout=0.1)
93
+ assert response in done, "completed inference is stuck cleaning up its watcher"
94
+ result = response.result()
95
+ assert result["answers"]["route"]["choice"] == "billing"
96
+ assert request.calls == 1
97
+ assert request.probe_task.done()
98
+ finally:
99
+ # Do not let a deliberately cancellation-resistant fake leak after red.
100
+ request.is_disconnected = disconnected_now
101
+ if request.probe_task is not None:
102
+ request.probe_task.cancel()
103
+ response.cancel()
104
+ await asyncio.gather(response, return_exceptions=True)
105
+
106
+
107
+ async def disconnected_now():
108
+ return True
109
+
110
+
111
+ @pytest.mark.asyncio
112
+ async def test_client_disconnect_still_cancels_unfinished_inference():
113
+ backend = ControlledBackend()
114
+ endpoint, body = endpoint_and_body(backend)
115
+
116
+ class DisconnectedRequest:
117
+ async def is_disconnected(self):
118
+ await backend.entered.wait()
119
+ return True
120
+
121
+ with pytest.raises(AdapterError) as error:
122
+ await asyncio.wait_for(endpoint(body, DisconnectedRequest()), timeout=1)
123
+ assert error.value.code == "client_disconnected"
124
+ assert error.value.status == 499
125
+ assert backend.cancelled.is_set()
126
+
127
+
128
+ @pytest.mark.asyncio
129
+ async def test_cancelled_handler_joins_work_and_cancellation_resistant_watcher():
130
+ backend = ControlledBackend()
131
+ endpoint, body = endpoint_and_body(backend)
132
+ request = CancellationSwallowingRequest()
133
+ response = asyncio.create_task(endpoint(body, request))
134
+ try:
135
+ await asyncio.wait_for(request.entered.wait(), timeout=1)
136
+ await asyncio.wait_for(backend.entered.wait(), timeout=1)
137
+ response.cancel()
138
+ done, _ = await asyncio.wait({response}, timeout=0.1)
139
+ assert response in done, "cancelled handler did not release its child tasks"
140
+ with pytest.raises(asyncio.CancelledError):
141
+ response.result()
142
+ assert backend.cancelled.is_set()
143
+ assert request.swallowed.is_set()
144
+ assert request.probe_task.done()
145
+ finally:
146
+ request.is_disconnected = disconnected_now
147
+ if request.probe_task is not None:
148
+ request.probe_task.cancel()
149
+ response.cancel()
150
+ await asyncio.gather(response, return_exceptions=True)
server/tests/test_main.py ADDED
@@ -0,0 +1,368 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """CLI wiring tests for --prompt-wording (argparse choices/env default, and
2
+ plumbing into SGLangBackend + create_app). uvicorn.run is monkeypatched so
3
+ main() never actually binds a socket; SGLangBackend/create_app/NativeTokenizer
4
+ are monkeypatched to capture their constructor arguments instead of talking to
5
+ a real engine or loading real tokenizer files (transformers is an optional,
6
+ not-installed-here dependency for this default venv)."""
7
+
8
+ import sys
9
+
10
+ import pytest
11
+
12
+ import jev_adapter.__main__ as main_module
13
+ import jev_adapter.native_tokenizer as native_tokenizer_module
14
+ import jev_adapter.server as server_module
15
+ import jev_adapter.sglang as sglang_module
16
+
17
+
18
+ class Capture:
19
+ def __init__(self):
20
+ self.backend_kwargs = None
21
+ self.app_kwargs = None
22
+
23
+ def fake_backend(self, *args, **kwargs):
24
+ self.backend_kwargs = kwargs
25
+ return object()
26
+
27
+ def fake_create_app(self, backend, **kwargs):
28
+ self.app_kwargs = kwargs
29
+ return object()
30
+
31
+
32
+ @pytest.fixture
33
+ def capture(monkeypatch):
34
+ capture = Capture()
35
+ monkeypatch.setattr(sglang_module, "SGLangBackend", capture.fake_backend)
36
+ monkeypatch.setattr(server_module, "create_app", capture.fake_create_app)
37
+ monkeypatch.setattr("uvicorn.run", lambda app, **kwargs: None)
38
+ return capture
39
+
40
+
41
+ def run_main(monkeypatch, argv, env=None):
42
+ monkeypatch.setattr(sys, "argv", ["jev-adapter", *argv])
43
+ for key in (
44
+ "JEV_PROMPT_WORDING",
45
+ "JEV_DEFAULT_TEMPERATURE",
46
+ "JEV_NATIVE_SYSTEM_PROMPT",
47
+ ):
48
+ monkeypatch.delenv(key, raising=False)
49
+ for key, value in (env or {}).items():
50
+ monkeypatch.setenv(key, value)
51
+ main_module.main()
52
+
53
+
54
+ def test_prompt_wording_defaults_to_served(monkeypatch, capture):
55
+ run_main(monkeypatch, ["--model", "decision-model"])
56
+ assert capture.backend_kwargs["prompt_wording"] == "served"
57
+ assert capture.backend_kwargs["native_system_prompt"] is None
58
+ assert capture.app_kwargs["prompt_wording"] == "served"
59
+
60
+
61
+ def test_prompt_wording_flag_selects_native(monkeypatch, capture):
62
+ run_main(monkeypatch, ["--model", "decision-model", "--prompt-wording", "native"])
63
+ assert capture.backend_kwargs["prompt_wording"] == "native"
64
+ assert capture.app_kwargs["prompt_wording"] == "native"
65
+
66
+
67
+ def test_prompt_wording_env_var_sets_the_default(monkeypatch, capture):
68
+ run_main(
69
+ monkeypatch,
70
+ ["--model", "decision-model"],
71
+ env={"JEV_PROMPT_WORDING": "native"},
72
+ )
73
+ assert capture.backend_kwargs["prompt_wording"] == "native"
74
+ assert capture.app_kwargs["prompt_wording"] == "native"
75
+
76
+
77
+ def test_explicit_flag_overrides_env_var(monkeypatch, capture):
78
+ run_main(
79
+ monkeypatch,
80
+ ["--model", "decision-model", "--prompt-wording", "served"],
81
+ env={"JEV_PROMPT_WORDING": "native"},
82
+ )
83
+ assert capture.backend_kwargs["prompt_wording"] == "served"
84
+
85
+
86
+ def test_invalid_prompt_wording_choice_rejected(monkeypatch, capture):
87
+ monkeypatch.setattr(sys, "argv", ["jev-adapter", "--model", "m", "--prompt-wording", "bogus"])
88
+ with pytest.raises(SystemExit):
89
+ main_module.main()
90
+
91
+
92
+ def test_native_wording_without_tokenizer_model_serves_text_only_and_warns(
93
+ monkeypatch, capture, caplog
94
+ ):
95
+ with caplog.at_level("WARNING"):
96
+ run_main(monkeypatch, ["--model", "decision-model", "--prompt-wording", "native"])
97
+ assert capture.backend_kwargs["native_system_prompt"] is None
98
+ assert any("native" in record.message for record in caplog.records)
99
+
100
+
101
+ def test_native_wording_with_tokenizer_model_extracts_the_system_prompt(
102
+ monkeypatch, capture
103
+ ):
104
+ calls = []
105
+
106
+ class FakeNativeTokenizer:
107
+ @classmethod
108
+ def from_pretrained(cls, model, revision):
109
+ calls.append(("from_pretrained", model, revision))
110
+ return object()
111
+
112
+ @classmethod
113
+ def native_default_system_prompt(cls, model, revision):
114
+ calls.append(("native_default_system_prompt", model, revision))
115
+ return "Default system text."
116
+
117
+ monkeypatch.setattr(native_tokenizer_module, "NativeTokenizer", FakeNativeTokenizer)
118
+ run_main(
119
+ monkeypatch,
120
+ [
121
+ "--model",
122
+ "decision-model",
123
+ "--prompt-wording",
124
+ "native",
125
+ "--tokenizer-model",
126
+ "mistralai/Ministral-3-8B-Instruct-2512-BF16",
127
+ "--tokenizer-revision",
128
+ "f" * 40,
129
+ ],
130
+ )
131
+ assert calls == [
132
+ ("from_pretrained", "mistralai/Ministral-3-8B-Instruct-2512-BF16", "f" * 40),
133
+ (
134
+ "native_default_system_prompt",
135
+ "mistralai/Ministral-3-8B-Instruct-2512-BF16",
136
+ "f" * 40,
137
+ ),
138
+ ]
139
+ assert capture.backend_kwargs["native_system_prompt"] == "Default system text."
140
+
141
+
142
+ def test_served_wording_with_tokenizer_model_never_extracts_a_system_prompt(
143
+ monkeypatch, capture
144
+ ):
145
+ calls = []
146
+
147
+ class FakeNativeTokenizer:
148
+ @classmethod
149
+ def from_pretrained(cls, model, revision):
150
+ calls.append("from_pretrained")
151
+ return object()
152
+
153
+ @classmethod
154
+ def native_default_system_prompt(cls, model, revision):
155
+ calls.append("native_default_system_prompt")
156
+ return "should not be reached"
157
+
158
+ monkeypatch.setattr(native_tokenizer_module, "NativeTokenizer", FakeNativeTokenizer)
159
+ run_main(
160
+ monkeypatch,
161
+ [
162
+ "--model",
163
+ "decision-model",
164
+ "--tokenizer-model",
165
+ "org/model",
166
+ "--tokenizer-revision",
167
+ "a" * 40,
168
+ ],
169
+ )
170
+ assert calls == ["from_pretrained"]
171
+ assert capture.backend_kwargs["prompt_wording"] == "served"
172
+ assert capture.backend_kwargs["native_system_prompt"] is None
173
+
174
+
175
+ def test_native_system_prompt_defaults_to_auto(monkeypatch, capture):
176
+ calls = []
177
+
178
+ class FakeNativeTokenizer:
179
+ @classmethod
180
+ def from_pretrained(cls, model, revision):
181
+ return object()
182
+
183
+ @classmethod
184
+ def native_default_system_prompt(cls, model, revision):
185
+ calls.append((model, revision))
186
+ return "Default system text."
187
+
188
+ monkeypatch.setattr(native_tokenizer_module, "NativeTokenizer", FakeNativeTokenizer)
189
+ run_main(
190
+ monkeypatch,
191
+ [
192
+ "--model",
193
+ "decision-model",
194
+ "--prompt-wording",
195
+ "native",
196
+ "--tokenizer-model",
197
+ "org/model",
198
+ "--tokenizer-revision",
199
+ "a" * 40,
200
+ ],
201
+ )
202
+ assert calls == [("org/model", "a" * 40)]
203
+ assert capture.backend_kwargs["native_system_prompt"] == "Default system text."
204
+
205
+
206
+ def test_native_system_prompt_none_skips_extraction_and_warning(
207
+ monkeypatch, capture, caplog
208
+ ):
209
+ calls = []
210
+
211
+ class FakeNativeTokenizer:
212
+ @classmethod
213
+ def from_pretrained(cls, model, revision):
214
+ return object()
215
+
216
+ @classmethod
217
+ def native_default_system_prompt(cls, model, revision):
218
+ calls.append((model, revision))
219
+ return "should not be reached"
220
+
221
+ monkeypatch.setattr(native_tokenizer_module, "NativeTokenizer", FakeNativeTokenizer)
222
+ with caplog.at_level("WARNING"):
223
+ run_main(
224
+ monkeypatch,
225
+ [
226
+ "--model",
227
+ "decision-model",
228
+ "--prompt-wording",
229
+ "native",
230
+ "--tokenizer-model",
231
+ "org/model",
232
+ "--tokenizer-revision",
233
+ "a" * 40,
234
+ "--native-system-prompt",
235
+ "none",
236
+ ],
237
+ )
238
+ assert calls == []
239
+ assert capture.backend_kwargs["native_system_prompt"] is None
240
+ assert not any("native" in record.message for record in caplog.records)
241
+
242
+
243
+ def test_native_system_prompt_none_without_tokenizer_never_warns(
244
+ monkeypatch, capture, caplog
245
+ ):
246
+ with caplog.at_level("WARNING"):
247
+ run_main(
248
+ monkeypatch,
249
+ [
250
+ "--model",
251
+ "decision-model",
252
+ "--prompt-wording",
253
+ "native",
254
+ "--native-system-prompt",
255
+ "none",
256
+ ],
257
+ )
258
+ assert capture.backend_kwargs["native_system_prompt"] is None
259
+ assert caplog.records == []
260
+
261
+
262
+ def test_native_system_prompt_env_var_sets_the_default(monkeypatch, capture, caplog):
263
+ with caplog.at_level("WARNING"):
264
+ run_main(
265
+ monkeypatch,
266
+ ["--model", "decision-model", "--prompt-wording", "native"],
267
+ env={"JEV_NATIVE_SYSTEM_PROMPT": "none"},
268
+ )
269
+ assert capture.backend_kwargs["native_system_prompt"] is None
270
+ assert caplog.records == []
271
+
272
+
273
+ def test_native_system_prompt_explicit_flag_overrides_env_var(monkeypatch, capture):
274
+ calls = []
275
+
276
+ class FakeNativeTokenizer:
277
+ @classmethod
278
+ def from_pretrained(cls, model, revision):
279
+ return object()
280
+
281
+ @classmethod
282
+ def native_default_system_prompt(cls, model, revision):
283
+ calls.append((model, revision))
284
+ return "Default system text."
285
+
286
+ monkeypatch.setattr(native_tokenizer_module, "NativeTokenizer", FakeNativeTokenizer)
287
+ run_main(
288
+ monkeypatch,
289
+ [
290
+ "--model",
291
+ "decision-model",
292
+ "--prompt-wording",
293
+ "native",
294
+ "--tokenizer-model",
295
+ "org/model",
296
+ "--tokenizer-revision",
297
+ "a" * 40,
298
+ "--native-system-prompt",
299
+ "auto",
300
+ ],
301
+ env={"JEV_NATIVE_SYSTEM_PROMPT": "none"},
302
+ )
303
+ assert calls == [("org/model", "a" * 40)]
304
+ assert capture.backend_kwargs["native_system_prompt"] == "Default system text."
305
+
306
+
307
+ def test_native_system_prompt_none_has_no_effect_on_served_wording(
308
+ monkeypatch, capture
309
+ ):
310
+ calls = []
311
+
312
+ class FakeNativeTokenizer:
313
+ @classmethod
314
+ def from_pretrained(cls, model, revision):
315
+ return object()
316
+
317
+ @classmethod
318
+ def native_default_system_prompt(cls, model, revision):
319
+ calls.append((model, revision))
320
+ return "should not be reached"
321
+
322
+ monkeypatch.setattr(native_tokenizer_module, "NativeTokenizer", FakeNativeTokenizer)
323
+ run_main(
324
+ monkeypatch,
325
+ [
326
+ "--model",
327
+ "decision-model",
328
+ "--tokenizer-model",
329
+ "org/model",
330
+ "--tokenizer-revision",
331
+ "a" * 40,
332
+ "--native-system-prompt",
333
+ "auto",
334
+ ],
335
+ )
336
+ assert calls == []
337
+ assert capture.backend_kwargs["prompt_wording"] == "served"
338
+ assert capture.backend_kwargs["native_system_prompt"] is None
339
+
340
+
341
+ def test_invalid_native_system_prompt_choice_rejected(monkeypatch, capture):
342
+ monkeypatch.setattr(
343
+ sys,
344
+ "argv",
345
+ ["jev-adapter", "--model", "m", "--native-system-prompt", "bogus"],
346
+ )
347
+ with pytest.raises(SystemExit):
348
+ main_module.main()
349
+
350
+
351
+ def test_native_system_prompt_startup_log_line(monkeypatch, capture, caplog):
352
+ with caplog.at_level("INFO"):
353
+ run_main(
354
+ monkeypatch,
355
+ [
356
+ "--model",
357
+ "decision-model",
358
+ "--prompt-wording",
359
+ "native",
360
+ "--native-system-prompt",
361
+ "none",
362
+ ],
363
+ )
364
+ info_records = [r for r in caplog.records if r.levelname == "INFO"]
365
+ assert len(info_records) == 1
366
+ message = info_records[0].getMessage()
367
+ assert "prompt_wording=native" in message
368
+ assert "native_system_prompt=none" in message
server/tests/test_service.py ADDED
@@ -0,0 +1,324 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import asyncio
2
+ import math
3
+ import unittest
4
+
5
+ from fastapi.testclient import TestClient
6
+
7
+ from jev_adapter.__main__ import positive_temperature
8
+ from jev_adapter.backend import AdapterError, ScoringResult
9
+ from jev_adapter.protocol import SystemOneRequest, probabilities_from_logprobs
10
+ from jev_adapter.server import create_app
11
+ from jev_adapter.service import SystemOneService
12
+
13
+
14
+ def payload():
15
+ return {
16
+ "model": "test-model",
17
+ "state": {"text": "결제가 두 번 되었어요. 환불해 주세요."},
18
+ "questions": {
19
+ "route": {
20
+ "type": "choice",
21
+ "instructions": "담당 부서",
22
+ "criteria": {"billing": "청구", "technical": "기술"},
23
+ },
24
+ "refund": {"type": "noul", "instructions": "환불 요청인가?"},
25
+ "urgency": {
26
+ "type": "score",
27
+ "instructions": "긴급도",
28
+ "criteria": ["낮음", "중간", "높음"],
29
+ },
30
+ },
31
+ }
32
+
33
+
34
+ class FakeBackend:
35
+ model = "test-model"
36
+
37
+ def __init__(self, delay=0, failure=False):
38
+ self.calls = []
39
+ self.active = self.peak = self.cancelled = 0
40
+ self.started = self.closed = False
41
+ self.delay, self.failure = delay, failure
42
+
43
+ async def start(self):
44
+ self.started = True
45
+
46
+ async def close(self):
47
+ self.closed = True
48
+
49
+ def labels(self, count):
50
+ return tuple(chr(65 + i) for i in range(count)), tuple(range(65, 65 + count))
51
+
52
+ async def evaluate(self, prompt, images, labels, token_ids, assistant_prefix):
53
+ self.calls.append((prompt, images, labels, token_ids, assistant_prefix))
54
+ self.active += 1
55
+ self.peak = max(self.peak, self.active)
56
+ try:
57
+ if self.failure:
58
+ raise AdapterError(
59
+ "backend_unavailable", "Engine unavailable.", status=502
60
+ )
61
+ await asyncio.sleep(self.delay)
62
+ return ScoringResult(
63
+ tuple(math.log(0.8 if i == 0 else 0.1) for i in range(len(labels))), 20
64
+ )
65
+ except asyncio.CancelledError:
66
+ self.cancelled += 1
67
+ raise
68
+ finally:
69
+ self.active -= 1
70
+
71
+
72
+ class TestService(unittest.IsolatedAsyncioTestCase):
73
+ async def test_mixed_questions_keep_images_and_build_typed_response(self):
74
+ backend = FakeBackend()
75
+ body = payload()
76
+ body["images"] = ["data:image/png;base64,aGVsbG8="]
77
+ result = await SystemOneService(backend).score(
78
+ SystemOneRequest.model_validate(body)
79
+ )
80
+ self.assertEqual(result["answers"]["route"]["choice"], "billing")
81
+ self.assertAlmostEqual(result["answers"]["refund"]["noul"], 8 / 9)
82
+ self.assertAlmostEqual(result["answers"]["urgency"]["score"], 0.3)
83
+ self.assertEqual(result["usage"], {"input_tokens": 60, "output_tokens": 0})
84
+ self.assertEqual(result["metadata"]["evaluations"], 3)
85
+ for call in backend.calls:
86
+ self.assertEqual(call[1], body["images"])
87
+ self.assertNotIn("담당 부서", backend.calls[1][0])
88
+
89
+ async def test_rotations_preserve_canonical_option_mapping(self):
90
+ body = payload()
91
+ body["questions"] = {"route": body["questions"]["route"]}
92
+ body["options"] = {"permutations": 2, "return_logprobs": True}
93
+ result = await SystemOneService(FakeBackend()).score(
94
+ SystemOneRequest.model_validate(body)
95
+ )
96
+ route = result["answers"]["route"]
97
+ self.assertAlmostEqual(route["probabilities"]["billing"], 0.5)
98
+ self.assertAlmostEqual(route["confidence"], 0)
99
+ self.assertEqual(len(route["logprobs"]), 2)
100
+
101
+ async def test_concurrency_limit_applies_across_requests(self):
102
+ backend = FakeBackend(delay=0.01)
103
+ service = SystemOneService(backend, max_concurrency=2)
104
+ await asyncio.gather(
105
+ *(
106
+ service.score(SystemOneRequest.model_validate(payload()))
107
+ for _ in range(3)
108
+ )
109
+ )
110
+ self.assertEqual(len(backend.calls), 9)
111
+ self.assertEqual(backend.peak, 2)
112
+
113
+ async def test_cancellation_closes_all_pending_work(self):
114
+ backend = FakeBackend(delay=10)
115
+ task = asyncio.create_task(
116
+ SystemOneService(backend, max_concurrency=2).score(
117
+ SystemOneRequest.model_validate(payload())
118
+ )
119
+ )
120
+ while backend.active != 2:
121
+ await asyncio.sleep(0)
122
+ task.cancel()
123
+ with self.assertRaises(asyncio.CancelledError):
124
+ await task
125
+ self.assertEqual(backend.cancelled, 2)
126
+ self.assertEqual(backend.active, 0)
127
+ self.assertEqual(len(backend.calls), 2)
128
+
129
+ async def test_wrong_model_rejected_before_inference(self):
130
+ backend = FakeBackend()
131
+ body = payload()
132
+ body["model"] = "wrong-model"
133
+ with self.assertRaises(AdapterError) as error:
134
+ await SystemOneService(backend).score(SystemOneRequest.model_validate(body))
135
+ self.assertEqual(error.exception.status, 404)
136
+ self.assertEqual(backend.calls, [])
137
+
138
+
139
+ class TestDefaultTemperature(unittest.IsolatedAsyncioTestCase):
140
+ """A server-side default fills an omitted options.temperature only."""
141
+
142
+ vector = tuple(math.log(0.8 if i == 0 else 0.1) for i in range(3))
143
+
144
+ async def score(self, body, **service_options):
145
+ service = SystemOneService(FakeBackend(), **service_options)
146
+ return await service.score(SystemOneRequest.model_validate(body))
147
+
148
+ def route_probabilities(self, result):
149
+ return list(result["answers"]["urgency"]["probabilities"].values())
150
+
151
+ async def test_unset_default_keeps_request_default_of_one(self):
152
+ result = await self.score(payload())
153
+ self.assertEqual(result["metadata"]["temperature"], 1.0)
154
+ expected = probabilities_from_logprobs(self.vector, 1.0)
155
+ for actual, wanted in zip(self.route_probabilities(result), expected):
156
+ self.assertAlmostEqual(actual, wanted)
157
+
158
+ async def test_omitted_temperature_takes_server_default(self):
159
+ for body in (payload(), {**payload(), "options": {"permutations": 1}}):
160
+ result = await self.score(body, default_temperature=2.4)
161
+ self.assertEqual(result["metadata"]["temperature"], 2.4)
162
+ expected = probabilities_from_logprobs(self.vector, 2.4)
163
+ for actual, wanted in zip(self.route_probabilities(result), expected):
164
+ self.assertAlmostEqual(actual, wanted)
165
+ self.assertNotAlmostEqual(
166
+ self.route_probabilities(result)[0],
167
+ probabilities_from_logprobs(self.vector, 1.0)[0],
168
+ )
169
+
170
+ async def test_explicit_request_temperature_wins_over_default(self):
171
+ body = {**payload(), "options": {"temperature": 1.0}}
172
+ result = await self.score(body, default_temperature=2.4)
173
+ self.assertEqual(result["metadata"]["temperature"], 1.0)
174
+ expected = probabilities_from_logprobs(self.vector, 1.0)
175
+ for actual, wanted in zip(self.route_probabilities(result), expected):
176
+ self.assertAlmostEqual(actual, wanted)
177
+ body = {**payload(), "options": {"temperature": 3.0}}
178
+ result = await self.score(body, default_temperature=2.4)
179
+ self.assertEqual(result["metadata"]["temperature"], 3.0)
180
+
181
+ async def test_disabled_scaling_ignores_default(self):
182
+ body = {**payload(), "options": {"temperature_scaling": False}}
183
+ result = await self.score(body, default_temperature=2.4)
184
+ self.assertEqual(result["metadata"]["temperature"], 1.0)
185
+ expected = probabilities_from_logprobs(self.vector, 1.0)
186
+ for actual, wanted in zip(self.route_probabilities(result), expected):
187
+ self.assertAlmostEqual(actual, wanted)
188
+
189
+ async def test_default_does_not_mutate_the_request(self):
190
+ request = SystemOneRequest.model_validate(payload())
191
+ service = SystemOneService(FakeBackend(), default_temperature=2.4)
192
+ await service.score(request)
193
+ self.assertEqual(request.options.temperature, 1.0)
194
+ self.assertNotIn("temperature", request.options.model_fields_set)
195
+ self.assertEqual(service.effective_options(request).temperature, 2.4)
196
+
197
+ def test_invalid_default_temperature_rejected_at_construction(self):
198
+ for value in (0, -1.0, float("inf"), float("nan"), True, "2.4", None):
199
+ with self.assertRaises((ValueError, TypeError)):
200
+ SystemOneService(FakeBackend(), default_temperature=value)
201
+ self.assertEqual(
202
+ SystemOneService(FakeBackend(), default_temperature=2).default_temperature,
203
+ 2.0,
204
+ )
205
+
206
+ def test_cli_default_temperature_type(self):
207
+ self.assertEqual(positive_temperature("2.4"), 2.4)
208
+ self.assertEqual(positive_temperature("1"), 1.0)
209
+ import argparse
210
+
211
+ for text in ("0", "-1", "inf", "nan", "abc", ""):
212
+ with self.assertRaises(argparse.ArgumentTypeError):
213
+ positive_temperature(text)
214
+
215
+
216
+ class TestPromptWording(unittest.IsolatedAsyncioTestCase):
217
+ """--prompt-wording is a server-wide setting (CLI/env, not per-request);
218
+ the default keeps today's served text unchanged."""
219
+
220
+ async def test_default_served_wording_is_unchanged(self):
221
+ backend = FakeBackend()
222
+ body = payload()
223
+ body["questions"] = {"route": body["questions"]["route"]}
224
+ await SystemOneService(backend).score(SystemOneRequest.model_validate(body))
225
+ prompt = backend.calls[0][0]
226
+ self.assertTrue(prompt.startswith("Context:\n"))
227
+
228
+ async def test_native_wording_is_plumbed_into_build_prompt(self):
229
+ backend = FakeBackend()
230
+ body = payload()
231
+ body["questions"] = {"route": body["questions"]["route"]}
232
+ service = SystemOneService(backend, prompt_wording="native")
233
+ await service.score(SystemOneRequest.model_validate(body))
234
+ prompt = backend.calls[0][0]
235
+ self.assertTrue(prompt.startswith("Read the state and question."))
236
+ self.assertIn("\n\nState:\n", prompt)
237
+ self.assertNotIn("Context:\n", prompt)
238
+
239
+ async def test_native_wording_requires_canonical_az_labels(self):
240
+ # FakeBackend.labels() already returns canonical A, B, C, ... so this
241
+ # documents the happy path; build_prompt itself enforces the
242
+ # requirement (see TestNativePromptWording in test_protocol.py) when a
243
+ # backend's labels are not canonical.
244
+ backend = FakeBackend()
245
+ labels, _ = backend.labels(3)
246
+ self.assertEqual(labels, ("A", "B", "C"))
247
+
248
+ def test_invalid_prompt_wording_rejected_at_construction(self):
249
+ for value in ("", "SERVED", "native ", None, 1):
250
+ with self.assertRaises(ValueError):
251
+ SystemOneService(FakeBackend(), prompt_wording=value)
252
+
253
+
254
+ class TestHTTP(unittest.TestCase):
255
+ def test_standalone_app_lifecycle_schema_and_alias(self):
256
+ backend = FakeBackend()
257
+ with TestClient(create_app(backend)) as client:
258
+ self.assertTrue(backend.started)
259
+ self.assertEqual(client.get("/health").json(), {"status": "ok"})
260
+ self.assertEqual(
261
+ client.get("/v1/models").json()["data"][0]["id"], backend.model
262
+ )
263
+ body = payload()
264
+ body["model"] = "jev-latest"
265
+ response = client.post("/v1/systemone", json=body)
266
+ self.assertEqual(response.status_code, 200)
267
+ self.assertEqual(response.json()["model"], backend.model)
268
+ self.assertEqual(
269
+ set(response.json()["answers"]), {"route", "refund", "urgency"}
270
+ )
271
+ bad = client.post("/v1/systemone", json={"model": backend.model})
272
+ self.assertEqual(bad.status_code, 422)
273
+ self.assertEqual(bad.json()["error"]["code"], "invalid_request")
274
+ self.assertTrue(backend.closed)
275
+
276
+ def test_app_default_temperature_applies_when_request_omits_it(self):
277
+ with TestClient(create_app(FakeBackend(), default_temperature=2.4)) as client:
278
+ response = client.post("/v1/systemone", json=payload())
279
+ self.assertEqual(response.status_code, 200)
280
+ self.assertEqual(response.json()["metadata"]["temperature"], 2.4)
281
+ body = {**payload(), "options": {"temperature": 1.5}}
282
+ response = client.post("/v1/systemone", json=body)
283
+ self.assertEqual(response.json()["metadata"]["temperature"], 1.5)
284
+ with self.assertRaises(ValueError):
285
+ create_app(FakeBackend(), default_temperature=0)
286
+
287
+ def test_app_prompt_wording_native_reaches_the_backend_over_http(self):
288
+ backend = FakeBackend()
289
+ with TestClient(create_app(backend, prompt_wording="native")) as client:
290
+ body = payload()
291
+ body["questions"] = {"route": body["questions"]["route"]}
292
+ response = client.post("/v1/systemone", json=body)
293
+ self.assertEqual(response.status_code, 200)
294
+ self.assertTrue(backend.calls[0][0].startswith("Read the state and question."))
295
+
296
+ def test_auth_guards_inference_and_model_discovery(self):
297
+ backend = FakeBackend()
298
+ with TestClient(create_app(backend, api_key="test-secret")) as client:
299
+ self.assertEqual(client.get("/v1/models").status_code, 401)
300
+ self.assertEqual(
301
+ client.post("/v1/systemone", json=payload()).status_code, 401
302
+ )
303
+ self.assertEqual(
304
+ client.post(
305
+ "/v1/systemone",
306
+ json=payload(),
307
+ headers={b"Authorization": b"Bearer caf\xe9"},
308
+ ).status_code,
309
+ 401,
310
+ )
311
+ self.assertEqual(backend.calls, [])
312
+ response = client.post(
313
+ "/v1/systemone",
314
+ json=payload(),
315
+ headers={"Authorization": "Bearer test-secret"},
316
+ )
317
+ self.assertEqual(response.status_code, 200)
318
+
319
+ def test_upstream_failure_is_not_returned_as_probabilities(self):
320
+ with TestClient(create_app(FakeBackend(failure=True))) as client:
321
+ response = client.post("/v1/systemone", json=payload())
322
+ self.assertEqual(response.status_code, 502)
323
+ self.assertEqual(response.json()["error"]["code"], "backend_unavailable")
324
+ self.assertNotIn("answers", response.json())
server/tests/test_sglang.py ADDED
@@ -0,0 +1,409 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import copy
2
+ import json
3
+
4
+ import httpx
5
+ import pytest
6
+
7
+ from jev_adapter.backend import AdapterError
8
+ from jev_adapter.sglang import SGLangBackend
9
+
10
+
11
+ class Engine:
12
+ def __init__(self):
13
+ self.calls = []
14
+ self.info = {
15
+ "served_model_name": "qwen",
16
+ "model_path": "org/qwen",
17
+ "is_generation": True,
18
+ "has_image_understanding": True,
19
+ "model_type": "qwen3_5",
20
+ "architectures": ["Qwen3_5ForConditionalGeneration"],
21
+ }
22
+ self.config = {"speculative_algorithm": None, "skip_tokenizer_init": False}
23
+ self.mutate_result = lambda result: result
24
+ self.invalid_boundary = False
25
+ self.lossy_roundtrip = False
26
+ self.special_label = None
27
+ self.fail_path = None
28
+ self.failure = None
29
+
30
+ @staticmethod
31
+ def encode(text):
32
+ return [ord(char) + 1000 for char in text]
33
+
34
+ @staticmethod
35
+ def decode(tokens):
36
+ return "".join(chr(token - 1000) for token in tokens)
37
+
38
+ def __call__(self, request):
39
+ path = request.url.path
40
+ body = json.loads(request.content) if request.content else None
41
+ self.calls.append((path, body, request))
42
+ if path == self.fail_path:
43
+ if isinstance(self.failure, Exception):
44
+ raise self.failure
45
+ return self.failure
46
+ if path == "/model_info":
47
+ return httpx.Response(200, json=self.info)
48
+ if path == "/server_info":
49
+ return httpx.Response(200, json=self.config)
50
+ if path == "/v1/tokenize":
51
+ if "messages" in body:
52
+ content = body["messages"][0]["content"]
53
+ if isinstance(content, list):
54
+ content = "<image>" * (len(content) - 1) + content[-1]["text"]
55
+ tokens = self.encode("<user>" + content + "</user><assistant>")
56
+ else:
57
+ assert body["add_special_tokens"] is False
58
+ tokens = [self.encode(text) for text in body["prompt"]]
59
+ if body["prompt"][0].startswith("<user>"):
60
+ if self.invalid_boundary:
61
+ tokens[-1][-1] += 1
62
+ if self.lossy_roundtrip:
63
+ tokens[0][-1] += 1
64
+ return httpx.Response(200, json={"tokens": tokens})
65
+ if path == "/v1/detokenize":
66
+ tokens = body["tokens"]
67
+ if isinstance(tokens[0], list):
68
+ assert body["skip_special_tokens"] is True
69
+ text = [self.decode(ids) for ids in tokens]
70
+ text = ["" if value == self.special_label else value for value in text]
71
+ else:
72
+ assert body["skip_special_tokens"] is False
73
+ text = self.decode(tokens)
74
+ return httpx.Response(200, json={"text": text})
75
+ if path == "/generate":
76
+ result = {
77
+ "text": "",
78
+ "meta_info": {
79
+ "completion_tokens": 0,
80
+ "prompt_tokens": 71,
81
+ # Return labels out of order to ensure alignment by token ID.
82
+ "output_token_ids_logprobs": [
83
+ [
84
+ [-index - 0.5, token, None]
85
+ for index, token in reversed(
86
+ list(enumerate(body["token_ids_logprob"]))
87
+ )
88
+ ]
89
+ ],
90
+ },
91
+ }
92
+ return httpx.Response(200, json=self.mutate_result(result))
93
+ raise AssertionError(f"Unexpected route: {path}")
94
+
95
+
96
+ async def make_backend(engine=None, **kwargs):
97
+ engine = engine or Engine()
98
+ client = httpx.AsyncClient(transport=httpx.MockTransport(engine))
99
+ backend = SGLangBackend("http://engine", "qwen", client=client, **kwargs)
100
+ await backend.start()
101
+ return backend, engine, client
102
+
103
+
104
+ @pytest.mark.asyncio
105
+ async def test_public_http_contract_and_zero_decode():
106
+ backend, engine, client = await make_backend(api_key="private-key")
107
+ try:
108
+ labels, tokens = backend.labels(3)
109
+ result = await backend.evaluate("pick an answer", [], labels, tokens, None)
110
+ assert result.logprobs == (-0.5, -1.5, -2.5)
111
+ assert result.input_tokens == 71
112
+ path, body, _ = engine.calls[-1]
113
+ assert path == "/generate"
114
+ assert "text" not in body and "image_data" not in body
115
+ assert body["input_ids"] == engine.encode(
116
+ "<user>pick an answer</user><assistant>"
117
+ )
118
+ assert body["sampling_params"] == {
119
+ "max_new_tokens": 0,
120
+ "temperature": 1,
121
+ "top_p": 1,
122
+ "top_k": -1,
123
+ }
124
+ assert body["token_ids_logprob"] == list(tokens)
125
+ assert body["logprob_start_len"] == -1
126
+ assert body["return_logprob"] is True
127
+ assert body["top_logprobs_num"] == 0
128
+ assert body["return_text_in_logprobs"] is False
129
+ assert body["stream"] is False
130
+ assert body["rid"].startswith("jev-")
131
+ assert all(
132
+ request.headers["Authorization"] == "Bearer private-key"
133
+ for _, _, request in engine.calls
134
+ )
135
+ assert engine.calls[-4][1]["chat_template_kwargs"] == {"enable_thinking": False}
136
+ finally:
137
+ await client.aclose()
138
+
139
+
140
+ def test_prompt_wording_rejects_unknown_value():
141
+ with pytest.raises(ValueError):
142
+ SGLangBackend("http://engine", "qwen", prompt_wording="bogus")
143
+
144
+
145
+ @pytest.mark.asyncio
146
+ async def test_native_wording_prepends_system_message_over_http():
147
+ backend, engine, client = await make_backend(
148
+ prompt_wording="native", native_system_prompt="Default system text."
149
+ )
150
+ try:
151
+ labels, tokens = backend.labels(3)
152
+ await backend.evaluate("pick an answer", [], labels, tokens, None)
153
+ # The messages-based /v1/tokenize call, three calls before /generate
154
+ # (detokenize, boundary-check tokenize, generate follow it).
155
+ path, body, _ = engine.calls[-4]
156
+ assert path == "/v1/tokenize"
157
+ assert body["messages"] == [
158
+ {"role": "system", "content": "Default system text."},
159
+ {"role": "user", "content": "pick an answer"},
160
+ ]
161
+ finally:
162
+ await client.aclose()
163
+
164
+
165
+ @pytest.mark.asyncio
166
+ async def test_served_wording_never_sends_a_system_message_even_if_configured():
167
+ # native_system_prompt is only honored when prompt_wording == "native";
168
+ # the default ("served") must stay byte-identical to today's behaviour.
169
+ backend, engine, client = await make_backend(
170
+ prompt_wording="served", native_system_prompt="Should be ignored."
171
+ )
172
+ try:
173
+ labels, tokens = backend.labels(2)
174
+ await backend.evaluate("look", [], labels, tokens, None)
175
+ path, body, _ = engine.calls[-4]
176
+ assert path == "/v1/tokenize"
177
+ assert body["messages"] == [{"role": "user", "content": "look"}]
178
+ finally:
179
+ await client.aclose()
180
+
181
+
182
+ @pytest.mark.asyncio
183
+ async def test_native_wording_passes_system_prompt_to_native_tokenizer():
184
+ class RecordingNativeTokenizer:
185
+ def __init__(self):
186
+ self.calls = []
187
+
188
+ def prepare(self, prompt, labels, token_ids, assistant_prefix, system_prompt=None):
189
+ self.calls.append((prompt, labels, token_ids, assistant_prefix, system_prompt))
190
+ return [1, 2, 3]
191
+
192
+ native_tokenizer = RecordingNativeTokenizer()
193
+ engine = Engine()
194
+ client = httpx.AsyncClient(transport=httpx.MockTransport(engine))
195
+ backend = SGLangBackend(
196
+ "http://engine",
197
+ "qwen",
198
+ client=client,
199
+ native_tokenizer=native_tokenizer,
200
+ prompt_wording="native",
201
+ native_system_prompt="Default system text.",
202
+ )
203
+ try:
204
+ await backend.start()
205
+ labels, tokens = backend.labels(2)
206
+ await backend.evaluate("pick", [], labels, tokens, "Answer:")
207
+ assert native_tokenizer.calls == [
208
+ ("pick", labels, tokens, "Answer:", "Default system text.")
209
+ ]
210
+ finally:
211
+ await client.aclose()
212
+
213
+
214
+ @pytest.mark.asyncio
215
+ async def test_images_remain_in_native_request_with_template_and_prefix():
216
+ backend, engine, client = await make_backend()
217
+ try:
218
+ images = ["data:image/png;base64,YQ==", "https://example.com/image.png"]
219
+ labels, tokens = backend.labels(2)
220
+ await backend.evaluate("look", images, labels, tokens, "Answer: ")
221
+ body = engine.calls[-1][1]
222
+ assert body["image_data"] == images
223
+ assert body["text"] == "<user><image><image>look</user><assistant>Answer: "
224
+ assert "input_ids" not in body
225
+ content = engine.calls[-4][1]["messages"][0]["content"]
226
+ assert content == [
227
+ {"type": "image_url", "image_url": {"url": images[0]}},
228
+ {"type": "image_url", "image_url": {"url": images[1]}},
229
+ {"type": "text", "text": "look"},
230
+ ]
231
+ finally:
232
+ await client.aclose()
233
+
234
+
235
+ @pytest.mark.asyncio
236
+ @pytest.mark.parametrize(
237
+ "field,value,code",
238
+ [
239
+ ("has_image_understanding", False, "images_not_supported"),
240
+ ("model_type", "kimi_k3", "image_model_not_supported"),
241
+ ],
242
+ )
243
+ async def test_image_capability_rejections(field, value, code):
244
+ engine = Engine()
245
+ engine.info[field] = value
246
+ engine.info["architectures"] = []
247
+ backend, engine, client = await make_backend(engine)
248
+ try:
249
+ labels, tokens = backend.labels(2)
250
+ with pytest.raises(AdapterError, match="image|Image") as error:
251
+ await backend.evaluate(
252
+ "look", ["https://example.com/a.png"], labels, tokens, None
253
+ )
254
+ assert error.value.code == code
255
+ assert not any(path == "/generate" for path, _, _ in engine.calls)
256
+ finally:
257
+ await client.aclose()
258
+
259
+
260
+ @pytest.mark.asyncio
261
+ @pytest.mark.parametrize(
262
+ "flag,code",
263
+ [
264
+ ("invalid_boundary", "invalid_label_boundary"),
265
+ ("lossy_roundtrip", "unsupported_tokenizer_roundtrip"),
266
+ ],
267
+ )
268
+ async def test_rejects_unsafe_tokenization_before_inference(flag, code):
269
+ backend, engine, client = await make_backend()
270
+ try:
271
+ setattr(engine, flag, True)
272
+ labels, tokens = backend.labels(2)
273
+ with pytest.raises(AdapterError) as error:
274
+ await backend.evaluate("look", [], labels, tokens, None)
275
+ assert error.value.code == code
276
+ assert not any(path == "/generate" for path, _, _ in engine.calls)
277
+ finally:
278
+ await client.aclose()
279
+
280
+
281
+ @pytest.mark.asyncio
282
+ @pytest.mark.parametrize(
283
+ "mutation",
284
+ [
285
+ "decoded",
286
+ "missing_label",
287
+ "duplicate_label",
288
+ "nan",
289
+ "positive",
290
+ "missing_usage",
291
+ "bad_usage",
292
+ "empty_positions",
293
+ "multiple_positions",
294
+ ],
295
+ )
296
+ async def test_rejects_unusable_engine_results(mutation):
297
+ backend, engine, client = await make_backend()
298
+
299
+ def mutate(original):
300
+ result = copy.deepcopy(original)
301
+ meta = result["meta_info"]
302
+ entries = meta["output_token_ids_logprobs"][0]
303
+ if mutation == "decoded":
304
+ meta["completion_tokens"] = 1
305
+ elif mutation == "missing_label":
306
+ entries.pop()
307
+ elif mutation == "duplicate_label":
308
+ entries.append(entries[0])
309
+ elif mutation == "nan":
310
+ # JSON null is also rejected; transport JSON forbids NaN values.
311
+ entries[0][0] = None
312
+ elif mutation == "positive":
313
+ entries[0][0] = 1.0
314
+ elif mutation == "missing_usage":
315
+ del meta["completion_tokens"]
316
+ elif mutation == "bad_usage":
317
+ meta["prompt_tokens"] = True
318
+ elif mutation == "empty_positions":
319
+ meta["output_token_ids_logprobs"] = []
320
+ elif mutation == "multiple_positions":
321
+ meta["output_token_ids_logprobs"].append(entries)
322
+ return result
323
+
324
+ engine.mutate_result = mutate
325
+ try:
326
+ labels, tokens = backend.labels(2)
327
+ with pytest.raises(AdapterError) as error:
328
+ await backend.evaluate("look", [], labels, tokens, None)
329
+ assert error.value.status == 502
330
+ assert error.value.code == "invalid_engine_response"
331
+ finally:
332
+ await client.aclose()
333
+
334
+
335
+ @pytest.mark.asyncio
336
+ @pytest.mark.parametrize(
337
+ "failure,status,code",
338
+ [
339
+ (httpx.ReadTimeout("secret upstream location"), 504, "engine_timeout"),
340
+ (httpx.ConnectError("secret upstream location"), 502, "engine_unavailable"),
341
+ (httpx.Response(401, text="secret-key"), 502, "engine_http_error"),
342
+ (httpx.Response(422, text="sensitive prompt"), 422, "engine_rejected_request"),
343
+ (httpx.Response(200, text="not-json secret"), 502, "invalid_engine_response"),
344
+ ],
345
+ )
346
+ async def test_transport_failures_are_sanitized(failure, status, code):
347
+ backend, engine, client = await make_backend()
348
+ engine.fail_path, engine.failure = "/generate", failure
349
+ try:
350
+ labels, tokens = backend.labels(2)
351
+ with pytest.raises(AdapterError) as error:
352
+ await backend.evaluate("look", [], labels, tokens, None)
353
+ assert error.value.status == status
354
+ assert error.value.code == code
355
+ assert "secret" not in str(error.value) and "sensitive" not in str(error.value)
356
+ finally:
357
+ await client.aclose()
358
+
359
+
360
+ @pytest.mark.asyncio
361
+ async def test_special_labels_removed_and_start_idempotent():
362
+ engine = Engine()
363
+ engine.special_label = "A"
364
+ backend, engine, client = await make_backend(engine)
365
+ try:
366
+ labels, _ = backend.labels(2)
367
+ assert labels == ("B", "C")
368
+ count = len(engine.calls)
369
+ await backend.start()
370
+ assert len(engine.calls) == count
371
+ with pytest.raises(AdapterError):
372
+ backend.labels(255)
373
+ finally:
374
+ await client.aclose()
375
+
376
+
377
+ @pytest.mark.asyncio
378
+ @pytest.mark.parametrize(
379
+ "config,code",
380
+ [
381
+ ({"speculative_algorithm": "EAGLE"}, "speculation_not_supported"),
382
+ ({"skip_tokenizer_init": True}, "tokenizer_unavailable"),
383
+ ],
384
+ )
385
+ async def test_startup_validates_engine_configuration(config, code):
386
+ engine = Engine()
387
+ engine.config.update(config)
388
+ async with httpx.AsyncClient(transport=httpx.MockTransport(engine)) as client:
389
+ backend = SGLangBackend("http://engine", "qwen", client=client)
390
+ with pytest.raises(AdapterError) as error:
391
+ await backend.start()
392
+ assert error.value.code == code
393
+
394
+
395
+ @pytest.mark.asyncio
396
+ async def test_nonfinite_raw_json_logprob_rejected():
397
+ backend, engine, client = await make_backend()
398
+ engine.fail_path = "/generate"
399
+ engine.failure = httpx.Response(
400
+ 200,
401
+ text='{"meta_info":{"completion_tokens":0,"prompt_tokens":1,"output_token_ids_logprobs":[[[NaN,1065],[-1.0,1066]]]}}',
402
+ )
403
+ try:
404
+ labels, tokens = backend.labels(2)
405
+ with pytest.raises(AdapterError) as error:
406
+ await backend.evaluate("look", [], labels, tokens, None)
407
+ assert error.value.code == "invalid_engine_response"
408
+ finally:
409
+ await client.aclose()
tekken.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:600bb27946565481ecf51ba8aee252e49b9a68507866080ac9c30185bb312843
3
+ size 16753784
tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d5f6046775b112f0e2d456ee9dba450684ab964fe5c4e231599bdc6773028135
3
+ size 17078128