Text Generation
Transformers
Safetensors
mistral3
image-text-to-text
decision-model
typed-decisions
jev
jevbench
calibration
decode-free
multilingual
vision-language
conversational
Instructions to use StandardThinking/StandardOne-8B with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use StandardThinking/StandardOne-8B with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="StandardThinking/StandardOne-8B") messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] pipe(text=messages)# pip install -U transformers accelerate # Load model directly from transformers import AutoProcessor, AutoModelForMultimodalLM processor = AutoProcessor.from_pretrained("StandardThinking/StandardOne-8B") model = AutoModelForMultimodalLM.from_pretrained("StandardThinking/StandardOne-8B", device_map="auto") messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] inputs = processor.apply_chat_template( messages, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt", ).to(model.device) outputs = model.generate(**inputs, max_new_tokens=256) print(processor.decode(outputs[0][inputs["input_ids"].shape[-1]:])) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use StandardThinking/StandardOne-8B with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "StandardThinking/StandardOne-8B" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "StandardThinking/StandardOne-8B", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker
docker model run hf.co/StandardThinking/StandardOne-8B
- SGLang
How to use StandardThinking/StandardOne-8B with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "StandardThinking/StandardOne-8B" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "StandardThinking/StandardOne-8B", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "StandardThinking/StandardOne-8B" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "StandardThinking/StandardOne-8B", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }' - Docker Model Runner
How to use StandardThinking/StandardOne-8B with Docker Model Runner:
docker model run hf.co/StandardThinking/StandardOne-8B
Add files using upload-large-folder tool
Browse files- .gitattributes +6 -0
- docs/assets/00-benchmark-card.png +3 -0
- docs/assets/02-latency-vs-qwen.png +3 -0
- docs/assets/05-gain-over-base.png +3 -0
- docs/assets/06-vs-jev.png +3 -0
- model-00001-of-00004.safetensors +3 -0
- model-00002-of-00004.safetensors +3 -0
- model-00003-of-00004.safetensors +3 -0
- model-00004-of-00004.safetensors +3 -0
- server/benchmarks/DECISION_BENCHMARK_SELECTION.md +77 -0
- server/benchmarks/METRIC_VALIDATION.md +71 -0
- server/benchmarks/PUBLIC_DATASETS.md +99 -0
- server/benchmarks/README.md +144 -0
- server/benchmarks/data/jevbench-easy/LICENSE +21 -0
- server/benchmarks/data/jevbench-easy/THIRD-PARTY.md +51 -0
- server/benchmarks/data/jevbench-easy/public.jsonl +48 -0
- server/benchmarks/data/jevbench-easy/public.manifest.json +95 -0
- server/benchmarks/data/jevbench-hard/LICENSE +21 -0
- server/benchmarks/data/jevbench-hard/THIRD-PARTY.md +51 -0
- server/benchmarks/data/jevbench-hard/public.jsonl +0 -0
- server/benchmarks/data/jevbench-hard/public.manifest.json +109 -0
- server/benchmarks/data/jevbench-original/LICENSE +21 -0
- server/benchmarks/data/jevbench-original/THIRD-PARTY.md +51 -0
- server/benchmarks/data/jevbench-original/public.manifest.json +101 -0
- server/benchmarks/run_matrix.py +277 -0
- server/benchmarks/verify_metrics.py +140 -0
- server/examples/request.json +13 -0
- server/examples/smoke.py +49 -0
- server/jev_adapter/benchmarks/__init__.py +1 -0
- server/jev_adapter/benchmarks/compare.py +72 -0
- server/jev_adapter/benchmarks/data.py +217 -0
- server/jev_adapter/benchmarks/jevbench.py +292 -0
- server/jev_adapter/benchmarks/metrics.py +326 -0
- server/jev_adapter/benchmarks/prepare.py +227 -0
- server/jev_adapter/benchmarks/run.py +441 -0
- server/tests/test_benchmark_data.py +220 -0
- server/tests/test_disconnect_cleanup.py +150 -0
- server/tests/test_main.py +368 -0
- server/tests/test_service.py +324 -0
- server/tests/test_sglang.py +409 -0
- tekken.json +3 -0
- tokenizer.json +3 -0
.gitattributes
CHANGED
|
@@ -33,3 +33,9 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
tekken.json filter=lfs diff=lfs merge=lfs -text
|
| 37 |
+
tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
| 38 |
+
docs/assets/05-gain-over-base.png filter=lfs diff=lfs merge=lfs -text
|
| 39 |
+
docs/assets/00-benchmark-card.png filter=lfs diff=lfs merge=lfs -text
|
| 40 |
+
docs/assets/06-vs-jev.png filter=lfs diff=lfs merge=lfs -text
|
| 41 |
+
docs/assets/02-latency-vs-qwen.png filter=lfs diff=lfs merge=lfs -text
|
docs/assets/00-benchmark-card.png
ADDED
|
Git LFS Details
|
docs/assets/02-latency-vs-qwen.png
ADDED
|
Git LFS Details
|
docs/assets/05-gain-over-base.png
ADDED
|
Git LFS Details
|
docs/assets/06-vs-jev.png
ADDED
|
Git LFS Details
|
model-00001-of-00004.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:41c1d711d80f96f9cd5ca8eb8b0b4375addd3174b2021bd8ab72dc1a9c9c7500
|
| 3 |
+
size 4999724576
|
model-00002-of-00004.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:57d50fd02ebafbb491d35d445c8279dda89ad8ad5034edf7637971564857b48d
|
| 3 |
+
size 4999820896
|
model-00003-of-00004.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:03616016aafb162714f7265f9e1a192f72bdc109e4d80d079c593fd0756c1cba
|
| 3 |
+
size 4915917688
|
model-00004-of-00004.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a77ea6878393937e46cead80bdfda98e090020cd93b746b89c38a5a81fbf823a
|
| 3 |
+
size 2920659992
|
server/benchmarks/DECISION_BENCHMARK_SELECTION.md
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# JEV 형태의 판단 성능을 위한 추가 벤치 선택
|
| 2 |
+
|
| 3 |
+
확인일: 2026-09-21. **현재 비교의 중심은 KEV의 고정 평가 세트와 JevBench**다.
|
| 4 |
+
추가 공개 데이터는 아래 두 개를 추천한다. 이 문서는 선정 근거이며,
|
| 5 |
+
CLINC150과 ContractNLI의 변환·실행·추가 지표는 아직 통합하지 않았다.
|
| 6 |
+
모두 텍스트 입력에서 정해진 선택지의 확률만 읽어 평가할 수 있다.
|
| 7 |
+
|
| 8 |
+
| 우선순위 | 데이터 | 측정하려는 판단 | 공식 테스트 규모 | 라이선스 |
|
| 9 |
+
|---|---|---|---|---|
|
| 10 |
+
| 1 | CLINC150 / oos-eval | 요청 라우팅과 지원 범위 밖 요청 거절 | 지원 범위 4,500 + 범위 밖 1,000 = 5,500문항 | CC BY 3.0 |
|
| 11 |
+
| 2 | ContractNLI | 문서가 조건을 지지하는지, 반박하는지, 근거가 없는지 | 123개 계약 × 17개 명제 = 2,091판단 | CC BY 4.0 |
|
| 12 |
+
|
| 13 |
+
## CLINC150: 라우팅과 범위 밖 요청
|
| 14 |
+
|
| 15 |
+
150개 intent와 `oos` 한 개를 합쳐 **151개 선택지**를 사용한다.
|
| 16 |
+
공식 `data/data_full.json`은 intent마다 train 100개·validation 20개·test 30개이며,
|
| 17 |
+
별도 OOS 분할은 train 100개·validation 100개·test 1,000개다.
|
| 18 |
+
따라서 검증 세트는 총 3,100개, 테스트는 5,500개다.
|
| 19 |
+
영어 단일 intent 요청을 사람이 작성·패러프레이즈한 데이터다.
|
| 20 |
+
[공식 설명](https://github.com/clinc/oos-eval/tree/828f8093932c8fe6ca7936c3d2e52903b1c523de)
|
| 21 |
+
[라이선스](https://github.com/clinc/oos-eval/blob/828f8093932c8fe6ca7936c3d2e52903b1c523de/LICENSE)
|
| 22 |
+
|
| 23 |
+
권장 매핑은 `state=사용자 발화`, `choice=150개 intent + oos`다.
|
| 24 |
+
공식 intent 이름의 밑줄을 공백으로 바꾸는 정도의 결정적 설명을 쓰고,
|
| 25 |
+
테스트 발화나 정답을 이용해 선택지 설명을 만들지 않는다.
|
| 26 |
+
모델마다 동일한 후보 목록·순서를 사용하며, 후보 일부를 정답에 맞춰 제거하지 않는다.
|
| 27 |
+
151개 선택지는 기존 어댑터의 255개 한도 내에 있다.
|
| 28 |
+
|
| 29 |
+
통합할 때 추가할 지표:
|
| 30 |
+
|
| 31 |
+
- 지원 범위 150개에 대한 accuracy·macro F1, 전체 151개 macro F1.
|
| 32 |
+
- `p(oos)`를 점수, OOS를 양성으로 정한 AUROC·AUPRC.
|
| 33 |
+
- FPR@95% OOS TPR: OOS의 95%를 탐지할 때 정상 요청을 잘못 거절하는 비율.
|
| 34 |
+
- 고정 임계값에서 OOS 탐지율과 정상 요청 오거절률, 수락한 요청의 정확도·coverage.
|
| 35 |
+
- Brier·NLL·ECE와 요청별 p50/p95 지연. 후보 151개의 비용을 함께 명시.
|
| 36 |
+
|
| 37 |
+
운영 임계값을 선택한다면 validation만 사용하고 test에는 고정해서 적용한다.
|
| 38 |
+
가중치 학습 없이 비교하되, 임계값 조정 여부는 별도로 기록한다.
|
| 39 |
+
OOS는 **지원하지 않는 intent**라는 뜻이다. 입력 근거가 부족해서 답을 알 수 없는
|
| 40 |
+
KEV `unknowable`과는 다른 능력을 측정하므로 둘을 합산하지 않는다.
|
| 41 |
+
|
| 42 |
+
직접 검증한 원본: 커밋 `828f8093932c8fe6ca7936c3d2e52903b1c523de`,
|
| 43 |
+
[`data/data_full.json`](https://github.com/clinc/oos-eval/blob/828f8093932c8fe6ca7936c3d2e52903b1c523de/data/data_full.json),
|
| 44 |
+
SHA256 `36923c3705a59e08fe9c3883d8bc2dd966ef93e22cb78ac41171782a698d56e0`.
|
| 45 |
+
다른 `small`·`plus` 변형과 혼합하지 않는다.
|
| 46 |
+
|
| 47 |
+
## ContractNLI: 명시된 문서 근거에 따른 판단
|
| 48 |
+
|
| 49 |
+
607개의 NDA에 동일한 17개 명제를 사람이 주석했다.
|
| 50 |
+
분할은 train 423개·development 61개·test 123개 계약이다.
|
| 51 |
+
정답은 `Entailment`, `Contradiction`, `NotMentioned`의 세 종류다.
|
| 52 |
+
예외에 의한 부정, 문서에 없는 내용, 긴 문서의 근거를 찾는 판단을 점검하기 좋다.
|
| 53 |
+
[공식 데이터·스키마·라이선스](https://stanfordnlp.github.io/contract-nli/)
|
| 54 |
+
[원 논문의 분할 표](https://aclanthology.org/2021.findings-emnlp.164.pdf)
|
| 55 |
+
|
| 56 |
+
`state=전체 계약 텍스트`, 각 명제를 `choice` 질문으로 바꾼다.
|
| 57 |
+
출력은 세 선택지의 확률이며, 근거 문장 생성은 요구하지 않는다.
|
| 58 |
+
`NotMentioned`를 `false`로 합치면 반박과 근거 부재를 구분할 수 없으므로
|
| 59 |
+
기본 비교에는 `noul` 대신 3-way `choice`를 권장한다.
|
| 60 |
+
원본의 정답 evidence span은 입력에 넣지 않는다.
|
| 61 |
+
|
| 62 |
+
권장 지표는 3개 레이블 macro F1·accuracy, 명제별 성능, Brier·NLL·ECE,
|
| 63 |
+
`NotMentioned` precision/recall이다. 계약 단위로 신뢰구간을 계산해 한 문서의
|
| 64 |
+
17개 질문을 독립 표본으로 취급하지 않는다.
|
| 65 |
+
명제당 지연과 계약 전체 17개 질문의 지연을 함께 보고한다.
|
| 66 |
+
|
| 67 |
+
긴 계약을 임의로 잘라 평가하지 않는다. 각 모델의 tokenizer로 입력 길이를 먼저
|
| 68 |
+
측정하고, 공통 context 한도 밖의 문서는 실패·제외 수를 명시한다.
|
| 69 |
+
본문 전체 판단만 수행한 결과를 원 논문의 evidence identification까지 포함한
|
| 70 |
+
전체 과제 결과와 동일하다고 표현하지 않는다.
|
| 71 |
+
|
| 72 |
+
## 해석 범위
|
| 73 |
+
|
| 74 |
+
두 데이터 모두 공개된 지 오래되어 최신 Qwen·Mistral의 사전학습 오염을 배제할 수 없다.
|
| 75 |
+
학습하지 않은 모델을 평가한다는 것은 **이번 실험에서 추가 튜닝하지 않는다**는 뜻이다.
|
| 76 |
+
공개 테스���를 처음 접했다는 보장은 아니다. 영어 중심 결과를 한국어 운영 성능으로
|
| 77 |
+
일반화하지 않으며, 모델 비교에서는 동일 입력·동일 분할·동일 판단 형식을 유지한다.
|
server/benchmarks/METRIC_VALIDATION.md
ADDED
|
@@ -0,0 +1,71 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# KEV metric parity validation
|
| 2 |
+
|
| 3 |
+
On 2026-09-21, the adapter's scalar quality metrics were compared numerically
|
| 4 |
+
with the original KEV implementations at commit
|
| 5 |
+
`4f8110a3f8620cc3a182ae9a708e4398492c4b1a`.
|
| 6 |
+
All common scalar metrics matched within `1e-12` absolute error. The largest
|
| 7 |
+
observed difference was `8.881784197001252e-16` (score MAE). Complete contrastive
|
| 8 |
+
pair metrics matched exactly.
|
| 9 |
+
|
| 10 |
+
This validates the metric calculations. It does not validate H200 execution,
|
| 11 |
+
inference probabilities, model accuracy, or serving latency.
|
| 12 |
+
|
| 13 |
+
## Method
|
| 14 |
+
|
| 15 |
+
The optional [verify_metrics.py](verify_metrics.py) script reads these original
|
| 16 |
+
files from a local KEV checkout and verifies their full SHA256 before execution:
|
| 17 |
+
|
| 18 |
+
| File | Extracted functions | SHA256 |
|
| 19 |
+
|---|---|---|
|
| 20 |
+
| `kev/evaluate.py` | `ece` | `1b20e3f9edf417aa8dae924b1526e52f74b710cadf7213c5ec68334f6e7f8fe1` |
|
| 21 |
+
| `kev/benchmark.py` | `coverage_at_error`, `metrics` | `4192ec3b26b065452f84bde38a091e6854a28fe185d2e0f39a0c91b2efe69df7` |
|
| 22 |
+
| `kev/contrastive.py` | `paired_flip` | `cbb979aa5d40265ad0e64695f94b281d91751fa405811ddfa8212fede111edcf` |
|
| 23 |
+
|
| 24 |
+
Python AST extraction keeps only those function definitions. KEV's package,
|
| 25 |
+
PyTorch, Transformers and model-loading code are not imported. The audit uses
|
| 26 |
+
NumPy in its own optional environment; NumPy is not an adapter dependency.
|
| 27 |
+
The original successful audit used NumPy `2.3.5`.
|
| 28 |
+
|
| 29 |
+
Input generation uses `numpy.random.default_rng(483)` for 100 batches of 50
|
| 30 |
+
rows, totaling 5,000 rows. Rows mix choice, boolean and ordinal score questions,
|
| 31 |
+
with 2–10 options. Probabilities come from Dirichlet distributions. The first
|
| 32 |
+
10 rows in each batch additionally cover binary probability endpoints and
|
| 33 |
+
decimal boundaries, including zero and one. Both implementations receive the
|
| 34 |
+
same rows in the same order, including confidence ties.
|
| 35 |
+
|
| 36 |
+
Compared metrics include accuracy, NLL, multiclass Brier score, ten-bin ECE,
|
| 37 |
+
mean confidence, confidence bias, confident errors, coverage/accuracy at 0.9,
|
| 38 |
+
coverage at 1%/5% empirical error, score MAE and ranked probability score.
|
| 39 |
+
Ten complete two-sibling pairs additionally check relevant and invariant pair
|
| 40 |
+
metrics. Five pairs have changed gold labels and five have unchanged labels.
|
| 41 |
+
|
| 42 |
+
## Reproduce
|
| 43 |
+
|
| 44 |
+
Use a Python environment that already has NumPy installed:
|
| 45 |
+
|
| 46 |
+
```bash
|
| 47 |
+
git clone https://github.com/jaredpalmer/kev.git /tmp/kev-reference
|
| 48 |
+
git -C /tmp/kev-reference checkout 4f8110a3f8620cc3a182ae9a708e4398492c4b1a
|
| 49 |
+
python benchmarks/verify_metrics.py --kev-root /tmp/kev-reference
|
| 50 |
+
```
|
| 51 |
+
|
| 52 |
+
The JSON output includes the reference hashes, current adapter metric-code hash,
|
| 53 |
+
NumPy version, sample counts, per-metric maximum differences and pair results.
|
| 54 |
+
A source mismatch or numerical discrepancy exits unsuccessfully. Reference
|
| 55 |
+
files are checked even if the local checkout has uncommitted changes.
|
| 56 |
+
|
| 57 |
+
## Scope differences
|
| 58 |
+
|
| 59 |
+
- Accuracy headlines use clean question rows, not all submitted records.
|
| 60 |
+
Source `unknowable` is excluded from accuracy and scored for confidence.
|
| 61 |
+
- Incomplete pairs are counted explicitly for smoke subsets; KEV's original
|
| 62 |
+
pair function rejects incomplete pairs.
|
| 63 |
+
- The adapter omits meaningless unknowable accuracy from grouped reports;
|
| 64 |
+
KEV's original code includes it in some diagnostic subreports.
|
| 65 |
+
- This audit compares common scalar metrics and complete pairs. It does not
|
| 66 |
+
certify every report field, input conversion, HTTP behavior, or SemIf's
|
| 67 |
+
separate family-balanced aggregation.
|
| 68 |
+
|
| 69 |
+
Original definitions: [KEV benchmark.py](https://github.com/jaredpalmer/kev/blob/4f8110a3f8620cc3a182ae9a708e4398492c4b1a/kev/benchmark.py),
|
| 70 |
+
[evaluate.py](https://github.com/jaredpalmer/kev/blob/4f8110a3f8620cc3a182ae9a708e4398492c4b1a/kev/evaluate.py),
|
| 71 |
+
[contrastive.py](https://github.com/jaredpalmer/kev/blob/4f8110a3f8620cc3a182ae9a708e4398492c4b1a/kev/contrastive.py).
|
server/benchmarks/PUBLIC_DATASETS.md
ADDED
|
@@ -0,0 +1,99 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Public decision benchmarks beyond the KEV suites
|
| 2 |
+
|
| 3 |
+
Verified on 2026-09-21. JevBench's 231 public decisions are integrated into the preparation tool and default H200 matrix. The other entries below are research candidates, not automatically downloaded.
|
| 4 |
+
|
| 5 |
+
The primary evaluation combines KEV and **JevBench's 231 public decisions** for explicit rule following and probability quality. **Typed Decisions' 400 test cases / 2,000 questions** is a secondary shared-state diagnostic because its labels measure teacher agreement. Image support is not a selection requirement. Additional objective intent/OOS and evidence-grounded candidates are assessed in [DECISION_BENCHMARK_SELECTION.md](DECISION_BENCHMARK_SELECTION.md). Keep each suite's results separate.
|
| 6 |
+
|
| 7 |
+
## Official TypeSafe data versus community benchmarks
|
| 8 |
+
|
| 9 |
+
No downloadable, labeled benchmark released by TypeSafe itself was located in its [official documentation](https://docs.typesafe.ai/introduction), [announcement](https://typesafe.ai/blog/introducing-system-one-models-and-jev), or [public GitHub organization](https://github.com/typesafe-ai). This is a search finding, not proof that no such data exists. TypeSafe publishes an [LLM comparison adapter](https://github.com/typesafe-ai/system-one-adapter-python), but an adapter is not an evaluation dataset.
|
| 10 |
+
|
| 11 |
+
All Jev-specific datasets below are independent community work. A repository name containing `Jev` or `TypeSafe` does not make it official. The community [TypeSafeAI playground](https://github.com/TypeSafeAI/typesafe-playground) explicitly says it is independent and its examples are not validated accuracy benchmarks.
|
| 12 |
+
|
| 13 |
+
KEV's current [4B model card](https://github.com/jaredpalmer/kev/blob/4f8110a3f8620cc3a182ae9a708e4398492c4b1a/docs/model-cards/kev-4b.md) already reports:
|
| 14 |
+
|
| 15 |
+
- `decision-v7`, `transfer-v4`, and newer `transfer-v9` evaluations;
|
| 16 |
+
- SemIf's 144 authored examples;
|
| 17 |
+
- scienthoon's 900 support tickets;
|
| 18 |
+
- ekzhang's 1,000-question MMLU-Pro sample.
|
| 19 |
+
|
| 20 |
+
SemIf, scienthoon, and that MMLU-Pro sample are therefore **external to KEV's original training suite, but not additional benchmarks that KEV has never run**. The referenced KEV suites are text based; they do not measure the vision encoder. “Never trained” in KEV's terminology concerns its own fine-tuning sources and is not evidence that a foundation model or Jev never saw the data.
|
| 21 |
+
|
| 22 |
+
## 1. JevBench public subset — first addition
|
| 23 |
+
|
| 24 |
+
[JevBench](https://github.com/fstandhartinger/jevbench) covers routing, policy checks, intent, enum extraction, ordinal severity, answer judging, and harder reasoning over supplied state. It explicitly disclaims TypeSafe affiliation. The full published evaluation has 534 decisions, while the downloadable public subset has **231**. Use a separate public-subset result rather than claiming reproduction of the full leaderboard.
|
| 25 |
+
|
| 26 |
+
Pinned repository: [`fd51755eb0c0b546ca206d764faf3302feca913e`](https://github.com/fstandhartinger/jevbench/tree/fd51755eb0c0b546ca206d764faf3302feca913e). Counts below were checked by parsing the pinned JSONL bytes.
|
| 27 |
+
|
| 28 |
+
| Download | Decisions | Choice | Noul | Score | License |
|
| 29 |
+
|---|---:|---:|---:|---:|---|
|
| 30 |
+
| [original.jsonl](https://raw.githubusercontent.com/fstandhartinger/jevbench/fd51755eb0c0b546ca206d764faf3302feca913e/datasets/public/original.jsonl) | 72 | 36 | 24 | 12 | MIT |
|
| 31 |
+
| [easy.jsonl](https://raw.githubusercontent.com/fstandhartinger/jevbench/fd51755eb0c0b546ca206d764faf3302feca913e/datasets/public/easy.jsonl) | 48 | 36 | 12 | 0 | MIT |
|
| 32 |
+
| [hard.jsonl](https://raw.githubusercontent.com/fstandhartinger/jevbench/fd51755eb0c0b546ca206d764faf3302feca913e/datasets/public/hard.jsonl) | 111 | 67 | 38 | 6 | MIT |
|
| 33 |
+
| Total | 231 | 139 | 74 | 18 | |
|
| 34 |
+
|
| 35 |
+
SHA-256 checksums, in the same order:
|
| 36 |
+
|
| 37 |
+
```text
|
| 38 |
+
5c2414edb3006b8bfcb70fda433f0f9ca015759433849f8d3104328a1f7c4180
|
| 39 |
+
231df3c2c8e88a1a8c137ebe85de96ba70fabd330849098ac7b3c52c70b7172b
|
| 40 |
+
89e9e6becb33ed88c1de7d42dcc87531b2fb64cfaef4e1986faf7c37b3f80ebb
|
| 41 |
+
```
|
| 42 |
+
|
| 43 |
+
Each row has `state`, one `question`, ordered `labels`, `expected`, `id`, `group`, and `provenance`. Send only `state` and `question`. Noul gold is `"no"`/`"yes"`; Score gold is an integer rubric index. The hard tier's `provenance` can contain the answer rationale and `gold_probs`: neither belongs in model input. Some hard cases have exact reference probabilities, unlike ordinary hard-label classification. The [hard-tier protocol](https://github.com/fstandhartinger/jevbench/blob/fd51755eb0c0b546ca206d764faf3302feca913e/datasets/HARD-TIER.md) describes synthetic authoring, cross-model review, and the freeze.
|
| 44 |
+
|
| 45 |
+
The 303 unavailable decisions comprise 157 deliberately private items and 146 imported items whose task text the maintainers do not redistribute. Hashes and aggregate results do not reconstruct those inputs or grant redistribution rights. See [third-party terms](https://github.com/fstandhartinger/jevbench/blob/fd51755eb0c0b546ca206d764faf3302feca913e/THIRD-PARTY.md) and the [license](https://github.com/fstandhartinger/jevbench/blob/fd51755eb0c0b546ca206d764faf3302feca913e/LICENSE).
|
| 46 |
+
|
| 47 |
+
Implementation notes: retain paraphrase `group` identifiers for grouped uncertainty estimates. Label our adapter's output as **token-logprob-derived**, since JevBench's published ordinary-LLM adapter uses verbalized probabilities. Its leaderboard also applies assumed latency adjustments to some endpoints; our H200 results should report actual local measurements without adopting those adjustments.
|
| 48 |
+
|
| 49 |
+
## 2. LocalLLaMA/typed-decisions — shared-state and soft labels
|
| 50 |
+
|
| 51 |
+
The [dataset card](https://huggingface.co/datasets/LocalLLaMA/typed-decisions/blob/ea9306458d6e9563628369a3d1e72e362fb381d2/README.md) specifies four workflows: agent traces, customer service, invoices, and security incidents. Each has 300 train and 100 test cases, with five Choice/Noul/Score questions per case. Evaluate **400 test cases / 2,000 questions**. The `all` configuration repeats the four workflow configurations; do not concatenate both.
|
| 52 |
+
|
| 53 |
+
- Dataset revision: `ea9306458d6e9563628369a3d1e72e362fb381d2`.
|
| 54 |
+
- [Combined test Parquet](https://huggingface.co/datasets/LocalLLaMA/typed-decisions/resolve/ea9306458d6e9563628369a3d1e72e362fb381d2/all/test-00000-of-00001.parquet), SHA-256 `4f294f218ea1da27f3efef936359389c62ea4d3973a41457732990f1d31b647c`.
|
| 55 |
+
- License: Apache-2.0, as declared in the pinned card.
|
| 56 |
+
- Parse JSON-string columns `state`, `questions`, and `gold`. Only the first two are model input; `factors` reveals the generating variables.
|
| 57 |
+
|
| 58 |
+
Gold averages three teacher-model probability samples. Consequently, this measures **agreement with a synthetic teacher**, not independently established correctness. Report hard-label agreement separately from soft-target Brier/KL and Score MAE. The teacher can be wrong, so a stronger model can receive a lower score. Its published Jev result is measured on this test set; its specialist baselines were trained on the corresponding training workflows.
|
| 59 |
+
|
| 60 |
+
## 3. NPC addressee benchmark — application behavior
|
| 61 |
+
|
| 62 |
+
[wondertwins/jev-benchmark](https://github.com/wondertwins/jev-benchmark/tree/1c2509ac7d5508b8a7ce00ae4df6d7652d05de8b) provides **79 hand-labeled utterances**, each with clean, punctuation-free STT, and misheard-name variants. That is 237 case variants, not 237 independent utterances. Its strict addressee metrics exclude four explicitly ambiguous originals. Questions combine per-NPC Noul, primary-addressee Choice, intent, and urgency Score.
|
| 63 |
+
|
| 64 |
+
- Pin: `1c2509ac7d5508b8a7ce00ae4df6d7652d05de8b`.
|
| 65 |
+
- Fixtures: [npcaddress/dataset.py](https://github.com/wondertwins/jev-benchmark/blob/1c2509ac7d5508b8a7ce00ae4df6d7652d05de8b/npcaddress/dataset.py).
|
| 66 |
+
- Request construction: [npcaddress/questions.py](https://github.com/wondertwins/jev-benchmark/blob/1c2509ac7d5508b8a7ce00ae4df6d7652d05de8b/npcaddress/questions.py).
|
| 67 |
+
- [MIT license](https://github.com/wondertwins/jev-benchmark/blob/1c2509ac7d5508b8a7ce00ae4df6d7652d05de8b/LICENSE).
|
| 68 |
+
|
| 69 |
+
The same repository contains 30 chess positions and 25 mate-in-one puzzles, with cached Stockfish references. Treat chess as a separate reasoning stress test. Keep raw-board, code-enriched, and filtered hybrid variants distinct. These inputs are text/structured state, not screenshots.
|
| 70 |
+
|
| 71 |
+
## 4. Prompt-injection decisions — domain-specific addition
|
| 72 |
+
|
| 73 |
+
[jev-sec-bench](https://github.com/Gaurav-Gosain/jev-sec-bench/tree/fdb16b94d37535db9bad77f8ef0faa971bd7d69a) evaluated all **662** public `deepset/prompt-injections` messages. The original data has 546 train and 116 test rows. Reproducing the 662-row community experiment requires explicitly naming the combined split; it must not be reported as 662 held-out test examples. Its policy context matters: the source labels concern a news assistant, not an unrestricted chatbot.
|
| 74 |
+
|
| 75 |
+
[Dataset and Apache-2.0 declaration](https://huggingface.co/datasets/deepset/prompt-injections/tree/4f61ecb038e9c3fb77e21034b22511b523772cdd), pinned revision `4f61ecb038e9c3fb77e21034b22511b523772cdd`. Schema: `text`, binary `label`; map to Noul using the documented deployment context. Report F1, false positives/negatives, ROC-AUC, Brier, and latency. This is an application-specific classification benchmark, not proof of general guardrail security.
|
| 76 |
+
|
| 77 |
+
## Lower-priority or overlapping sources
|
| 78 |
+
|
| 79 |
+
[AbdelStark/jev-benchmarks](https://github.com/AbdelStark/jev-benchmarks/tree/0d610cc53e79bcbec691312b0c4adb4a0e371642) has a reproducible 300-example BTZSC pilot: AG News, Banking77, and Emotion. Those source families already occur in KEV training/evaluation, so the new harness is useful, but the data source is not independent of KEV's source selection. It pins BTZSC to `fef2a2ac62b69c58670047dddf045c53d7c3cb5e`; its Banking configuration has 72 available labels and excludes rows with no positive candidate. The harness is Apache-2.0; underlying datasets retain their own terms.
|
| 80 |
+
|
| 81 |
+
Small scenario collections such as [souvikr/jev-test](https://github.com/souvikr/jev-test) are useful smoke tests, but 17 cases / 23 checks are too small and easy to serve as the main accuracy benchmark. Documentation-derived instruction corpora and unlabeled live Jev outputs are not independent gold evaluation data.
|
| 82 |
+
|
| 83 |
+
## Future vision additions — general benchmarks, not Jev releases
|
| 84 |
+
|
| 85 |
+
### POPE: object-presence Noul
|
| 86 |
+
|
| 87 |
+
The [original POPE project](https://github.com/AoiDragon/POPE/tree/08d957b917e5a378a2f99d35b6293c536a66298b) supplies image filenames, questions, and yes/no labels. At the pinned revision, each of [random](https://raw.githubusercontent.com/AoiDragon/POPE/08d957b917e5a378a2f99d35b6293c536a66298b/output/coco/coco_pope_random.json), [popular](https://raw.githubusercontent.com/AoiDragon/POPE/08d957b917e5a378a2f99d35b6293c536a66298b/output/coco/coco_pope_popular.json), and [adversarial](https://raw.githubusercontent.com/AoiDragon/POPE/08d957b917e5a378a2f99d35b6293c536a66298b/output/coco/coco_pope_adversarial.json) contains **3,000 questions over 500 images**; counts were checked directly. These are related sampling variants, not three independent image datasets. Schema: `question_id`, `image`, `text`, `label`.
|
| 88 |
+
|
| 89 |
+
Use Noul for “is this object present?” and report each variant separately. The [repository license](https://github.com/AoiDragon/POPE/blob/08d957b917e5a378a2f99d35b6293c536a66298b/LICENSE) is MIT; COCO image rights remain separate. Obtain the named COCO 2014 images according to the [COCO terms](https://cocodataset.org/#termsofuse). Start with a deterministic image-grouped subset before a full run.
|
| 90 |
+
|
| 91 |
+
### VSR: spatial-relation Noul
|
| 92 |
+
|
| 93 |
+
The original [VSR benchmark](https://github.com/cambridgeltl/visual-spatial-reasoning/tree/b27a0af0ee1462d2b6b92c8c83e869d9254a241a) asks whether a caption describes the spatial relationship in an image. Recommend the [random test file](https://huggingface.co/datasets/cambridgeltl/vsr_random/resolve/b2053328fafdd018ff56cf1dfa9643caaa4e69b8/test.jsonl): **2,195** image/caption pairs, SHA-256 `8ade82a0b93ac9dc1e53f6cf1f11e9d5536776a3715102b6b4d27d4f81d551cc`. Its [official dataset card](https://huggingface.co/datasets/cambridgeltl/vsr_random/tree/b2053328fafdd018ff56cf1dfa9643caaa4e69b8) declares CC-BY-4.0; the implementation repository is Apache-2.0, and COCO image terms remain separate.
|
| 94 |
+
|
| 95 |
+
Schema includes `image`, `image_link`, `caption`, binary `label`, and `relation`. Fetch images using the original project's [image instructions](https://github.com/cambridgeltl/visual-spatial-reasoning/blob/b27a0af0ee1462d2b6b92c8c83e869d9254a241a/data/README.md); only image and caption belong in input. Score by relation and overall.
|
| 96 |
+
|
| 97 |
+
Version pitfall: the repository README's zero-shot table says 616 test examples, but the pinned GitHub and [HF zero-shot](https://huggingface.co/datasets/cambridgeltl/vsr_zeroshot/tree/148b3777ceef4a1bfe46377614980836db7d12f2) `test.jsonl` both contain **1,222** rows, SHA-256 `914d9156723b912d8f794b179248af865a68c5f81c9d8e3a21d007778cacaad4`. Do not copy the README count into a report for those files.
|
| 98 |
+
|
| 99 |
+
For vision timing, record image dimensions, processed image-token counts, resize policy, and cache condition. Measure the same fixed examples on every model. A no-image ablation tests image dependence but does not make the text-only task equally answerable. Keep prompt tuning, temperature calibration, and final test evaluation separate; public availability does not establish absence from any model's pretraining.
|
server/benchmarks/README.md
ADDED
|
@@ -0,0 +1,144 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# H200 한 장에서 학습 전 판단 성능 비교
|
| 2 |
+
|
| 3 |
+
추가 학습·LoRA·확률 보정 없이 공식 모델을 실행해 KEV의 공개 평가 문항을 평가합니다. 모델과 SGLang은 서버에서 실행하고, 이 저장소의 어댑터와 벤치 클라이언트는 별도 Python 환경을 사용합니다. 기본 엔진을 수정하지 않습니다.
|
| 4 |
+
|
| 5 |
+
현재 상태: **H200 실측 완료.** 네 모델 각각 4,006개 요청·4,852개 질문을 오류 없이 처리했습니다. [결과 보고서](../../outputs/h200-baseline-benchmark-2026-09-21/RESULTS.md)에서 정확도와 HTTP 지연을 확인할 수 있습니다. 제공된 모델은 추가 학습 전 공식 지시학습 배포본이며, 사전학습 전용 `*-Base` 모델을 뜻하지 않습니다.
|
| 6 |
+
|
| 7 |
+
2026-09-21 로컬 검증: pytest 97개와 하위 사례 74개, H200 실행 도구 CPU 테스트 12개 통과. KEV 5종과 JevBench 공개 3종의 총 4,006개 요청·4,852개 질문을 준비했습니다. 합산 숫자는 실행 작업량이며 세트 간 중복을 제거한 독립 평가 문항 수가 아닙니다.
|
| 8 |
+
|
| 9 |
+
## 모델과 비교 조건
|
| 10 |
+
|
| 11 |
+
| 프로필 | 공식 모델 | 기본 정밀도 |
|
| 12 |
+
|---|---|---|
|
| 13 |
+
| `ministral3-3b-bf16` | `mistralai/Ministral-3-3B-Instruct-2512-BF16` | BF16 |
|
| 14 |
+
| `qwen35-4b-bf16` | `Qwen/Qwen3.5-4B` | BF16 |
|
| 15 |
+
| `qwen36-27b-bf16` | `Qwen/Qwen3.6-27B` | BF16 |
|
| 16 |
+
| `qwen36-35b-a3b-bf16` | `Qwen/Qwen3.6-35B-A3B` | BF16 |
|
| 17 |
+
|
| 18 |
+
원래 다운로드했던 Ministral의 공식 FP8 배포본은 `ministral3-3b-fp8` 선택형 프로필입니다. 기본 비교는 BF16으로 맞추며, FP8 결과는 별도 행으로 비교합니다. 모델 리비전, 엔진 리비전과 실행 옵션은 [h200/models.json](h200/models.json)에 고정했습니다. H200 1장, TP=1, 컨텍스트 8,192, 입력 절단 없음, speculative decoding/MTP 없음, 비전 인코더 유지 조건입니다. 최대 동시 요청은 32이며 기본 클라이언트 동시성은 1입니다.
|
| 19 |
+
|
| 20 |
+
공식 지시학습 모델의 chat template에 `enable_thinking=False`를 전달하고, 선택지 라벨의 다음 토큰 확률을 읽습니다. 문장을 생성하지 않으며 `output_tokens=0`을 검증합니다. 질문마다 별도 프리필을 실행하므로 여러 질문을 하나의 특수 pointer head로 처리하는 KEV와 실행 구조는 다릅니다. 문항과 지표를 맞춘 **학습 전 어댑터 기준선**이며 KEV의 학습 결과나 기존 기본 모델 probe의 프롬프트를 그대로 복제한 실험은 아닙니다.
|
| 21 |
+
|
| 22 |
+
Ministral은 SGLang 0.5.20의 native chat 문자열 재인코딩 오류를 피하기 위해
|
| 23 |
+
어댑터에서 공식 토크나이저의 `tokenize=True` 결과를 전달합니다. 전체 4,852문항의
|
| 24 |
+
토큰 경계와 최대 입력 길이(4,037토큰)를 CPU에서 확인했습니다. Qwen은 엔진의
|
| 25 |
+
HTTP 토큰화 경로를 사용합니다. 따라서 모델 간 HTTP 지연에는 토큰화 위치와
|
| 26 |
+
호출 횟수 차이가 포함되며, 순수 GPU 프리필 시간 비교가 아닙니다.
|
| 27 |
+
초기 Ministral HTTP 점검은 `invalidated.json`으로 무효 표시하며 비교에서 제외합니다.
|
| 28 |
+
|
| 29 |
+
## 준비된 KEV 데이터
|
| 30 |
+
|
| 31 |
+
원본 커밋은 [`4f8110a3f8620cc3a182ae9a708e4398492c4b1a`](https://github.com/jaredpalmer/kev/tree/4f8110a3f8620cc3a182ae9a708e4398492c4b1a)입니다. 원본 manifest와 JSONL의 SHA256을 검증하고, 정답·출처·변형 정보를 모델 요청에서 제거합니다. 모든 질문을 유지하며 Banking77의 78개 선택지 변형도 포함합니다.
|
| 32 |
+
|
| 33 |
+
| 개발 세트 | 요청 수 | 전체 질문 | 기본 정확도에 포함되는 질문 |
|
| 34 |
+
|---|---:|---:|---:|
|
| 35 |
+
| `decision-v7` | 1,204 | 1,468 | 1,264 |
|
| 36 |
+
| `transfer-v4` | 764 | 764 | 656 |
|
| 37 |
+
| `transfer-v9` | 1,264 | 1,264 | 1,046 |
|
| 38 |
+
| `semif-v1` — 선택형 | 252 | 252 | 144 |
|
| 39 |
+
| `scienthoon-v1` — 선택형 | 291 | 873 | 873, 작업별 구분 필요 |
|
| 40 |
+
|
| 41 |
+
`decision-v7`은 KEV 학습에 사용된 출처의 평가 문항, `transfer-v4`는 KEV 미세조정에서 사용하지 않은 출처·정책 구조입니다. 이는 원래 Qwen/Mistral 사전학습에서 보지 않았다는 보장은 아닙니다. `transfer-v9`는 v4에 MMLU-Pro·긴 문맥·근거 제거 문항을 더한 것이므로 두 결과를 독립 데이터처럼 합산하지 않습니다.
|
| 42 |
+
|
| 43 |
+
모델카드의 정확도와 비교할 값은 `report.json`의 `suites.<suite>/development.clean.acc`입니다. 순서 변형과 none-of-the-above 문항은 `variants`, 조건 반전은 `paired_flip`, 선택지 순서 영향은 `permutation`에 기록합니다. v9의 근거 제거 110문항은 정확도에서 제외하고 `unknowable`의 과신 비율로 평가합니다.
|
| 44 |
+
|
| 45 |
+
SemIf는 원래 가족별 balanced accuracy를 사용하므로 여기의 clean micro accuracy를 원래 지표와 혼동하지 않습니다. Scienthoon의 원본 900개 질문 행은 KEV 변환 과정에서 291개 고유 상태·873개 질문이 됐습니다. priority 라벨은 입력에 없는 조직 규칙에 의존하므로 queue·angry와 분리해 `tasks`를 확인합니다. 두 외부 세트도 KEV가 이미 평가한 데이터���니다.
|
| 46 |
+
|
| 47 |
+
## JevBench 공개 데이터 — 기본 실행에 추가
|
| 48 |
+
|
| 49 |
+
[JevBench](https://github.com/fstandhartinger/jevbench/tree/fd51755eb0c0b546ca206d764faf3302feca913e)는 JEV 계열 판단 모델용 커뮤니티 벤치입니다. TypeSafe 공식 벤치는 아닙니다. 전체 결과표의 534문항 중 실제 공개된 231문항을 커밋과 SHA256으로 고정했습니다.
|
| 50 |
+
|
| 51 |
+
| 공개 세트 | 문항 | 주요 평가 |
|
| 52 |
+
|---|---:|---|
|
| 53 |
+
| `jevbench-original` | 72 | 정책, 라우팅, 의도, 등급, 추출, 답변 적합성 |
|
| 54 |
+
| `jevbench-easy` | 48 | 명확한 사실·의도·도구 선택 |
|
| 55 |
+
| `jevbench-hard` | 111 | 복합 규칙, 시간·수량, 확률, 모호성, 함정, 답변 판단 |
|
| 56 |
+
|
| 57 |
+
정답과 해설·생성 근거는 모델 입력에 넣지 않습니다. 데이터는 `public.jsonl`로 저장하며 테스트나 비공개 문항으로 부르지 않습니다. 확률을 정확히 계산할 수 있는 hard 10문항은 `reference_distribution`에서 확률 분포 오차를 별도로 측정합니다. 최고 확률 동률은 JevBench 원본처럼 라벨 사전순으로 처리합니다. 기존 KEV 지표와 공개 문항별 지표를 기록하며, JevBench의 전체 534문항 종합 점수나 네트워크 지연 보정 점수는 재현했다고 주장하지 않습니다. 확률 출력 방식도 `token-logprob-derived`로 구분합니다.
|
| 58 |
+
|
| 59 |
+
평가 선정 기준은 **정책·분류·근거 부족·다중 질문 판단과 확률 품질**이며 비전 여부는 조건이 아닙니다. 기본 준비 검사는 텍스트만 사용하고, `--probe-vision`을 추가했을 때만 이미지도 점검합니다. Typed Decisions는 작은 교사 모델과의 일치도를 측정하므로 보조 후보입니다. 추가 후보와 선정 근거는 [PUBLIC_DATASETS.md](PUBLIC_DATASETS.md), [DECISION_BENCHMARK_SELECTION.md](DECISION_BENCHMARK_SELECTION.md)에 있습니다.
|
| 60 |
+
|
| 61 |
+
## H200 서버에서 실행
|
| 62 |
+
|
| 63 |
+
이 저장소를 서버로 옮긴 뒤 저장소 루트에서 실행합니다. 모델 가중치는 실행 시 Hugging Face에서 받으며, 네 모델을 동시에 GPU에 올리지 않습니다. 전체 BF16 체크포인트 캐시를 위해 충분한 로컬 디스크 공간을 확보합니다. 엔진 설치 조건과 CUDA 13 환경은 [h200/README.md](h200/README.md)를 따릅니다.
|
| 64 |
+
|
| 65 |
+
어댑터 환경:
|
| 66 |
+
|
| 67 |
+
```bash
|
| 68 |
+
python3.12 -m venv .venv
|
| 69 |
+
.venv/bin/python -m pip install -e '.[dev,native-tokenizer]'
|
| 70 |
+
```
|
| 71 |
+
|
| 72 |
+
평가 데이터만 다운로드합니다. 함께 전달된 `benchmarks/data`가 있다면 이 단계는 다시 실행해도 동일 파일인지 확인합니다.
|
| 73 |
+
|
| 74 |
+
```bash
|
| 75 |
+
.venv/bin/python -m jev_adapter.benchmarks.prepare \
|
| 76 |
+
--output benchmarks/data \
|
| 77 |
+
--suite decision-v7 --suite transfer-v4 --suite transfer-v9 \
|
| 78 |
+
--suite jevbench-original --suite jevbench-easy --suite jevbench-hard \
|
| 79 |
+
--suite semif-v1 --suite scienthoon-v1
|
| 80 |
+
```
|
| 81 |
+
|
| 82 |
+
오프라인에서는 같은 커밋의 KEV 체크아웃을 `--source-root /path/to/kev`, JevBench 체크아웃을 `--jevbench-source-root /path/to/jevbench`로 지정할 수 있습니다. 학습 데이터는 다운로드하지 않습니다.
|
| 83 |
+
|
| 84 |
+
먼저 네 모델의 실행 명령만 확인합니다. `--execute`가 없으면 모델을 받거나 실행하지 않습니다.
|
| 85 |
+
|
| 86 |
+
```bash
|
| 87 |
+
.venv/bin/python benchmarks/run_matrix.py \
|
| 88 |
+
--engine-python .venv-sglang/bin/python \
|
| 89 |
+
--output benchmarks/results/preview --include-external
|
| 90 |
+
```
|
| 91 |
+
|
| 92 |
+
서버 통합 점검은 소량 문항으로 진행합니다. 이 결과는 정확도 비교용으로 쓰지 않습니다.
|
| 93 |
+
|
| 94 |
+
```bash
|
| 95 |
+
.venv/bin/python benchmarks/run_matrix.py \
|
| 96 |
+
--engine-python .venv-sglang/bin/python \
|
| 97 |
+
--output benchmarks/results/smoke \
|
| 98 |
+
--limit 8 --warmup 2 --execute
|
| 99 |
+
```
|
| 100 |
+
|
| 101 |
+
사전 점검이 통과하면 네 모델을 순서대로 평가합니다. 각 모델의 텍스트 입력, 토큰 경계, 0토큰 출력, 모델 리비전과 캐시 설정을 먼저 검증하고, 모델마다 어댑터를 다시 시작합니다. 기존 프로세스가 포트를 사용하면 중단합니다. JevBench만 실행할 때는 `--suite jevbench-original --suite jevbench-easy --suite jevbench-hard`를 지정하세요.
|
| 102 |
+
|
| 103 |
+
```bash
|
| 104 |
+
.venv/bin/python benchmarks/run_matrix.py \
|
| 105 |
+
--engine-python .venv-sglang/bin/python \
|
| 106 |
+
--output benchmarks/results/bf16-c1 \
|
| 107 |
+
--include-external --execute
|
| 108 |
+
```
|
| 109 |
+
|
| 110 |
+
동시성 8에서의 처리량도 보려면 별도 출력 폴더로 실행합니다. 동시성 1의 지연시간과 구분해서 해석합니다.
|
| 111 |
+
|
| 112 |
+
```bash
|
| 113 |
+
.venv/bin/python benchmarks/run_matrix.py \
|
| 114 |
+
--engine-python .venv-sglang/bin/python \
|
| 115 |
+
--output benchmarks/results/bf16-c8 \
|
| 116 |
+
--concurrency 8 --include-external --execute
|
| 117 |
+
```
|
| 118 |
+
|
| 119 |
+
`--profile qwen36-35b-a3b-bf16`처럼 모델 하나만 선택할 수 있습니다. `--repeats 3`은 지연 측정을 반복하고 정확도는 첫 반복만 사용합니다. 모든 결과 디렉터리는 새 경로여야 하며 기존 결과를 덮어쓰지 않습니다.
|
| 120 |
+
|
| 121 |
+
## 결과와 측정 범위
|
| 122 |
+
|
| 123 |
+
각 모델 아래 `engine/launch.json`, `engine/preflight.json`, `engine/engine.log`, `c1/manifest.json`, `c1/predictions.jsonl`, `c1/report.json`이 생성됩니다. 전체 실행이 성공하면 `comparison.csv`에 모델·세트별 결과를 모읍니다.
|
| 124 |
+
|
| 125 |
+
- 정확도: clean 질문별 micro accuracy, 출처·작업별 정확도, 작업별 macro accuracy.
|
| 126 |
+
- 확률 품질: multiclass Brier, NLL, 10-bin ECE, 최대 확률 0.9 이상 오답 비율, 경험적 오류 예산별 coverage. 어댑터의 entropy confidence 대신 **최대 선택지 확률**로 ECE를 계산합니다.
|
| 127 |
+
- 순서·정책 반전: 라벨 정렬 후 flip rate, 최소 대조쌍 양쪽 정답률. 부분 집합의 누락된 짝은 별도 집계합니다.
|
| 128 |
+
- 시간: 요청별 HTTP 왕복 p50/p95/p99, 어댑터 내부 벽시계 시간, 전체 요청·질문 처리량, 입력 토큰 수. 모델 로딩과 20회 준비 요청은 제외합니다.
|
| 129 |
+
|
| 130 |
+
지연시간에는 HTTP, 토큰화, 엔진 스케줄링과 프리필이 포함됩니다. **순수 GPU 커널 시간이나 프리필 FLOPS 측정값이 아닙니다.** 질문 수와 입력 토큰 수도 함께 보세요. 어댑터는 질문마다 엔진 HTTP 호출 4회를 사용합니다. 정확도 평가를 위해 지정한 라벨 logprob를 반환받는 비용도 포함됩니다.
|
| 131 |
+
|
| 132 |
+
기본 실행은 radix cache와 이미지 전처리·비전 특징 재사용을 끕니다. 반복 입력의 캐시 효과가 섞이지 않는 조건입니다. 반대로 실제 서비스의 캐시 효과를 측정하려면 별도 엔진 설정과 `--cache-mode server-default`의 개별 runner 실행으로 기록해야 합니다.
|
| 133 |
+
|
| 134 |
+
오류·시간 초과·확률 누락·출력 토큰 발생 시 재시도나 문항 제외로 성공률을 높이지 않습니다. 실패한 실행은 `status=failed`로 저장하고 기본 정확도를 출력하지 않습니다. 데이터 SHA256·부분 집합 여부·코드 SHA256·실행 옵션·엔진/모델 정보를 기록합니다. 실패한 실행은 비교 CSV에 포함하지 않습니다.
|
| 135 |
+
|
| 136 |
+
개발 세트로 조건을 정한 뒤 최종 확인에만 test를 사용합니다. test 준비와 개별 실행은 모두 `--allow-test`를 명시해야 합니다. 행렬 실행기는 KEV 개발 세트와 JevBench 공개 세트를 사용합니다. 학습·프롬프트 선택·온도 보정 없이 현재 고정 조건을 그대로 비교하는 것이 이 기준선의 목적입니다.
|
| 137 |
+
|
| 138 |
+
```bash
|
| 139 |
+
.venv/bin/python -m pytest
|
| 140 |
+
.venv/bin/python -m unittest discover -s benchmarks/h200 -p 'test_*.py'
|
| 141 |
+
.venv/bin/ruff check .
|
| 142 |
+
```
|
| 143 |
+
|
| 144 |
+
지표 수식의 원본 대조 결과와 재현 방법은 [METRIC_VALIDATION.md](METRIC_VALIDATION.md)를 참고하세요. 공개 데이터마다 원래 라이선스가 다르며 변환 manifest가 해당 출처를 보존합니다. JEV 유료 API 호출이나 학습은 실행 과정에 포함되지 않습니다.
|
server/benchmarks/data/jevbench-easy/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
MIT License
|
| 2 |
+
|
| 3 |
+
Copyright (c) 2026 Florian Standhartinger and contributors
|
| 4 |
+
|
| 5 |
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
| 6 |
+
of this software and associated documentation files (the "Software"), to deal
|
| 7 |
+
in the Software without restriction, including without limitation the rights
|
| 8 |
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
| 9 |
+
copies of the Software, and to permit persons to whom the Software is
|
| 10 |
+
furnished to do so, subject to the following conditions:
|
| 11 |
+
|
| 12 |
+
The above copyright notice and this permission notice shall be included in all
|
| 13 |
+
copies or substantial portions of the Software.
|
| 14 |
+
|
| 15 |
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
| 16 |
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
| 17 |
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
| 18 |
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
| 19 |
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
| 20 |
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
| 21 |
+
SOFTWARE.
|
server/benchmarks/data/jevbench-easy/THIRD-PARTY.md
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Third-party projects, data and services
|
| 2 |
+
|
| 3 |
+
MIT (see `LICENSE`) covers this harness and the 72 original public decisions in
|
| 4 |
+
`datasets/public/original.jsonl`. Nothing else in this list is ours to license.
|
| 5 |
+
|
| 6 |
+
## Systems evaluated
|
| 7 |
+
|
| 8 |
+
Each was reached through the interface its author published. We used public
|
| 9 |
+
endpoints as ordinary clients, at one request at a time, and never sent a
|
| 10 |
+
provider's API key to anyone else's endpoint.
|
| 11 |
+
|
| 12 |
+
| System | Author | Source |
|
| 13 |
+
|---|---|---|
|
| 14 |
+
| Jev 1.13.0 | TypeSafe AI | <https://docs.typesafe.ai> |
|
| 15 |
+
| openjev-sglang | ekzhang | <https://github.com/ekzhang/openjev-sglang> |
|
| 16 |
+
| system-one-open | mithalouni | <https://github.com/mithalouni/system-one-open> |
|
| 17 |
+
| open-alternative-jev | IkerMoel | <https://github.com/ikermoel/open-alternative-jev>, <https://huggingface.co/spaces/IkerMoel/open-alternative-jev> |
|
| 18 |
+
| open-jev-deberta-v3-large | Kotoba Labs | <https://huggingface.co/com-kotobalabs/open-jev-deberta-v3-large>, <https://github.com/kotoba-lang/typed-decisions> |
|
| 19 |
+
| GPT-5.6 Luna | OpenAI | <https://platform.openai.com> |
|
| 20 |
+
| Gemini 3.1 Flash-Lite | Google | <https://ai.google.dev> |
|
| 21 |
+
| DeepSeek V4.1 Flash | DeepSeek | <https://api-docs.deepseek.com> |
|
| 22 |
+
| Qwen3.8 27B | Qwen, served by Chutes | <https://chutes.ai> |
|
| 23 |
+
|
| 24 |
+
Model weights, base models and each project's own code keep their own licences.
|
| 25 |
+
A permissive licence on a repository is not a licence for the base model it
|
| 26 |
+
fine-tunes, and we do not restate either.
|
| 27 |
+
|
| 28 |
+
## Wire format
|
| 29 |
+
|
| 30 |
+
The typed-decision request shape (`state`, `questions`, the `noul` / `choice` /
|
| 31 |
+
`score` primitives) is TypeSafe's public HTTP API, documented at
|
| 32 |
+
<https://docs.typesafe.ai/api>. Several of the open rebuilds implement it
|
| 33 |
+
deliberately, which is why one adapter reaches more than one of them. JevBench is
|
| 34 |
+
not affiliated with or endorsed by TypeSafe AI.
|
| 35 |
+
|
| 36 |
+
## Imported decisions
|
| 37 |
+
|
| 38 |
+
146 of the 242 decisions come from our own earlier auto-router experiment
|
| 39 |
+
(<https://github.com/fstandhartinger/auto-model-router>): 78 routing requests and
|
| 40 |
+
68 answer-adequacy judgements whose ground truth is a deterministic grader's
|
| 41 |
+
verdict on a saved answer. The upstream task text is **not** redistributed here,
|
| 42 |
+
because the datasets it was drawn from keep their own terms. `datasets/manifest.json`
|
| 43 |
+
pins the hashes; `scripts/import_router.py` shows exactly what was taken and what
|
| 44 |
+
was excluded.
|
| 45 |
+
|
| 46 |
+
## Held-out decisions
|
| 47 |
+
|
| 48 |
+
24 scenarios are written by us and deliberately unpublished, so the suite cannot
|
| 49 |
+
be trained on in full. Only their whole-split hash and aggregate results appear
|
| 50 |
+
here. They are sent to the services being evaluated, which is not the same thing
|
| 51 |
+
as being public - see the limits section of the README.
|
server/benchmarks/data/jevbench-easy/public.jsonl
ADDED
|
@@ -0,0 +1,48 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"id": "easy-intent-00", "suite": "jevbench-easy", "split": "public", "source": "intent", "variant": "clean", "record": {"state": "Where is my package? I ordered it last week and it still hasn't arrived.", "questions": {"decision": {"type": "choice", "instructions": "Which intent does the user's message express?", "criteria": {"billing_question": "Asks about a charge, invoice or payment", "cancel_order": "Wants to cancel an order", "change_address": "Wants to change the delivery address", "report_damage": "Received an item that is broken or damaged", "track_order": "Wants to know where an order is or when it arrives"}}}}, "expected": {"decision": {"labels": ["track_order", "cancel_order", "change_address", "report_damage", "billing_question"], "target": [1.0, 0.0, 0.0, 0.0, 0.0], "label": 0, "type": "choice", "task": "intent"}}, "metadata": {"id": "easy-intent-00", "source": "intent", "variant": "clean", "group_id": "easy-intent-00", "upstream_group": null, "canonical_labels": ["track_order", "cancel_order", "change_address", "report_damage", "billing_question"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 2 |
+
{"id": "easy-intent-01", "suite": "jevbench-easy", "split": "public", "source": "intent", "variant": "clean", "record": {"state": "Please cancel my order #4471, I don't need it anymore.", "questions": {"decision": {"type": "choice", "instructions": "Which intent does the user's message express?", "criteria": {"billing_question": "Asks about a charge, invoice or payment", "cancel_order": "Wants to cancel an order", "change_address": "Wants to change the delivery address", "report_damage": "Received an item that is broken or damaged", "track_order": "Wants to know where an order is or when it arrives"}}}}, "expected": {"decision": {"labels": ["track_order", "cancel_order", "change_address", "report_damage", "billing_question"], "target": [0.0, 1.0, 0.0, 0.0, 0.0], "label": 1, "type": "choice", "task": "intent"}}, "metadata": {"id": "easy-intent-01", "source": "intent", "variant": "clean", "group_id": "easy-intent-01", "upstream_group": null, "canonical_labels": ["track_order", "cancel_order", "change_address", "report_damage", "billing_question"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 3 |
+
{"id": "easy-intent-02", "suite": "jevbench-easy", "split": "public", "source": "intent", "variant": "clean", "record": {"state": "I moved. Can you ship my order to 12 Elm Street instead of the old address?", "questions": {"decision": {"type": "choice", "instructions": "Which intent does the user's message express?", "criteria": {"billing_question": "Asks about a charge, invoice or payment", "cancel_order": "Wants to cancel an order", "change_address": "Wants to change the delivery address", "report_damage": "Received an item that is broken or damaged", "track_order": "Wants to know where an order is or when it arrives"}}}}, "expected": {"decision": {"labels": ["track_order", "cancel_order", "change_address", "report_damage", "billing_question"], "target": [0.0, 0.0, 1.0, 0.0, 0.0], "label": 2, "type": "choice", "task": "intent"}}, "metadata": {"id": "easy-intent-02", "source": "intent", "variant": "clean", "group_id": "easy-intent-02", "upstream_group": null, "canonical_labels": ["track_order", "cancel_order", "change_address", "report_damage", "billing_question"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 4 |
+
{"id": "easy-intent-03", "suite": "jevbench-easy", "split": "public", "source": "intent", "variant": "clean", "record": {"state": "The mug I received arrived smashed into pieces.", "questions": {"decision": {"type": "choice", "instructions": "Which intent does the user's message express?", "criteria": {"billing_question": "Asks about a charge, invoice or payment", "cancel_order": "Wants to cancel an order", "change_address": "Wants to change the delivery address", "report_damage": "Received an item that is broken or damaged", "track_order": "Wants to know where an order is or when it arrives"}}}}, "expected": {"decision": {"labels": ["track_order", "cancel_order", "change_address", "report_damage", "billing_question"], "target": [0.0, 0.0, 0.0, 1.0, 0.0], "label": 3, "type": "choice", "task": "intent"}}, "metadata": {"id": "easy-intent-03", "source": "intent", "variant": "clean", "group_id": "easy-intent-03", "upstream_group": null, "canonical_labels": ["track_order", "cancel_order", "change_address", "report_damage", "billing_question"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 5 |
+
{"id": "easy-intent-04", "suite": "jevbench-easy", "split": "public", "source": "intent", "variant": "clean", "record": {"state": "Why was I charged twice on my credit card statement for one order?", "questions": {"decision": {"type": "choice", "instructions": "Which intent does the user's message express?", "criteria": {"billing_question": "Asks about a charge, invoice or payment", "cancel_order": "Wants to cancel an order", "change_address": "Wants to change the delivery address", "report_damage": "Received an item that is broken or damaged", "track_order": "Wants to know where an order is or when it arrives"}}}}, "expected": {"decision": {"labels": ["track_order", "cancel_order", "change_address", "report_damage", "billing_question"], "target": [0.0, 0.0, 0.0, 0.0, 1.0], "label": 4, "type": "choice", "task": "intent"}}, "metadata": {"id": "easy-intent-04", "source": "intent", "variant": "clean", "group_id": "easy-intent-04", "upstream_group": null, "canonical_labels": ["track_order", "cancel_order", "change_address", "report_damage", "billing_question"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 6 |
+
{"id": "easy-intent-05", "suite": "jevbench-easy", "split": "public", "source": "intent", "variant": "clean", "record": {"state": "When will my order be delivered? The tracking page shows nothing.", "questions": {"decision": {"type": "choice", "instructions": "Which intent does the user's message express?", "criteria": {"billing_question": "Asks about a charge, invoice or payment", "cancel_order": "Wants to cancel an order", "change_address": "Wants to change the delivery address", "report_damage": "Received an item that is broken or damaged", "track_order": "Wants to know where an order is or when it arrives"}}}}, "expected": {"decision": {"labels": ["track_order", "cancel_order", "change_address", "report_damage", "billing_question"], "target": [1.0, 0.0, 0.0, 0.0, 0.0], "label": 0, "type": "choice", "task": "intent"}}, "metadata": {"id": "easy-intent-05", "source": "intent", "variant": "clean", "group_id": "easy-intent-05", "upstream_group": null, "canonical_labels": ["track_order", "cancel_order", "change_address", "report_damage", "billing_question"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 7 |
+
{"id": "easy-intent-06", "suite": "jevbench-easy", "split": "public", "source": "intent", "variant": "clean", "record": {"state": "Wake me up at 6:30 tomorrow morning.", "questions": {"decision": {"type": "choice", "instructions": "Which intent does the user's message express?", "criteria": {"play_music": "Play a song, album, artist or playlist", "send_message": "Send a text message to someone", "set_alarm": "Set an alarm or wake-up time", "turn_off_lights": "Switch lights off", "weather": "Ask about the weather or forecast"}}}}, "expected": {"decision": {"labels": ["set_alarm", "play_music", "weather", "send_message", "turn_off_lights"], "target": [1.0, 0.0, 0.0, 0.0, 0.0], "label": 0, "type": "choice", "task": "intent"}}, "metadata": {"id": "easy-intent-06", "source": "intent", "variant": "clean", "group_id": "easy-intent-06", "upstream_group": null, "canonical_labels": ["set_alarm", "play_music", "weather", "send_message", "turn_off_lights"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 8 |
+
{"id": "easy-intent-07", "suite": "jevbench-easy", "split": "public", "source": "intent", "variant": "clean", "record": {"state": "Play some Taylor Swift.", "questions": {"decision": {"type": "choice", "instructions": "Which intent does the user's message express?", "criteria": {"play_music": "Play a song, album, artist or playlist", "send_message": "Send a text message to someone", "set_alarm": "Set an alarm or wake-up time", "turn_off_lights": "Switch lights off", "weather": "Ask about the weather or forecast"}}}}, "expected": {"decision": {"labels": ["set_alarm", "play_music", "weather", "send_message", "turn_off_lights"], "target": [0.0, 1.0, 0.0, 0.0, 0.0], "label": 1, "type": "choice", "task": "intent"}}, "metadata": {"id": "easy-intent-07", "source": "intent", "variant": "clean", "group_id": "easy-intent-07", "upstream_group": null, "canonical_labels": ["set_alarm", "play_music", "weather", "send_message", "turn_off_lights"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 9 |
+
{"id": "easy-intent-08", "suite": "jevbench-easy", "split": "public", "source": "intent", "variant": "clean", "record": {"state": "Will it rain in Berlin tomorrow?", "questions": {"decision": {"type": "choice", "instructions": "Which intent does the user's message express?", "criteria": {"play_music": "Play a song, album, artist or playlist", "send_message": "Send a text message to someone", "set_alarm": "Set an alarm or wake-up time", "turn_off_lights": "Switch lights off", "weather": "Ask about the weather or forecast"}}}}, "expected": {"decision": {"labels": ["set_alarm", "play_music", "weather", "send_message", "turn_off_lights"], "target": [0.0, 0.0, 1.0, 0.0, 0.0], "label": 2, "type": "choice", "task": "intent"}}, "metadata": {"id": "easy-intent-08", "source": "intent", "variant": "clean", "group_id": "easy-intent-08", "upstream_group": null, "canonical_labels": ["set_alarm", "play_music", "weather", "send_message", "turn_off_lights"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 10 |
+
{"id": "easy-intent-09", "suite": "jevbench-easy", "split": "public", "source": "intent", "variant": "clean", "record": {"state": "Text Anna that I'll be ten minutes late.", "questions": {"decision": {"type": "choice", "instructions": "Which intent does the user's message express?", "criteria": {"play_music": "Play a song, album, artist or playlist", "send_message": "Send a text message to someone", "set_alarm": "Set an alarm or wake-up time", "turn_off_lights": "Switch lights off", "weather": "Ask about the weather or forecast"}}}}, "expected": {"decision": {"labels": ["set_alarm", "play_music", "weather", "send_message", "turn_off_lights"], "target": [0.0, 0.0, 0.0, 1.0, 0.0], "label": 3, "type": "choice", "task": "intent"}}, "metadata": {"id": "easy-intent-09", "source": "intent", "variant": "clean", "group_id": "easy-intent-09", "upstream_group": null, "canonical_labels": ["set_alarm", "play_music", "weather", "send_message", "turn_off_lights"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 11 |
+
{"id": "easy-intent-10", "suite": "jevbench-easy", "split": "public", "source": "intent", "variant": "clean", "record": {"state": "Turn off the lights in the bedroom.", "questions": {"decision": {"type": "choice", "instructions": "Which intent does the user's message express?", "criteria": {"play_music": "Play a song, album, artist or playlist", "send_message": "Send a text message to someone", "set_alarm": "Set an alarm or wake-up time", "turn_off_lights": "Switch lights off", "weather": "Ask about the weather or forecast"}}}}, "expected": {"decision": {"labels": ["set_alarm", "play_music", "weather", "send_message", "turn_off_lights"], "target": [0.0, 0.0, 0.0, 0.0, 1.0], "label": 4, "type": "choice", "task": "intent"}}, "metadata": {"id": "easy-intent-10", "source": "intent", "variant": "clean", "group_id": "easy-intent-10", "upstream_group": null, "canonical_labels": ["set_alarm", "play_music", "weather", "send_message", "turn_off_lights"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 12 |
+
{"id": "easy-intent-11", "suite": "jevbench-easy", "split": "public", "source": "intent", "variant": "clean", "record": {"state": "Set an alarm for 7 am.", "questions": {"decision": {"type": "choice", "instructions": "Which intent does the user's message express?", "criteria": {"play_music": "Play a song, album, artist or playlist", "send_message": "Send a text message to someone", "set_alarm": "Set an alarm or wake-up time", "turn_off_lights": "Switch lights off", "weather": "Ask about the weather or forecast"}}}}, "expected": {"decision": {"labels": ["set_alarm", "play_music", "weather", "send_message", "turn_off_lights"], "target": [1.0, 0.0, 0.0, 0.0, 0.0], "label": 0, "type": "choice", "task": "intent"}}, "metadata": {"id": "easy-intent-11", "source": "intent", "variant": "clean", "group_id": "easy-intent-11", "upstream_group": null, "canonical_labels": ["set_alarm", "play_music", "weather", "send_message", "turn_off_lights"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 13 |
+
{"id": "easy-fact-00", "suite": "jevbench-easy", "split": "public", "source": "fact", "variant": "clean", "record": {"state": "Order #1182. Status: shipped on 3 September. Carrier: DHL.", "questions": {"decision": {"type": "noul", "instructions": "Has the order been shipped? Answer strictly from the facts stated in the text.", "criteria": {"false": "The text states that this is not so", "true": "The text states that this is so"}}}}, "expected": {"decision": {"labels": ["false", "true"], "target": [0.0, 1.0], "label": 1, "type": "noul", "task": "fact"}}, "metadata": {"id": "easy-fact-00", "source": "fact", "variant": "clean", "group_id": "easy-fact-00", "upstream_group": null, "canonical_labels": ["no", "yes"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 14 |
+
{"id": "easy-fact-01", "suite": "jevbench-easy", "split": "public", "source": "fact", "variant": "clean", "record": {"state": "Order #1183. Status: not yet shipped, waiting for stock.", "questions": {"decision": {"type": "noul", "instructions": "Has the order been shipped? Answer strictly from the facts stated in the text.", "criteria": {"false": "The text states that this is not so", "true": "The text states that this is so"}}}}, "expected": {"decision": {"labels": ["false", "true"], "target": [1.0, 0.0], "label": 0, "type": "noul", "task": "fact"}}, "metadata": {"id": "easy-fact-01", "source": "fact", "variant": "clean", "group_id": "easy-fact-01", "upstream_group": null, "canonical_labels": ["no", "yes"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 15 |
+
{"id": "easy-fact-02", "suite": "jevbench-easy", "split": "public", "source": "fact", "variant": "clean", "record": {"state": "Invoice 2026-044. Amount: 120 EUR. Payment status: paid in full.", "questions": {"decision": {"type": "noul", "instructions": "Is the invoice paid? Answer strictly from the facts stated in the text.", "criteria": {"false": "The text states that this is not so", "true": "The text states that this is so"}}}}, "expected": {"decision": {"labels": ["false", "true"], "target": [0.0, 1.0], "label": 1, "type": "noul", "task": "fact"}}, "metadata": {"id": "easy-fact-02", "source": "fact", "variant": "clean", "group_id": "easy-fact-02", "upstream_group": null, "canonical_labels": ["no", "yes"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 16 |
+
{"id": "easy-fact-03", "suite": "jevbench-easy", "split": "public", "source": "fact", "variant": "clean", "record": {"state": "Invoice 2026-045. Amount: 80 EUR. Payment status: unpaid, overdue since 1 August.", "questions": {"decision": {"type": "noul", "instructions": "Is the invoice paid? Answer strictly from the facts stated in the text.", "criteria": {"false": "The text states that this is not so", "true": "The text states that this is so"}}}}, "expected": {"decision": {"labels": ["false", "true"], "target": [1.0, 0.0], "label": 0, "type": "noul", "task": "fact"}}, "metadata": {"id": "easy-fact-03", "source": "fact", "variant": "clean", "group_id": "easy-fact-03", "upstream_group": null, "canonical_labels": ["no", "yes"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 17 |
+
{"id": "easy-fact-04", "suite": "jevbench-easy", "split": "public", "source": "fact", "variant": "clean", "record": {"state": "Guest profile: Maria Lopez. Allergies: peanuts. Diet: vegetarian.", "questions": {"decision": {"type": "noul", "instructions": "Is the guest allergic to peanuts? Answer strictly from the facts stated in the text.", "criteria": {"false": "The text states that this is not so", "true": "The text states that this is so"}}}}, "expected": {"decision": {"labels": ["false", "true"], "target": [0.0, 1.0], "label": 1, "type": "noul", "task": "fact"}}, "metadata": {"id": "easy-fact-04", "source": "fact", "variant": "clean", "group_id": "easy-fact-04", "upstream_group": null, "canonical_labels": ["no", "yes"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 18 |
+
{"id": "easy-fact-05", "suite": "jevbench-easy", "split": "public", "source": "fact", "variant": "clean", "record": {"state": "Guest profile: Jonas Weber. Allergies: none. Diet: no restrictions.", "questions": {"decision": {"type": "noul", "instructions": "Is the guest allergic to peanuts? Answer strictly from the facts stated in the text.", "criteria": {"false": "The text states that this is not so", "true": "The text states that this is so"}}}}, "expected": {"decision": {"labels": ["false", "true"], "target": [1.0, 0.0], "label": 0, "type": "noul", "task": "fact"}}, "metadata": {"id": "easy-fact-05", "source": "fact", "variant": "clean", "group_id": "easy-fact-05", "upstream_group": null, "canonical_labels": ["no", "yes"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 19 |
+
{"id": "easy-fact-06", "suite": "jevbench-easy", "split": "public", "source": "fact", "variant": "clean", "record": {"state": "Meeting room B: booked from 14:00 to 15:00 by the sales team.", "questions": {"decision": {"type": "noul", "instructions": "Is meeting room B booked from 14:00 to 15:00? Answer strictly from the facts stated in the text.", "criteria": {"false": "The text states that this is not so", "true": "The text states that this is so"}}}}, "expected": {"decision": {"labels": ["false", "true"], "target": [0.0, 1.0], "label": 1, "type": "noul", "task": "fact"}}, "metadata": {"id": "easy-fact-06", "source": "fact", "variant": "clean", "group_id": "easy-fact-06", "upstream_group": null, "canonical_labels": ["no", "yes"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 20 |
+
{"id": "easy-fact-07", "suite": "jevbench-easy", "split": "public", "source": "fact", "variant": "clean", "record": {"state": "Meeting room C: free all afternoon, no bookings.", "questions": {"decision": {"type": "noul", "instructions": "Is meeting room C booked this afternoon? Answer strictly from the facts stated in the text.", "criteria": {"false": "The text states that this is not so", "true": "The text states that this is so"}}}}, "expected": {"decision": {"labels": ["false", "true"], "target": [1.0, 0.0], "label": 0, "type": "noul", "task": "fact"}}, "metadata": {"id": "easy-fact-07", "source": "fact", "variant": "clean", "group_id": "easy-fact-07", "upstream_group": null, "canonical_labels": ["no", "yes"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 21 |
+
{"id": "easy-fact-08", "suite": "jevbench-easy", "split": "public", "source": "fact", "variant": "clean", "record": {"state": "Account settings: two-factor authentication is enabled.", "questions": {"decision": {"type": "noul", "instructions": "Is two-factor authentication enabled? Answer strictly from the facts stated in the text.", "criteria": {"false": "The text states that this is not so", "true": "The text states that this is so"}}}}, "expected": {"decision": {"labels": ["false", "true"], "target": [0.0, 1.0], "label": 1, "type": "noul", "task": "fact"}}, "metadata": {"id": "easy-fact-08", "source": "fact", "variant": "clean", "group_id": "easy-fact-08", "upstream_group": null, "canonical_labels": ["no", "yes"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 22 |
+
{"id": "easy-fact-09", "suite": "jevbench-easy", "split": "public", "source": "fact", "variant": "clean", "record": {"state": "Account settings: two-factor authentication is disabled.", "questions": {"decision": {"type": "noul", "instructions": "Is two-factor authentication enabled? Answer strictly from the facts stated in the text.", "criteria": {"false": "The text states that this is not so", "true": "The text states that this is so"}}}}, "expected": {"decision": {"labels": ["false", "true"], "target": [1.0, 0.0], "label": 0, "type": "noul", "task": "fact"}}, "metadata": {"id": "easy-fact-09", "source": "fact", "variant": "clean", "group_id": "easy-fact-09", "upstream_group": null, "canonical_labels": ["no", "yes"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 23 |
+
{"id": "easy-fact-10", "suite": "jevbench-easy", "split": "public", "source": "fact", "variant": "clean", "record": {"state": "The user wrote: 'Yes, please subscribe me to the newsletter.'", "questions": {"decision": {"type": "noul", "instructions": "Did the user agree to subscribe to the newsletter? Answer strictly from the facts stated in the text.", "criteria": {"false": "The text states that this is not so", "true": "The text states that this is so"}}}}, "expected": {"decision": {"labels": ["false", "true"], "target": [0.0, 1.0], "label": 1, "type": "noul", "task": "fact"}}, "metadata": {"id": "easy-fact-10", "source": "fact", "variant": "clean", "group_id": "easy-fact-10", "upstream_group": null, "canonical_labels": ["no", "yes"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 24 |
+
{"id": "easy-fact-11", "suite": "jevbench-easy", "split": "public", "source": "fact", "variant": "clean", "record": {"state": "The user wrote: 'No thanks, I do not want the newsletter.'", "questions": {"decision": {"type": "noul", "instructions": "Did the user agree to subscribe to the newsletter? Answer strictly from the facts stated in the text.", "criteria": {"false": "The text states that this is not so", "true": "The text states that this is so"}}}}, "expected": {"decision": {"labels": ["false", "true"], "target": [1.0, 0.0], "label": 0, "type": "noul", "task": "fact"}}, "metadata": {"id": "easy-fact-11", "source": "fact", "variant": "clean", "group_id": "easy-fact-11", "upstream_group": null, "canonical_labels": ["no", "yes"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 25 |
+
{"id": "easy-extraction-00", "suite": "jevbench-easy", "split": "public", "source": "extraction", "variant": "clean", "record": {"state": "I paid with PayPal yesterday evening.", "questions": {"decision": {"type": "choice", "instructions": "Which payment method does the customer name?", "criteria": {"bank_transfer": "Bank transfer / wire", "cash": "Cash", "credit_card": "Paid or wants to pay by credit card", "paypal": "PayPal"}}}}, "expected": {"decision": {"labels": ["credit_card", "paypal", "bank_transfer", "cash"], "target": [0.0, 1.0, 0.0, 0.0], "label": 1, "type": "choice", "task": "extraction"}}, "metadata": {"id": "easy-extraction-00", "source": "extraction", "variant": "clean", "group_id": "easy-extraction-00", "upstream_group": null, "canonical_labels": ["credit_card", "paypal", "bank_transfer", "cash"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 26 |
+
{"id": "easy-extraction-01", "suite": "jevbench-easy", "split": "public", "source": "extraction", "variant": "clean", "record": {"state": "I'll pay cash when the courier arrives.", "questions": {"decision": {"type": "choice", "instructions": "Which payment method does the customer name?", "criteria": {"bank_transfer": "Bank transfer / wire", "cash": "Cash", "credit_card": "Paid or wants to pay by credit card", "paypal": "PayPal"}}}}, "expected": {"decision": {"labels": ["credit_card", "paypal", "bank_transfer", "cash"], "target": [0.0, 0.0, 0.0, 1.0], "label": 3, "type": "choice", "task": "extraction"}}, "metadata": {"id": "easy-extraction-01", "source": "extraction", "variant": "clean", "group_id": "easy-extraction-01", "upstream_group": null, "canonical_labels": ["credit_card", "paypal", "bank_transfer", "cash"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 27 |
+
{"id": "easy-extraction-02", "suite": "jevbench-easy", "split": "public", "source": "extraction", "variant": "clean", "record": {"state": "I sent the money by bank transfer on Monday.", "questions": {"decision": {"type": "choice", "instructions": "Which payment method does the customer name?", "criteria": {"bank_transfer": "Bank transfer / wire", "cash": "Cash", "credit_card": "Paid or wants to pay by credit card", "paypal": "PayPal"}}}}, "expected": {"decision": {"labels": ["credit_card", "paypal", "bank_transfer", "cash"], "target": [0.0, 0.0, 1.0, 0.0], "label": 2, "type": "choice", "task": "extraction"}}, "metadata": {"id": "easy-extraction-02", "source": "extraction", "variant": "clean", "group_id": "easy-extraction-02", "upstream_group": null, "canonical_labels": ["credit_card", "paypal", "bank_transfer", "cash"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 28 |
+
{"id": "easy-extraction-03", "suite": "jevbench-easy", "split": "public", "source": "extraction", "variant": "clean", "record": {"state": "Ticket #88 - Priority: HIGH - Printer on floor 2 is jammed.", "questions": {"decision": {"type": "choice", "instructions": "Which priority does the ticket state?", "criteria": {"high": "High priority", "low": "Low priority", "medium": "Medium priority", "urgent": "Urgent / critical"}}}}, "expected": {"decision": {"labels": ["low", "medium", "high", "urgent"], "target": [0.0, 0.0, 1.0, 0.0], "label": 2, "type": "choice", "task": "extraction"}}, "metadata": {"id": "easy-extraction-03", "source": "extraction", "variant": "clean", "group_id": "easy-extraction-03", "upstream_group": null, "canonical_labels": ["low", "medium", "high", "urgent"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 29 |
+
{"id": "easy-extraction-04", "suite": "jevbench-easy", "split": "public", "source": "extraction", "variant": "clean", "record": {"state": "Ticket #89 - Priority: low - Please update my desk phone label.", "questions": {"decision": {"type": "choice", "instructions": "Which priority does the ticket state?", "criteria": {"high": "High priority", "low": "Low priority", "medium": "Medium priority", "urgent": "Urgent / critical"}}}}, "expected": {"decision": {"labels": ["low", "medium", "high", "urgent"], "target": [1.0, 0.0, 0.0, 0.0], "label": 0, "type": "choice", "task": "extraction"}}, "metadata": {"id": "easy-extraction-04", "source": "extraction", "variant": "clean", "group_id": "easy-extraction-04", "upstream_group": null, "canonical_labels": ["low", "medium", "high", "urgent"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 30 |
+
{"id": "easy-extraction-05", "suite": "jevbench-easy", "split": "public", "source": "extraction", "variant": "clean", "record": {"state": "Ticket #90 - Priority: URGENT - Production database is down.", "questions": {"decision": {"type": "choice", "instructions": "Which priority does the ticket state?", "criteria": {"high": "High priority", "low": "Low priority", "medium": "Medium priority", "urgent": "Urgent / critical"}}}}, "expected": {"decision": {"labels": ["low", "medium", "high", "urgent"], "target": [0.0, 0.0, 0.0, 1.0], "label": 3, "type": "choice", "task": "extraction"}}, "metadata": {"id": "easy-extraction-05", "source": "extraction", "variant": "clean", "group_id": "easy-extraction-05", "upstream_group": null, "canonical_labels": ["low", "medium", "high", "urgent"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 31 |
+
{"id": "easy-extraction-06", "suite": "jevbench-easy", "split": "public", "source": "extraction", "variant": "clean", "record": {"state": "Could I get the blue shirt in size M, please?", "questions": {"decision": {"type": "choice", "instructions": "Which shirt size does the customer ask for?", "criteria": {"L": "Large", "M": "Medium", "S": "Small", "XL": "Extra large"}}}}, "expected": {"decision": {"labels": ["S", "M", "L", "XL"], "target": [0.0, 1.0, 0.0, 0.0], "label": 1, "type": "choice", "task": "extraction"}}, "metadata": {"id": "easy-extraction-06", "source": "extraction", "variant": "clean", "group_id": "easy-extraction-06", "upstream_group": null, "canonical_labels": ["S", "M", "L", "XL"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 32 |
+
{"id": "easy-extraction-07", "suite": "jevbench-easy", "split": "public", "source": "extraction", "variant": "clean", "record": {"state": "I need the jacket in extra large (XL).", "questions": {"decision": {"type": "choice", "instructions": "Which shirt size does the customer ask for?", "criteria": {"L": "Large", "M": "Medium", "S": "Small", "XL": "Extra large"}}}}, "expected": {"decision": {"labels": ["S", "M", "L", "XL"], "target": [0.0, 0.0, 0.0, 1.0], "label": 3, "type": "choice", "task": "extraction"}}, "metadata": {"id": "easy-extraction-07", "source": "extraction", "variant": "clean", "group_id": "easy-extraction-07", "upstream_group": null, "canonical_labels": ["S", "M", "L", "XL"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 33 |
+
{"id": "easy-extraction-08", "suite": "jevbench-easy", "split": "public", "source": "extraction", "variant": "clean", "record": {"state": "Let's meet on Thursday at 10.", "questions": {"decision": {"type": "choice", "instructions": "Which day does the person propose for the meeting?", "criteria": {"friday": "Friday", "monday": "Monday", "thursday": "Thursday", "tuesday": "Tuesday", "wednesday": "Wednesday"}}}}, "expected": {"decision": {"labels": ["monday", "tuesday", "wednesday", "thursday", "friday"], "target": [0.0, 0.0, 0.0, 1.0, 0.0], "label": 3, "type": "choice", "task": "extraction"}}, "metadata": {"id": "easy-extraction-08", "source": "extraction", "variant": "clean", "group_id": "easy-extraction-08", "upstream_group": null, "canonical_labels": ["monday", "tuesday", "wednesday", "thursday", "friday"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 34 |
+
{"id": "easy-extraction-09", "suite": "jevbench-easy", "split": "public", "source": "extraction", "variant": "clean", "record": {"state": "Does Monday work for you? I'm free all day.", "questions": {"decision": {"type": "choice", "instructions": "Which day does the person propose for the meeting?", "criteria": {"friday": "Friday", "monday": "Monday", "thursday": "Thursday", "tuesday": "Tuesday", "wednesday": "Wednesday"}}}}, "expected": {"decision": {"labels": ["monday", "tuesday", "wednesday", "thursday", "friday"], "target": [1.0, 0.0, 0.0, 0.0, 0.0], "label": 0, "type": "choice", "task": "extraction"}}, "metadata": {"id": "easy-extraction-09", "source": "extraction", "variant": "clean", "group_id": "easy-extraction-09", "upstream_group": null, "canonical_labels": ["monday", "tuesday", "wednesday", "thursday", "friday"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 35 |
+
{"id": "easy-extraction-10", "suite": "jevbench-easy", "split": "public", "source": "extraction", "variant": "clean", "record": {"state": "How about Friday afternoon for the review?", "questions": {"decision": {"type": "choice", "instructions": "Which day does the person propose for the meeting?", "criteria": {"friday": "Friday", "monday": "Monday", "thursday": "Thursday", "tuesday": "Tuesday", "wednesday": "Wednesday"}}}}, "expected": {"decision": {"labels": ["monday", "tuesday", "wednesday", "thursday", "friday"], "target": [0.0, 0.0, 0.0, 0.0, 1.0], "label": 4, "type": "choice", "task": "extraction"}}, "metadata": {"id": "easy-extraction-10", "source": "extraction", "variant": "clean", "group_id": "easy-extraction-10", "upstream_group": null, "canonical_labels": ["monday", "tuesday", "wednesday", "thursday", "friday"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 36 |
+
{"id": "easy-extraction-11", "suite": "jevbench-easy", "split": "public", "source": "extraction", "variant": "clean", "record": {"state": "Please charge my credit card ending in 4242.", "questions": {"decision": {"type": "choice", "instructions": "Which payment method does the customer name?", "criteria": {"bank_transfer": "Bank transfer / wire", "cash": "Cash", "credit_card": "Paid or wants to pay by credit card", "paypal": "PayPal"}}}}, "expected": {"decision": {"labels": ["credit_card", "paypal", "bank_transfer", "cash"], "target": [1.0, 0.0, 0.0, 0.0], "label": 0, "type": "choice", "task": "extraction"}}, "metadata": {"id": "easy-extraction-11", "source": "extraction", "variant": "clean", "group_id": "easy-extraction-11", "upstream_group": null, "canonical_labels": ["credit_card", "paypal", "bank_transfer", "cash"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 37 |
+
{"id": "easy-tool_selection-00", "suite": "jevbench-easy", "split": "public", "source": "tool_selection", "variant": "clean", "record": {"state": "What's the weather like in Paris right now?", "questions": {"decision": {"type": "choice", "instructions": "Which single tool should be called to handle this request?", "criteria": {"convert_currency": "Convert an amount from one currency to another", "create_calendar_event": "Put an appointment or meeting into the calendar", "get_weather": "Current weather or forecast for a place", "send_email": "Send an email to a recipient", "translate_text": "Translate text into another language"}}}}, "expected": {"decision": {"labels": ["get_weather", "send_email", "create_calendar_event", "convert_currency", "translate_text"], "target": [1.0, 0.0, 0.0, 0.0, 0.0], "label": 0, "type": "choice", "task": "tool_selection"}}, "metadata": {"id": "easy-tool_selection-00", "source": "tool_selection", "variant": "clean", "group_id": "easy-tool_selection-00", "upstream_group": null, "canonical_labels": ["get_weather", "send_email", "create_calendar_event", "convert_currency", "translate_text"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 38 |
+
{"id": "easy-tool_selection-01", "suite": "jevbench-easy", "split": "public", "source": "tool_selection", "variant": "clean", "record": {"state": "Email Sarah the quarterly report and say it's attached.", "questions": {"decision": {"type": "choice", "instructions": "Which single tool should be called to handle this request?", "criteria": {"convert_currency": "Convert an amount from one currency to another", "create_calendar_event": "Put an appointment or meeting into the calendar", "get_weather": "Current weather or forecast for a place", "send_email": "Send an email to a recipient", "translate_text": "Translate text into another language"}}}}, "expected": {"decision": {"labels": ["get_weather", "send_email", "create_calendar_event", "convert_currency", "translate_text"], "target": [0.0, 1.0, 0.0, 0.0, 0.0], "label": 1, "type": "choice", "task": "tool_selection"}}, "metadata": {"id": "easy-tool_selection-01", "source": "tool_selection", "variant": "clean", "group_id": "easy-tool_selection-01", "upstream_group": null, "canonical_labels": ["get_weather", "send_email", "create_calendar_event", "convert_currency", "translate_text"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 39 |
+
{"id": "easy-tool_selection-02", "suite": "jevbench-easy", "split": "public", "source": "tool_selection", "variant": "clean", "record": {"state": "Add a dentist appointment on Friday at 3 pm to my calendar.", "questions": {"decision": {"type": "choice", "instructions": "Which single tool should be called to handle this request?", "criteria": {"convert_currency": "Convert an amount from one currency to another", "create_calendar_event": "Put an appointment or meeting into the calendar", "get_weather": "Current weather or forecast for a place", "send_email": "Send an email to a recipient", "translate_text": "Translate text into another language"}}}}, "expected": {"decision": {"labels": ["get_weather", "send_email", "create_calendar_event", "convert_currency", "translate_text"], "target": [0.0, 0.0, 1.0, 0.0, 0.0], "label": 2, "type": "choice", "task": "tool_selection"}}, "metadata": {"id": "easy-tool_selection-02", "source": "tool_selection", "variant": "clean", "group_id": "easy-tool_selection-02", "upstream_group": null, "canonical_labels": ["get_weather", "send_email", "create_calendar_event", "convert_currency", "translate_text"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 40 |
+
{"id": "easy-tool_selection-03", "suite": "jevbench-easy", "split": "public", "source": "tool_selection", "variant": "clean", "record": {"state": "How much is 250 US dollars in euros?", "questions": {"decision": {"type": "choice", "instructions": "Which single tool should be called to handle this request?", "criteria": {"convert_currency": "Convert an amount from one currency to another", "create_calendar_event": "Put an appointment or meeting into the calendar", "get_weather": "Current weather or forecast for a place", "send_email": "Send an email to a recipient", "translate_text": "Translate text into another language"}}}}, "expected": {"decision": {"labels": ["get_weather", "send_email", "create_calendar_event", "convert_currency", "translate_text"], "target": [0.0, 0.0, 0.0, 1.0, 0.0], "label": 3, "type": "choice", "task": "tool_selection"}}, "metadata": {"id": "easy-tool_selection-03", "source": "tool_selection", "variant": "clean", "group_id": "easy-tool_selection-03", "upstream_group": null, "canonical_labels": ["get_weather", "send_email", "create_calendar_event", "convert_currency", "translate_text"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 41 |
+
{"id": "easy-tool_selection-04", "suite": "jevbench-easy", "split": "public", "source": "tool_selection", "variant": "clean", "record": {"state": "Translate 'good morning' into Japanese.", "questions": {"decision": {"type": "choice", "instructions": "Which single tool should be called to handle this request?", "criteria": {"convert_currency": "Convert an amount from one currency to another", "create_calendar_event": "Put an appointment or meeting into the calendar", "get_weather": "Current weather or forecast for a place", "send_email": "Send an email to a recipient", "translate_text": "Translate text into another language"}}}}, "expected": {"decision": {"labels": ["get_weather", "send_email", "create_calendar_event", "convert_currency", "translate_text"], "target": [0.0, 0.0, 0.0, 0.0, 1.0], "label": 4, "type": "choice", "task": "tool_selection"}}, "metadata": {"id": "easy-tool_selection-04", "source": "tool_selection", "variant": "clean", "group_id": "easy-tool_selection-04", "upstream_group": null, "canonical_labels": ["get_weather", "send_email", "create_calendar_event", "convert_currency", "translate_text"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 42 |
+
{"id": "easy-tool_selection-05", "suite": "jevbench-easy", "split": "public", "source": "tool_selection", "variant": "clean", "record": {"state": "Schedule a team meeting next Tuesday at 10 in my calendar.", "questions": {"decision": {"type": "choice", "instructions": "Which single tool should be called to handle this request?", "criteria": {"convert_currency": "Convert an amount from one currency to another", "create_calendar_event": "Put an appointment or meeting into the calendar", "get_weather": "Current weather or forecast for a place", "send_email": "Send an email to a recipient", "translate_text": "Translate text into another language"}}}}, "expected": {"decision": {"labels": ["get_weather", "send_email", "create_calendar_event", "convert_currency", "translate_text"], "target": [0.0, 0.0, 1.0, 0.0, 0.0], "label": 2, "type": "choice", "task": "tool_selection"}}, "metadata": {"id": "easy-tool_selection-05", "source": "tool_selection", "variant": "clean", "group_id": "easy-tool_selection-05", "upstream_group": null, "canonical_labels": ["get_weather", "send_email", "create_calendar_event", "convert_currency", "translate_text"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 43 |
+
{"id": "easy-tool_selection-06", "suite": "jevbench-easy", "split": "public", "source": "tool_selection", "variant": "clean", "record": {"state": "Find me flights from Munich to Lisbon on 12 October.", "questions": {"decision": {"type": "choice", "instructions": "Which single tool should be called to handle this request?", "criteria": {"book_restaurant": "Reserve a table at a restaurant", "call_taxi": "Order a taxi to a location", "get_stock_price": "Look up the current price of a stock", "search_flights": "Find flights between two cities", "set_reminder": "Remind the user of something at a given time"}}}}, "expected": {"decision": {"labels": ["search_flights", "book_restaurant", "set_reminder", "get_stock_price", "call_taxi"], "target": [1.0, 0.0, 0.0, 0.0, 0.0], "label": 0, "type": "choice", "task": "tool_selection"}}, "metadata": {"id": "easy-tool_selection-06", "source": "tool_selection", "variant": "clean", "group_id": "easy-tool_selection-06", "upstream_group": null, "canonical_labels": ["search_flights", "book_restaurant", "set_reminder", "get_stock_price", "call_taxi"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 44 |
+
{"id": "easy-tool_selection-07", "suite": "jevbench-easy", "split": "public", "source": "tool_selection", "variant": "clean", "record": {"state": "Book a table for four at Luigi's tonight at 8.", "questions": {"decision": {"type": "choice", "instructions": "Which single tool should be called to handle this request?", "criteria": {"book_restaurant": "Reserve a table at a restaurant", "call_taxi": "Order a taxi to a location", "get_stock_price": "Look up the current price of a stock", "search_flights": "Find flights between two cities", "set_reminder": "Remind the user of something at a given time"}}}}, "expected": {"decision": {"labels": ["search_flights", "book_restaurant", "set_reminder", "get_stock_price", "call_taxi"], "target": [0.0, 1.0, 0.0, 0.0, 0.0], "label": 1, "type": "choice", "task": "tool_selection"}}, "metadata": {"id": "easy-tool_selection-07", "source": "tool_selection", "variant": "clean", "group_id": "easy-tool_selection-07", "upstream_group": null, "canonical_labels": ["search_flights", "book_restaurant", "set_reminder", "get_stock_price", "call_taxi"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 45 |
+
{"id": "easy-tool_selection-08", "suite": "jevbench-easy", "split": "public", "source": "tool_selection", "variant": "clean", "record": {"state": "Remind me to call my mother at 6 pm.", "questions": {"decision": {"type": "choice", "instructions": "Which single tool should be called to handle this request?", "criteria": {"book_restaurant": "Reserve a table at a restaurant", "call_taxi": "Order a taxi to a location", "get_stock_price": "Look up the current price of a stock", "search_flights": "Find flights between two cities", "set_reminder": "Remind the user of something at a given time"}}}}, "expected": {"decision": {"labels": ["search_flights", "book_restaurant", "set_reminder", "get_stock_price", "call_taxi"], "target": [0.0, 0.0, 1.0, 0.0, 0.0], "label": 2, "type": "choice", "task": "tool_selection"}}, "metadata": {"id": "easy-tool_selection-08", "source": "tool_selection", "variant": "clean", "group_id": "easy-tool_selection-08", "upstream_group": null, "canonical_labels": ["search_flights", "book_restaurant", "set_reminder", "get_stock_price", "call_taxi"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 46 |
+
{"id": "easy-tool_selection-09", "suite": "jevbench-easy", "split": "public", "source": "tool_selection", "variant": "clean", "record": {"state": "What is Apple's stock price right now?", "questions": {"decision": {"type": "choice", "instructions": "Which single tool should be called to handle this request?", "criteria": {"book_restaurant": "Reserve a table at a restaurant", "call_taxi": "Order a taxi to a location", "get_stock_price": "Look up the current price of a stock", "search_flights": "Find flights between two cities", "set_reminder": "Remind the user of something at a given time"}}}}, "expected": {"decision": {"labels": ["search_flights", "book_restaurant", "set_reminder", "get_stock_price", "call_taxi"], "target": [0.0, 0.0, 0.0, 1.0, 0.0], "label": 3, "type": "choice", "task": "tool_selection"}}, "metadata": {"id": "easy-tool_selection-09", "source": "tool_selection", "variant": "clean", "group_id": "easy-tool_selection-09", "upstream_group": null, "canonical_labels": ["search_flights", "book_restaurant", "set_reminder", "get_stock_price", "call_taxi"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 47 |
+
{"id": "easy-tool_selection-10", "suite": "jevbench-easy", "split": "public", "source": "tool_selection", "variant": "clean", "record": {"state": "Get me a taxi to the main station.", "questions": {"decision": {"type": "choice", "instructions": "Which single tool should be called to handle this request?", "criteria": {"book_restaurant": "Reserve a table at a restaurant", "call_taxi": "Order a taxi to a location", "get_stock_price": "Look up the current price of a stock", "search_flights": "Find flights between two cities", "set_reminder": "Remind the user of something at a given time"}}}}, "expected": {"decision": {"labels": ["search_flights", "book_restaurant", "set_reminder", "get_stock_price", "call_taxi"], "target": [0.0, 0.0, 0.0, 0.0, 1.0], "label": 4, "type": "choice", "task": "tool_selection"}}, "metadata": {"id": "easy-tool_selection-10", "source": "tool_selection", "variant": "clean", "group_id": "easy-tool_selection-10", "upstream_group": null, "canonical_labels": ["search_flights", "book_restaurant", "set_reminder", "get_stock_price", "call_taxi"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
| 48 |
+
{"id": "easy-tool_selection-11", "suite": "jevbench-easy", "split": "public", "source": "tool_selection", "variant": "clean", "record": {"state": "Remind me tomorrow at 9 to water the plants.", "questions": {"decision": {"type": "choice", "instructions": "Which single tool should be called to handle this request?", "criteria": {"book_restaurant": "Reserve a table at a restaurant", "call_taxi": "Order a taxi to a location", "get_stock_price": "Look up the current price of a stock", "search_flights": "Find flights between two cities", "set_reminder": "Remind the user of something at a given time"}}}}, "expected": {"decision": {"labels": ["search_flights", "book_restaurant", "set_reminder", "get_stock_price", "call_taxi"], "target": [0.0, 0.0, 1.0, 0.0, 0.0], "label": 2, "type": "choice", "task": "tool_selection"}}, "metadata": {"id": "easy-tool_selection-11", "source": "tool_selection", "variant": "clean", "group_id": "easy-tool_selection-11", "upstream_group": null, "canonical_labels": ["search_flights", "book_restaurant", "set_reminder", "get_stock_price", "call_taxi"], "provenance": {"exclude_reason": null, "label_basis": "Answer named or stated explicitly in the text; authored and reviewed before inference", "license": "MIT", "source": "JevBench v1.1 easy tier, original authored item"}, "gold_policy": {"hard_label": "authored_reviewed_rubric", "reference_probs": null, "argmax_tie_break": "lexicographic_label"}}}
|
server/benchmarks/data/jevbench-easy/public.manifest.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"schema_version": 1,
|
| 3 |
+
"suite": "jevbench-easy",
|
| 4 |
+
"split": "public",
|
| 5 |
+
"description": "Public easy tier: 48 explicit facts, intents and tool selections.",
|
| 6 |
+
"data_file": "public.jsonl",
|
| 7 |
+
"data_sha256": "e186c6dd90b3a7a177ef15f2fd3673d7bbfce95340bfdc6062dc15bfa0e01a6a",
|
| 8 |
+
"upstream": {
|
| 9 |
+
"repository": "https://github.com/fstandhartinger/jevbench",
|
| 10 |
+
"commit": "fd51755eb0c0b546ca206d764faf3302feca913e",
|
| 11 |
+
"path": "datasets/public/easy.jsonl",
|
| 12 |
+
"url": "https://raw.githubusercontent.com/fstandhartinger/jevbench/fd51755eb0c0b546ca206d764faf3302feca913e/datasets/public/easy.jsonl",
|
| 13 |
+
"sha256": "231df3c2c8e88a1a8c137ebe85de96ba70fabd330849098ac7b3c52c70b7172b",
|
| 14 |
+
"notice_sha256": {
|
| 15 |
+
"LICENSE": "3e5beed774bb0bcbfb2fcf24ba9554212c6ae112937d308040989465ab0c5784",
|
| 16 |
+
"THIRD-PARTY.md": "396422e29055bba4073f4a9e5a7163cc724f13e970a34ef58e8ea1fb439b9b38"
|
| 17 |
+
}
|
| 18 |
+
},
|
| 19 |
+
"full_partition": {
|
| 20 |
+
"records": 48,
|
| 21 |
+
"questions": 48,
|
| 22 |
+
"clean_records": 48,
|
| 23 |
+
"clean_questions": 48,
|
| 24 |
+
"headline_questions": 48,
|
| 25 |
+
"variants": {
|
| 26 |
+
"clean": 48
|
| 27 |
+
},
|
| 28 |
+
"clean_sources": {
|
| 29 |
+
"intent": 12,
|
| 30 |
+
"fact": 12,
|
| 31 |
+
"extraction": 12,
|
| 32 |
+
"tool_selection": 12
|
| 33 |
+
},
|
| 34 |
+
"question_types": {
|
| 35 |
+
"choice": 36,
|
| 36 |
+
"noul": 12
|
| 37 |
+
},
|
| 38 |
+
"maximum_options": 5
|
| 39 |
+
},
|
| 40 |
+
"selected": {
|
| 41 |
+
"records": 48,
|
| 42 |
+
"questions": 48,
|
| 43 |
+
"clean_records": 48,
|
| 44 |
+
"clean_questions": 48,
|
| 45 |
+
"headline_questions": 48,
|
| 46 |
+
"variants": {
|
| 47 |
+
"clean": 48
|
| 48 |
+
},
|
| 49 |
+
"clean_sources": {
|
| 50 |
+
"intent": 12,
|
| 51 |
+
"fact": 12,
|
| 52 |
+
"extraction": 12,
|
| 53 |
+
"tool_selection": 12
|
| 54 |
+
},
|
| 55 |
+
"question_types": {
|
| 56 |
+
"choice": 36,
|
| 57 |
+
"noul": 12
|
| 58 |
+
},
|
| 59 |
+
"maximum_options": 5
|
| 60 |
+
},
|
| 61 |
+
"selection": {
|
| 62 |
+
"method": "full",
|
| 63 |
+
"requested_limit": null,
|
| 64 |
+
"is_full_partition": true,
|
| 65 |
+
"note": "Full frozen public tier; not the complete published leaderboard."
|
| 66 |
+
},
|
| 67 |
+
"protocol": {
|
| 68 |
+
"calibration_applied": false,
|
| 69 |
+
"training_data_downloaded": false,
|
| 70 |
+
"gold_labels_sent_to_model": false,
|
| 71 |
+
"headline_variant": "clean",
|
| 72 |
+
"exclude_from_headline_sources": [],
|
| 73 |
+
"locked_test": false,
|
| 74 |
+
"notes": [
|
| 75 |
+
"Community benchmark; not an official TypeSafe dataset release.",
|
| 76 |
+
"Public original/easy/hard have 72/48/111 questions; report separately.",
|
| 77 |
+
"The 534-item leaderboard contains private/untracked data not downloaded here.",
|
| 78 |
+
"Choice request criteria retain their native order; scoring labels retain canonical order.",
|
| 79 |
+
"Native accuracy uses argmax with lexicographic label tie-break, including Score.",
|
| 80 |
+
"Score also supports expected-value MAE; original has 36 paraphrase groups.",
|
| 81 |
+
"One-hot labels and ten explicit mathematical probability references are separate targets.",
|
| 82 |
+
"Provenance, rationale and gold probabilities are never sent in requests."
|
| 83 |
+
]
|
| 84 |
+
},
|
| 85 |
+
"provenance": {
|
| 86 |
+
"dataset_license": "MIT",
|
| 87 |
+
"license_url": "https://github.com/fstandhartinger/jevbench/blob/fd51755eb0c0b546ca206d764faf3302feca913e/LICENSE",
|
| 88 |
+
"notices": [
|
| 89 |
+
"LICENSE",
|
| 90 |
+
"THIRD-PARTY.md"
|
| 91 |
+
],
|
| 92 |
+
"reference_probability_questions": 0,
|
| 93 |
+
"gold_policy": "Authored rubric labels reviewed before inference; hard items retain author and review metadata. Explicit mathematical distributions are not teacher model confidence or population frequency estimates."
|
| 94 |
+
}
|
| 95 |
+
}
|
server/benchmarks/data/jevbench-hard/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
MIT License
|
| 2 |
+
|
| 3 |
+
Copyright (c) 2026 Florian Standhartinger and contributors
|
| 4 |
+
|
| 5 |
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
| 6 |
+
of this software and associated documentation files (the "Software"), to deal
|
| 7 |
+
in the Software without restriction, including without limitation the rights
|
| 8 |
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
| 9 |
+
copies of the Software, and to permit persons to whom the Software is
|
| 10 |
+
furnished to do so, subject to the following conditions:
|
| 11 |
+
|
| 12 |
+
The above copyright notice and this permission notice shall be included in all
|
| 13 |
+
copies or substantial portions of the Software.
|
| 14 |
+
|
| 15 |
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
| 16 |
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
| 17 |
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
| 18 |
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
| 19 |
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
| 20 |
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
| 21 |
+
SOFTWARE.
|
server/benchmarks/data/jevbench-hard/THIRD-PARTY.md
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Third-party projects, data and services
|
| 2 |
+
|
| 3 |
+
MIT (see `LICENSE`) covers this harness and the 72 original public decisions in
|
| 4 |
+
`datasets/public/original.jsonl`. Nothing else in this list is ours to license.
|
| 5 |
+
|
| 6 |
+
## Systems evaluated
|
| 7 |
+
|
| 8 |
+
Each was reached through the interface its author published. We used public
|
| 9 |
+
endpoints as ordinary clients, at one request at a time, and never sent a
|
| 10 |
+
provider's API key to anyone else's endpoint.
|
| 11 |
+
|
| 12 |
+
| System | Author | Source |
|
| 13 |
+
|---|---|---|
|
| 14 |
+
| Jev 1.13.0 | TypeSafe AI | <https://docs.typesafe.ai> |
|
| 15 |
+
| openjev-sglang | ekzhang | <https://github.com/ekzhang/openjev-sglang> |
|
| 16 |
+
| system-one-open | mithalouni | <https://github.com/mithalouni/system-one-open> |
|
| 17 |
+
| open-alternative-jev | IkerMoel | <https://github.com/ikermoel/open-alternative-jev>, <https://huggingface.co/spaces/IkerMoel/open-alternative-jev> |
|
| 18 |
+
| open-jev-deberta-v3-large | Kotoba Labs | <https://huggingface.co/com-kotobalabs/open-jev-deberta-v3-large>, <https://github.com/kotoba-lang/typed-decisions> |
|
| 19 |
+
| GPT-5.6 Luna | OpenAI | <https://platform.openai.com> |
|
| 20 |
+
| Gemini 3.1 Flash-Lite | Google | <https://ai.google.dev> |
|
| 21 |
+
| DeepSeek V4.1 Flash | DeepSeek | <https://api-docs.deepseek.com> |
|
| 22 |
+
| Qwen3.8 27B | Qwen, served by Chutes | <https://chutes.ai> |
|
| 23 |
+
|
| 24 |
+
Model weights, base models and each project's own code keep their own licences.
|
| 25 |
+
A permissive licence on a repository is not a licence for the base model it
|
| 26 |
+
fine-tunes, and we do not restate either.
|
| 27 |
+
|
| 28 |
+
## Wire format
|
| 29 |
+
|
| 30 |
+
The typed-decision request shape (`state`, `questions`, the `noul` / `choice` /
|
| 31 |
+
`score` primitives) is TypeSafe's public HTTP API, documented at
|
| 32 |
+
<https://docs.typesafe.ai/api>. Several of the open rebuilds implement it
|
| 33 |
+
deliberately, which is why one adapter reaches more than one of them. JevBench is
|
| 34 |
+
not affiliated with or endorsed by TypeSafe AI.
|
| 35 |
+
|
| 36 |
+
## Imported decisions
|
| 37 |
+
|
| 38 |
+
146 of the 242 decisions come from our own earlier auto-router experiment
|
| 39 |
+
(<https://github.com/fstandhartinger/auto-model-router>): 78 routing requests and
|
| 40 |
+
68 answer-adequacy judgements whose ground truth is a deterministic grader's
|
| 41 |
+
verdict on a saved answer. The upstream task text is **not** redistributed here,
|
| 42 |
+
because the datasets it was drawn from keep their own terms. `datasets/manifest.json`
|
| 43 |
+
pins the hashes; `scripts/import_router.py` shows exactly what was taken and what
|
| 44 |
+
was excluded.
|
| 45 |
+
|
| 46 |
+
## Held-out decisions
|
| 47 |
+
|
| 48 |
+
24 scenarios are written by us and deliberately unpublished, so the suite cannot
|
| 49 |
+
be trained on in full. Only their whole-split hash and aggregate results appear
|
| 50 |
+
here. They are sent to the services being evaluated, which is not the same thing
|
| 51 |
+
as being public - see the limits section of the README.
|
server/benchmarks/data/jevbench-hard/public.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
server/benchmarks/data/jevbench-hard/public.manifest.json
ADDED
|
@@ -0,0 +1,109 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"schema_version": 1,
|
| 3 |
+
"suite": "jevbench-hard",
|
| 4 |
+
"split": "public",
|
| 5 |
+
"description": "Public hard tier: 111 decisions; ten supply probability references.",
|
| 6 |
+
"data_file": "public.jsonl",
|
| 7 |
+
"data_sha256": "6c93c82edf10d08b70ee0c3addcf7c1ea8fd19957e94fb8021969ff4c0027f95",
|
| 8 |
+
"upstream": {
|
| 9 |
+
"repository": "https://github.com/fstandhartinger/jevbench",
|
| 10 |
+
"commit": "fd51755eb0c0b546ca206d764faf3302feca913e",
|
| 11 |
+
"path": "datasets/public/hard.jsonl",
|
| 12 |
+
"url": "https://raw.githubusercontent.com/fstandhartinger/jevbench/fd51755eb0c0b546ca206d764faf3302feca913e/datasets/public/hard.jsonl",
|
| 13 |
+
"sha256": "89e9e6becb33ed88c1de7d42dcc87531b2fb64cfaef4e1986faf7c37b3f80ebb",
|
| 14 |
+
"notice_sha256": {
|
| 15 |
+
"LICENSE": "3e5beed774bb0bcbfb2fcf24ba9554212c6ae112937d308040989465ab0c5784",
|
| 16 |
+
"THIRD-PARTY.md": "396422e29055bba4073f4a9e5a7163cc724f13e970a34ef58e8ea1fb439b9b38"
|
| 17 |
+
}
|
| 18 |
+
},
|
| 19 |
+
"full_partition": {
|
| 20 |
+
"records": 111,
|
| 21 |
+
"questions": 111,
|
| 22 |
+
"clean_records": 111,
|
| 23 |
+
"clean_questions": 111,
|
| 24 |
+
"headline_questions": 111,
|
| 25 |
+
"variants": {
|
| 26 |
+
"clean": 111
|
| 27 |
+
},
|
| 28 |
+
"clean_sources": {
|
| 29 |
+
"long_policy": 19,
|
| 30 |
+
"probability": 10,
|
| 31 |
+
"temporal_numeric": 15,
|
| 32 |
+
"ambiguous": 7,
|
| 33 |
+
"multi_hop": 18,
|
| 34 |
+
"tradeoff": 6,
|
| 35 |
+
"adversarial": 6,
|
| 36 |
+
"trap": 8,
|
| 37 |
+
"judge_hard": 17,
|
| 38 |
+
"routing_hard": 5
|
| 39 |
+
},
|
| 40 |
+
"question_types": {
|
| 41 |
+
"choice": 67,
|
| 42 |
+
"noul": 38,
|
| 43 |
+
"score": 6
|
| 44 |
+
},
|
| 45 |
+
"maximum_options": 6
|
| 46 |
+
},
|
| 47 |
+
"selected": {
|
| 48 |
+
"records": 111,
|
| 49 |
+
"questions": 111,
|
| 50 |
+
"clean_records": 111,
|
| 51 |
+
"clean_questions": 111,
|
| 52 |
+
"headline_questions": 111,
|
| 53 |
+
"variants": {
|
| 54 |
+
"clean": 111
|
| 55 |
+
},
|
| 56 |
+
"clean_sources": {
|
| 57 |
+
"long_policy": 19,
|
| 58 |
+
"probability": 10,
|
| 59 |
+
"temporal_numeric": 15,
|
| 60 |
+
"ambiguous": 7,
|
| 61 |
+
"multi_hop": 18,
|
| 62 |
+
"tradeoff": 6,
|
| 63 |
+
"adversarial": 6,
|
| 64 |
+
"trap": 8,
|
| 65 |
+
"judge_hard": 17,
|
| 66 |
+
"routing_hard": 5
|
| 67 |
+
},
|
| 68 |
+
"question_types": {
|
| 69 |
+
"choice": 67,
|
| 70 |
+
"noul": 38,
|
| 71 |
+
"score": 6
|
| 72 |
+
},
|
| 73 |
+
"maximum_options": 6
|
| 74 |
+
},
|
| 75 |
+
"selection": {
|
| 76 |
+
"method": "full",
|
| 77 |
+
"requested_limit": null,
|
| 78 |
+
"is_full_partition": true,
|
| 79 |
+
"note": "Full frozen public tier; not the complete published leaderboard."
|
| 80 |
+
},
|
| 81 |
+
"protocol": {
|
| 82 |
+
"calibration_applied": false,
|
| 83 |
+
"training_data_downloaded": false,
|
| 84 |
+
"gold_labels_sent_to_model": false,
|
| 85 |
+
"headline_variant": "clean",
|
| 86 |
+
"exclude_from_headline_sources": [],
|
| 87 |
+
"locked_test": false,
|
| 88 |
+
"notes": [
|
| 89 |
+
"Community benchmark; not an official TypeSafe dataset release.",
|
| 90 |
+
"Public original/easy/hard have 72/48/111 questions; report separately.",
|
| 91 |
+
"The 534-item leaderboard contains private/untracked data not downloaded here.",
|
| 92 |
+
"Choice request criteria retain their native order; scoring labels retain canonical order.",
|
| 93 |
+
"Native accuracy uses argmax with lexicographic label tie-break, including Score.",
|
| 94 |
+
"Score also supports expected-value MAE; original has 36 paraphrase groups.",
|
| 95 |
+
"One-hot labels and ten explicit mathematical probability references are separate targets.",
|
| 96 |
+
"Provenance, rationale and gold probabilities are never sent in requests."
|
| 97 |
+
]
|
| 98 |
+
},
|
| 99 |
+
"provenance": {
|
| 100 |
+
"dataset_license": "MIT",
|
| 101 |
+
"license_url": "https://github.com/fstandhartinger/jevbench/blob/fd51755eb0c0b546ca206d764faf3302feca913e/LICENSE",
|
| 102 |
+
"notices": [
|
| 103 |
+
"LICENSE",
|
| 104 |
+
"THIRD-PARTY.md"
|
| 105 |
+
],
|
| 106 |
+
"reference_probability_questions": 10,
|
| 107 |
+
"gold_policy": "Authored rubric labels reviewed before inference; hard items retain author and review metadata. Explicit mathematical distributions are not teacher model confidence or population frequency estimates."
|
| 108 |
+
}
|
| 109 |
+
}
|
server/benchmarks/data/jevbench-original/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
MIT License
|
| 2 |
+
|
| 3 |
+
Copyright (c) 2026 Florian Standhartinger and contributors
|
| 4 |
+
|
| 5 |
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
| 6 |
+
of this software and associated documentation files (the "Software"), to deal
|
| 7 |
+
in the Software without restriction, including without limitation the rights
|
| 8 |
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
| 9 |
+
copies of the Software, and to permit persons to whom the Software is
|
| 10 |
+
furnished to do so, subject to the following conditions:
|
| 11 |
+
|
| 12 |
+
The above copyright notice and this permission notice shall be included in all
|
| 13 |
+
copies or substantial portions of the Software.
|
| 14 |
+
|
| 15 |
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
| 16 |
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
| 17 |
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
| 18 |
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
| 19 |
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
| 20 |
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
| 21 |
+
SOFTWARE.
|
server/benchmarks/data/jevbench-original/THIRD-PARTY.md
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Third-party projects, data and services
|
| 2 |
+
|
| 3 |
+
MIT (see `LICENSE`) covers this harness and the 72 original public decisions in
|
| 4 |
+
`datasets/public/original.jsonl`. Nothing else in this list is ours to license.
|
| 5 |
+
|
| 6 |
+
## Systems evaluated
|
| 7 |
+
|
| 8 |
+
Each was reached through the interface its author published. We used public
|
| 9 |
+
endpoints as ordinary clients, at one request at a time, and never sent a
|
| 10 |
+
provider's API key to anyone else's endpoint.
|
| 11 |
+
|
| 12 |
+
| System | Author | Source |
|
| 13 |
+
|---|---|---|
|
| 14 |
+
| Jev 1.13.0 | TypeSafe AI | <https://docs.typesafe.ai> |
|
| 15 |
+
| openjev-sglang | ekzhang | <https://github.com/ekzhang/openjev-sglang> |
|
| 16 |
+
| system-one-open | mithalouni | <https://github.com/mithalouni/system-one-open> |
|
| 17 |
+
| open-alternative-jev | IkerMoel | <https://github.com/ikermoel/open-alternative-jev>, <https://huggingface.co/spaces/IkerMoel/open-alternative-jev> |
|
| 18 |
+
| open-jev-deberta-v3-large | Kotoba Labs | <https://huggingface.co/com-kotobalabs/open-jev-deberta-v3-large>, <https://github.com/kotoba-lang/typed-decisions> |
|
| 19 |
+
| GPT-5.6 Luna | OpenAI | <https://platform.openai.com> |
|
| 20 |
+
| Gemini 3.1 Flash-Lite | Google | <https://ai.google.dev> |
|
| 21 |
+
| DeepSeek V4.1 Flash | DeepSeek | <https://api-docs.deepseek.com> |
|
| 22 |
+
| Qwen3.8 27B | Qwen, served by Chutes | <https://chutes.ai> |
|
| 23 |
+
|
| 24 |
+
Model weights, base models and each project's own code keep their own licences.
|
| 25 |
+
A permissive licence on a repository is not a licence for the base model it
|
| 26 |
+
fine-tunes, and we do not restate either.
|
| 27 |
+
|
| 28 |
+
## Wire format
|
| 29 |
+
|
| 30 |
+
The typed-decision request shape (`state`, `questions`, the `noul` / `choice` /
|
| 31 |
+
`score` primitives) is TypeSafe's public HTTP API, documented at
|
| 32 |
+
<https://docs.typesafe.ai/api>. Several of the open rebuilds implement it
|
| 33 |
+
deliberately, which is why one adapter reaches more than one of them. JevBench is
|
| 34 |
+
not affiliated with or endorsed by TypeSafe AI.
|
| 35 |
+
|
| 36 |
+
## Imported decisions
|
| 37 |
+
|
| 38 |
+
146 of the 242 decisions come from our own earlier auto-router experiment
|
| 39 |
+
(<https://github.com/fstandhartinger/auto-model-router>): 78 routing requests and
|
| 40 |
+
68 answer-adequacy judgements whose ground truth is a deterministic grader's
|
| 41 |
+
verdict on a saved answer. The upstream task text is **not** redistributed here,
|
| 42 |
+
because the datasets it was drawn from keep their own terms. `datasets/manifest.json`
|
| 43 |
+
pins the hashes; `scripts/import_router.py` shows exactly what was taken and what
|
| 44 |
+
was excluded.
|
| 45 |
+
|
| 46 |
+
## Held-out decisions
|
| 47 |
+
|
| 48 |
+
24 scenarios are written by us and deliberately unpublished, so the suite cannot
|
| 49 |
+
be trained on in full. Only their whole-split hash and aggregate results appear
|
| 50 |
+
here. They are sent to the services being evaluated, which is not the same thing
|
| 51 |
+
as being public - see the limits section of the README.
|
server/benchmarks/data/jevbench-original/public.manifest.json
ADDED
|
@@ -0,0 +1,101 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"schema_version": 1,
|
| 3 |
+
"suite": "jevbench-original",
|
| 4 |
+
"split": "public",
|
| 5 |
+
"description": "Public original tier: 72 decisions in 36 paraphrase pairs.",
|
| 6 |
+
"data_file": "public.jsonl",
|
| 7 |
+
"data_sha256": "ad453e78624b1c150d5eb3f0f47c5c14caa79542329b4f7e1bcc6199e2eb7007",
|
| 8 |
+
"upstream": {
|
| 9 |
+
"repository": "https://github.com/fstandhartinger/jevbench",
|
| 10 |
+
"commit": "fd51755eb0c0b546ca206d764faf3302feca913e",
|
| 11 |
+
"path": "datasets/public/original.jsonl",
|
| 12 |
+
"url": "https://raw.githubusercontent.com/fstandhartinger/jevbench/fd51755eb0c0b546ca206d764faf3302feca913e/datasets/public/original.jsonl",
|
| 13 |
+
"sha256": "5c2414edb3006b8bfcb70fda433f0f9ca015759433849f8d3104328a1f7c4180",
|
| 14 |
+
"notice_sha256": {
|
| 15 |
+
"LICENSE": "3e5beed774bb0bcbfb2fcf24ba9554212c6ae112937d308040989465ab0c5784",
|
| 16 |
+
"THIRD-PARTY.md": "396422e29055bba4073f4a9e5a7163cc724f13e970a34ef58e8ea1fb439b9b38"
|
| 17 |
+
}
|
| 18 |
+
},
|
| 19 |
+
"full_partition": {
|
| 20 |
+
"records": 72,
|
| 21 |
+
"questions": 72,
|
| 22 |
+
"clean_records": 72,
|
| 23 |
+
"clean_questions": 72,
|
| 24 |
+
"headline_questions": 72,
|
| 25 |
+
"variants": {
|
| 26 |
+
"clean": 72
|
| 27 |
+
},
|
| 28 |
+
"clean_sources": {
|
| 29 |
+
"policy": 12,
|
| 30 |
+
"intent": 12,
|
| 31 |
+
"ordinal": 12,
|
| 32 |
+
"extraction": 12,
|
| 33 |
+
"adequacy": 12,
|
| 34 |
+
"routing": 12
|
| 35 |
+
},
|
| 36 |
+
"question_types": {
|
| 37 |
+
"noul": 24,
|
| 38 |
+
"choice": 36,
|
| 39 |
+
"score": 12
|
| 40 |
+
},
|
| 41 |
+
"maximum_options": 6
|
| 42 |
+
},
|
| 43 |
+
"selected": {
|
| 44 |
+
"records": 72,
|
| 45 |
+
"questions": 72,
|
| 46 |
+
"clean_records": 72,
|
| 47 |
+
"clean_questions": 72,
|
| 48 |
+
"headline_questions": 72,
|
| 49 |
+
"variants": {
|
| 50 |
+
"clean": 72
|
| 51 |
+
},
|
| 52 |
+
"clean_sources": {
|
| 53 |
+
"policy": 12,
|
| 54 |
+
"intent": 12,
|
| 55 |
+
"ordinal": 12,
|
| 56 |
+
"extraction": 12,
|
| 57 |
+
"adequacy": 12,
|
| 58 |
+
"routing": 12
|
| 59 |
+
},
|
| 60 |
+
"question_types": {
|
| 61 |
+
"noul": 24,
|
| 62 |
+
"choice": 36,
|
| 63 |
+
"score": 12
|
| 64 |
+
},
|
| 65 |
+
"maximum_options": 6
|
| 66 |
+
},
|
| 67 |
+
"selection": {
|
| 68 |
+
"method": "full",
|
| 69 |
+
"requested_limit": null,
|
| 70 |
+
"is_full_partition": true,
|
| 71 |
+
"note": "Full frozen public tier; not the complete published leaderboard."
|
| 72 |
+
},
|
| 73 |
+
"protocol": {
|
| 74 |
+
"calibration_applied": false,
|
| 75 |
+
"training_data_downloaded": false,
|
| 76 |
+
"gold_labels_sent_to_model": false,
|
| 77 |
+
"headline_variant": "clean",
|
| 78 |
+
"exclude_from_headline_sources": [],
|
| 79 |
+
"locked_test": false,
|
| 80 |
+
"notes": [
|
| 81 |
+
"Community benchmark; not an official TypeSafe dataset release.",
|
| 82 |
+
"Public original/easy/hard have 72/48/111 questions; report separately.",
|
| 83 |
+
"The 534-item leaderboard contains private/untracked data not downloaded here.",
|
| 84 |
+
"Choice request criteria retain their native order; scoring labels retain canonical order.",
|
| 85 |
+
"Native accuracy uses argmax with lexicographic label tie-break, including Score.",
|
| 86 |
+
"Score also supports expected-value MAE; original has 36 paraphrase groups.",
|
| 87 |
+
"One-hot labels and ten explicit mathematical probability references are separate targets.",
|
| 88 |
+
"Provenance, rationale and gold probabilities are never sent in requests."
|
| 89 |
+
]
|
| 90 |
+
},
|
| 91 |
+
"provenance": {
|
| 92 |
+
"dataset_license": "MIT",
|
| 93 |
+
"license_url": "https://github.com/fstandhartinger/jevbench/blob/fd51755eb0c0b546ca206d764faf3302feca913e/LICENSE",
|
| 94 |
+
"notices": [
|
| 95 |
+
"LICENSE",
|
| 96 |
+
"THIRD-PARTY.md"
|
| 97 |
+
],
|
| 98 |
+
"reference_probability_questions": 0,
|
| 99 |
+
"gold_policy": "Authored rubric labels reviewed before inference; hard items retain author and review metadata. Explicit mathematical distributions are not teacher model confidence or population frequency estimates."
|
| 100 |
+
}
|
| 101 |
+
}
|
server/benchmarks/run_matrix.py
ADDED
|
@@ -0,0 +1,277 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
"""Run one pinned model at a time on one H200. Default is a command preview."""
|
| 3 |
+
|
| 4 |
+
import argparse
|
| 5 |
+
import json
|
| 6 |
+
import os
|
| 7 |
+
import shlex
|
| 8 |
+
import signal
|
| 9 |
+
import socket
|
| 10 |
+
import subprocess
|
| 11 |
+
import sys
|
| 12 |
+
import time
|
| 13 |
+
from pathlib import Path
|
| 14 |
+
|
| 15 |
+
ROOT = Path(__file__).resolve().parents[1]
|
| 16 |
+
RUNTIME = ROOT / "benchmarks/h200/runtime.py"
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
def commands(args, manifest, profile):
|
| 20 |
+
root = args.output.resolve() / profile
|
| 21 |
+
engine = [
|
| 22 |
+
sys.executable,
|
| 23 |
+
str(RUNTIME),
|
| 24 |
+
"launch",
|
| 25 |
+
profile,
|
| 26 |
+
"--engine-python",
|
| 27 |
+
# Resolving a venv Python symlink loses pyvenv.cfg and its installed engine.
|
| 28 |
+
str(args.engine_python.absolute()),
|
| 29 |
+
"--gpu",
|
| 30 |
+
args.gpu,
|
| 31 |
+
"--output",
|
| 32 |
+
str(root / "engine"),
|
| 33 |
+
"--timeout",
|
| 34 |
+
str(args.startup_timeout),
|
| 35 |
+
"--execute",
|
| 36 |
+
]
|
| 37 |
+
if args.probe_vision:
|
| 38 |
+
engine.append("--probe-vision")
|
| 39 |
+
adapter = [
|
| 40 |
+
sys.executable,
|
| 41 |
+
"-m",
|
| 42 |
+
"jev_adapter",
|
| 43 |
+
"--engine-url",
|
| 44 |
+
"http://127.0.0.1:30000",
|
| 45 |
+
"--model",
|
| 46 |
+
"decision-model",
|
| 47 |
+
"--port",
|
| 48 |
+
"30120",
|
| 49 |
+
"--max-concurrency",
|
| 50 |
+
"32",
|
| 51 |
+
]
|
| 52 |
+
model = manifest["models"][profile]
|
| 53 |
+
if model.get("tokenization") == "native_official_text":
|
| 54 |
+
adapter += [
|
| 55 |
+
"--tokenizer-model",
|
| 56 |
+
model["repo_id"],
|
| 57 |
+
"--tokenizer-revision",
|
| 58 |
+
model["revision"],
|
| 59 |
+
]
|
| 60 |
+
suites = [
|
| 61 |
+
"decision-v7",
|
| 62 |
+
"transfer-v4",
|
| 63 |
+
"transfer-v9",
|
| 64 |
+
"jevbench-original",
|
| 65 |
+
"jevbench-easy",
|
| 66 |
+
"jevbench-hard",
|
| 67 |
+
]
|
| 68 |
+
if args.suite:
|
| 69 |
+
suites = args.suite
|
| 70 |
+
if args.include_external:
|
| 71 |
+
suites += [
|
| 72 |
+
suite for suite in ["semif-v1", "scienthoon-v1"] if suite not in suites
|
| 73 |
+
]
|
| 74 |
+
evaluations = []
|
| 75 |
+
for concurrency in args.concurrency:
|
| 76 |
+
run = [
|
| 77 |
+
sys.executable,
|
| 78 |
+
"-m",
|
| 79 |
+
"jev_adapter.benchmarks.run",
|
| 80 |
+
"--output",
|
| 81 |
+
str(root / f"c{concurrency}"),
|
| 82 |
+
"--engine-manifest",
|
| 83 |
+
str(root / "engine/launch.json"),
|
| 84 |
+
"--concurrency",
|
| 85 |
+
str(concurrency),
|
| 86 |
+
"--warmup",
|
| 87 |
+
str(args.warmup),
|
| 88 |
+
"--repeats",
|
| 89 |
+
str(args.repeats),
|
| 90 |
+
]
|
| 91 |
+
for suite in suites:
|
| 92 |
+
split = "public" if suite.startswith("jevbench-") else "development"
|
| 93 |
+
run += [
|
| 94 |
+
"--data",
|
| 95 |
+
str(args.data_root.resolve() / suite / f"{split}.jsonl"),
|
| 96 |
+
]
|
| 97 |
+
if args.limit is not None:
|
| 98 |
+
run += ["--limit", str(args.limit)]
|
| 99 |
+
evaluations.append(run)
|
| 100 |
+
return {
|
| 101 |
+
"profile": profile,
|
| 102 |
+
"model": manifest["models"][profile],
|
| 103 |
+
"engine": engine,
|
| 104 |
+
"adapter": adapter,
|
| 105 |
+
"evaluations": evaluations,
|
| 106 |
+
}
|
| 107 |
+
|
| 108 |
+
|
| 109 |
+
def stop_owned(process):
|
| 110 |
+
if process is not None and process.poll() is None:
|
| 111 |
+
# SIGINT can be inherited as ignored under nohup; the runtime handles TERM.
|
| 112 |
+
os.killpg(process.pid, signal.SIGTERM)
|
| 113 |
+
try:
|
| 114 |
+
process.wait(timeout=40)
|
| 115 |
+
except subprocess.TimeoutExpired as exc:
|
| 116 |
+
raise RuntimeError(
|
| 117 |
+
f"Owned process {process.pid} did not stop; check its log."
|
| 118 |
+
) from exc
|
| 119 |
+
|
| 120 |
+
|
| 121 |
+
def wait_engine(process, ready_path, timeout):
|
| 122 |
+
deadline = time.monotonic() + timeout
|
| 123 |
+
while not ready_path.exists():
|
| 124 |
+
if process.poll() is not None:
|
| 125 |
+
raise RuntimeError(
|
| 126 |
+
"Engine launcher exited; see launcher.log and engine/engine.log"
|
| 127 |
+
)
|
| 128 |
+
if time.monotonic() > deadline:
|
| 129 |
+
raise TimeoutError("Engine startup deadline exceeded")
|
| 130 |
+
time.sleep(1)
|
| 131 |
+
|
| 132 |
+
|
| 133 |
+
def wait_adapter(process, timeout=60):
|
| 134 |
+
import httpx
|
| 135 |
+
|
| 136 |
+
deadline = time.monotonic() + timeout
|
| 137 |
+
headers = {}
|
| 138 |
+
if key := os.environ.get("JEV_API_KEY"):
|
| 139 |
+
headers["Authorization"] = f"Bearer {key}"
|
| 140 |
+
with httpx.Client(timeout=2) as client:
|
| 141 |
+
while True:
|
| 142 |
+
if process.poll() is not None:
|
| 143 |
+
raise RuntimeError("Adapter startup failed; see adapter.log")
|
| 144 |
+
try:
|
| 145 |
+
response = client.get(
|
| 146 |
+
"http://127.0.0.1:30120/v1/models", headers=headers
|
| 147 |
+
)
|
| 148 |
+
if response.is_success:
|
| 149 |
+
return
|
| 150 |
+
except httpx.HTTPError:
|
| 151 |
+
pass
|
| 152 |
+
if time.monotonic() > deadline:
|
| 153 |
+
raise TimeoutError("Adapter startup deadline exceeded")
|
| 154 |
+
time.sleep(1)
|
| 155 |
+
|
| 156 |
+
|
| 157 |
+
def main():
|
| 158 |
+
manifest = json.loads((RUNTIME.parent / "models.json").read_text())
|
| 159 |
+
parser = argparse.ArgumentParser(description=__doc__)
|
| 160 |
+
parser.add_argument("--output", type=Path, required=True)
|
| 161 |
+
parser.add_argument(
|
| 162 |
+
"--engine-python", type=Path, default=ROOT / ".venv-sglang/bin/python"
|
| 163 |
+
)
|
| 164 |
+
parser.add_argument("--data-root", type=Path, default=ROOT / "benchmarks/data")
|
| 165 |
+
parser.add_argument("--profile", action="append", choices=list(manifest["models"]))
|
| 166 |
+
parser.add_argument("--concurrency", type=int, action="append")
|
| 167 |
+
parser.add_argument("--warmup", type=int, default=20)
|
| 168 |
+
parser.add_argument("--repeats", type=int, default=1)
|
| 169 |
+
parser.add_argument("--limit", type=int)
|
| 170 |
+
parser.add_argument("--startup-timeout", type=int, default=3600)
|
| 171 |
+
parser.add_argument("--gpu", default="0")
|
| 172 |
+
parser.add_argument("--include-external", action="store_true")
|
| 173 |
+
parser.add_argument(
|
| 174 |
+
"--suite",
|
| 175 |
+
action="append",
|
| 176 |
+
choices=[
|
| 177 |
+
"decision-v7",
|
| 178 |
+
"transfer-v4",
|
| 179 |
+
"transfer-v9",
|
| 180 |
+
"jevbench-original",
|
| 181 |
+
"jevbench-easy",
|
| 182 |
+
"jevbench-hard",
|
| 183 |
+
"semif-v1",
|
| 184 |
+
"scienthoon-v1",
|
| 185 |
+
],
|
| 186 |
+
)
|
| 187 |
+
parser.add_argument(
|
| 188 |
+
"--probe-vision", action="store_true", help="optional image readiness probe"
|
| 189 |
+
)
|
| 190 |
+
parser.add_argument("--execute", action="store_true")
|
| 191 |
+
args = parser.parse_args()
|
| 192 |
+
args.concurrency = args.concurrency or [1]
|
| 193 |
+
if (
|
| 194 |
+
any(c < 1 or c > 32 for c in args.concurrency)
|
| 195 |
+
or len(set(args.concurrency)) != len(args.concurrency)
|
| 196 |
+
or args.warmup < 0
|
| 197 |
+
or args.repeats < 1
|
| 198 |
+
or args.startup_timeout < 1
|
| 199 |
+
or (args.limit is not None and args.limit < 1)
|
| 200 |
+
):
|
| 201 |
+
parser.error("invalid counts; concurrency must be unique values from 1 to 32")
|
| 202 |
+
profiles = args.profile or manifest["default_matrix"]
|
| 203 |
+
if len(set(profiles)) != len(profiles):
|
| 204 |
+
parser.error("profiles must be unique")
|
| 205 |
+
plans = [commands(args, manifest, profile) for profile in profiles]
|
| 206 |
+
for plan in plans:
|
| 207 |
+
print(f"\n{plan['profile']}")
|
| 208 |
+
for command in [plan["engine"], plan["adapter"], *plan["evaluations"]]:
|
| 209 |
+
print(shlex.join(command))
|
| 210 |
+
if not args.execute:
|
| 211 |
+
print("\nPreview only. Add --execute on the prepared H200 server.")
|
| 212 |
+
return
|
| 213 |
+
if not args.engine_python.is_file():
|
| 214 |
+
parser.error("engine Python is missing; follow benchmarks/h200/README.md")
|
| 215 |
+
for plan in plans:
|
| 216 |
+
for command in plan["evaluations"]:
|
| 217 |
+
for i, value in enumerate(command):
|
| 218 |
+
if value == "--data" and not Path(command[i + 1]).is_file():
|
| 219 |
+
parser.error(f"missing prepared dataset: {command[i + 1]}")
|
| 220 |
+
args.output.mkdir(parents=True, exist_ok=False)
|
| 221 |
+
(args.output / "matrix.json").write_text(json.dumps(plans, indent=2) + "\n")
|
| 222 |
+
for plan in plans:
|
| 223 |
+
# Never attach to or stop unrelated services using the benchmark ports.
|
| 224 |
+
for port in (30000, 30120):
|
| 225 |
+
with socket.socket() as listener:
|
| 226 |
+
# Match server bind semantics: ignore TIME_WAIT, reject live listeners.
|
| 227 |
+
listener.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1)
|
| 228 |
+
listener.bind(("127.0.0.1", port))
|
| 229 |
+
root = args.output.resolve() / plan["profile"]
|
| 230 |
+
root.mkdir()
|
| 231 |
+
engine, adapter = None, None
|
| 232 |
+
with (
|
| 233 |
+
(root / "launcher.log").open("x") as engine_log,
|
| 234 |
+
(root / "adapter.log").open("x") as adapter_log,
|
| 235 |
+
):
|
| 236 |
+
try:
|
| 237 |
+
engine = subprocess.Popen(
|
| 238 |
+
plan["engine"],
|
| 239 |
+
cwd=ROOT,
|
| 240 |
+
stdout=engine_log,
|
| 241 |
+
stderr=subprocess.STDOUT,
|
| 242 |
+
start_new_session=True,
|
| 243 |
+
)
|
| 244 |
+
wait_engine(
|
| 245 |
+
engine, root / "engine/ready.json", args.startup_timeout + 90
|
| 246 |
+
)
|
| 247 |
+
adapter = subprocess.Popen(
|
| 248 |
+
plan["adapter"],
|
| 249 |
+
cwd=ROOT,
|
| 250 |
+
stdout=adapter_log,
|
| 251 |
+
stderr=subprocess.STDOUT,
|
| 252 |
+
start_new_session=True,
|
| 253 |
+
)
|
| 254 |
+
wait_adapter(adapter)
|
| 255 |
+
for command in plan["evaluations"]:
|
| 256 |
+
subprocess.run(command, cwd=ROOT, check=True)
|
| 257 |
+
finally:
|
| 258 |
+
try:
|
| 259 |
+
stop_owned(adapter)
|
| 260 |
+
finally:
|
| 261 |
+
stop_owned(engine)
|
| 262 |
+
subprocess.run(
|
| 263 |
+
[
|
| 264 |
+
sys.executable,
|
| 265 |
+
"-m",
|
| 266 |
+
"jev_adapter.benchmarks.compare",
|
| 267 |
+
str(args.output.resolve()),
|
| 268 |
+
"--output",
|
| 269 |
+
str(args.output.resolve() / "comparison.csv"),
|
| 270 |
+
],
|
| 271 |
+
cwd=ROOT,
|
| 272 |
+
check=True,
|
| 273 |
+
)
|
| 274 |
+
|
| 275 |
+
|
| 276 |
+
if __name__ == "__main__":
|
| 277 |
+
main()
|
server/benchmarks/verify_metrics.py
ADDED
|
@@ -0,0 +1,140 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Audit metric parity against verified KEV source, without importing KEV or torch.
|
| 2 |
+
|
| 3 |
+
Requires NumPy only in the optional audit environment, not in jev-adapter.
|
| 4 |
+
"""
|
| 5 |
+
|
| 6 |
+
from __future__ import annotations
|
| 7 |
+
|
| 8 |
+
import argparse
|
| 9 |
+
import ast
|
| 10 |
+
import hashlib
|
| 11 |
+
import json
|
| 12 |
+
import math
|
| 13 |
+
import sys
|
| 14 |
+
from pathlib import Path
|
| 15 |
+
|
| 16 |
+
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
| 17 |
+
|
| 18 |
+
from jev_adapter.benchmarks import metrics as adapter_metrics
|
| 19 |
+
from jev_adapter.benchmarks.data import KEV_COMMIT
|
| 20 |
+
|
| 21 |
+
REFERENCE_FILES = {
|
| 22 |
+
"kev/evaluate.py": (
|
| 23 |
+
"1b20e3f9edf417aa8dae924b1526e52f74b710cadf7213c5ec68334f6e7f8fe1",
|
| 24 |
+
{"ece"},
|
| 25 |
+
),
|
| 26 |
+
"kev/benchmark.py": (
|
| 27 |
+
"4192ec3b26b065452f84bde38a091e6854a28fe185d2e0f39a0c91b2efe69df7",
|
| 28 |
+
{"coverage_at_error", "metrics"},
|
| 29 |
+
),
|
| 30 |
+
"kev/contrastive.py": (
|
| 31 |
+
"cbb979aa5d40265ad0e64695f94b281d91751fa405811ddfa8212fede111edcf",
|
| 32 |
+
{"paired_flip"},
|
| 33 |
+
),
|
| 34 |
+
}
|
| 35 |
+
|
| 36 |
+
|
| 37 |
+
def extract_reference(root, numpy):
|
| 38 |
+
scope = {"np": numpy, "math": math, "EPSILON": 1e-9}
|
| 39 |
+
for relative, (expected_hash, names) in REFERENCE_FILES.items():
|
| 40 |
+
payload = (root / relative).read_bytes()
|
| 41 |
+
if hashlib.sha256(payload).hexdigest() != expected_hash:
|
| 42 |
+
raise ValueError(
|
| 43 |
+
f"reference file differs from KEV {KEV_COMMIT}: {relative}"
|
| 44 |
+
)
|
| 45 |
+
tree = ast.parse(payload, filename=relative)
|
| 46 |
+
tree.body = [
|
| 47 |
+
node
|
| 48 |
+
for node in tree.body
|
| 49 |
+
if isinstance(node, ast.FunctionDef) and node.name in names
|
| 50 |
+
]
|
| 51 |
+
if {node.name for node in tree.body} != names:
|
| 52 |
+
raise ValueError(f"missing reference functions in {relative}")
|
| 53 |
+
# Execute only the hash-verified function definitions. Top-level imports,
|
| 54 |
+
# model loading, decorators elsewhere and benchmark entry points are absent.
|
| 55 |
+
exec(compile(tree, relative, "exec"), scope) # noqa: S102
|
| 56 |
+
return scope
|
| 57 |
+
|
| 58 |
+
|
| 59 |
+
def audit(root, numpy):
|
| 60 |
+
reference = extract_reference(root, numpy)
|
| 61 |
+
rng = numpy.random.default_rng(483)
|
| 62 |
+
maximum_delta = {}
|
| 63 |
+
tolerance = 1e-12
|
| 64 |
+
for batch in range(100):
|
| 65 |
+
rows = []
|
| 66 |
+
for index in range(50):
|
| 67 |
+
kind = ["choice", "noul", "score"][index % 3]
|
| 68 |
+
options = 2 if kind == "noul" else int(rng.integers(2, 11))
|
| 69 |
+
probabilities = rng.dirichlet(numpy.ones(options)).tolist()
|
| 70 |
+
label = int(rng.integers(options))
|
| 71 |
+
if index < 10:
|
| 72 |
+
probabilities = [index / 10, 1 - index / 10]
|
| 73 |
+
label %= 2
|
| 74 |
+
kind = "noul"
|
| 75 |
+
rows.append({"p": probabilities, "label": label, "type": kind})
|
| 76 |
+
expected = reference["metrics"](rows)
|
| 77 |
+
actual = adapter_metrics.metrics(rows)
|
| 78 |
+
for key, value in actual.items():
|
| 79 |
+
if value is None or not isinstance(value, (int, float)):
|
| 80 |
+
continue
|
| 81 |
+
delta = abs(value - expected[key])
|
| 82 |
+
maximum_delta[key] = max(maximum_delta.get(key, 0), delta)
|
| 83 |
+
if delta > tolerance:
|
| 84 |
+
raise ValueError(
|
| 85 |
+
f"batch {batch}, {key}: adapter={value}, KEV={expected[key]}"
|
| 86 |
+
)
|
| 87 |
+
rows = []
|
| 88 |
+
for index in range(10):
|
| 89 |
+
for sibling, label in (("a", 0), ("b", index % 2)):
|
| 90 |
+
rows.append(
|
| 91 |
+
{
|
| 92 |
+
"pair_id": str(index),
|
| 93 |
+
"question": "q",
|
| 94 |
+
"keys": ["x", "y"],
|
| 95 |
+
"label": label,
|
| 96 |
+
"p": [0.7, 0.3] if index % 3 else [0.4, 0.6],
|
| 97 |
+
"sibling": sibling,
|
| 98 |
+
}
|
| 99 |
+
)
|
| 100 |
+
expected = reference["paired_flip"](rows)
|
| 101 |
+
actual = adapter_metrics.paired_flip(rows)
|
| 102 |
+
for key, value in expected.items():
|
| 103 |
+
if actual[key] != value:
|
| 104 |
+
raise ValueError(f"pair metric differs: {key}: {actual[key]} != {value}")
|
| 105 |
+
return {
|
| 106 |
+
"status": "passed",
|
| 107 |
+
"kev_commit": KEV_COMMIT,
|
| 108 |
+
"reference_sha256": {path: entry[0] for path, entry in REFERENCE_FILES.items()},
|
| 109 |
+
"adapter_metrics_sha256": hashlib.sha256(
|
| 110 |
+
Path(adapter_metrics.__file__).read_bytes()
|
| 111 |
+
).hexdigest(),
|
| 112 |
+
"numpy_version": numpy.__version__,
|
| 113 |
+
"seed": 483,
|
| 114 |
+
"batches": 100,
|
| 115 |
+
"rows_per_batch": 50,
|
| 116 |
+
"total_metric_rows": 5000,
|
| 117 |
+
"absolute_tolerance": tolerance,
|
| 118 |
+
"maximum_absolute_delta": maximum_delta,
|
| 119 |
+
"complete_pair_metrics_exact": expected,
|
| 120 |
+
"limitations": [
|
| 121 |
+
"Numerical metric parity, not GPU/model inference validation.",
|
| 122 |
+
"Only common returned scalar metrics and complete pair metrics compared.",
|
| 123 |
+
"Partial-pair handling and unknowable reporting intentionally differ.",
|
| 124 |
+
],
|
| 125 |
+
}
|
| 126 |
+
|
| 127 |
+
|
| 128 |
+
def main():
|
| 129 |
+
parser = argparse.ArgumentParser(description=__doc__)
|
| 130 |
+
parser.add_argument("--kev-root", required=True, type=Path)
|
| 131 |
+
args = parser.parse_args()
|
| 132 |
+
try:
|
| 133 |
+
import numpy
|
| 134 |
+
except ImportError:
|
| 135 |
+
parser.error("NumPy is required in this optional audit environment")
|
| 136 |
+
print(json.dumps(audit(args.kev_root, numpy), indent=2, allow_nan=False))
|
| 137 |
+
|
| 138 |
+
|
| 139 |
+
if __name__ == "__main__":
|
| 140 |
+
main()
|
server/examples/request.json
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"model": "jev-latest",
|
| 3 |
+
"state": "결제가 두 번 되었습니다. 중복 결제 건을 환불해 주세요.",
|
| 4 |
+
"questions": {
|
| 5 |
+
"route": {
|
| 6 |
+
"type": "choice",
|
| 7 |
+
"instructions": "어느 부서로 보내야 하나?",
|
| 8 |
+
"criteria": {"billing": "결제 및 환불", "technical": "기술 지원", "sales": "상품 문의"}
|
| 9 |
+
},
|
| 10 |
+
"refund": {"type": "noul", "instructions": "환불을 요청했나?"},
|
| 11 |
+
"urgency": {"type": "score", "instructions": "대응 긴급도", "criteria": ["낮음", "보통", "높음"]}
|
| 12 |
+
}
|
| 13 |
+
}
|
server/examples/smoke.py
ADDED
|
@@ -0,0 +1,49 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Smoke-test a running adapter, optionally including actual image bytes."""
|
| 2 |
+
|
| 3 |
+
import argparse
|
| 4 |
+
import base64
|
| 5 |
+
import json
|
| 6 |
+
import mimetypes
|
| 7 |
+
import os
|
| 8 |
+
import time
|
| 9 |
+
from pathlib import Path
|
| 10 |
+
from urllib.request import Request, urlopen
|
| 11 |
+
|
| 12 |
+
|
| 13 |
+
def main():
|
| 14 |
+
parser = argparse.ArgumentParser(description=__doc__)
|
| 15 |
+
parser.add_argument("--base-url", default="http://127.0.0.1:30120")
|
| 16 |
+
parser.add_argument(
|
| 17 |
+
"--request", type=Path, default=Path(__file__).with_name("request.json")
|
| 18 |
+
)
|
| 19 |
+
parser.add_argument("--image", type=Path, action="append", default=[])
|
| 20 |
+
args = parser.parse_args()
|
| 21 |
+
body = json.loads(args.request.read_text())
|
| 22 |
+
for path in args.image:
|
| 23 |
+
mime = mimetypes.guess_type(path.name)[0]
|
| 24 |
+
if not mime or not mime.startswith("image/"):
|
| 25 |
+
parser.error(f"Cannot identify image type: {path.name}")
|
| 26 |
+
encoded = base64.b64encode(path.read_bytes()).decode("ascii")
|
| 27 |
+
body.setdefault("images", []).append(f"data:{mime};base64,{encoded}")
|
| 28 |
+
headers = {"Content-Type": "application/json"}
|
| 29 |
+
if key := os.environ.get("JEV_API_KEY"):
|
| 30 |
+
headers["Authorization"] = f"Bearer {key}"
|
| 31 |
+
request = Request(
|
| 32 |
+
args.base_url.rstrip("/") + "/v1/systemone",
|
| 33 |
+
data=json.dumps(body).encode(),
|
| 34 |
+
headers=headers,
|
| 35 |
+
)
|
| 36 |
+
started = time.perf_counter()
|
| 37 |
+
with urlopen(request, timeout=120) as response:
|
| 38 |
+
result = json.load(response)
|
| 39 |
+
elapsed = (time.perf_counter() - started) * 1000
|
| 40 |
+
if result.get("usage", {}).get("output_tokens") != 0:
|
| 41 |
+
raise RuntimeError("Expected zero generated tokens")
|
| 42 |
+
if set(result.get("answers", {})) != set(body["questions"]):
|
| 43 |
+
raise RuntimeError("Missing decision answers")
|
| 44 |
+
print(json.dumps(result, indent=2, ensure_ascii=False))
|
| 45 |
+
print(f"End-to-end HTTP latency: {elapsed:.1f} ms (not a GPU benchmark)")
|
| 46 |
+
|
| 47 |
+
|
| 48 |
+
if __name__ == "__main__":
|
| 49 |
+
main()
|
server/jev_adapter/benchmarks/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
"""Reproducible, training-free decision benchmarks over the adapter HTTP API."""
|
server/jev_adapter/benchmarks/compare.py
ADDED
|
@@ -0,0 +1,72 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Collect complete benchmark reports into a comparison table, one row per suite."""
|
| 2 |
+
|
| 3 |
+
import argparse
|
| 4 |
+
import csv
|
| 5 |
+
import json
|
| 6 |
+
from pathlib import Path
|
| 7 |
+
|
| 8 |
+
|
| 9 |
+
def collect(root):
|
| 10 |
+
rows = []
|
| 11 |
+
for path in sorted(root.rglob("report.json")):
|
| 12 |
+
if path.with_name("invalidated.json").exists():
|
| 13 |
+
continue
|
| 14 |
+
report = json.loads(path.read_text())
|
| 15 |
+
if report.get("status") != "complete" or not report.get("suites"):
|
| 16 |
+
continue
|
| 17 |
+
manifest = json.loads(path.with_name("manifest.json").read_text())
|
| 18 |
+
launch = manifest.get("launch_manifest", {})
|
| 19 |
+
engine_model = (
|
| 20 |
+
manifest.get("engine", {}).get("model_info", {}).get("model_path")
|
| 21 |
+
)
|
| 22 |
+
engine_revision = (
|
| 23 |
+
manifest.get("engine", {}).get("server_info", {}).get("revision")
|
| 24 |
+
)
|
| 25 |
+
for suite, result in report["suites"].items():
|
| 26 |
+
clean = result.get("clean")
|
| 27 |
+
if clean is None:
|
| 28 |
+
continue
|
| 29 |
+
latency = result["latency_ms"]
|
| 30 |
+
rows.append(
|
| 31 |
+
{
|
| 32 |
+
"profile": launch.get("profile", manifest["requested_model"]),
|
| 33 |
+
"model_repo": launch.get("model", {}).get("repo_id", engine_model),
|
| 34 |
+
"model_revision": launch.get("model", {}).get(
|
| 35 |
+
"revision", engine_revision
|
| 36 |
+
),
|
| 37 |
+
"suite": suite,
|
| 38 |
+
"concurrency": manifest["concurrency"],
|
| 39 |
+
"partial_dataset": report["partial_dataset"],
|
| 40 |
+
"clean_questions": clean["n"],
|
| 41 |
+
"accuracy": clean["acc"],
|
| 42 |
+
"brier": clean["brier"],
|
| 43 |
+
"ece": clean["ece"],
|
| 44 |
+
"confident_error_rate": clean["confident_error_rate"],
|
| 45 |
+
"p50_ms": latency["p50"],
|
| 46 |
+
"p95_ms": latency["p95"],
|
| 47 |
+
"p99_ms": latency["p99"],
|
| 48 |
+
"cache_mode": manifest["cache_mode"],
|
| 49 |
+
"report": str(path.resolve()),
|
| 50 |
+
}
|
| 51 |
+
)
|
| 52 |
+
return rows
|
| 53 |
+
|
| 54 |
+
|
| 55 |
+
def main():
|
| 56 |
+
parser = argparse.ArgumentParser(description=__doc__)
|
| 57 |
+
parser.add_argument("root", type=Path)
|
| 58 |
+
parser.add_argument("--output", type=Path, required=True)
|
| 59 |
+
args = parser.parse_args()
|
| 60 |
+
rows = collect(args.root)
|
| 61 |
+
if not rows:
|
| 62 |
+
parser.error("no complete benchmark reports found; failed runs are not ranked")
|
| 63 |
+
args.output.parent.mkdir(parents=True, exist_ok=True)
|
| 64 |
+
with args.output.open("x", newline="") as handle:
|
| 65 |
+
writer = csv.DictWriter(handle, fieldnames=list(rows[0]))
|
| 66 |
+
writer.writeheader()
|
| 67 |
+
writer.writerows(rows)
|
| 68 |
+
print(f"Wrote {len(rows)} suite rows to {args.output}; latency is HTTP wall time.")
|
| 69 |
+
|
| 70 |
+
|
| 71 |
+
if __name__ == "__main__":
|
| 72 |
+
main()
|
server/jev_adapter/benchmarks/data.py
ADDED
|
@@ -0,0 +1,217 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Normalize pinned, public KEV evaluation records without changing their tasks.
|
| 2 |
+
|
| 3 |
+
Only the request is sent to a model. Labels and source metadata remain outside
|
| 4 |
+
that request, in a separate expected/metadata envelope used by the evaluator.
|
| 5 |
+
This module is an independent format conversion, not imported KEV model code.
|
| 6 |
+
"""
|
| 7 |
+
|
| 8 |
+
from __future__ import annotations
|
| 9 |
+
|
| 10 |
+
import copy
|
| 11 |
+
import hashlib
|
| 12 |
+
import json
|
| 13 |
+
from collections import Counter
|
| 14 |
+
from dataclasses import dataclass
|
| 15 |
+
from typing import Any
|
| 16 |
+
|
| 17 |
+
KEV_COMMIT = "4f8110a3f8620cc3a182ae9a708e4398492c4b1a"
|
| 18 |
+
KEV_REPOSITORY = "https://github.com/jaredpalmer/kev"
|
| 19 |
+
KEV_RAW = f"https://raw.githubusercontent.com/jaredpalmer/kev/{KEV_COMMIT}"
|
| 20 |
+
DEFAULT_SUITES = ("decision-v7", "transfer-v4", "transfer-v9")
|
| 21 |
+
|
| 22 |
+
|
| 23 |
+
@dataclass(frozen=True)
|
| 24 |
+
class SuiteSpec:
|
| 25 |
+
path: str
|
| 26 |
+
manifest_sha256: str
|
| 27 |
+
description: str
|
| 28 |
+
notes: tuple[str, ...] = ()
|
| 29 |
+
|
| 30 |
+
|
| 31 |
+
SUITES = {
|
| 32 |
+
"decision-v7": SuiteSpec(
|
| 33 |
+
"evals/v7/decision-v7",
|
| 34 |
+
"a8f50e481b7d90b97da049e0ff6a01cee2f1ed204aed61a8265af0edbb5514d2",
|
| 35 |
+
"Ten public sources and generated policies; KEV trained-source evaluation.",
|
| 36 |
+
("Includes up to 78 choices with none-of-the-above variants.",),
|
| 37 |
+
),
|
| 38 |
+
"transfer-v4": SuiteSpec(
|
| 39 |
+
"evals/v4/transfer-v4",
|
| 40 |
+
"31677c2256b406222e7d94ffdc0a02a70ce05746b9efe307876024c4e77291d1",
|
| 41 |
+
"Six sources unseen in KEV fine-tuning and held-out policy structures.",
|
| 42 |
+
("Unseen means unseen in KEV fine-tuning, not in base-model pretraining.",),
|
| 43 |
+
),
|
| 44 |
+
"transfer-v9": SuiteSpec(
|
| 45 |
+
"evals/v9/transfer-v9",
|
| 46 |
+
"3c4f0be94509a3612678bfd3a30fd99a8d0ca3c47ddfe7318075d95b2fa365e4",
|
| 47 |
+
"Transfer-v4 plus MMLU-Pro, buried evidence and unknowable/control pairs.",
|
| 48 |
+
(
|
| 49 |
+
"Contains transfer-v4 records; do not pool both suites as independent data.",
|
| 50 |
+
"Source 'unknowable' is evaluated for confidence, not accuracy.",
|
| 51 |
+
),
|
| 52 |
+
),
|
| 53 |
+
"semif-v1": SuiteSpec(
|
| 54 |
+
"evals/external/semif-v1",
|
| 55 |
+
"0de05eac16b0ddeeb2719c50a94a9148d6ae195f66303aec74aa103a3845ad11",
|
| 56 |
+
"SemIf's 144 authored choices plus 108 perturbations, frozen by KEV.",
|
| 57 |
+
(
|
| 58 |
+
"Already evaluated by KEV; this is not an additional independent suite.",
|
| 59 |
+
"SemIf's own headline is mean family balanced accuracy.",
|
| 60 |
+
),
|
| 61 |
+
),
|
| 62 |
+
"scienthoon-v1": SuiteSpec(
|
| 63 |
+
"evals/external/scienthoon-v1",
|
| 64 |
+
"ef31183425bf9d3c2d8ac5d245a14d30fa44d1e531c945e4405edcb1f75ae0d5",
|
| 65 |
+
"KEV's frozen conversion of scienthoon's synthetic support tickets.",
|
| 66 |
+
(
|
| 67 |
+
"Already evaluated by KEV; this is not an additional independent suite.",
|
| 68 |
+
"Original 900 question rows become 291 unique states / 873 questions in this conversion.",
|
| 69 |
+
"Priority labels depend on an organizational rule absent from the state; report separately.",
|
| 70 |
+
),
|
| 71 |
+
),
|
| 72 |
+
}
|
| 73 |
+
|
| 74 |
+
|
| 75 |
+
def sha256(data: bytes) -> str:
|
| 76 |
+
return hashlib.sha256(data).hexdigest()
|
| 77 |
+
|
| 78 |
+
|
| 79 |
+
def require_sha256(data: bytes, expected: str, name: str) -> None:
|
| 80 |
+
actual = sha256(data)
|
| 81 |
+
if actual != expected:
|
| 82 |
+
raise ValueError(
|
| 83 |
+
f"SHA256 mismatch for {name}: expected {expected}, got {actual}"
|
| 84 |
+
)
|
| 85 |
+
|
| 86 |
+
|
| 87 |
+
def normalize_record(raw: dict[str, Any], suite: str, split: str) -> dict[str, Any]:
|
| 88 |
+
"""Retain option order, all sibling questions and KEV's original provenance."""
|
| 89 |
+
meta = copy.deepcopy(raw["_meta"])
|
| 90 |
+
if not isinstance(meta.get("id"), str) or not meta["id"]:
|
| 91 |
+
raise ValueError("record requires a nonempty _meta.id")
|
| 92 |
+
questions, expected = {}, {}
|
| 93 |
+
for qid, question in raw["questions"].items():
|
| 94 |
+
kind = question["type"]
|
| 95 |
+
if kind == "choice":
|
| 96 |
+
labels = list(question["criteria"])
|
| 97 |
+
try:
|
| 98 |
+
label = labels.index(question["label"])
|
| 99 |
+
except ValueError as exc:
|
| 100 |
+
raise ValueError(f"{meta['id']}/{qid}: label is not an option") from exc
|
| 101 |
+
elif kind == "noul":
|
| 102 |
+
if type(question["label"]) is not bool:
|
| 103 |
+
raise ValueError(f"{meta['id']}/{qid}: noul label must be a boolean")
|
| 104 |
+
labels, label = ["false", "true"], int(question["label"])
|
| 105 |
+
elif kind == "score":
|
| 106 |
+
labels = [str(i) for i in range(len(question["criteria"]))]
|
| 107 |
+
label = question["label"]
|
| 108 |
+
if type(label) is not int or not 0 <= label < len(labels):
|
| 109 |
+
raise ValueError(f"{meta['id']}/{qid}: score label is out of range")
|
| 110 |
+
else:
|
| 111 |
+
raise ValueError(f"unsupported question type: {kind}")
|
| 112 |
+
if len(labels) < 2 or len(set(labels)) != len(labels):
|
| 113 |
+
raise ValueError(f"{meta['id']}/{qid}: invalid option labels")
|
| 114 |
+
questions[qid] = {
|
| 115 |
+
key: copy.deepcopy(question[key])
|
| 116 |
+
for key in ("type", "instructions", "criteria")
|
| 117 |
+
if key in question
|
| 118 |
+
}
|
| 119 |
+
expected[qid] = {
|
| 120 |
+
"labels": labels,
|
| 121 |
+
"target": [float(i == label) for i in range(len(labels))],
|
| 122 |
+
"label": label,
|
| 123 |
+
"type": kind,
|
| 124 |
+
"task": question.get("src", meta["source"]),
|
| 125 |
+
}
|
| 126 |
+
if not questions:
|
| 127 |
+
raise ValueError(f"{meta['id']}: no questions")
|
| 128 |
+
record = {"state": copy.deepcopy(raw["state"]), "questions": questions}
|
| 129 |
+
for key in ("images", "options"):
|
| 130 |
+
if key in raw:
|
| 131 |
+
record[key] = copy.deepcopy(raw[key])
|
| 132 |
+
return {
|
| 133 |
+
"id": meta["id"],
|
| 134 |
+
"suite": suite,
|
| 135 |
+
"split": split,
|
| 136 |
+
"source": meta["source"],
|
| 137 |
+
"variant": meta.get("variant", "clean"),
|
| 138 |
+
"record": record,
|
| 139 |
+
"expected": expected,
|
| 140 |
+
"metadata": meta,
|
| 141 |
+
}
|
| 142 |
+
|
| 143 |
+
|
| 144 |
+
def parse_partition(data: bytes, suite: str, split: str) -> list[dict[str, Any]]:
|
| 145 |
+
records, seen = [], set()
|
| 146 |
+
for number, line in enumerate(data.decode("utf-8").splitlines(), 1):
|
| 147 |
+
if not line.strip():
|
| 148 |
+
raise ValueError(f"{suite}/{split}:{number}: blank JSONL record")
|
| 149 |
+
row = normalize_record(json.loads(line), suite, split)
|
| 150 |
+
if row["id"] in seen:
|
| 151 |
+
raise ValueError(f"duplicate record id: {row['id']}")
|
| 152 |
+
seen.add(row["id"])
|
| 153 |
+
records.append(row)
|
| 154 |
+
return records
|
| 155 |
+
|
| 156 |
+
|
| 157 |
+
def population_counts(records: list[dict[str, Any]]) -> dict[str, Any]:
|
| 158 |
+
clean = [row for row in records if row["variant"] == "clean"]
|
| 159 |
+
return {
|
| 160 |
+
"records": len(records),
|
| 161 |
+
"questions": sum(len(row["expected"]) for row in records),
|
| 162 |
+
"clean_records": len(clean),
|
| 163 |
+
"clean_questions": sum(len(row["expected"]) for row in clean),
|
| 164 |
+
"headline_questions": sum(
|
| 165 |
+
len(row["expected"]) for row in clean if row["source"] != "unknowable"
|
| 166 |
+
),
|
| 167 |
+
"variants": dict(Counter(row["variant"] for row in records)),
|
| 168 |
+
"clean_sources": dict(Counter(row["source"] for row in clean)),
|
| 169 |
+
"question_types": dict(
|
| 170 |
+
Counter(q["type"] for row in records for q in row["expected"].values())
|
| 171 |
+
),
|
| 172 |
+
"maximum_options": max(
|
| 173 |
+
(len(q["labels"]) for row in records for q in row["expected"].values()),
|
| 174 |
+
default=0,
|
| 175 |
+
),
|
| 176 |
+
}
|
| 177 |
+
|
| 178 |
+
|
| 179 |
+
def source_provenance(
|
| 180 |
+
records: list[dict[str, Any]], upstream_manifest: dict[str, Any]
|
| 181 |
+
) -> dict[str, Any]:
|
| 182 |
+
"""Point to underlying dataset licenses; do not relicense mixed source data."""
|
| 183 |
+
datasets = {}
|
| 184 |
+
for row in records:
|
| 185 |
+
meta = row["metadata"]
|
| 186 |
+
repo = meta.get("repo")
|
| 187 |
+
if not repo:
|
| 188 |
+
continue
|
| 189 |
+
revision = meta.get("revision")
|
| 190 |
+
key = (repo, revision)
|
| 191 |
+
external = upstream_manifest.get("external", {})
|
| 192 |
+
is_external = external.get("repo", "").endswith("/" + repo)
|
| 193 |
+
datasets[key] = {
|
| 194 |
+
"repository": repo,
|
| 195 |
+
"revision": revision,
|
| 196 |
+
"source_url": (
|
| 197 |
+
external["repo"]
|
| 198 |
+
if is_external
|
| 199 |
+
else f"https://huggingface.co/datasets/{repo}"
|
| 200 |
+
),
|
| 201 |
+
"license": external.get("license")
|
| 202 |
+
if is_external
|
| 203 |
+
else "see upstream dataset",
|
| 204 |
+
}
|
| 205 |
+
return {
|
| 206 |
+
"kev_repository_license": "Apache-2.0",
|
| 207 |
+
"kev_license_url": f"{KEV_REPOSITORY}/blob/{KEV_COMMIT}/LICENSE",
|
| 208 |
+
"dataset_notice": (
|
| 209 |
+
"Public and downloadable does not mean all source datasets share Apache-2.0. "
|
| 210 |
+
"Their individual licenses and attribution terms continue to apply."
|
| 211 |
+
),
|
| 212 |
+
"datasets": list(datasets.values()),
|
| 213 |
+
"external": upstream_manifest.get("external"),
|
| 214 |
+
"dataset_revisions_from_manifest": upstream_manifest.get(
|
| 215 |
+
"dataset_revisions", {}
|
| 216 |
+
),
|
| 217 |
+
}
|
server/jev_adapter/benchmarks/jevbench.py
ADDED
|
@@ -0,0 +1,292 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Prepare JevBench's three pinned public tiers without exposing answer metadata.
|
| 2 |
+
|
| 3 |
+
This is an independent format conversion. The public 231 questions are not the
|
| 4 |
+
534-question leaderboard, whose remaining questions are private or untracked.
|
| 5 |
+
"""
|
| 6 |
+
|
| 7 |
+
from __future__ import annotations
|
| 8 |
+
|
| 9 |
+
import copy
|
| 10 |
+
import json
|
| 11 |
+
import math
|
| 12 |
+
from dataclasses import dataclass
|
| 13 |
+
from pathlib import Path
|
| 14 |
+
from typing import Any
|
| 15 |
+
|
| 16 |
+
import httpx
|
| 17 |
+
|
| 18 |
+
from jev_adapter.protocol import SystemOneRequest
|
| 19 |
+
|
| 20 |
+
from .data import normalize_record, population_counts, require_sha256, sha256
|
| 21 |
+
|
| 22 |
+
JEVBENCH_COMMIT = "fd51755eb0c0b546ca206d764faf3302feca913e"
|
| 23 |
+
JEVBENCH_REPOSITORY = "https://github.com/fstandhartinger/jevbench"
|
| 24 |
+
JEVBENCH_RAW = (
|
| 25 |
+
f"https://raw.githubusercontent.com/fstandhartinger/jevbench/{JEVBENCH_COMMIT}"
|
| 26 |
+
)
|
| 27 |
+
JEVBENCH_NOTICES = {
|
| 28 |
+
"LICENSE": "3e5beed774bb0bcbfb2fcf24ba9554212c6ae112937d308040989465ab0c5784",
|
| 29 |
+
"THIRD-PARTY.md": "396422e29055bba4073f4a9e5a7163cc724f13e970a34ef58e8ea1fb439b9b38",
|
| 30 |
+
}
|
| 31 |
+
|
| 32 |
+
|
| 33 |
+
@dataclass(frozen=True)
|
| 34 |
+
class JevBenchSpec:
|
| 35 |
+
path: str
|
| 36 |
+
sha256: str
|
| 37 |
+
records: int
|
| 38 |
+
description: str
|
| 39 |
+
|
| 40 |
+
|
| 41 |
+
JEVBENCH_SUITES = {
|
| 42 |
+
"jevbench-original": JevBenchSpec(
|
| 43 |
+
"datasets/public/original.jsonl",
|
| 44 |
+
"5c2414edb3006b8bfcb70fda433f0f9ca015759433849f8d3104328a1f7c4180",
|
| 45 |
+
72,
|
| 46 |
+
"Public original tier: 72 decisions in 36 paraphrase pairs.",
|
| 47 |
+
),
|
| 48 |
+
"jevbench-easy": JevBenchSpec(
|
| 49 |
+
"datasets/public/easy.jsonl",
|
| 50 |
+
"231df3c2c8e88a1a8c137ebe85de96ba70fabd330849098ac7b3c52c70b7172b",
|
| 51 |
+
48,
|
| 52 |
+
"Public easy tier: 48 explicit facts, intents and tool selections.",
|
| 53 |
+
),
|
| 54 |
+
"jevbench-hard": JevBenchSpec(
|
| 55 |
+
"datasets/public/hard.jsonl",
|
| 56 |
+
"89e9e6becb33ed88c1de7d42dcc87531b2fb64cfaef4e1986faf7c37b3f80ebb",
|
| 57 |
+
111,
|
| 58 |
+
"Public hard tier: 111 decisions; ten supply probability references.",
|
| 59 |
+
),
|
| 60 |
+
}
|
| 61 |
+
|
| 62 |
+
|
| 63 |
+
def normalize_jevbench_record(raw: dict[str, Any], suite: str) -> dict[str, Any]:
|
| 64 |
+
"""Keep native question/option order and canonical scoring label order."""
|
| 65 |
+
if raw.get("split") != "public":
|
| 66 |
+
raise ValueError("JevBench converter accepts only the public subset")
|
| 67 |
+
if not isinstance(raw.get("id"), str) or not raw["id"]:
|
| 68 |
+
raise ValueError("JevBench record requires a nonempty id")
|
| 69 |
+
labels = raw["labels"]
|
| 70 |
+
if (
|
| 71 |
+
not isinstance(labels, list)
|
| 72 |
+
or len(labels) < 2
|
| 73 |
+
or any(not isinstance(label, str) or not label for label in labels)
|
| 74 |
+
or len(set(labels)) != len(labels)
|
| 75 |
+
):
|
| 76 |
+
raise ValueError(f"{raw['id']}: invalid canonical labels")
|
| 77 |
+
question = raw["question"]
|
| 78 |
+
kind, gold = question["type"], raw["expected"]
|
| 79 |
+
if kind == "choice":
|
| 80 |
+
if not isinstance(question["criteria"], dict) or set(labels) != set(
|
| 81 |
+
question["criteria"]
|
| 82 |
+
):
|
| 83 |
+
raise ValueError(f"{raw['id']}: choice criteria/labels mismatch")
|
| 84 |
+
if not isinstance(gold, str) or gold not in labels:
|
| 85 |
+
raise ValueError(f"{raw['id']}: invalid choice expected label")
|
| 86 |
+
canonical_labels, canonical_gold = list(labels), gold
|
| 87 |
+
elif kind == "noul":
|
| 88 |
+
if labels != ["no", "yes"] or gold not in ("no", "yes"):
|
| 89 |
+
raise ValueError(f"{raw['id']}: noul requires no/yes canonical labels")
|
| 90 |
+
canonical_labels = ["false", "true"]
|
| 91 |
+
canonical_gold, gold = ("true" if gold == "yes" else "false"), gold == "yes"
|
| 92 |
+
elif kind == "score":
|
| 93 |
+
if not isinstance(question["criteria"], list) or labels != [
|
| 94 |
+
str(i) for i in range(len(question["criteria"]))
|
| 95 |
+
]:
|
| 96 |
+
raise ValueError(f"{raw['id']}: score criteria/labels mismatch")
|
| 97 |
+
if type(gold) is not int or not 0 <= gold < len(labels):
|
| 98 |
+
raise ValueError(f"{raw['id']}: invalid score expected label")
|
| 99 |
+
canonical_labels, canonical_gold = list(labels), str(gold)
|
| 100 |
+
else:
|
| 101 |
+
raise ValueError(f"unsupported JevBench question type: {kind}")
|
| 102 |
+
provenance = copy.deepcopy(raw["provenance"])
|
| 103 |
+
reference = provenance.get("gold_probs")
|
| 104 |
+
if provenance.get("exclude_reason"):
|
| 105 |
+
raise ValueError(f"{raw['id']}: excluded upstream item is not evaluable")
|
| 106 |
+
family = raw["family"]
|
| 107 |
+
if not isinstance(family, str) or not family:
|
| 108 |
+
raise ValueError(f"{raw['id']}: missing task family")
|
| 109 |
+
# Only these three native question fields are allowed into the request.
|
| 110 |
+
clean_question = {
|
| 111 |
+
key: copy.deepcopy(question[key])
|
| 112 |
+
for key in ("type", "instructions", "criteria")
|
| 113 |
+
if key in question
|
| 114 |
+
}
|
| 115 |
+
clean_question["label"] = gold
|
| 116 |
+
row = normalize_record(
|
| 117 |
+
{
|
| 118 |
+
"state": raw["state"],
|
| 119 |
+
"questions": {"decision": clean_question},
|
| 120 |
+
"_meta": {
|
| 121 |
+
"id": raw["id"],
|
| 122 |
+
"source": family,
|
| 123 |
+
"variant": "clean",
|
| 124 |
+
"group_id": raw.get("group") or raw["id"],
|
| 125 |
+
"upstream_group": raw.get("group"),
|
| 126 |
+
"canonical_labels": list(labels),
|
| 127 |
+
"provenance": provenance,
|
| 128 |
+
"gold_policy": {
|
| 129 |
+
"hard_label": "authored_reviewed_rubric",
|
| 130 |
+
"reference_probs": (
|
| 131 |
+
"countable_mathematical_probability"
|
| 132 |
+
if reference is not None
|
| 133 |
+
else None
|
| 134 |
+
),
|
| 135 |
+
"argmax_tie_break": "lexicographic_label",
|
| 136 |
+
},
|
| 137 |
+
},
|
| 138 |
+
},
|
| 139 |
+
suite,
|
| 140 |
+
"public",
|
| 141 |
+
)
|
| 142 |
+
expected = row["expected"]["decision"]
|
| 143 |
+
expected["labels"] = canonical_labels
|
| 144 |
+
expected["label"] = canonical_labels.index(canonical_gold)
|
| 145 |
+
expected["target"] = [float(label == canonical_gold) for label in canonical_labels]
|
| 146 |
+
if reference is not None:
|
| 147 |
+
if not isinstance(reference, dict) or set(reference) != set(labels):
|
| 148 |
+
raise ValueError(f"{raw['id']}: reference probability labels mismatch")
|
| 149 |
+
probabilities = [reference[label] for label in labels]
|
| 150 |
+
if any(
|
| 151 |
+
isinstance(value, bool)
|
| 152 |
+
or not isinstance(value, (float, int))
|
| 153 |
+
or not math.isfinite(value)
|
| 154 |
+
or not 0 <= value <= 1
|
| 155 |
+
for value in probabilities
|
| 156 |
+
) or not math.isclose(sum(probabilities), 1.0, rel_tol=0, abs_tol=1e-9):
|
| 157 |
+
raise ValueError(f"{raw['id']}: invalid reference probabilities")
|
| 158 |
+
expected["reference_probs"] = [float(value) for value in probabilities]
|
| 159 |
+
SystemOneRequest.model_validate({"model": "validation-only", **row["record"]})
|
| 160 |
+
return row
|
| 161 |
+
|
| 162 |
+
|
| 163 |
+
def parse_jevbench(data: bytes, suite: str) -> list[dict[str, Any]]:
|
| 164 |
+
records, seen = [], set()
|
| 165 |
+
for number, line in enumerate(data.decode("utf-8").splitlines(), 1):
|
| 166 |
+
if not line.strip():
|
| 167 |
+
raise ValueError(f"{suite}/public:{number}: blank JSONL record")
|
| 168 |
+
row = normalize_jevbench_record(json.loads(line), suite)
|
| 169 |
+
if row["id"] in seen:
|
| 170 |
+
raise ValueError(f"duplicate record id: {row['id']}")
|
| 171 |
+
seen.add(row["id"])
|
| 172 |
+
records.append(row)
|
| 173 |
+
return records
|
| 174 |
+
|
| 175 |
+
|
| 176 |
+
def prepare_jevbench(
|
| 177 |
+
suite: str,
|
| 178 |
+
output: Path,
|
| 179 |
+
*,
|
| 180 |
+
client: httpx.Client | None = None,
|
| 181 |
+
source_root: Path | None = None,
|
| 182 |
+
limit: int | None = None,
|
| 183 |
+
) -> dict[str, Any]:
|
| 184 |
+
"""Verify all source bytes before writing an immutable prepared public tier."""
|
| 185 |
+
if suite not in JEVBENCH_SUITES:
|
| 186 |
+
raise ValueError(f"unknown JevBench suite: {suite}")
|
| 187 |
+
if limit is not None and (type(limit) is not int or limit < 1):
|
| 188 |
+
raise ValueError("--limit must be positive")
|
| 189 |
+
|
| 190 |
+
def read_source(path: str) -> bytes:
|
| 191 |
+
if source_root is not None:
|
| 192 |
+
return (source_root / path).read_bytes()
|
| 193 |
+
if client is None:
|
| 194 |
+
raise ValueError("an HTTP client or --source-root is required")
|
| 195 |
+
response = client.get(f"{JEVBENCH_RAW}/{path}")
|
| 196 |
+
response.raise_for_status()
|
| 197 |
+
return response.content
|
| 198 |
+
|
| 199 |
+
spec = JEVBENCH_SUITES[suite]
|
| 200 |
+
raw_bytes = read_source(spec.path)
|
| 201 |
+
require_sha256(raw_bytes, spec.sha256, spec.path)
|
| 202 |
+
notices = {}
|
| 203 |
+
for path, digest in JEVBENCH_NOTICES.items():
|
| 204 |
+
content = read_source(path)
|
| 205 |
+
require_sha256(content, digest, path)
|
| 206 |
+
notices[path] = content
|
| 207 |
+
records = parse_jevbench(raw_bytes, suite)
|
| 208 |
+
counts = population_counts(records)
|
| 209 |
+
if counts["records"] != spec.records or counts["questions"] != spec.records:
|
| 210 |
+
raise ValueError(f"upstream record/question count mismatch: {suite}/public")
|
| 211 |
+
selected = records if limit is None else records[:limit]
|
| 212 |
+
serialized = "".join(
|
| 213 |
+
json.dumps(row, ensure_ascii=False, allow_nan=False) + "\n" for row in selected
|
| 214 |
+
).encode("utf-8")
|
| 215 |
+
full = len(selected) == len(records)
|
| 216 |
+
manifest = {
|
| 217 |
+
"schema_version": 1,
|
| 218 |
+
"suite": suite,
|
| 219 |
+
"split": "public",
|
| 220 |
+
"description": spec.description,
|
| 221 |
+
"data_file": "public.jsonl",
|
| 222 |
+
"data_sha256": sha256(serialized),
|
| 223 |
+
"upstream": {
|
| 224 |
+
"repository": JEVBENCH_REPOSITORY,
|
| 225 |
+
"commit": JEVBENCH_COMMIT,
|
| 226 |
+
"path": spec.path,
|
| 227 |
+
"url": f"{JEVBENCH_RAW}/{spec.path}",
|
| 228 |
+
"sha256": spec.sha256,
|
| 229 |
+
"notice_sha256": dict(JEVBENCH_NOTICES),
|
| 230 |
+
},
|
| 231 |
+
"full_partition": counts,
|
| 232 |
+
"selected": population_counts(selected),
|
| 233 |
+
"selection": {
|
| 234 |
+
"method": "full" if full else "prefix",
|
| 235 |
+
"requested_limit": limit,
|
| 236 |
+
"is_full_partition": full,
|
| 237 |
+
"note": (
|
| 238 |
+
"Full frozen public tier; not the complete published leaderboard."
|
| 239 |
+
if full
|
| 240 |
+
else "Smoke subset only; not a full-tier result. Pairs may be incomplete."
|
| 241 |
+
),
|
| 242 |
+
},
|
| 243 |
+
"protocol": {
|
| 244 |
+
"calibration_applied": False,
|
| 245 |
+
"training_data_downloaded": False,
|
| 246 |
+
"gold_labels_sent_to_model": False,
|
| 247 |
+
"headline_variant": "clean",
|
| 248 |
+
"exclude_from_headline_sources": [],
|
| 249 |
+
"locked_test": False,
|
| 250 |
+
"notes": [
|
| 251 |
+
"Community benchmark; not an official TypeSafe dataset release.",
|
| 252 |
+
"Public original/easy/hard have 72/48/111 questions; report separately.",
|
| 253 |
+
"The 534-item leaderboard contains private/untracked data not downloaded here.",
|
| 254 |
+
"Choice request criteria retain their native order; scoring labels retain canonical order.",
|
| 255 |
+
"Native accuracy uses argmax with lexicographic label tie-break, including Score.",
|
| 256 |
+
"Score also supports expected-value MAE; original has 36 paraphrase groups.",
|
| 257 |
+
"One-hot labels and ten explicit mathematical probability references are separate targets.",
|
| 258 |
+
"Provenance, rationale and gold probabilities are never sent in requests.",
|
| 259 |
+
],
|
| 260 |
+
},
|
| 261 |
+
"provenance": {
|
| 262 |
+
"dataset_license": "MIT",
|
| 263 |
+
"license_url": f"{JEVBENCH_REPOSITORY}/blob/{JEVBENCH_COMMIT}/LICENSE",
|
| 264 |
+
"notices": list(JEVBENCH_NOTICES),
|
| 265 |
+
"reference_probability_questions": sum(
|
| 266 |
+
"reference_probs" in row["expected"]["decision"] for row in records
|
| 267 |
+
),
|
| 268 |
+
"gold_policy": (
|
| 269 |
+
"Authored rubric labels reviewed before inference; hard items retain author "
|
| 270 |
+
"and review metadata. Explicit mathematical distributions are not teacher "
|
| 271 |
+
"model confidence or population frequency estimates."
|
| 272 |
+
),
|
| 273 |
+
},
|
| 274 |
+
}
|
| 275 |
+
artifacts = {
|
| 276 |
+
"public.jsonl": serialized,
|
| 277 |
+
"public.manifest.json": (
|
| 278 |
+
json.dumps(manifest, ensure_ascii=False, indent=2, allow_nan=False) + "\n"
|
| 279 |
+
).encode("utf-8"),
|
| 280 |
+
**notices,
|
| 281 |
+
}
|
| 282 |
+
directory = output / suite
|
| 283 |
+
for name, content in artifacts.items():
|
| 284 |
+
path = directory / name
|
| 285 |
+
if path.exists() and path.read_bytes() != content:
|
| 286 |
+
raise FileExistsError(
|
| 287 |
+
f"refusing to overwrite a different prepared artifact: {path}"
|
| 288 |
+
)
|
| 289 |
+
directory.mkdir(parents=True, exist_ok=True)
|
| 290 |
+
for name, content in artifacts.items():
|
| 291 |
+
(directory / name).write_bytes(content)
|
| 292 |
+
return manifest
|
server/jev_adapter/benchmarks/metrics.py
ADDED
|
@@ -0,0 +1,326 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Dependency-free metrics following KEV's frozen benchmark definitions.
|
| 2 |
+
|
| 3 |
+
Reimplemented against jaredpalmer/kev commit
|
| 4 |
+
4f8110a3f8620cc3a182ae9a708e4398492c4b1a, kev/benchmark.py,
|
| 5 |
+
kev/evaluate.py and kev/contrastive.py (Apache-2.0). Differences are explicit:
|
| 6 |
+
partial pairs are counted, and unknowable items never receive accuracy metrics.
|
| 7 |
+
"""
|
| 8 |
+
|
| 9 |
+
import math
|
| 10 |
+
from collections import defaultdict
|
| 11 |
+
from statistics import fmean
|
| 12 |
+
|
| 13 |
+
|
| 14 |
+
def argmax(values):
|
| 15 |
+
return max(range(len(values)), key=values.__getitem__)
|
| 16 |
+
|
| 17 |
+
|
| 18 |
+
def predicted_index(row):
|
| 19 |
+
if row.get("argmax_tie_break") == "lexicographic_label":
|
| 20 |
+
return min(range(len(row["p"])), key=lambda i: (-row["p"][i], row["keys"][i]))
|
| 21 |
+
return argmax(row["p"])
|
| 22 |
+
|
| 23 |
+
|
| 24 |
+
def distribution(raw, labels):
|
| 25 |
+
if not isinstance(raw, dict) or set(raw) != set(labels):
|
| 26 |
+
raise ValueError("probability keys differ from requested labels")
|
| 27 |
+
p = [raw[key] for key in labels]
|
| 28 |
+
if any(
|
| 29 |
+
type(value) not in (int, float)
|
| 30 |
+
or not math.isfinite(value)
|
| 31 |
+
or not 0 <= value <= 1
|
| 32 |
+
for value in p
|
| 33 |
+
):
|
| 34 |
+
raise ValueError("invalid probability value")
|
| 35 |
+
total = math.fsum(p)
|
| 36 |
+
if total <= 0 or abs(total - 1) > max(1e-5, len(labels) * 0.005 + 1e-8):
|
| 37 |
+
raise ValueError("invalid probability sum")
|
| 38 |
+
return [value / total for value in p], total
|
| 39 |
+
|
| 40 |
+
|
| 41 |
+
def prediction_rows(item, response):
|
| 42 |
+
expected = item["expected"]
|
| 43 |
+
answers = response.get("answers")
|
| 44 |
+
if not isinstance(answers, dict) or set(answers) != set(expected):
|
| 45 |
+
raise ValueError("response question IDs differ from request")
|
| 46 |
+
rows = []
|
| 47 |
+
meta = item["metadata"]
|
| 48 |
+
for qid, target in expected.items():
|
| 49 |
+
answer = answers[qid]
|
| 50 |
+
kind = target["type"]
|
| 51 |
+
if not isinstance(answer, dict) or answer.get("type") != kind:
|
| 52 |
+
raise ValueError("response answer type differs from request")
|
| 53 |
+
if kind == "noul":
|
| 54 |
+
value = answer.get("noul")
|
| 55 |
+
if type(value) not in (int, float):
|
| 56 |
+
raise ValueError("invalid noul probability")
|
| 57 |
+
raw = {"true": value, "false": 1 - value}
|
| 58 |
+
else:
|
| 59 |
+
raw = answer.get("probabilities")
|
| 60 |
+
labels = target["labels"]
|
| 61 |
+
p, total = distribution(raw, labels)
|
| 62 |
+
rows.append(
|
| 63 |
+
{
|
| 64 |
+
"id": item["id"],
|
| 65 |
+
"suite": item["suite"],
|
| 66 |
+
"split": item["split"],
|
| 67 |
+
"question": qid,
|
| 68 |
+
"source": item["source"],
|
| 69 |
+
"task": target["task"],
|
| 70 |
+
"type": kind,
|
| 71 |
+
"variant": item["variant"],
|
| 72 |
+
"keys": labels,
|
| 73 |
+
"label": target["label"],
|
| 74 |
+
"p": p,
|
| 75 |
+
"group": meta.get("group_id", item["id"]),
|
| 76 |
+
"pair_id": meta.get("pair_id"),
|
| 77 |
+
"sibling": meta.get("sibling"),
|
| 78 |
+
"control_id": meta.get("control_id"),
|
| 79 |
+
"parent": meta.get("parent_id")
|
| 80 |
+
or (item["id"] if item["variant"] == "clean" else meta.get("group_id")),
|
| 81 |
+
"raw_probability_sum": total,
|
| 82 |
+
"zero_count": sum(value == 0 for value in p),
|
| 83 |
+
"reference_probs": target.get("reference_probs"),
|
| 84 |
+
"argmax_tie_break": meta.get("gold_policy", {}).get("argmax_tie_break"),
|
| 85 |
+
}
|
| 86 |
+
)
|
| 87 |
+
return rows
|
| 88 |
+
|
| 89 |
+
|
| 90 |
+
def coverage_at_error(confidence, correct, budget):
|
| 91 |
+
# Stable ties deliberately match KEV. This empirical, retrospective cutoff
|
| 92 |
+
# is not a risk guarantee and must not be deployed as a fitted threshold.
|
| 93 |
+
order = sorted(range(len(confidence)), key=lambda i: -confidence[i])
|
| 94 |
+
wrong, accepted = 0, 0
|
| 95 |
+
for rank, index in enumerate(order, 1):
|
| 96 |
+
wrong += not correct[index]
|
| 97 |
+
if wrong <= budget * rank:
|
| 98 |
+
accepted = rank
|
| 99 |
+
return accepted / len(order)
|
| 100 |
+
|
| 101 |
+
|
| 102 |
+
def metrics(rows):
|
| 103 |
+
if not rows:
|
| 104 |
+
return None
|
| 105 |
+
nll, correct, confidence, brier, mae, rps = [], [], [], [], [], []
|
| 106 |
+
for row in rows:
|
| 107 |
+
p, y = row["p"], row["label"]
|
| 108 |
+
nll.append(-math.log(max(p[y], 1e-9)))
|
| 109 |
+
correct.append(predicted_index(row) == y)
|
| 110 |
+
confidence.append(max(p))
|
| 111 |
+
brier.append(math.fsum((value - (i == y)) ** 2 for i, value in enumerate(p)))
|
| 112 |
+
if row["type"] == "score":
|
| 113 |
+
mae.append(abs(math.fsum(i * value for i, value in enumerate(p)) - y))
|
| 114 |
+
cumulative = 0.0
|
| 115 |
+
errors = []
|
| 116 |
+
for i, value in enumerate(p[:-1]):
|
| 117 |
+
cumulative += value
|
| 118 |
+
errors.append((cumulative - (i >= y)) ** 2)
|
| 119 |
+
rps.append(fmean(errors))
|
| 120 |
+
ece = 0.0
|
| 121 |
+
for index in range(10):
|
| 122 |
+
# Match numpy.linspace(0, 1, 11), including floating-point bin edges.
|
| 123 |
+
lo, hi = index * 0.1, (index + 1) * 0.1
|
| 124 |
+
bucket = [
|
| 125 |
+
i
|
| 126 |
+
for i, c in enumerate(confidence)
|
| 127 |
+
if lo <= c and (c < hi if index < 9 else c <= 1)
|
| 128 |
+
]
|
| 129 |
+
if bucket:
|
| 130 |
+
ece += (
|
| 131 |
+
len(bucket)
|
| 132 |
+
/ len(rows)
|
| 133 |
+
* abs(
|
| 134 |
+
fmean(correct[i] for i in bucket)
|
| 135 |
+
- fmean(confidence[i] for i in bucket)
|
| 136 |
+
)
|
| 137 |
+
)
|
| 138 |
+
high = [i for i, value in enumerate(confidence) if value >= 0.9]
|
| 139 |
+
result = {
|
| 140 |
+
"n": len(rows),
|
| 141 |
+
"acc": fmean(correct),
|
| 142 |
+
"nll": fmean(nll),
|
| 143 |
+
"brier": fmean(brier),
|
| 144 |
+
"ece": ece,
|
| 145 |
+
"mean_conf": fmean(confidence),
|
| 146 |
+
"confidence_bias": fmean(confidence) - fmean(correct),
|
| 147 |
+
"confident_error_rate": sum(not correct[i] for i in high) / len(rows),
|
| 148 |
+
"coverage_at_0_9": len(high) / len(rows),
|
| 149 |
+
"accuracy_at_0_9": fmean(correct[i] for i in high) if high else None,
|
| 150 |
+
"coverage_at_5pct_error": coverage_at_error(confidence, correct, 0.05),
|
| 151 |
+
"coverage_at_1pct_error": coverage_at_error(confidence, correct, 0.01),
|
| 152 |
+
}
|
| 153 |
+
if mae:
|
| 154 |
+
result.update(score_mae=fmean(mae), ranked_probability_score=fmean(rps))
|
| 155 |
+
return result
|
| 156 |
+
|
| 157 |
+
|
| 158 |
+
def grouped_metrics(rows, key):
|
| 159 |
+
grouped = defaultdict(list)
|
| 160 |
+
for row in rows:
|
| 161 |
+
if row["source"] != "unknowable":
|
| 162 |
+
grouped[row[key]].append(row)
|
| 163 |
+
return {name: metrics(group) for name, group in sorted(grouped.items())}
|
| 164 |
+
|
| 165 |
+
|
| 166 |
+
def paired_flip(rows):
|
| 167 |
+
pairs = defaultdict(dict)
|
| 168 |
+
for row in rows:
|
| 169 |
+
if row.get("pair_id"):
|
| 170 |
+
pair = pairs[row["pair_id"], row["question"]]
|
| 171 |
+
if row["sibling"] in pair:
|
| 172 |
+
raise ValueError("duplicate contrastive sibling")
|
| 173 |
+
pair[row["sibling"]] = row
|
| 174 |
+
complete = [pair for pair in pairs.values() if set(pair) == {"a", "b"}]
|
| 175 |
+
|
| 176 |
+
def truth(row):
|
| 177 |
+
return row["keys"][row["label"]]
|
| 178 |
+
|
| 179 |
+
def prediction(row):
|
| 180 |
+
return row["keys"][predicted_index(row)]
|
| 181 |
+
|
| 182 |
+
relevant = [p for p in complete if truth(p["a"]) != truth(p["b"])]
|
| 183 |
+
invariant = [p for p in complete if truth(p["a"]) == truth(p["b"])]
|
| 184 |
+
|
| 185 |
+
def both(group):
|
| 186 |
+
return (
|
| 187 |
+
fmean(all(prediction(r) == truth(r) for r in p.values()) for p in group)
|
| 188 |
+
if group
|
| 189 |
+
else None
|
| 190 |
+
)
|
| 191 |
+
|
| 192 |
+
return {
|
| 193 |
+
"pairs": len(relevant),
|
| 194 |
+
"incomplete_pairs": len(pairs) - len(complete),
|
| 195 |
+
"flip_rate": fmean(prediction(p["a"]) != prediction(p["b"]) for p in relevant)
|
| 196 |
+
if relevant
|
| 197 |
+
else None,
|
| 198 |
+
"both_correct_rate": both(relevant),
|
| 199 |
+
"invariant_pairs": len(invariant),
|
| 200 |
+
"invariant_both_correct_rate": both(invariant),
|
| 201 |
+
"invariance_rate": fmean(
|
| 202 |
+
prediction(p["a"]) == prediction(p["b"]) for p in invariant
|
| 203 |
+
)
|
| 204 |
+
if invariant
|
| 205 |
+
else None,
|
| 206 |
+
}
|
| 207 |
+
|
| 208 |
+
|
| 209 |
+
def summarize(rows):
|
| 210 |
+
clean = [r for r in rows if r["variant"] == "clean"]
|
| 211 |
+
knowable = [r for r in clean if r["source"] != "unknowable"]
|
| 212 |
+
controls = {r["id"]: r for r in clean if r["source"] == "unknowable_control"}
|
| 213 |
+
unknown = [r for r in clean if r["source"] == "unknowable"]
|
| 214 |
+
unknowable = None
|
| 215 |
+
if unknown:
|
| 216 |
+
paired = [
|
| 217 |
+
(max(r["p"]), max(controls[r["control_id"]]["p"]))
|
| 218 |
+
for r in unknown
|
| 219 |
+
if r.get("control_id") in controls
|
| 220 |
+
]
|
| 221 |
+
unknowable = {
|
| 222 |
+
"n": len(unknown),
|
| 223 |
+
"mean_max_p": fmean(max(r["p"]) for r in unknown),
|
| 224 |
+
"share_at_0_9": fmean(max(r["p"]) >= 0.9 for r in unknown),
|
| 225 |
+
"control_acc": (
|
| 226 |
+
fmean(predicted_index(r) == r["label"] for r in controls.values())
|
| 227 |
+
if controls
|
| 228 |
+
else None
|
| 229 |
+
),
|
| 230 |
+
"control_mean_max_p": (
|
| 231 |
+
fmean(max(r["p"]) for r in controls.values()) if controls else None
|
| 232 |
+
),
|
| 233 |
+
"control_share_at_0_9": (
|
| 234 |
+
fmean(max(r["p"]) >= 0.9 for r in controls.values())
|
| 235 |
+
if controls
|
| 236 |
+
else None
|
| 237 |
+
),
|
| 238 |
+
"paired_confidence_drop": fmean(c - u for u, c in paired)
|
| 239 |
+
if paired
|
| 240 |
+
else None,
|
| 241 |
+
"share_less_confident_than_control": (
|
| 242 |
+
fmean(u < c for u, c in paired) if paired else None
|
| 243 |
+
),
|
| 244 |
+
}
|
| 245 |
+
originals = {(r["id"], r["question"]): r for r in clean}
|
| 246 |
+
deltas, flips, missing = [], [], 0
|
| 247 |
+
for row in rows:
|
| 248 |
+
if row["variant"] != "permuted" or row["type"] != "choice":
|
| 249 |
+
continue
|
| 250 |
+
original = originals.get((row["parent"], row["question"]))
|
| 251 |
+
if original is None:
|
| 252 |
+
missing += 1
|
| 253 |
+
continue
|
| 254 |
+
aligned = [row["p"][row["keys"].index(key)] for key in original["keys"]]
|
| 255 |
+
deltas.append(max(abs(a - b) for a, b in zip(aligned, original["p"])))
|
| 256 |
+
flips.append(argmax(aligned) != argmax(original["p"]))
|
| 257 |
+
tasks = grouped_metrics(clean, "task")
|
| 258 |
+
return {
|
| 259 |
+
"clean": metrics(knowable),
|
| 260 |
+
"tasks": tasks,
|
| 261 |
+
"sources": grouped_metrics(clean, "source"),
|
| 262 |
+
"variants": grouped_metrics(rows, "variant"),
|
| 263 |
+
"macro_task_acc": fmean(t["acc"] for t in tasks.values()) if tasks else None,
|
| 264 |
+
"objective": -fmean(t["nll"] for t in tasks.values()) if tasks else None,
|
| 265 |
+
"paired_flip": paired_flip(clean),
|
| 266 |
+
"reference_distribution": reference_metrics(clean),
|
| 267 |
+
"unknowable": unknowable,
|
| 268 |
+
"permutation": {
|
| 269 |
+
"n": len(flips),
|
| 270 |
+
"missing_parents": missing,
|
| 271 |
+
"flip_rate": fmean(flips) if flips else None,
|
| 272 |
+
"mean_max_delta": fmean(deltas) if deltas else None,
|
| 273 |
+
},
|
| 274 |
+
"metric_policy": {
|
| 275 |
+
"nll_floor": 1e-9,
|
| 276 |
+
"ece_bins": 10,
|
| 277 |
+
"confidence_for_calibration": "maximum label probability",
|
| 278 |
+
"calibration_applied": False,
|
| 279 |
+
"unknown_accuracy_excluded": True,
|
| 280 |
+
"coverage_error_budget_is_retrospective": True,
|
| 281 |
+
"raw_sums_outside_1e_5": sum(
|
| 282 |
+
abs(r["raw_probability_sum"] - 1) > 1e-5 for r in rows
|
| 283 |
+
),
|
| 284 |
+
"returned_zeros": sum(r["zero_count"] for r in rows),
|
| 285 |
+
},
|
| 286 |
+
}
|
| 287 |
+
|
| 288 |
+
|
| 289 |
+
def reference_metrics(rows):
|
| 290 |
+
selected = [row for row in rows if row.get("reference_probs") is not None]
|
| 291 |
+
if not selected:
|
| 292 |
+
return None
|
| 293 |
+
squared, variation, kl = [], [], []
|
| 294 |
+
for row in selected:
|
| 295 |
+
p, q = row["p"], row["reference_probs"]
|
| 296 |
+
squared.append(math.fsum((a - b) ** 2 for a, b in zip(p, q)))
|
| 297 |
+
variation.append(0.5 * math.fsum(abs(a - b) for a, b in zip(p, q)))
|
| 298 |
+
kl.append(
|
| 299 |
+
math.fsum(b * math.log(b / max(a, 1e-9)) for a, b in zip(p, q) if b > 0)
|
| 300 |
+
)
|
| 301 |
+
return {
|
| 302 |
+
"n": len(selected),
|
| 303 |
+
"squared_l2": fmean(squared),
|
| 304 |
+
"total_variation": fmean(variation),
|
| 305 |
+
"kl_reference_to_model": fmean(kl),
|
| 306 |
+
"note": "Exact reference-distribution fidelity, separate from hard-label Brier/ECE.",
|
| 307 |
+
}
|
| 308 |
+
|
| 309 |
+
|
| 310 |
+
def quantile(values, q):
|
| 311 |
+
if not values:
|
| 312 |
+
return None
|
| 313 |
+
ordered = sorted(values)
|
| 314 |
+
position = (len(ordered) - 1) * q
|
| 315 |
+
lo, hi = math.floor(position), math.ceil(position)
|
| 316 |
+
return ordered[lo] + (ordered[hi] - ordered[lo]) * (position - lo)
|
| 317 |
+
|
| 318 |
+
|
| 319 |
+
def latency_summary(values):
|
| 320 |
+
return {
|
| 321 |
+
"n": len(values),
|
| 322 |
+
"mean": fmean(values) if values else None,
|
| 323 |
+
"p50": quantile(values, 0.5),
|
| 324 |
+
"p95": quantile(values, 0.95),
|
| 325 |
+
"p99": quantile(values, 0.99),
|
| 326 |
+
}
|
server/jev_adapter/benchmarks/prepare.py
ADDED
|
@@ -0,0 +1,227 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Download pinned KEV and JevBench evaluation data, without training data."""
|
| 2 |
+
|
| 3 |
+
from __future__ import annotations
|
| 4 |
+
|
| 5 |
+
import argparse
|
| 6 |
+
import json
|
| 7 |
+
from pathlib import Path
|
| 8 |
+
|
| 9 |
+
import httpx
|
| 10 |
+
|
| 11 |
+
from jev_adapter.protocol import SystemOneRequest
|
| 12 |
+
|
| 13 |
+
from .data import (
|
| 14 |
+
DEFAULT_SUITES,
|
| 15 |
+
KEV_COMMIT,
|
| 16 |
+
KEV_RAW,
|
| 17 |
+
KEV_REPOSITORY,
|
| 18 |
+
SUITES,
|
| 19 |
+
parse_partition,
|
| 20 |
+
population_counts,
|
| 21 |
+
require_sha256,
|
| 22 |
+
sha256,
|
| 23 |
+
source_provenance,
|
| 24 |
+
)
|
| 25 |
+
from .jevbench import JEVBENCH_SUITES, prepare_jevbench
|
| 26 |
+
|
| 27 |
+
|
| 28 |
+
def read_source(
|
| 29 |
+
path: str, client: httpx.Client | None, source_root: Path | None
|
| 30 |
+
) -> bytes:
|
| 31 |
+
if source_root is not None:
|
| 32 |
+
return (source_root / path).read_bytes()
|
| 33 |
+
if client is None:
|
| 34 |
+
raise ValueError("an HTTP client or --source-root is required")
|
| 35 |
+
response = client.get(f"{KEV_RAW}/{path}")
|
| 36 |
+
response.raise_for_status()
|
| 37 |
+
return response.content
|
| 38 |
+
|
| 39 |
+
|
| 40 |
+
def prepare_suite(
|
| 41 |
+
suite: str,
|
| 42 |
+
split: str,
|
| 43 |
+
output: Path,
|
| 44 |
+
*,
|
| 45 |
+
client: httpx.Client | None = None,
|
| 46 |
+
source_root: Path | None = None,
|
| 47 |
+
allow_test: bool = False,
|
| 48 |
+
limit: int | None = None,
|
| 49 |
+
) -> dict:
|
| 50 |
+
if suite not in SUITES:
|
| 51 |
+
raise ValueError(f"unknown suite: {suite}")
|
| 52 |
+
if split not in ("development", "test"):
|
| 53 |
+
raise ValueError("only development and test are evaluation partitions")
|
| 54 |
+
if split == "test" and not allow_test:
|
| 55 |
+
raise ValueError(
|
| 56 |
+
"locked test requires --allow-test; use development for iteration"
|
| 57 |
+
)
|
| 58 |
+
if limit is not None and limit < 1:
|
| 59 |
+
raise ValueError("--limit must be positive")
|
| 60 |
+
spec = SUITES[suite]
|
| 61 |
+
manifest_bytes = read_source(f"{spec.path}/manifest.json", client, source_root)
|
| 62 |
+
require_sha256(manifest_bytes, spec.manifest_sha256, f"{suite}/manifest.json")
|
| 63 |
+
upstream = json.loads(manifest_bytes)
|
| 64 |
+
expected_file = upstream["files"][f"{split}.jsonl"]
|
| 65 |
+
raw_bytes = read_source(f"{spec.path}/{split}.jsonl", client, source_root)
|
| 66 |
+
require_sha256(raw_bytes, expected_file["sha256"], f"{suite}/{split}.jsonl")
|
| 67 |
+
records = parse_partition(raw_bytes, suite, split)
|
| 68 |
+
counts = population_counts(records)
|
| 69 |
+
if counts["records"] != expected_file["records"]:
|
| 70 |
+
raise ValueError(f"upstream record count mismatch: {suite}/{split}")
|
| 71 |
+
if (
|
| 72 |
+
"questions" in expected_file
|
| 73 |
+
and counts["questions"] != expected_file["questions"]
|
| 74 |
+
):
|
| 75 |
+
raise ValueError(f"upstream question count mismatch: {suite}/{split}")
|
| 76 |
+
if not records:
|
| 77 |
+
raise ValueError(
|
| 78 |
+
f"{suite}/{split} is empty; this suite may be development-only"
|
| 79 |
+
)
|
| 80 |
+
selected = records if limit is None else records[:limit]
|
| 81 |
+
for row in selected:
|
| 82 |
+
SystemOneRequest.model_validate({"model": "validation-only", **row["record"]})
|
| 83 |
+
serialized = (
|
| 84 |
+
"".join(
|
| 85 |
+
json.dumps(row, ensure_ascii=False, allow_nan=False) + "\n"
|
| 86 |
+
for row in selected
|
| 87 |
+
)
|
| 88 |
+
).encode("utf-8")
|
| 89 |
+
directory = output / suite
|
| 90 |
+
data_path = directory / f"{split}.jsonl"
|
| 91 |
+
manifest_path = directory / f"{split}.manifest.json"
|
| 92 |
+
upstream_path = directory / "upstream.manifest.json"
|
| 93 |
+
manifest = {
|
| 94 |
+
"schema_version": 1,
|
| 95 |
+
"suite": suite,
|
| 96 |
+
"split": split,
|
| 97 |
+
"description": spec.description,
|
| 98 |
+
"data_file": data_path.name,
|
| 99 |
+
"data_sha256": sha256(serialized),
|
| 100 |
+
"upstream": {
|
| 101 |
+
"repository": KEV_REPOSITORY,
|
| 102 |
+
"commit": KEV_COMMIT,
|
| 103 |
+
"path": f"{spec.path}/{split}.jsonl",
|
| 104 |
+
"url": f"{KEV_RAW}/{spec.path}/{split}.jsonl",
|
| 105 |
+
"sha256": sha256(raw_bytes),
|
| 106 |
+
"manifest_sha256": spec.manifest_sha256,
|
| 107 |
+
},
|
| 108 |
+
"full_partition": counts,
|
| 109 |
+
"selected": population_counts(selected),
|
| 110 |
+
"selection": {
|
| 111 |
+
"method": "full" if len(selected) == len(records) else "prefix",
|
| 112 |
+
"requested_limit": limit,
|
| 113 |
+
"is_full_partition": len(selected) == len(records),
|
| 114 |
+
"note": (
|
| 115 |
+
"Full frozen partition."
|
| 116 |
+
if len(selected) == len(records)
|
| 117 |
+
else "Smoke subset only; not a full-suite result. Pairs may be incomplete."
|
| 118 |
+
),
|
| 119 |
+
},
|
| 120 |
+
"protocol": {
|
| 121 |
+
"calibration_applied": False,
|
| 122 |
+
"training_data_downloaded": False,
|
| 123 |
+
"gold_labels_sent_to_model": False,
|
| 124 |
+
"headline_variant": "clean",
|
| 125 |
+
"exclude_from_headline_sources": ["unknowable"],
|
| 126 |
+
"locked_test": split == "test",
|
| 127 |
+
"notes": list(spec.notes),
|
| 128 |
+
"upstream_context_policy": upstream.get("context"),
|
| 129 |
+
},
|
| 130 |
+
"provenance": source_provenance(records, upstream),
|
| 131 |
+
}
|
| 132 |
+
manifest_payload = (
|
| 133 |
+
json.dumps(manifest, ensure_ascii=False, indent=2) + "\n"
|
| 134 |
+
).encode()
|
| 135 |
+
# Idempotent preparation is safe; refuse to replace any changed frozen artifact.
|
| 136 |
+
for path, content in (
|
| 137 |
+
(data_path, serialized),
|
| 138 |
+
(manifest_path, manifest_payload),
|
| 139 |
+
(upstream_path, manifest_bytes),
|
| 140 |
+
):
|
| 141 |
+
if path.exists() and path.read_bytes() != content:
|
| 142 |
+
raise FileExistsError(
|
| 143 |
+
f"refusing to overwrite a different prepared artifact: {path}"
|
| 144 |
+
)
|
| 145 |
+
directory.mkdir(parents=True, exist_ok=True)
|
| 146 |
+
data_path.write_bytes(serialized)
|
| 147 |
+
manifest_path.write_bytes(manifest_payload)
|
| 148 |
+
upstream_path.write_bytes(manifest_bytes)
|
| 149 |
+
return manifest
|
| 150 |
+
|
| 151 |
+
|
| 152 |
+
def main(argv: list[str] | None = None) -> None:
|
| 153 |
+
parser = argparse.ArgumentParser(description=__doc__)
|
| 154 |
+
parser.add_argument("--output", type=Path, required=True)
|
| 155 |
+
parser.add_argument(
|
| 156 |
+
"--suite", action="append", choices=tuple(SUITES) + tuple(JEVBENCH_SUITES)
|
| 157 |
+
)
|
| 158 |
+
parser.add_argument(
|
| 159 |
+
"--split", choices=("development", "test"), default="development"
|
| 160 |
+
)
|
| 161 |
+
parser.add_argument("--allow-test", action="store_true")
|
| 162 |
+
parser.add_argument(
|
| 163 |
+
"--limit", type=int, help="deterministic prefix smoke subset per suite"
|
| 164 |
+
)
|
| 165 |
+
parser.add_argument(
|
| 166 |
+
"--source-root", type=Path, help="offline KEV checkout; same SHA256 checks"
|
| 167 |
+
)
|
| 168 |
+
parser.add_argument(
|
| 169 |
+
"--jevbench-source-root", type=Path, help="offline pinned JevBench checkout"
|
| 170 |
+
)
|
| 171 |
+
args = parser.parse_args(argv)
|
| 172 |
+
if args.split == "test" and not args.allow_test:
|
| 173 |
+
parser.error("--split test requires --allow-test")
|
| 174 |
+
if args.limit is not None and args.limit < 1:
|
| 175 |
+
parser.error("--limit must be positive")
|
| 176 |
+
suites = args.suite or (
|
| 177 |
+
DEFAULT_SUITES if args.split == "test" else (*DEFAULT_SUITES, *JEVBENCH_SUITES)
|
| 178 |
+
)
|
| 179 |
+
if args.split == "test" and any(suite in JEVBENCH_SUITES for suite in suites):
|
| 180 |
+
parser.error("JevBench supplies a public subset, not a locked test split")
|
| 181 |
+
with httpx.Client(timeout=60, follow_redirects=True) as client:
|
| 182 |
+
for suite in dict.fromkeys(suites):
|
| 183 |
+
if suite in JEVBENCH_SUITES:
|
| 184 |
+
manifest = prepare_jevbench(
|
| 185 |
+
suite,
|
| 186 |
+
args.output,
|
| 187 |
+
client=client,
|
| 188 |
+
source_root=args.jevbench_source_root,
|
| 189 |
+
limit=args.limit,
|
| 190 |
+
)
|
| 191 |
+
print(
|
| 192 |
+
json.dumps(
|
| 193 |
+
{
|
| 194 |
+
"suite": suite,
|
| 195 |
+
"split": "public",
|
| 196 |
+
"selected": manifest["selected"],
|
| 197 |
+
"is_full_partition": manifest["selection"][
|
| 198 |
+
"is_full_partition"
|
| 199 |
+
],
|
| 200 |
+
}
|
| 201 |
+
)
|
| 202 |
+
)
|
| 203 |
+
continue
|
| 204 |
+
manifest = prepare_suite(
|
| 205 |
+
suite,
|
| 206 |
+
args.split,
|
| 207 |
+
args.output,
|
| 208 |
+
client=client,
|
| 209 |
+
source_root=args.source_root,
|
| 210 |
+
allow_test=args.allow_test,
|
| 211 |
+
limit=args.limit,
|
| 212 |
+
)
|
| 213 |
+
print(
|
| 214 |
+
json.dumps(
|
| 215 |
+
{
|
| 216 |
+
"suite": suite,
|
| 217 |
+
"split": args.split,
|
| 218 |
+
"data": str(args.output / suite / f"{args.split}.jsonl"),
|
| 219 |
+
"selected": manifest["selected"],
|
| 220 |
+
"is_full_partition": manifest["selection"]["is_full_partition"],
|
| 221 |
+
}
|
| 222 |
+
)
|
| 223 |
+
)
|
| 224 |
+
|
| 225 |
+
|
| 226 |
+
if __name__ == "__main__":
|
| 227 |
+
main()
|
server/jev_adapter/benchmarks/run.py
ADDED
|
@@ -0,0 +1,441 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Evaluate frozen decisions against a running, untrained-model adapter.
|
| 2 |
+
|
| 3 |
+
No retries, generation, calibration, truncation, or training. Latency includes
|
| 4 |
+
adapter HTTP, tokenization, engine scheduling and forward work; it is NOT CUDA
|
| 5 |
+
kernel time. Model load and warmup are excluded. Failed runs have no headline.
|
| 6 |
+
"""
|
| 7 |
+
|
| 8 |
+
import argparse
|
| 9 |
+
import asyncio
|
| 10 |
+
import hashlib
|
| 11 |
+
import json
|
| 12 |
+
import math
|
| 13 |
+
import os
|
| 14 |
+
import platform
|
| 15 |
+
import random
|
| 16 |
+
import sys
|
| 17 |
+
import time
|
| 18 |
+
from collections import defaultdict
|
| 19 |
+
from datetime import UTC, datetime
|
| 20 |
+
from pathlib import Path
|
| 21 |
+
|
| 22 |
+
import httpx
|
| 23 |
+
|
| 24 |
+
from jev_adapter.protocol import SystemOneRequest
|
| 25 |
+
|
| 26 |
+
from .metrics import latency_summary, prediction_rows, summarize
|
| 27 |
+
|
| 28 |
+
|
| 29 |
+
def digest(data):
|
| 30 |
+
return hashlib.sha256(data).hexdigest()
|
| 31 |
+
|
| 32 |
+
|
| 33 |
+
def write_json(path, value):
|
| 34 |
+
path.write_text(
|
| 35 |
+
json.dumps(value, ensure_ascii=False, indent=2, allow_nan=False) + "\n"
|
| 36 |
+
)
|
| 37 |
+
|
| 38 |
+
|
| 39 |
+
def load_data(paths, model, allow_test=False, limit=None):
|
| 40 |
+
items, seen, sources = [], set(), []
|
| 41 |
+
for path in paths:
|
| 42 |
+
raw = path.read_bytes()
|
| 43 |
+
loaded = [json.loads(line) for line in raw.splitlines() if line.strip()]
|
| 44 |
+
source = {
|
| 45 |
+
"path": str(path.resolve()),
|
| 46 |
+
"sha256": digest(raw),
|
| 47 |
+
"records": len(loaded),
|
| 48 |
+
}
|
| 49 |
+
manifest_path = path.with_suffix(".manifest.json")
|
| 50 |
+
if manifest_path.exists():
|
| 51 |
+
prepared = json.loads(manifest_path.read_text())
|
| 52 |
+
if prepared.get("data_sha256") != digest(raw):
|
| 53 |
+
raise ValueError(f"prepared dataset checksum mismatch: {path}")
|
| 54 |
+
if prepared.get("selected", {}).get("records") != len(loaded):
|
| 55 |
+
raise ValueError(f"prepared dataset count mismatch: {path}")
|
| 56 |
+
source["prepared_manifest"] = prepared
|
| 57 |
+
sources.append(source)
|
| 58 |
+
for item in loaded:
|
| 59 |
+
key = item["suite"], item["split"], item["id"]
|
| 60 |
+
if key in seen:
|
| 61 |
+
raise ValueError(f"duplicate input record: {key}")
|
| 62 |
+
seen.add(key)
|
| 63 |
+
if item["split"] == "test" and not allow_test:
|
| 64 |
+
raise ValueError(
|
| 65 |
+
"test data requires --allow-test; use development first"
|
| 66 |
+
)
|
| 67 |
+
request = SystemOneRequest.model_validate(
|
| 68 |
+
{**item["record"], "model": model}
|
| 69 |
+
)
|
| 70 |
+
if set(request.questions) != set(item["expected"]):
|
| 71 |
+
raise ValueError(f"question/label mismatch: {key}")
|
| 72 |
+
for qid, expected in item["expected"].items():
|
| 73 |
+
labels, target, label = (
|
| 74 |
+
expected["labels"],
|
| 75 |
+
expected["target"],
|
| 76 |
+
expected["label"],
|
| 77 |
+
)
|
| 78 |
+
if (
|
| 79 |
+
len(labels) < 2
|
| 80 |
+
or len(labels) != len(set(labels))
|
| 81 |
+
or type(label) is not int
|
| 82 |
+
or not 0 <= label < len(labels)
|
| 83 |
+
or target != [int(i == label) for i in range(len(labels))]
|
| 84 |
+
):
|
| 85 |
+
raise ValueError(
|
| 86 |
+
"this runner requires valid one-hot objective labels"
|
| 87 |
+
)
|
| 88 |
+
question = request.questions[qid]
|
| 89 |
+
actual = (
|
| 90 |
+
list(question.criteria)
|
| 91 |
+
if question.type == "choice"
|
| 92 |
+
else ["false", "true"]
|
| 93 |
+
if question.type == "noul"
|
| 94 |
+
else [str(i) for i in range(len(question.criteria))]
|
| 95 |
+
)
|
| 96 |
+
if set(labels) != set(actual) or expected["type"] != question.type:
|
| 97 |
+
raise ValueError(
|
| 98 |
+
f"expected labels/type differ from request: {key}/{qid}"
|
| 99 |
+
)
|
| 100 |
+
reference = expected.get("reference_probs")
|
| 101 |
+
if reference is not None and (
|
| 102 |
+
not isinstance(reference, list)
|
| 103 |
+
or len(reference) != len(labels)
|
| 104 |
+
or any(
|
| 105 |
+
type(p) not in (int, float)
|
| 106 |
+
or not math.isfinite(p)
|
| 107 |
+
or not 0 <= p <= 1
|
| 108 |
+
for p in reference
|
| 109 |
+
)
|
| 110 |
+
or abs(math.fsum(reference) - 1) > 1e-8
|
| 111 |
+
):
|
| 112 |
+
raise ValueError("invalid reference probability distribution")
|
| 113 |
+
items.append(item)
|
| 114 |
+
if limit is not None:
|
| 115 |
+
items = items[:limit]
|
| 116 |
+
if not items:
|
| 117 |
+
raise ValueError("no input records")
|
| 118 |
+
return items, sources
|
| 119 |
+
|
| 120 |
+
|
| 121 |
+
async def engine_snapshot(client, base_url, cache_mode):
|
| 122 |
+
headers = {}
|
| 123 |
+
if key := os.environ.get("SGLANG_API_KEY"):
|
| 124 |
+
headers["Authorization"] = f"Bearer {key}"
|
| 125 |
+
results = {}
|
| 126 |
+
allowed = {
|
| 127 |
+
"model_path",
|
| 128 |
+
"model_type",
|
| 129 |
+
"architectures",
|
| 130 |
+
"served_model_name",
|
| 131 |
+
"revision",
|
| 132 |
+
"tokenizer_revision",
|
| 133 |
+
"dtype",
|
| 134 |
+
"quantization",
|
| 135 |
+
"tp_size",
|
| 136 |
+
"tp",
|
| 137 |
+
"version",
|
| 138 |
+
"context_length",
|
| 139 |
+
"max_total_tokens",
|
| 140 |
+
"max_running_requests",
|
| 141 |
+
"chunked_prefill_size",
|
| 142 |
+
"mem_fraction_static",
|
| 143 |
+
"disable_radix_cache",
|
| 144 |
+
"mm_preprocess_cache_size_mb",
|
| 145 |
+
"enable_prefix_mm_cache",
|
| 146 |
+
"enable_mm_global_cache",
|
| 147 |
+
"speculative_algorithm",
|
| 148 |
+
"is_generation",
|
| 149 |
+
"has_image_understanding",
|
| 150 |
+
"attention_backend",
|
| 151 |
+
"moe_runner_backend",
|
| 152 |
+
}
|
| 153 |
+
for name in ("model_info", "server_info"):
|
| 154 |
+
response = await client.get(base_url.rstrip("/") + "/" + name, headers=headers)
|
| 155 |
+
if response.status_code == 404:
|
| 156 |
+
response = await client.get(
|
| 157 |
+
base_url.rstrip("/") + "/get_" + name, headers=headers
|
| 158 |
+
)
|
| 159 |
+
response.raise_for_status()
|
| 160 |
+
value = response.json()
|
| 161 |
+
if not isinstance(value, dict):
|
| 162 |
+
raise TypeError(f"invalid engine {name}")
|
| 163 |
+
results[name] = {k: v for k, v in value.items() if k in allowed}
|
| 164 |
+
config = results["server_info"]
|
| 165 |
+
if config.get("speculative_algorithm") is not None:
|
| 166 |
+
raise ValueError("disable speculative decoding for this benchmark")
|
| 167 |
+
if cache_mode == "full-prefill":
|
| 168 |
+
if config.get("disable_radix_cache") is not True:
|
| 169 |
+
raise ValueError("full-prefill requires engine --disable-radix-cache")
|
| 170 |
+
if config.get("mm_preprocess_cache_size_mb") != 0:
|
| 171 |
+
raise ValueError("full-prefill requires --mm-preprocess-cache-size-mb 0")
|
| 172 |
+
if config.get("enable_prefix_mm_cache") or config.get("enable_mm_global_cache"):
|
| 173 |
+
raise ValueError("full-prefill requires multimodal feature caches disabled")
|
| 174 |
+
return results
|
| 175 |
+
|
| 176 |
+
|
| 177 |
+
def checked_usage(response):
|
| 178 |
+
usage, meta = response.get("usage", {}), response.get("metadata", {})
|
| 179 |
+
if type(usage.get("output_tokens")) is not int or usage["output_tokens"] != 0:
|
| 180 |
+
raise ValueError("engine must return exactly zero output tokens")
|
| 181 |
+
if type(usage.get("input_tokens")) is not int or usage["input_tokens"] < 1:
|
| 182 |
+
raise ValueError("missing or invalid input-token accounting")
|
| 183 |
+
if type(meta.get("evaluations")) is not int or meta["evaluations"] < 1:
|
| 184 |
+
raise ValueError("missing forward evaluation count")
|
| 185 |
+
elapsed = meta.get("adapter_elapsed_ms")
|
| 186 |
+
if type(elapsed) not in (int, float) or not math.isfinite(elapsed) or elapsed < 0:
|
| 187 |
+
raise ValueError("missing or invalid adapter elapsed time")
|
| 188 |
+
if not isinstance(response.get("model"), str) or not response["model"]:
|
| 189 |
+
raise ValueError("missing served model identity")
|
| 190 |
+
return usage, meta
|
| 191 |
+
|
| 192 |
+
|
| 193 |
+
def validate_launch(snapshot, launch):
|
| 194 |
+
model, engine = launch["model"], launch["engine"]
|
| 195 |
+
server, info = snapshot["server_info"], snapshot["model_info"]
|
| 196 |
+
for field, expected in {
|
| 197 |
+
"model_path": model["repo_id"],
|
| 198 |
+
"revision": model["revision"],
|
| 199 |
+
"dtype": model["dtype"],
|
| 200 |
+
"quantization": model["quantization"],
|
| 201 |
+
}.items():
|
| 202 |
+
if server.get(field) != expected:
|
| 203 |
+
raise ValueError(f"launch manifest {field} differs from running engine")
|
| 204 |
+
if info.get("model_path") != model["repo_id"]:
|
| 205 |
+
raise ValueError("launch manifest model differs from running engine model_info")
|
| 206 |
+
if not engine.get("revision") or not launch.get("profile"):
|
| 207 |
+
raise ValueError("launch manifest lacks pinned engine/profile identity")
|
| 208 |
+
|
| 209 |
+
|
| 210 |
+
async def evaluate(args, client=None):
|
| 211 |
+
items, inputs = load_data(args.data, args.model, args.allow_test, args.limit)
|
| 212 |
+
args.output.mkdir(parents=True, exist_ok=False)
|
| 213 |
+
manifest = {
|
| 214 |
+
"created_at": datetime.now(UTC).isoformat(),
|
| 215 |
+
"status": "started",
|
| 216 |
+
"python": sys.version,
|
| 217 |
+
"platform": platform.platform(),
|
| 218 |
+
"requested_model": args.model,
|
| 219 |
+
"base_url": args.base_url,
|
| 220 |
+
"engine_url": args.engine_url,
|
| 221 |
+
"data": inputs,
|
| 222 |
+
"limit": args.limit,
|
| 223 |
+
"concurrency": args.concurrency,
|
| 224 |
+
"warmup_requests": args.warmup,
|
| 225 |
+
"repeats": args.repeats,
|
| 226 |
+
"seed": args.seed,
|
| 227 |
+
"cache_mode": args.cache_mode,
|
| 228 |
+
"options": {"temperature": 1.0, "permutations": 1, "enable_thinking": False},
|
| 229 |
+
"latency_scope": "client HTTP wall time, excluding queue before dispatch; NOT GPU kernel time",
|
| 230 |
+
"quality_repeats": "first measured repetition only",
|
| 231 |
+
"assistant_prefix": args.assistant_prefix,
|
| 232 |
+
"code_sha256": {
|
| 233 |
+
str(p.relative_to(Path(__file__).parents[1])): digest(p.read_bytes())
|
| 234 |
+
for p in Path(__file__).parents[1].rglob("*.py")
|
| 235 |
+
},
|
| 236 |
+
}
|
| 237 |
+
write_json(args.output / "manifest.json", manifest)
|
| 238 |
+
owned = client is None
|
| 239 |
+
if owned:
|
| 240 |
+
client = httpx.AsyncClient(
|
| 241 |
+
timeout=args.timeout, limits=httpx.Limits(max_connections=args.concurrency)
|
| 242 |
+
)
|
| 243 |
+
headers = {}
|
| 244 |
+
if key := os.environ.get("JEV_API_KEY"):
|
| 245 |
+
headers["Authorization"] = f"Bearer {key}"
|
| 246 |
+
results, failures = [], []
|
| 247 |
+
try:
|
| 248 |
+
manifest["engine"] = await engine_snapshot(
|
| 249 |
+
client, args.engine_url, args.cache_mode
|
| 250 |
+
)
|
| 251 |
+
expected_served = manifest["engine"]["model_info"].get("served_model_name")
|
| 252 |
+
if not expected_served:
|
| 253 |
+
raise ValueError("engine did not identify its served model")
|
| 254 |
+
if args.engine_manifest:
|
| 255 |
+
manifest["launch_manifest"] = json.loads(args.engine_manifest.read_text())
|
| 256 |
+
validate_launch(manifest["engine"], manifest["launch_manifest"])
|
| 257 |
+
write_json(args.output / "manifest.json", manifest)
|
| 258 |
+
|
| 259 |
+
async def call(item):
|
| 260 |
+
request = {
|
| 261 |
+
**item["record"],
|
| 262 |
+
"model": args.model,
|
| 263 |
+
"options": {
|
| 264 |
+
"temperature": 1.0,
|
| 265 |
+
"permutations": 1,
|
| 266 |
+
"return_logprobs": True,
|
| 267 |
+
},
|
| 268 |
+
}
|
| 269 |
+
if args.assistant_prefix is not None:
|
| 270 |
+
request["options"]["assistant_prefix"] = args.assistant_prefix
|
| 271 |
+
started = time.perf_counter()
|
| 272 |
+
response = await client.post(
|
| 273 |
+
args.base_url.rstrip("/") + "/v1/systemone",
|
| 274 |
+
headers=headers,
|
| 275 |
+
json=request,
|
| 276 |
+
)
|
| 277 |
+
response.raise_for_status()
|
| 278 |
+
body = response.json()
|
| 279 |
+
elapsed = (time.perf_counter() - started) * 1000
|
| 280 |
+
usage, meta = checked_usage(body)
|
| 281 |
+
if body["model"] != expected_served:
|
| 282 |
+
raise ValueError("adapter response model differs from engine snapshot")
|
| 283 |
+
if meta["evaluations"] != len(item["record"]["questions"]):
|
| 284 |
+
raise ValueError("expected one forward evaluation per question")
|
| 285 |
+
rows = prediction_rows(item, body)
|
| 286 |
+
return {
|
| 287 |
+
"request_sha256": digest(json.dumps(request, sort_keys=True).encode()),
|
| 288 |
+
"id": item["id"],
|
| 289 |
+
"suite": item["suite"],
|
| 290 |
+
"split": item["split"],
|
| 291 |
+
"latency_ms": elapsed,
|
| 292 |
+
"adapter_ms": meta["adapter_elapsed_ms"],
|
| 293 |
+
"input_tokens": usage["input_tokens"],
|
| 294 |
+
"evaluations": meta["evaluations"],
|
| 295 |
+
"served_model": body["model"],
|
| 296 |
+
"response": body,
|
| 297 |
+
"rows": rows,
|
| 298 |
+
}
|
| 299 |
+
|
| 300 |
+
for i in range(args.warmup):
|
| 301 |
+
# Span the full input so warmup includes more than the first task.
|
| 302 |
+
item = items[(i * len(items) // max(args.warmup, 1)) % len(items)]
|
| 303 |
+
print(
|
| 304 |
+
f"Warmup {i + 1}/{args.warmup}: {item['suite']} {item['id']}",
|
| 305 |
+
flush=True,
|
| 306 |
+
)
|
| 307 |
+
await call(item)
|
| 308 |
+
jobs = [
|
| 309 |
+
(i, repeat) for repeat in range(args.repeats) for i in range(len(items))
|
| 310 |
+
]
|
| 311 |
+
random.Random(args.seed).shuffle(jobs)
|
| 312 |
+
semaphore = asyncio.Semaphore(args.concurrency)
|
| 313 |
+
started = time.perf_counter()
|
| 314 |
+
with (args.output / "predictions.jsonl").open("w") as handle:
|
| 315 |
+
|
| 316 |
+
async def job(index, repeat):
|
| 317 |
+
async with semaphore:
|
| 318 |
+
try:
|
| 319 |
+
result = await call(items[index])
|
| 320 |
+
result.update(index=index, repeat=repeat)
|
| 321 |
+
results.append(result)
|
| 322 |
+
handle.write(
|
| 323 |
+
json.dumps(result, ensure_ascii=False, allow_nan=False)
|
| 324 |
+
+ "\n"
|
| 325 |
+
)
|
| 326 |
+
handle.flush()
|
| 327 |
+
except (httpx.HTTPError, ValueError, KeyError, TypeError) as error:
|
| 328 |
+
failures.append(
|
| 329 |
+
{
|
| 330 |
+
"index": index,
|
| 331 |
+
"repeat": repeat,
|
| 332 |
+
"id": items[index]["id"],
|
| 333 |
+
"suite": items[index]["suite"],
|
| 334 |
+
"error_type": type(error).__name__,
|
| 335 |
+
"error": str(error),
|
| 336 |
+
}
|
| 337 |
+
)
|
| 338 |
+
done = len(results) + len(failures)
|
| 339 |
+
if done % 50 == 0 or done == len(jobs):
|
| 340 |
+
print(
|
| 341 |
+
f"{done}/{len(jobs)} requests; errors={len(failures)}",
|
| 342 |
+
flush=True,
|
| 343 |
+
)
|
| 344 |
+
|
| 345 |
+
await asyncio.gather(*(job(index, repeat) for index, repeat in jobs))
|
| 346 |
+
elapsed = time.perf_counter() - started
|
| 347 |
+
served = {r["served_model"] for r in results}
|
| 348 |
+
if len(served) > 1:
|
| 349 |
+
failures.append({"error": "served model changed during run"})
|
| 350 |
+
report = {
|
| 351 |
+
"status": "failed" if failures else "complete",
|
| 352 |
+
"requested_records": len(items),
|
| 353 |
+
"requested_questions": sum(len(i["expected"]) for i in items),
|
| 354 |
+
"attempted_requests": len(jobs),
|
| 355 |
+
"successful_requests": len(results),
|
| 356 |
+
"errors": len(failures),
|
| 357 |
+
"served_models": sorted(served),
|
| 358 |
+
"elapsed_measurement_s": elapsed,
|
| 359 |
+
"requests_per_second": len(results) / elapsed,
|
| 360 |
+
"questions_per_second": sum(r["evaluations"] for r in results) / elapsed,
|
| 361 |
+
"latency_ms": latency_summary([r["latency_ms"] for r in results]),
|
| 362 |
+
"adapter_ms": latency_summary([r["adapter_ms"] for r in results]),
|
| 363 |
+
"input_tokens_per_request": latency_summary(
|
| 364 |
+
[r["input_tokens"] for r in results]
|
| 365 |
+
),
|
| 366 |
+
"suites": {},
|
| 367 |
+
"latency_scope": manifest["latency_scope"],
|
| 368 |
+
"partial_dataset": args.limit is not None
|
| 369 |
+
or any(
|
| 370 |
+
not source.get("prepared_manifest", {})
|
| 371 |
+
.get("selection", {})
|
| 372 |
+
.get("is_full_partition", False)
|
| 373 |
+
for source in inputs
|
| 374 |
+
),
|
| 375 |
+
}
|
| 376 |
+
if not failures:
|
| 377 |
+
grouped, times = defaultdict(list), defaultdict(list)
|
| 378 |
+
for result in sorted(results, key=lambda r: (r["index"], r["repeat"])):
|
| 379 |
+
key = result["suite"] + "/" + result["split"]
|
| 380 |
+
times[key].append(result["latency_ms"])
|
| 381 |
+
if result["repeat"] == 0:
|
| 382 |
+
grouped[key].extend(result["rows"])
|
| 383 |
+
report["suites"] = {
|
| 384 |
+
key: {**summarize(rows), "latency_ms": latency_summary(times[key])}
|
| 385 |
+
for key, rows in grouped.items()
|
| 386 |
+
}
|
| 387 |
+
write_json(args.output / "report.json", report)
|
| 388 |
+
if failures:
|
| 389 |
+
write_json(args.output / "failures.json", failures)
|
| 390 |
+
manifest.update(status=report["status"], served_models=sorted(served))
|
| 391 |
+
write_json(args.output / "manifest.json", manifest)
|
| 392 |
+
return report
|
| 393 |
+
except BaseException as error:
|
| 394 |
+
manifest.update(
|
| 395 |
+
status="failed", failure_type=type(error).__name__, failure=str(error)
|
| 396 |
+
)
|
| 397 |
+
write_json(args.output / "manifest.json", manifest)
|
| 398 |
+
raise
|
| 399 |
+
finally:
|
| 400 |
+
if owned:
|
| 401 |
+
await client.aclose()
|
| 402 |
+
|
| 403 |
+
|
| 404 |
+
def main():
|
| 405 |
+
parser = argparse.ArgumentParser(description=__doc__)
|
| 406 |
+
parser.add_argument("--data", type=Path, action="append", required=True)
|
| 407 |
+
parser.add_argument("--output", type=Path, required=True)
|
| 408 |
+
parser.add_argument("--base-url", default="http://127.0.0.1:30120")
|
| 409 |
+
parser.add_argument("--engine-url", default="http://127.0.0.1:30000")
|
| 410 |
+
parser.add_argument("--engine-manifest", type=Path)
|
| 411 |
+
parser.add_argument("--model", default="decision-model")
|
| 412 |
+
parser.add_argument("--concurrency", type=int, default=1)
|
| 413 |
+
parser.add_argument("--warmup", type=int, default=20)
|
| 414 |
+
parser.add_argument("--repeats", type=int, default=1)
|
| 415 |
+
parser.add_argument("--seed", type=int, default=42)
|
| 416 |
+
parser.add_argument("--limit", type=int)
|
| 417 |
+
parser.add_argument("--timeout", type=float, default=120)
|
| 418 |
+
parser.add_argument("--assistant-prefix")
|
| 419 |
+
parser.add_argument("--allow-test", action="store_true")
|
| 420 |
+
parser.add_argument(
|
| 421 |
+
"--cache-mode",
|
| 422 |
+
choices=["full-prefill", "server-default"],
|
| 423 |
+
default="full-prefill",
|
| 424 |
+
)
|
| 425 |
+
args = parser.parse_args()
|
| 426 |
+
if (
|
| 427 |
+
args.concurrency < 1
|
| 428 |
+
or args.repeats < 1
|
| 429 |
+
or args.warmup < 0
|
| 430 |
+
or (args.limit is not None and args.limit < 1)
|
| 431 |
+
or not math.isfinite(args.timeout)
|
| 432 |
+
or args.timeout <= 0
|
| 433 |
+
):
|
| 434 |
+
parser.error("counts and timeout must be positive; warmup may be zero")
|
| 435 |
+
report = asyncio.run(evaluate(args))
|
| 436 |
+
print(json.dumps(report, ensure_ascii=False, indent=2))
|
| 437 |
+
raise SystemExit(0 if report["status"] == "complete" else 1)
|
| 438 |
+
|
| 439 |
+
|
| 440 |
+
if __name__ == "__main__":
|
| 441 |
+
main()
|
server/tests/test_benchmark_data.py
ADDED
|
@@ -0,0 +1,220 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Check benchmark integrity and prevent labels from entering inference requests."""
|
| 2 |
+
|
| 3 |
+
import copy
|
| 4 |
+
import json
|
| 5 |
+
import tempfile
|
| 6 |
+
import unittest
|
| 7 |
+
from pathlib import Path
|
| 8 |
+
from unittest.mock import patch
|
| 9 |
+
|
| 10 |
+
import httpx
|
| 11 |
+
|
| 12 |
+
from jev_adapter.benchmarks.data import (
|
| 13 |
+
SuiteSpec,
|
| 14 |
+
normalize_record,
|
| 15 |
+
parse_partition,
|
| 16 |
+
population_counts,
|
| 17 |
+
sha256,
|
| 18 |
+
)
|
| 19 |
+
from jev_adapter.benchmarks.prepare import prepare_suite
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
def fixture_record(name="example/1"):
|
| 23 |
+
return {
|
| 24 |
+
"state": {"message": "Needs a refund", "count": 4},
|
| 25 |
+
"questions": {
|
| 26 |
+
"queue": {
|
| 27 |
+
"type": "choice",
|
| 28 |
+
"instructions": "Choose the queue.",
|
| 29 |
+
"criteria": {"other": None, "billing": "Refunds"},
|
| 30 |
+
"label": "billing",
|
| 31 |
+
"src": "routing",
|
| 32 |
+
},
|
| 33 |
+
"urgent": {
|
| 34 |
+
"type": "noul",
|
| 35 |
+
"instructions": "Is the message urgent?",
|
| 36 |
+
"label": False,
|
| 37 |
+
"src": "urgency",
|
| 38 |
+
},
|
| 39 |
+
"priority": {
|
| 40 |
+
"type": "score",
|
| 41 |
+
"instructions": "Select priority.",
|
| 42 |
+
"criteria": ["low", "medium", "high"],
|
| 43 |
+
"label": 2,
|
| 44 |
+
"src": "priority",
|
| 45 |
+
},
|
| 46 |
+
},
|
| 47 |
+
"_meta": {
|
| 48 |
+
"id": name,
|
| 49 |
+
"source": "example",
|
| 50 |
+
"variant": "clean",
|
| 51 |
+
"group_id": "pair-1",
|
| 52 |
+
"pair_id": "pair-1",
|
| 53 |
+
"sibling": "a",
|
| 54 |
+
},
|
| 55 |
+
}
|
| 56 |
+
|
| 57 |
+
|
| 58 |
+
def fake_source(records):
|
| 59 |
+
payload = ("".join(json.dumps(r) + "\n" for r in records)).encode()
|
| 60 |
+
manifest = json.dumps(
|
| 61 |
+
{
|
| 62 |
+
"files": {
|
| 63 |
+
name: {
|
| 64 |
+
"sha256": sha256(payload),
|
| 65 |
+
"records": len(records),
|
| 66 |
+
"questions": sum(len(r["questions"]) for r in records),
|
| 67 |
+
}
|
| 68 |
+
for name in ("development.jsonl", "test.jsonl")
|
| 69 |
+
},
|
| 70 |
+
"dataset_revisions": {},
|
| 71 |
+
}
|
| 72 |
+
).encode()
|
| 73 |
+
spec = SuiteSpec("evals/example", sha256(manifest), "Test fixture")
|
| 74 |
+
return spec, manifest, payload
|
| 75 |
+
|
| 76 |
+
|
| 77 |
+
class TestBenchmarkNormalization(unittest.TestCase):
|
| 78 |
+
def test_all_question_types_preserve_order_and_hide_gold(self):
|
| 79 |
+
original = fixture_record()
|
| 80 |
+
before = copy.deepcopy(original)
|
| 81 |
+
row = normalize_record(original, "example", "development")
|
| 82 |
+
self.assertEqual(original, before)
|
| 83 |
+
self.assertEqual(
|
| 84 |
+
list(row["record"]["questions"]), ["queue", "urgent", "priority"]
|
| 85 |
+
)
|
| 86 |
+
self.assertNotIn("model", row["record"])
|
| 87 |
+
self.assertNotIn("_meta", row["record"])
|
| 88 |
+
for q in row["record"]["questions"].values():
|
| 89 |
+
self.assertNotIn("label", q)
|
| 90 |
+
self.assertNotIn("src", q)
|
| 91 |
+
self.assertEqual(row["expected"]["queue"]["labels"], ["other", "billing"])
|
| 92 |
+
self.assertEqual(row["expected"]["queue"]["target"], [0, 1])
|
| 93 |
+
self.assertEqual(row["expected"]["urgent"]["labels"], ["false", "true"])
|
| 94 |
+
self.assertEqual(row["expected"]["urgent"]["target"], [1, 0])
|
| 95 |
+
self.assertEqual(row["expected"]["priority"]["target"], [0, 0, 1])
|
| 96 |
+
self.assertIsNone(row["record"]["questions"]["queue"]["criteria"]["other"])
|
| 97 |
+
self.assertEqual(row["metadata"]["pair_id"], "pair-1")
|
| 98 |
+
|
| 99 |
+
def test_large_choice_space_is_not_truncated(self):
|
| 100 |
+
raw = fixture_record()
|
| 101 |
+
raw["questions"] = {
|
| 102 |
+
"intent": {
|
| 103 |
+
"type": "choice",
|
| 104 |
+
"instructions": "Which intent?",
|
| 105 |
+
"criteria": {f"intent_{i}": None for i in range(78)},
|
| 106 |
+
"label": "intent_77",
|
| 107 |
+
}
|
| 108 |
+
}
|
| 109 |
+
row = normalize_record(raw, "example", "development")
|
| 110 |
+
self.assertEqual(len(row["expected"]["intent"]["labels"]), 78)
|
| 111 |
+
self.assertEqual(row["expected"]["intent"]["label"], 77)
|
| 112 |
+
|
| 113 |
+
def test_duplicate_ids_and_malformed_targets_fail(self):
|
| 114 |
+
raw = fixture_record()
|
| 115 |
+
payload = (json.dumps(raw) + "\n") * 2
|
| 116 |
+
with self.assertRaisesRegex(ValueError, "duplicate record"):
|
| 117 |
+
parse_partition(payload.encode(), "example", "development")
|
| 118 |
+
for question, label in (
|
| 119 |
+
("queue", "absent"),
|
| 120 |
+
("urgent", "false"),
|
| 121 |
+
("priority", 9),
|
| 122 |
+
):
|
| 123 |
+
modified = copy.deepcopy(raw)
|
| 124 |
+
modified["questions"][question]["label"] = label
|
| 125 |
+
with self.subTest(question=question), self.assertRaises(ValueError):
|
| 126 |
+
normalize_record(modified, "example", "development")
|
| 127 |
+
|
| 128 |
+
def test_population_separates_perturbations_and_unknowable(self):
|
| 129 |
+
raws = [fixture_record(str(i)) for i in range(3)]
|
| 130 |
+
raws[1]["_meta"]["variant"] = "permuted"
|
| 131 |
+
raws[2]["_meta"]["source"] = "unknowable"
|
| 132 |
+
counts = population_counts(
|
| 133 |
+
[normalize_record(raw, "example", "development") for raw in raws]
|
| 134 |
+
)
|
| 135 |
+
self.assertEqual(counts["records"], 3)
|
| 136 |
+
self.assertEqual(counts["questions"], 9)
|
| 137 |
+
self.assertEqual(counts["clean_questions"], 6)
|
| 138 |
+
self.assertEqual(counts["headline_questions"], 3)
|
| 139 |
+
|
| 140 |
+
|
| 141 |
+
class TestBenchmarkPreparation(unittest.TestCase):
|
| 142 |
+
def test_verified_download_subset_and_idempotency(self):
|
| 143 |
+
spec, manifest, payload = fake_source(
|
| 144 |
+
[fixture_record("one"), fixture_record("two")]
|
| 145 |
+
)
|
| 146 |
+
fetched = []
|
| 147 |
+
|
| 148 |
+
def transport(request):
|
| 149 |
+
fetched.append(request.url.path)
|
| 150 |
+
return httpx.Response(
|
| 151 |
+
200,
|
| 152 |
+
content=manifest
|
| 153 |
+
if request.url.path.endswith("manifest.json")
|
| 154 |
+
else payload,
|
| 155 |
+
)
|
| 156 |
+
|
| 157 |
+
with (
|
| 158 |
+
tempfile.TemporaryDirectory() as tmp,
|
| 159 |
+
patch.dict("jev_adapter.benchmarks.prepare.SUITES", {"example": spec}),
|
| 160 |
+
httpx.Client(transport=httpx.MockTransport(transport)) as client,
|
| 161 |
+
):
|
| 162 |
+
out = Path(tmp)
|
| 163 |
+
result = prepare_suite(
|
| 164 |
+
"example", "development", out, client=client, limit=1
|
| 165 |
+
)
|
| 166 |
+
again = prepare_suite("example", "development", out, client=client, limit=1)
|
| 167 |
+
self.assertEqual(result, again)
|
| 168 |
+
self.assertFalse(result["selection"]["is_full_partition"])
|
| 169 |
+
self.assertEqual(result["full_partition"]["questions"], 6)
|
| 170 |
+
self.assertEqual(result["selected"]["questions"], 3)
|
| 171 |
+
self.assertEqual(
|
| 172 |
+
result["data_sha256"],
|
| 173 |
+
sha256((out / "example/development.jsonl").read_bytes()),
|
| 174 |
+
)
|
| 175 |
+
self.assertTrue(all("train" not in path for path in fetched))
|
| 176 |
+
with self.assertRaises(FileExistsError):
|
| 177 |
+
prepare_suite("example", "development", out, client=client)
|
| 178 |
+
|
| 179 |
+
def test_tampered_partition_rejected_without_writing(self):
|
| 180 |
+
spec, manifest, payload = fake_source([fixture_record()])
|
| 181 |
+
|
| 182 |
+
def transport(request):
|
| 183 |
+
return httpx.Response(
|
| 184 |
+
200,
|
| 185 |
+
content=manifest
|
| 186 |
+
if request.url.path.endswith("manifest.json")
|
| 187 |
+
else payload + b" ",
|
| 188 |
+
)
|
| 189 |
+
|
| 190 |
+
with (
|
| 191 |
+
tempfile.TemporaryDirectory() as tmp,
|
| 192 |
+
patch.dict("jev_adapter.benchmarks.prepare.SUITES", {"example": spec}),
|
| 193 |
+
httpx.Client(transport=httpx.MockTransport(transport)) as client,
|
| 194 |
+
):
|
| 195 |
+
out = Path(tmp)
|
| 196 |
+
with self.assertRaisesRegex(ValueError, "SHA256 mismatch"):
|
| 197 |
+
prepare_suite("example", "development", out, client=client)
|
| 198 |
+
self.assertFalse((out / "example").exists())
|
| 199 |
+
|
| 200 |
+
def test_offline_source_has_same_integrity_and_test_gate(self):
|
| 201 |
+
spec, manifest, payload = fake_source([fixture_record()])
|
| 202 |
+
with (
|
| 203 |
+
tempfile.TemporaryDirectory() as tmp,
|
| 204 |
+
patch.dict("jev_adapter.benchmarks.prepare.SUITES", {"example": spec}),
|
| 205 |
+
):
|
| 206 |
+
root = Path(tmp) / "source"
|
| 207 |
+
directory = root / "evals/example"
|
| 208 |
+
directory.mkdir(parents=True)
|
| 209 |
+
(directory / "manifest.json").write_bytes(manifest)
|
| 210 |
+
(directory / "test.jsonl").write_bytes(payload)
|
| 211 |
+
out = Path(tmp) / "out"
|
| 212 |
+
with self.assertRaisesRegex(ValueError, "locked test"):
|
| 213 |
+
prepare_suite("example", "test", out, source_root=root)
|
| 214 |
+
result = prepare_suite(
|
| 215 |
+
"example", "test", out, source_root=root, allow_test=True
|
| 216 |
+
)
|
| 217 |
+
self.assertTrue(result["protocol"]["locked_test"])
|
| 218 |
+
(directory / "manifest.json").write_bytes(manifest + b" ")
|
| 219 |
+
with self.assertRaisesRegex(ValueError, "SHA256 mismatch"):
|
| 220 |
+
prepare_suite("example", "test", out, source_root=root, allow_test=True)
|
server/tests/test_disconnect_cleanup.py
ADDED
|
@@ -0,0 +1,150 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""A completed score must not wait forever for a cancelled disconnect poll."""
|
| 2 |
+
|
| 3 |
+
import asyncio
|
| 4 |
+
|
| 5 |
+
import pytest
|
| 6 |
+
|
| 7 |
+
from jev_adapter.backend import AdapterError, ScoringResult
|
| 8 |
+
from jev_adapter.protocol import SystemOneRequest
|
| 9 |
+
from jev_adapter.server import create_app
|
| 10 |
+
|
| 11 |
+
|
| 12 |
+
class ControlledBackend:
|
| 13 |
+
model = "test-model"
|
| 14 |
+
|
| 15 |
+
def __init__(self):
|
| 16 |
+
self.release = asyncio.Event()
|
| 17 |
+
self.entered = asyncio.Event()
|
| 18 |
+
self.cancelled = asyncio.Event()
|
| 19 |
+
|
| 20 |
+
def labels(self, count):
|
| 21 |
+
return tuple("AB"[:count]), tuple(range(count))
|
| 22 |
+
|
| 23 |
+
async def evaluate(self, prompt, images, labels, token_ids, assistant_prefix):
|
| 24 |
+
self.entered.set()
|
| 25 |
+
try:
|
| 26 |
+
await self.release.wait()
|
| 27 |
+
except asyncio.CancelledError:
|
| 28 |
+
self.cancelled.set()
|
| 29 |
+
raise
|
| 30 |
+
return ScoringResult((-0.1, -2.0), 10)
|
| 31 |
+
|
| 32 |
+
|
| 33 |
+
class CancellationSwallowingRequest:
|
| 34 |
+
"""Reproduce a request probe whose internal cancellation scope wins a race.
|
| 35 |
+
|
| 36 |
+
CancelledError can be suppressed within a probe, leaving its caller alive.
|
| 37 |
+
A real Request.is_disconnected uses a self-cancelling AnyIO CancelScope;
|
| 38 |
+
deliberately control that boundary here rather than rely on scheduler luck.
|
| 39 |
+
"""
|
| 40 |
+
|
| 41 |
+
def __init__(self):
|
| 42 |
+
self.entered = asyncio.Event()
|
| 43 |
+
self.swallowed = asyncio.Event()
|
| 44 |
+
self.probe_task = None
|
| 45 |
+
self.calls = 0
|
| 46 |
+
|
| 47 |
+
async def is_disconnected(self):
|
| 48 |
+
self.probe_task = asyncio.current_task()
|
| 49 |
+
self.calls += 1
|
| 50 |
+
self.entered.set()
|
| 51 |
+
try:
|
| 52 |
+
await asyncio.Future()
|
| 53 |
+
except asyncio.CancelledError:
|
| 54 |
+
self.swallowed.set()
|
| 55 |
+
return False
|
| 56 |
+
|
| 57 |
+
|
| 58 |
+
def endpoint_and_body(backend):
|
| 59 |
+
app = create_app(backend)
|
| 60 |
+
endpoint = next(
|
| 61 |
+
route.endpoint
|
| 62 |
+
for route in app.routes
|
| 63 |
+
if getattr(route, "path", None) == "/v1/systemone"
|
| 64 |
+
)
|
| 65 |
+
body = SystemOneRequest.model_validate(
|
| 66 |
+
{
|
| 67 |
+
"model": backend.model,
|
| 68 |
+
"state": "A billing question.",
|
| 69 |
+
"questions": {
|
| 70 |
+
"route": {
|
| 71 |
+
"type": "choice",
|
| 72 |
+
"instructions": "Choose the team.",
|
| 73 |
+
"criteria": {"billing": None, "technical": None},
|
| 74 |
+
}
|
| 75 |
+
},
|
| 76 |
+
}
|
| 77 |
+
)
|
| 78 |
+
return endpoint, body
|
| 79 |
+
|
| 80 |
+
|
| 81 |
+
@pytest.mark.asyncio
|
| 82 |
+
async def test_completed_score_returns_when_disconnect_probe_swallows_cancel():
|
| 83 |
+
backend = ControlledBackend()
|
| 84 |
+
endpoint, body = endpoint_and_body(backend)
|
| 85 |
+
request = CancellationSwallowingRequest()
|
| 86 |
+
response = asyncio.create_task(endpoint(body, request))
|
| 87 |
+
try:
|
| 88 |
+
await asyncio.wait_for(request.entered.wait(), timeout=1)
|
| 89 |
+
await asyncio.wait_for(backend.entered.wait(), timeout=1)
|
| 90 |
+
backend.release.set()
|
| 91 |
+
await asyncio.wait_for(request.swallowed.wait(), timeout=1)
|
| 92 |
+
done, _ = await asyncio.wait({response}, timeout=0.1)
|
| 93 |
+
assert response in done, "completed inference is stuck cleaning up its watcher"
|
| 94 |
+
result = response.result()
|
| 95 |
+
assert result["answers"]["route"]["choice"] == "billing"
|
| 96 |
+
assert request.calls == 1
|
| 97 |
+
assert request.probe_task.done()
|
| 98 |
+
finally:
|
| 99 |
+
# Do not let a deliberately cancellation-resistant fake leak after red.
|
| 100 |
+
request.is_disconnected = disconnected_now
|
| 101 |
+
if request.probe_task is not None:
|
| 102 |
+
request.probe_task.cancel()
|
| 103 |
+
response.cancel()
|
| 104 |
+
await asyncio.gather(response, return_exceptions=True)
|
| 105 |
+
|
| 106 |
+
|
| 107 |
+
async def disconnected_now():
|
| 108 |
+
return True
|
| 109 |
+
|
| 110 |
+
|
| 111 |
+
@pytest.mark.asyncio
|
| 112 |
+
async def test_client_disconnect_still_cancels_unfinished_inference():
|
| 113 |
+
backend = ControlledBackend()
|
| 114 |
+
endpoint, body = endpoint_and_body(backend)
|
| 115 |
+
|
| 116 |
+
class DisconnectedRequest:
|
| 117 |
+
async def is_disconnected(self):
|
| 118 |
+
await backend.entered.wait()
|
| 119 |
+
return True
|
| 120 |
+
|
| 121 |
+
with pytest.raises(AdapterError) as error:
|
| 122 |
+
await asyncio.wait_for(endpoint(body, DisconnectedRequest()), timeout=1)
|
| 123 |
+
assert error.value.code == "client_disconnected"
|
| 124 |
+
assert error.value.status == 499
|
| 125 |
+
assert backend.cancelled.is_set()
|
| 126 |
+
|
| 127 |
+
|
| 128 |
+
@pytest.mark.asyncio
|
| 129 |
+
async def test_cancelled_handler_joins_work_and_cancellation_resistant_watcher():
|
| 130 |
+
backend = ControlledBackend()
|
| 131 |
+
endpoint, body = endpoint_and_body(backend)
|
| 132 |
+
request = CancellationSwallowingRequest()
|
| 133 |
+
response = asyncio.create_task(endpoint(body, request))
|
| 134 |
+
try:
|
| 135 |
+
await asyncio.wait_for(request.entered.wait(), timeout=1)
|
| 136 |
+
await asyncio.wait_for(backend.entered.wait(), timeout=1)
|
| 137 |
+
response.cancel()
|
| 138 |
+
done, _ = await asyncio.wait({response}, timeout=0.1)
|
| 139 |
+
assert response in done, "cancelled handler did not release its child tasks"
|
| 140 |
+
with pytest.raises(asyncio.CancelledError):
|
| 141 |
+
response.result()
|
| 142 |
+
assert backend.cancelled.is_set()
|
| 143 |
+
assert request.swallowed.is_set()
|
| 144 |
+
assert request.probe_task.done()
|
| 145 |
+
finally:
|
| 146 |
+
request.is_disconnected = disconnected_now
|
| 147 |
+
if request.probe_task is not None:
|
| 148 |
+
request.probe_task.cancel()
|
| 149 |
+
response.cancel()
|
| 150 |
+
await asyncio.gather(response, return_exceptions=True)
|
server/tests/test_main.py
ADDED
|
@@ -0,0 +1,368 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""CLI wiring tests for --prompt-wording (argparse choices/env default, and
|
| 2 |
+
plumbing into SGLangBackend + create_app). uvicorn.run is monkeypatched so
|
| 3 |
+
main() never actually binds a socket; SGLangBackend/create_app/NativeTokenizer
|
| 4 |
+
are monkeypatched to capture their constructor arguments instead of talking to
|
| 5 |
+
a real engine or loading real tokenizer files (transformers is an optional,
|
| 6 |
+
not-installed-here dependency for this default venv)."""
|
| 7 |
+
|
| 8 |
+
import sys
|
| 9 |
+
|
| 10 |
+
import pytest
|
| 11 |
+
|
| 12 |
+
import jev_adapter.__main__ as main_module
|
| 13 |
+
import jev_adapter.native_tokenizer as native_tokenizer_module
|
| 14 |
+
import jev_adapter.server as server_module
|
| 15 |
+
import jev_adapter.sglang as sglang_module
|
| 16 |
+
|
| 17 |
+
|
| 18 |
+
class Capture:
|
| 19 |
+
def __init__(self):
|
| 20 |
+
self.backend_kwargs = None
|
| 21 |
+
self.app_kwargs = None
|
| 22 |
+
|
| 23 |
+
def fake_backend(self, *args, **kwargs):
|
| 24 |
+
self.backend_kwargs = kwargs
|
| 25 |
+
return object()
|
| 26 |
+
|
| 27 |
+
def fake_create_app(self, backend, **kwargs):
|
| 28 |
+
self.app_kwargs = kwargs
|
| 29 |
+
return object()
|
| 30 |
+
|
| 31 |
+
|
| 32 |
+
@pytest.fixture
|
| 33 |
+
def capture(monkeypatch):
|
| 34 |
+
capture = Capture()
|
| 35 |
+
monkeypatch.setattr(sglang_module, "SGLangBackend", capture.fake_backend)
|
| 36 |
+
monkeypatch.setattr(server_module, "create_app", capture.fake_create_app)
|
| 37 |
+
monkeypatch.setattr("uvicorn.run", lambda app, **kwargs: None)
|
| 38 |
+
return capture
|
| 39 |
+
|
| 40 |
+
|
| 41 |
+
def run_main(monkeypatch, argv, env=None):
|
| 42 |
+
monkeypatch.setattr(sys, "argv", ["jev-adapter", *argv])
|
| 43 |
+
for key in (
|
| 44 |
+
"JEV_PROMPT_WORDING",
|
| 45 |
+
"JEV_DEFAULT_TEMPERATURE",
|
| 46 |
+
"JEV_NATIVE_SYSTEM_PROMPT",
|
| 47 |
+
):
|
| 48 |
+
monkeypatch.delenv(key, raising=False)
|
| 49 |
+
for key, value in (env or {}).items():
|
| 50 |
+
monkeypatch.setenv(key, value)
|
| 51 |
+
main_module.main()
|
| 52 |
+
|
| 53 |
+
|
| 54 |
+
def test_prompt_wording_defaults_to_served(monkeypatch, capture):
|
| 55 |
+
run_main(monkeypatch, ["--model", "decision-model"])
|
| 56 |
+
assert capture.backend_kwargs["prompt_wording"] == "served"
|
| 57 |
+
assert capture.backend_kwargs["native_system_prompt"] is None
|
| 58 |
+
assert capture.app_kwargs["prompt_wording"] == "served"
|
| 59 |
+
|
| 60 |
+
|
| 61 |
+
def test_prompt_wording_flag_selects_native(monkeypatch, capture):
|
| 62 |
+
run_main(monkeypatch, ["--model", "decision-model", "--prompt-wording", "native"])
|
| 63 |
+
assert capture.backend_kwargs["prompt_wording"] == "native"
|
| 64 |
+
assert capture.app_kwargs["prompt_wording"] == "native"
|
| 65 |
+
|
| 66 |
+
|
| 67 |
+
def test_prompt_wording_env_var_sets_the_default(monkeypatch, capture):
|
| 68 |
+
run_main(
|
| 69 |
+
monkeypatch,
|
| 70 |
+
["--model", "decision-model"],
|
| 71 |
+
env={"JEV_PROMPT_WORDING": "native"},
|
| 72 |
+
)
|
| 73 |
+
assert capture.backend_kwargs["prompt_wording"] == "native"
|
| 74 |
+
assert capture.app_kwargs["prompt_wording"] == "native"
|
| 75 |
+
|
| 76 |
+
|
| 77 |
+
def test_explicit_flag_overrides_env_var(monkeypatch, capture):
|
| 78 |
+
run_main(
|
| 79 |
+
monkeypatch,
|
| 80 |
+
["--model", "decision-model", "--prompt-wording", "served"],
|
| 81 |
+
env={"JEV_PROMPT_WORDING": "native"},
|
| 82 |
+
)
|
| 83 |
+
assert capture.backend_kwargs["prompt_wording"] == "served"
|
| 84 |
+
|
| 85 |
+
|
| 86 |
+
def test_invalid_prompt_wording_choice_rejected(monkeypatch, capture):
|
| 87 |
+
monkeypatch.setattr(sys, "argv", ["jev-adapter", "--model", "m", "--prompt-wording", "bogus"])
|
| 88 |
+
with pytest.raises(SystemExit):
|
| 89 |
+
main_module.main()
|
| 90 |
+
|
| 91 |
+
|
| 92 |
+
def test_native_wording_without_tokenizer_model_serves_text_only_and_warns(
|
| 93 |
+
monkeypatch, capture, caplog
|
| 94 |
+
):
|
| 95 |
+
with caplog.at_level("WARNING"):
|
| 96 |
+
run_main(monkeypatch, ["--model", "decision-model", "--prompt-wording", "native"])
|
| 97 |
+
assert capture.backend_kwargs["native_system_prompt"] is None
|
| 98 |
+
assert any("native" in record.message for record in caplog.records)
|
| 99 |
+
|
| 100 |
+
|
| 101 |
+
def test_native_wording_with_tokenizer_model_extracts_the_system_prompt(
|
| 102 |
+
monkeypatch, capture
|
| 103 |
+
):
|
| 104 |
+
calls = []
|
| 105 |
+
|
| 106 |
+
class FakeNativeTokenizer:
|
| 107 |
+
@classmethod
|
| 108 |
+
def from_pretrained(cls, model, revision):
|
| 109 |
+
calls.append(("from_pretrained", model, revision))
|
| 110 |
+
return object()
|
| 111 |
+
|
| 112 |
+
@classmethod
|
| 113 |
+
def native_default_system_prompt(cls, model, revision):
|
| 114 |
+
calls.append(("native_default_system_prompt", model, revision))
|
| 115 |
+
return "Default system text."
|
| 116 |
+
|
| 117 |
+
monkeypatch.setattr(native_tokenizer_module, "NativeTokenizer", FakeNativeTokenizer)
|
| 118 |
+
run_main(
|
| 119 |
+
monkeypatch,
|
| 120 |
+
[
|
| 121 |
+
"--model",
|
| 122 |
+
"decision-model",
|
| 123 |
+
"--prompt-wording",
|
| 124 |
+
"native",
|
| 125 |
+
"--tokenizer-model",
|
| 126 |
+
"mistralai/Ministral-3-8B-Instruct-2512-BF16",
|
| 127 |
+
"--tokenizer-revision",
|
| 128 |
+
"f" * 40,
|
| 129 |
+
],
|
| 130 |
+
)
|
| 131 |
+
assert calls == [
|
| 132 |
+
("from_pretrained", "mistralai/Ministral-3-8B-Instruct-2512-BF16", "f" * 40),
|
| 133 |
+
(
|
| 134 |
+
"native_default_system_prompt",
|
| 135 |
+
"mistralai/Ministral-3-8B-Instruct-2512-BF16",
|
| 136 |
+
"f" * 40,
|
| 137 |
+
),
|
| 138 |
+
]
|
| 139 |
+
assert capture.backend_kwargs["native_system_prompt"] == "Default system text."
|
| 140 |
+
|
| 141 |
+
|
| 142 |
+
def test_served_wording_with_tokenizer_model_never_extracts_a_system_prompt(
|
| 143 |
+
monkeypatch, capture
|
| 144 |
+
):
|
| 145 |
+
calls = []
|
| 146 |
+
|
| 147 |
+
class FakeNativeTokenizer:
|
| 148 |
+
@classmethod
|
| 149 |
+
def from_pretrained(cls, model, revision):
|
| 150 |
+
calls.append("from_pretrained")
|
| 151 |
+
return object()
|
| 152 |
+
|
| 153 |
+
@classmethod
|
| 154 |
+
def native_default_system_prompt(cls, model, revision):
|
| 155 |
+
calls.append("native_default_system_prompt")
|
| 156 |
+
return "should not be reached"
|
| 157 |
+
|
| 158 |
+
monkeypatch.setattr(native_tokenizer_module, "NativeTokenizer", FakeNativeTokenizer)
|
| 159 |
+
run_main(
|
| 160 |
+
monkeypatch,
|
| 161 |
+
[
|
| 162 |
+
"--model",
|
| 163 |
+
"decision-model",
|
| 164 |
+
"--tokenizer-model",
|
| 165 |
+
"org/model",
|
| 166 |
+
"--tokenizer-revision",
|
| 167 |
+
"a" * 40,
|
| 168 |
+
],
|
| 169 |
+
)
|
| 170 |
+
assert calls == ["from_pretrained"]
|
| 171 |
+
assert capture.backend_kwargs["prompt_wording"] == "served"
|
| 172 |
+
assert capture.backend_kwargs["native_system_prompt"] is None
|
| 173 |
+
|
| 174 |
+
|
| 175 |
+
def test_native_system_prompt_defaults_to_auto(monkeypatch, capture):
|
| 176 |
+
calls = []
|
| 177 |
+
|
| 178 |
+
class FakeNativeTokenizer:
|
| 179 |
+
@classmethod
|
| 180 |
+
def from_pretrained(cls, model, revision):
|
| 181 |
+
return object()
|
| 182 |
+
|
| 183 |
+
@classmethod
|
| 184 |
+
def native_default_system_prompt(cls, model, revision):
|
| 185 |
+
calls.append((model, revision))
|
| 186 |
+
return "Default system text."
|
| 187 |
+
|
| 188 |
+
monkeypatch.setattr(native_tokenizer_module, "NativeTokenizer", FakeNativeTokenizer)
|
| 189 |
+
run_main(
|
| 190 |
+
monkeypatch,
|
| 191 |
+
[
|
| 192 |
+
"--model",
|
| 193 |
+
"decision-model",
|
| 194 |
+
"--prompt-wording",
|
| 195 |
+
"native",
|
| 196 |
+
"--tokenizer-model",
|
| 197 |
+
"org/model",
|
| 198 |
+
"--tokenizer-revision",
|
| 199 |
+
"a" * 40,
|
| 200 |
+
],
|
| 201 |
+
)
|
| 202 |
+
assert calls == [("org/model", "a" * 40)]
|
| 203 |
+
assert capture.backend_kwargs["native_system_prompt"] == "Default system text."
|
| 204 |
+
|
| 205 |
+
|
| 206 |
+
def test_native_system_prompt_none_skips_extraction_and_warning(
|
| 207 |
+
monkeypatch, capture, caplog
|
| 208 |
+
):
|
| 209 |
+
calls = []
|
| 210 |
+
|
| 211 |
+
class FakeNativeTokenizer:
|
| 212 |
+
@classmethod
|
| 213 |
+
def from_pretrained(cls, model, revision):
|
| 214 |
+
return object()
|
| 215 |
+
|
| 216 |
+
@classmethod
|
| 217 |
+
def native_default_system_prompt(cls, model, revision):
|
| 218 |
+
calls.append((model, revision))
|
| 219 |
+
return "should not be reached"
|
| 220 |
+
|
| 221 |
+
monkeypatch.setattr(native_tokenizer_module, "NativeTokenizer", FakeNativeTokenizer)
|
| 222 |
+
with caplog.at_level("WARNING"):
|
| 223 |
+
run_main(
|
| 224 |
+
monkeypatch,
|
| 225 |
+
[
|
| 226 |
+
"--model",
|
| 227 |
+
"decision-model",
|
| 228 |
+
"--prompt-wording",
|
| 229 |
+
"native",
|
| 230 |
+
"--tokenizer-model",
|
| 231 |
+
"org/model",
|
| 232 |
+
"--tokenizer-revision",
|
| 233 |
+
"a" * 40,
|
| 234 |
+
"--native-system-prompt",
|
| 235 |
+
"none",
|
| 236 |
+
],
|
| 237 |
+
)
|
| 238 |
+
assert calls == []
|
| 239 |
+
assert capture.backend_kwargs["native_system_prompt"] is None
|
| 240 |
+
assert not any("native" in record.message for record in caplog.records)
|
| 241 |
+
|
| 242 |
+
|
| 243 |
+
def test_native_system_prompt_none_without_tokenizer_never_warns(
|
| 244 |
+
monkeypatch, capture, caplog
|
| 245 |
+
):
|
| 246 |
+
with caplog.at_level("WARNING"):
|
| 247 |
+
run_main(
|
| 248 |
+
monkeypatch,
|
| 249 |
+
[
|
| 250 |
+
"--model",
|
| 251 |
+
"decision-model",
|
| 252 |
+
"--prompt-wording",
|
| 253 |
+
"native",
|
| 254 |
+
"--native-system-prompt",
|
| 255 |
+
"none",
|
| 256 |
+
],
|
| 257 |
+
)
|
| 258 |
+
assert capture.backend_kwargs["native_system_prompt"] is None
|
| 259 |
+
assert caplog.records == []
|
| 260 |
+
|
| 261 |
+
|
| 262 |
+
def test_native_system_prompt_env_var_sets_the_default(monkeypatch, capture, caplog):
|
| 263 |
+
with caplog.at_level("WARNING"):
|
| 264 |
+
run_main(
|
| 265 |
+
monkeypatch,
|
| 266 |
+
["--model", "decision-model", "--prompt-wording", "native"],
|
| 267 |
+
env={"JEV_NATIVE_SYSTEM_PROMPT": "none"},
|
| 268 |
+
)
|
| 269 |
+
assert capture.backend_kwargs["native_system_prompt"] is None
|
| 270 |
+
assert caplog.records == []
|
| 271 |
+
|
| 272 |
+
|
| 273 |
+
def test_native_system_prompt_explicit_flag_overrides_env_var(monkeypatch, capture):
|
| 274 |
+
calls = []
|
| 275 |
+
|
| 276 |
+
class FakeNativeTokenizer:
|
| 277 |
+
@classmethod
|
| 278 |
+
def from_pretrained(cls, model, revision):
|
| 279 |
+
return object()
|
| 280 |
+
|
| 281 |
+
@classmethod
|
| 282 |
+
def native_default_system_prompt(cls, model, revision):
|
| 283 |
+
calls.append((model, revision))
|
| 284 |
+
return "Default system text."
|
| 285 |
+
|
| 286 |
+
monkeypatch.setattr(native_tokenizer_module, "NativeTokenizer", FakeNativeTokenizer)
|
| 287 |
+
run_main(
|
| 288 |
+
monkeypatch,
|
| 289 |
+
[
|
| 290 |
+
"--model",
|
| 291 |
+
"decision-model",
|
| 292 |
+
"--prompt-wording",
|
| 293 |
+
"native",
|
| 294 |
+
"--tokenizer-model",
|
| 295 |
+
"org/model",
|
| 296 |
+
"--tokenizer-revision",
|
| 297 |
+
"a" * 40,
|
| 298 |
+
"--native-system-prompt",
|
| 299 |
+
"auto",
|
| 300 |
+
],
|
| 301 |
+
env={"JEV_NATIVE_SYSTEM_PROMPT": "none"},
|
| 302 |
+
)
|
| 303 |
+
assert calls == [("org/model", "a" * 40)]
|
| 304 |
+
assert capture.backend_kwargs["native_system_prompt"] == "Default system text."
|
| 305 |
+
|
| 306 |
+
|
| 307 |
+
def test_native_system_prompt_none_has_no_effect_on_served_wording(
|
| 308 |
+
monkeypatch, capture
|
| 309 |
+
):
|
| 310 |
+
calls = []
|
| 311 |
+
|
| 312 |
+
class FakeNativeTokenizer:
|
| 313 |
+
@classmethod
|
| 314 |
+
def from_pretrained(cls, model, revision):
|
| 315 |
+
return object()
|
| 316 |
+
|
| 317 |
+
@classmethod
|
| 318 |
+
def native_default_system_prompt(cls, model, revision):
|
| 319 |
+
calls.append((model, revision))
|
| 320 |
+
return "should not be reached"
|
| 321 |
+
|
| 322 |
+
monkeypatch.setattr(native_tokenizer_module, "NativeTokenizer", FakeNativeTokenizer)
|
| 323 |
+
run_main(
|
| 324 |
+
monkeypatch,
|
| 325 |
+
[
|
| 326 |
+
"--model",
|
| 327 |
+
"decision-model",
|
| 328 |
+
"--tokenizer-model",
|
| 329 |
+
"org/model",
|
| 330 |
+
"--tokenizer-revision",
|
| 331 |
+
"a" * 40,
|
| 332 |
+
"--native-system-prompt",
|
| 333 |
+
"auto",
|
| 334 |
+
],
|
| 335 |
+
)
|
| 336 |
+
assert calls == []
|
| 337 |
+
assert capture.backend_kwargs["prompt_wording"] == "served"
|
| 338 |
+
assert capture.backend_kwargs["native_system_prompt"] is None
|
| 339 |
+
|
| 340 |
+
|
| 341 |
+
def test_invalid_native_system_prompt_choice_rejected(monkeypatch, capture):
|
| 342 |
+
monkeypatch.setattr(
|
| 343 |
+
sys,
|
| 344 |
+
"argv",
|
| 345 |
+
["jev-adapter", "--model", "m", "--native-system-prompt", "bogus"],
|
| 346 |
+
)
|
| 347 |
+
with pytest.raises(SystemExit):
|
| 348 |
+
main_module.main()
|
| 349 |
+
|
| 350 |
+
|
| 351 |
+
def test_native_system_prompt_startup_log_line(monkeypatch, capture, caplog):
|
| 352 |
+
with caplog.at_level("INFO"):
|
| 353 |
+
run_main(
|
| 354 |
+
monkeypatch,
|
| 355 |
+
[
|
| 356 |
+
"--model",
|
| 357 |
+
"decision-model",
|
| 358 |
+
"--prompt-wording",
|
| 359 |
+
"native",
|
| 360 |
+
"--native-system-prompt",
|
| 361 |
+
"none",
|
| 362 |
+
],
|
| 363 |
+
)
|
| 364 |
+
info_records = [r for r in caplog.records if r.levelname == "INFO"]
|
| 365 |
+
assert len(info_records) == 1
|
| 366 |
+
message = info_records[0].getMessage()
|
| 367 |
+
assert "prompt_wording=native" in message
|
| 368 |
+
assert "native_system_prompt=none" in message
|
server/tests/test_service.py
ADDED
|
@@ -0,0 +1,324 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import asyncio
|
| 2 |
+
import math
|
| 3 |
+
import unittest
|
| 4 |
+
|
| 5 |
+
from fastapi.testclient import TestClient
|
| 6 |
+
|
| 7 |
+
from jev_adapter.__main__ import positive_temperature
|
| 8 |
+
from jev_adapter.backend import AdapterError, ScoringResult
|
| 9 |
+
from jev_adapter.protocol import SystemOneRequest, probabilities_from_logprobs
|
| 10 |
+
from jev_adapter.server import create_app
|
| 11 |
+
from jev_adapter.service import SystemOneService
|
| 12 |
+
|
| 13 |
+
|
| 14 |
+
def payload():
|
| 15 |
+
return {
|
| 16 |
+
"model": "test-model",
|
| 17 |
+
"state": {"text": "결제가 두 번 되었어요. 환불해 주세요."},
|
| 18 |
+
"questions": {
|
| 19 |
+
"route": {
|
| 20 |
+
"type": "choice",
|
| 21 |
+
"instructions": "담당 부서",
|
| 22 |
+
"criteria": {"billing": "청구", "technical": "기술"},
|
| 23 |
+
},
|
| 24 |
+
"refund": {"type": "noul", "instructions": "환불 요청인가?"},
|
| 25 |
+
"urgency": {
|
| 26 |
+
"type": "score",
|
| 27 |
+
"instructions": "긴급도",
|
| 28 |
+
"criteria": ["낮음", "중간", "높음"],
|
| 29 |
+
},
|
| 30 |
+
},
|
| 31 |
+
}
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
class FakeBackend:
|
| 35 |
+
model = "test-model"
|
| 36 |
+
|
| 37 |
+
def __init__(self, delay=0, failure=False):
|
| 38 |
+
self.calls = []
|
| 39 |
+
self.active = self.peak = self.cancelled = 0
|
| 40 |
+
self.started = self.closed = False
|
| 41 |
+
self.delay, self.failure = delay, failure
|
| 42 |
+
|
| 43 |
+
async def start(self):
|
| 44 |
+
self.started = True
|
| 45 |
+
|
| 46 |
+
async def close(self):
|
| 47 |
+
self.closed = True
|
| 48 |
+
|
| 49 |
+
def labels(self, count):
|
| 50 |
+
return tuple(chr(65 + i) for i in range(count)), tuple(range(65, 65 + count))
|
| 51 |
+
|
| 52 |
+
async def evaluate(self, prompt, images, labels, token_ids, assistant_prefix):
|
| 53 |
+
self.calls.append((prompt, images, labels, token_ids, assistant_prefix))
|
| 54 |
+
self.active += 1
|
| 55 |
+
self.peak = max(self.peak, self.active)
|
| 56 |
+
try:
|
| 57 |
+
if self.failure:
|
| 58 |
+
raise AdapterError(
|
| 59 |
+
"backend_unavailable", "Engine unavailable.", status=502
|
| 60 |
+
)
|
| 61 |
+
await asyncio.sleep(self.delay)
|
| 62 |
+
return ScoringResult(
|
| 63 |
+
tuple(math.log(0.8 if i == 0 else 0.1) for i in range(len(labels))), 20
|
| 64 |
+
)
|
| 65 |
+
except asyncio.CancelledError:
|
| 66 |
+
self.cancelled += 1
|
| 67 |
+
raise
|
| 68 |
+
finally:
|
| 69 |
+
self.active -= 1
|
| 70 |
+
|
| 71 |
+
|
| 72 |
+
class TestService(unittest.IsolatedAsyncioTestCase):
|
| 73 |
+
async def test_mixed_questions_keep_images_and_build_typed_response(self):
|
| 74 |
+
backend = FakeBackend()
|
| 75 |
+
body = payload()
|
| 76 |
+
body["images"] = ["data:image/png;base64,aGVsbG8="]
|
| 77 |
+
result = await SystemOneService(backend).score(
|
| 78 |
+
SystemOneRequest.model_validate(body)
|
| 79 |
+
)
|
| 80 |
+
self.assertEqual(result["answers"]["route"]["choice"], "billing")
|
| 81 |
+
self.assertAlmostEqual(result["answers"]["refund"]["noul"], 8 / 9)
|
| 82 |
+
self.assertAlmostEqual(result["answers"]["urgency"]["score"], 0.3)
|
| 83 |
+
self.assertEqual(result["usage"], {"input_tokens": 60, "output_tokens": 0})
|
| 84 |
+
self.assertEqual(result["metadata"]["evaluations"], 3)
|
| 85 |
+
for call in backend.calls:
|
| 86 |
+
self.assertEqual(call[1], body["images"])
|
| 87 |
+
self.assertNotIn("담당 부서", backend.calls[1][0])
|
| 88 |
+
|
| 89 |
+
async def test_rotations_preserve_canonical_option_mapping(self):
|
| 90 |
+
body = payload()
|
| 91 |
+
body["questions"] = {"route": body["questions"]["route"]}
|
| 92 |
+
body["options"] = {"permutations": 2, "return_logprobs": True}
|
| 93 |
+
result = await SystemOneService(FakeBackend()).score(
|
| 94 |
+
SystemOneRequest.model_validate(body)
|
| 95 |
+
)
|
| 96 |
+
route = result["answers"]["route"]
|
| 97 |
+
self.assertAlmostEqual(route["probabilities"]["billing"], 0.5)
|
| 98 |
+
self.assertAlmostEqual(route["confidence"], 0)
|
| 99 |
+
self.assertEqual(len(route["logprobs"]), 2)
|
| 100 |
+
|
| 101 |
+
async def test_concurrency_limit_applies_across_requests(self):
|
| 102 |
+
backend = FakeBackend(delay=0.01)
|
| 103 |
+
service = SystemOneService(backend, max_concurrency=2)
|
| 104 |
+
await asyncio.gather(
|
| 105 |
+
*(
|
| 106 |
+
service.score(SystemOneRequest.model_validate(payload()))
|
| 107 |
+
for _ in range(3)
|
| 108 |
+
)
|
| 109 |
+
)
|
| 110 |
+
self.assertEqual(len(backend.calls), 9)
|
| 111 |
+
self.assertEqual(backend.peak, 2)
|
| 112 |
+
|
| 113 |
+
async def test_cancellation_closes_all_pending_work(self):
|
| 114 |
+
backend = FakeBackend(delay=10)
|
| 115 |
+
task = asyncio.create_task(
|
| 116 |
+
SystemOneService(backend, max_concurrency=2).score(
|
| 117 |
+
SystemOneRequest.model_validate(payload())
|
| 118 |
+
)
|
| 119 |
+
)
|
| 120 |
+
while backend.active != 2:
|
| 121 |
+
await asyncio.sleep(0)
|
| 122 |
+
task.cancel()
|
| 123 |
+
with self.assertRaises(asyncio.CancelledError):
|
| 124 |
+
await task
|
| 125 |
+
self.assertEqual(backend.cancelled, 2)
|
| 126 |
+
self.assertEqual(backend.active, 0)
|
| 127 |
+
self.assertEqual(len(backend.calls), 2)
|
| 128 |
+
|
| 129 |
+
async def test_wrong_model_rejected_before_inference(self):
|
| 130 |
+
backend = FakeBackend()
|
| 131 |
+
body = payload()
|
| 132 |
+
body["model"] = "wrong-model"
|
| 133 |
+
with self.assertRaises(AdapterError) as error:
|
| 134 |
+
await SystemOneService(backend).score(SystemOneRequest.model_validate(body))
|
| 135 |
+
self.assertEqual(error.exception.status, 404)
|
| 136 |
+
self.assertEqual(backend.calls, [])
|
| 137 |
+
|
| 138 |
+
|
| 139 |
+
class TestDefaultTemperature(unittest.IsolatedAsyncioTestCase):
|
| 140 |
+
"""A server-side default fills an omitted options.temperature only."""
|
| 141 |
+
|
| 142 |
+
vector = tuple(math.log(0.8 if i == 0 else 0.1) for i in range(3))
|
| 143 |
+
|
| 144 |
+
async def score(self, body, **service_options):
|
| 145 |
+
service = SystemOneService(FakeBackend(), **service_options)
|
| 146 |
+
return await service.score(SystemOneRequest.model_validate(body))
|
| 147 |
+
|
| 148 |
+
def route_probabilities(self, result):
|
| 149 |
+
return list(result["answers"]["urgency"]["probabilities"].values())
|
| 150 |
+
|
| 151 |
+
async def test_unset_default_keeps_request_default_of_one(self):
|
| 152 |
+
result = await self.score(payload())
|
| 153 |
+
self.assertEqual(result["metadata"]["temperature"], 1.0)
|
| 154 |
+
expected = probabilities_from_logprobs(self.vector, 1.0)
|
| 155 |
+
for actual, wanted in zip(self.route_probabilities(result), expected):
|
| 156 |
+
self.assertAlmostEqual(actual, wanted)
|
| 157 |
+
|
| 158 |
+
async def test_omitted_temperature_takes_server_default(self):
|
| 159 |
+
for body in (payload(), {**payload(), "options": {"permutations": 1}}):
|
| 160 |
+
result = await self.score(body, default_temperature=2.4)
|
| 161 |
+
self.assertEqual(result["metadata"]["temperature"], 2.4)
|
| 162 |
+
expected = probabilities_from_logprobs(self.vector, 2.4)
|
| 163 |
+
for actual, wanted in zip(self.route_probabilities(result), expected):
|
| 164 |
+
self.assertAlmostEqual(actual, wanted)
|
| 165 |
+
self.assertNotAlmostEqual(
|
| 166 |
+
self.route_probabilities(result)[0],
|
| 167 |
+
probabilities_from_logprobs(self.vector, 1.0)[0],
|
| 168 |
+
)
|
| 169 |
+
|
| 170 |
+
async def test_explicit_request_temperature_wins_over_default(self):
|
| 171 |
+
body = {**payload(), "options": {"temperature": 1.0}}
|
| 172 |
+
result = await self.score(body, default_temperature=2.4)
|
| 173 |
+
self.assertEqual(result["metadata"]["temperature"], 1.0)
|
| 174 |
+
expected = probabilities_from_logprobs(self.vector, 1.0)
|
| 175 |
+
for actual, wanted in zip(self.route_probabilities(result), expected):
|
| 176 |
+
self.assertAlmostEqual(actual, wanted)
|
| 177 |
+
body = {**payload(), "options": {"temperature": 3.0}}
|
| 178 |
+
result = await self.score(body, default_temperature=2.4)
|
| 179 |
+
self.assertEqual(result["metadata"]["temperature"], 3.0)
|
| 180 |
+
|
| 181 |
+
async def test_disabled_scaling_ignores_default(self):
|
| 182 |
+
body = {**payload(), "options": {"temperature_scaling": False}}
|
| 183 |
+
result = await self.score(body, default_temperature=2.4)
|
| 184 |
+
self.assertEqual(result["metadata"]["temperature"], 1.0)
|
| 185 |
+
expected = probabilities_from_logprobs(self.vector, 1.0)
|
| 186 |
+
for actual, wanted in zip(self.route_probabilities(result), expected):
|
| 187 |
+
self.assertAlmostEqual(actual, wanted)
|
| 188 |
+
|
| 189 |
+
async def test_default_does_not_mutate_the_request(self):
|
| 190 |
+
request = SystemOneRequest.model_validate(payload())
|
| 191 |
+
service = SystemOneService(FakeBackend(), default_temperature=2.4)
|
| 192 |
+
await service.score(request)
|
| 193 |
+
self.assertEqual(request.options.temperature, 1.0)
|
| 194 |
+
self.assertNotIn("temperature", request.options.model_fields_set)
|
| 195 |
+
self.assertEqual(service.effective_options(request).temperature, 2.4)
|
| 196 |
+
|
| 197 |
+
def test_invalid_default_temperature_rejected_at_construction(self):
|
| 198 |
+
for value in (0, -1.0, float("inf"), float("nan"), True, "2.4", None):
|
| 199 |
+
with self.assertRaises((ValueError, TypeError)):
|
| 200 |
+
SystemOneService(FakeBackend(), default_temperature=value)
|
| 201 |
+
self.assertEqual(
|
| 202 |
+
SystemOneService(FakeBackend(), default_temperature=2).default_temperature,
|
| 203 |
+
2.0,
|
| 204 |
+
)
|
| 205 |
+
|
| 206 |
+
def test_cli_default_temperature_type(self):
|
| 207 |
+
self.assertEqual(positive_temperature("2.4"), 2.4)
|
| 208 |
+
self.assertEqual(positive_temperature("1"), 1.0)
|
| 209 |
+
import argparse
|
| 210 |
+
|
| 211 |
+
for text in ("0", "-1", "inf", "nan", "abc", ""):
|
| 212 |
+
with self.assertRaises(argparse.ArgumentTypeError):
|
| 213 |
+
positive_temperature(text)
|
| 214 |
+
|
| 215 |
+
|
| 216 |
+
class TestPromptWording(unittest.IsolatedAsyncioTestCase):
|
| 217 |
+
"""--prompt-wording is a server-wide setting (CLI/env, not per-request);
|
| 218 |
+
the default keeps today's served text unchanged."""
|
| 219 |
+
|
| 220 |
+
async def test_default_served_wording_is_unchanged(self):
|
| 221 |
+
backend = FakeBackend()
|
| 222 |
+
body = payload()
|
| 223 |
+
body["questions"] = {"route": body["questions"]["route"]}
|
| 224 |
+
await SystemOneService(backend).score(SystemOneRequest.model_validate(body))
|
| 225 |
+
prompt = backend.calls[0][0]
|
| 226 |
+
self.assertTrue(prompt.startswith("Context:\n"))
|
| 227 |
+
|
| 228 |
+
async def test_native_wording_is_plumbed_into_build_prompt(self):
|
| 229 |
+
backend = FakeBackend()
|
| 230 |
+
body = payload()
|
| 231 |
+
body["questions"] = {"route": body["questions"]["route"]}
|
| 232 |
+
service = SystemOneService(backend, prompt_wording="native")
|
| 233 |
+
await service.score(SystemOneRequest.model_validate(body))
|
| 234 |
+
prompt = backend.calls[0][0]
|
| 235 |
+
self.assertTrue(prompt.startswith("Read the state and question."))
|
| 236 |
+
self.assertIn("\n\nState:\n", prompt)
|
| 237 |
+
self.assertNotIn("Context:\n", prompt)
|
| 238 |
+
|
| 239 |
+
async def test_native_wording_requires_canonical_az_labels(self):
|
| 240 |
+
# FakeBackend.labels() already returns canonical A, B, C, ... so this
|
| 241 |
+
# documents the happy path; build_prompt itself enforces the
|
| 242 |
+
# requirement (see TestNativePromptWording in test_protocol.py) when a
|
| 243 |
+
# backend's labels are not canonical.
|
| 244 |
+
backend = FakeBackend()
|
| 245 |
+
labels, _ = backend.labels(3)
|
| 246 |
+
self.assertEqual(labels, ("A", "B", "C"))
|
| 247 |
+
|
| 248 |
+
def test_invalid_prompt_wording_rejected_at_construction(self):
|
| 249 |
+
for value in ("", "SERVED", "native ", None, 1):
|
| 250 |
+
with self.assertRaises(ValueError):
|
| 251 |
+
SystemOneService(FakeBackend(), prompt_wording=value)
|
| 252 |
+
|
| 253 |
+
|
| 254 |
+
class TestHTTP(unittest.TestCase):
|
| 255 |
+
def test_standalone_app_lifecycle_schema_and_alias(self):
|
| 256 |
+
backend = FakeBackend()
|
| 257 |
+
with TestClient(create_app(backend)) as client:
|
| 258 |
+
self.assertTrue(backend.started)
|
| 259 |
+
self.assertEqual(client.get("/health").json(), {"status": "ok"})
|
| 260 |
+
self.assertEqual(
|
| 261 |
+
client.get("/v1/models").json()["data"][0]["id"], backend.model
|
| 262 |
+
)
|
| 263 |
+
body = payload()
|
| 264 |
+
body["model"] = "jev-latest"
|
| 265 |
+
response = client.post("/v1/systemone", json=body)
|
| 266 |
+
self.assertEqual(response.status_code, 200)
|
| 267 |
+
self.assertEqual(response.json()["model"], backend.model)
|
| 268 |
+
self.assertEqual(
|
| 269 |
+
set(response.json()["answers"]), {"route", "refund", "urgency"}
|
| 270 |
+
)
|
| 271 |
+
bad = client.post("/v1/systemone", json={"model": backend.model})
|
| 272 |
+
self.assertEqual(bad.status_code, 422)
|
| 273 |
+
self.assertEqual(bad.json()["error"]["code"], "invalid_request")
|
| 274 |
+
self.assertTrue(backend.closed)
|
| 275 |
+
|
| 276 |
+
def test_app_default_temperature_applies_when_request_omits_it(self):
|
| 277 |
+
with TestClient(create_app(FakeBackend(), default_temperature=2.4)) as client:
|
| 278 |
+
response = client.post("/v1/systemone", json=payload())
|
| 279 |
+
self.assertEqual(response.status_code, 200)
|
| 280 |
+
self.assertEqual(response.json()["metadata"]["temperature"], 2.4)
|
| 281 |
+
body = {**payload(), "options": {"temperature": 1.5}}
|
| 282 |
+
response = client.post("/v1/systemone", json=body)
|
| 283 |
+
self.assertEqual(response.json()["metadata"]["temperature"], 1.5)
|
| 284 |
+
with self.assertRaises(ValueError):
|
| 285 |
+
create_app(FakeBackend(), default_temperature=0)
|
| 286 |
+
|
| 287 |
+
def test_app_prompt_wording_native_reaches_the_backend_over_http(self):
|
| 288 |
+
backend = FakeBackend()
|
| 289 |
+
with TestClient(create_app(backend, prompt_wording="native")) as client:
|
| 290 |
+
body = payload()
|
| 291 |
+
body["questions"] = {"route": body["questions"]["route"]}
|
| 292 |
+
response = client.post("/v1/systemone", json=body)
|
| 293 |
+
self.assertEqual(response.status_code, 200)
|
| 294 |
+
self.assertTrue(backend.calls[0][0].startswith("Read the state and question."))
|
| 295 |
+
|
| 296 |
+
def test_auth_guards_inference_and_model_discovery(self):
|
| 297 |
+
backend = FakeBackend()
|
| 298 |
+
with TestClient(create_app(backend, api_key="test-secret")) as client:
|
| 299 |
+
self.assertEqual(client.get("/v1/models").status_code, 401)
|
| 300 |
+
self.assertEqual(
|
| 301 |
+
client.post("/v1/systemone", json=payload()).status_code, 401
|
| 302 |
+
)
|
| 303 |
+
self.assertEqual(
|
| 304 |
+
client.post(
|
| 305 |
+
"/v1/systemone",
|
| 306 |
+
json=payload(),
|
| 307 |
+
headers={b"Authorization": b"Bearer caf\xe9"},
|
| 308 |
+
).status_code,
|
| 309 |
+
401,
|
| 310 |
+
)
|
| 311 |
+
self.assertEqual(backend.calls, [])
|
| 312 |
+
response = client.post(
|
| 313 |
+
"/v1/systemone",
|
| 314 |
+
json=payload(),
|
| 315 |
+
headers={"Authorization": "Bearer test-secret"},
|
| 316 |
+
)
|
| 317 |
+
self.assertEqual(response.status_code, 200)
|
| 318 |
+
|
| 319 |
+
def test_upstream_failure_is_not_returned_as_probabilities(self):
|
| 320 |
+
with TestClient(create_app(FakeBackend(failure=True))) as client:
|
| 321 |
+
response = client.post("/v1/systemone", json=payload())
|
| 322 |
+
self.assertEqual(response.status_code, 502)
|
| 323 |
+
self.assertEqual(response.json()["error"]["code"], "backend_unavailable")
|
| 324 |
+
self.assertNotIn("answers", response.json())
|
server/tests/test_sglang.py
ADDED
|
@@ -0,0 +1,409 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import copy
|
| 2 |
+
import json
|
| 3 |
+
|
| 4 |
+
import httpx
|
| 5 |
+
import pytest
|
| 6 |
+
|
| 7 |
+
from jev_adapter.backend import AdapterError
|
| 8 |
+
from jev_adapter.sglang import SGLangBackend
|
| 9 |
+
|
| 10 |
+
|
| 11 |
+
class Engine:
|
| 12 |
+
def __init__(self):
|
| 13 |
+
self.calls = []
|
| 14 |
+
self.info = {
|
| 15 |
+
"served_model_name": "qwen",
|
| 16 |
+
"model_path": "org/qwen",
|
| 17 |
+
"is_generation": True,
|
| 18 |
+
"has_image_understanding": True,
|
| 19 |
+
"model_type": "qwen3_5",
|
| 20 |
+
"architectures": ["Qwen3_5ForConditionalGeneration"],
|
| 21 |
+
}
|
| 22 |
+
self.config = {"speculative_algorithm": None, "skip_tokenizer_init": False}
|
| 23 |
+
self.mutate_result = lambda result: result
|
| 24 |
+
self.invalid_boundary = False
|
| 25 |
+
self.lossy_roundtrip = False
|
| 26 |
+
self.special_label = None
|
| 27 |
+
self.fail_path = None
|
| 28 |
+
self.failure = None
|
| 29 |
+
|
| 30 |
+
@staticmethod
|
| 31 |
+
def encode(text):
|
| 32 |
+
return [ord(char) + 1000 for char in text]
|
| 33 |
+
|
| 34 |
+
@staticmethod
|
| 35 |
+
def decode(tokens):
|
| 36 |
+
return "".join(chr(token - 1000) for token in tokens)
|
| 37 |
+
|
| 38 |
+
def __call__(self, request):
|
| 39 |
+
path = request.url.path
|
| 40 |
+
body = json.loads(request.content) if request.content else None
|
| 41 |
+
self.calls.append((path, body, request))
|
| 42 |
+
if path == self.fail_path:
|
| 43 |
+
if isinstance(self.failure, Exception):
|
| 44 |
+
raise self.failure
|
| 45 |
+
return self.failure
|
| 46 |
+
if path == "/model_info":
|
| 47 |
+
return httpx.Response(200, json=self.info)
|
| 48 |
+
if path == "/server_info":
|
| 49 |
+
return httpx.Response(200, json=self.config)
|
| 50 |
+
if path == "/v1/tokenize":
|
| 51 |
+
if "messages" in body:
|
| 52 |
+
content = body["messages"][0]["content"]
|
| 53 |
+
if isinstance(content, list):
|
| 54 |
+
content = "<image>" * (len(content) - 1) + content[-1]["text"]
|
| 55 |
+
tokens = self.encode("<user>" + content + "</user><assistant>")
|
| 56 |
+
else:
|
| 57 |
+
assert body["add_special_tokens"] is False
|
| 58 |
+
tokens = [self.encode(text) for text in body["prompt"]]
|
| 59 |
+
if body["prompt"][0].startswith("<user>"):
|
| 60 |
+
if self.invalid_boundary:
|
| 61 |
+
tokens[-1][-1] += 1
|
| 62 |
+
if self.lossy_roundtrip:
|
| 63 |
+
tokens[0][-1] += 1
|
| 64 |
+
return httpx.Response(200, json={"tokens": tokens})
|
| 65 |
+
if path == "/v1/detokenize":
|
| 66 |
+
tokens = body["tokens"]
|
| 67 |
+
if isinstance(tokens[0], list):
|
| 68 |
+
assert body["skip_special_tokens"] is True
|
| 69 |
+
text = [self.decode(ids) for ids in tokens]
|
| 70 |
+
text = ["" if value == self.special_label else value for value in text]
|
| 71 |
+
else:
|
| 72 |
+
assert body["skip_special_tokens"] is False
|
| 73 |
+
text = self.decode(tokens)
|
| 74 |
+
return httpx.Response(200, json={"text": text})
|
| 75 |
+
if path == "/generate":
|
| 76 |
+
result = {
|
| 77 |
+
"text": "",
|
| 78 |
+
"meta_info": {
|
| 79 |
+
"completion_tokens": 0,
|
| 80 |
+
"prompt_tokens": 71,
|
| 81 |
+
# Return labels out of order to ensure alignment by token ID.
|
| 82 |
+
"output_token_ids_logprobs": [
|
| 83 |
+
[
|
| 84 |
+
[-index - 0.5, token, None]
|
| 85 |
+
for index, token in reversed(
|
| 86 |
+
list(enumerate(body["token_ids_logprob"]))
|
| 87 |
+
)
|
| 88 |
+
]
|
| 89 |
+
],
|
| 90 |
+
},
|
| 91 |
+
}
|
| 92 |
+
return httpx.Response(200, json=self.mutate_result(result))
|
| 93 |
+
raise AssertionError(f"Unexpected route: {path}")
|
| 94 |
+
|
| 95 |
+
|
| 96 |
+
async def make_backend(engine=None, **kwargs):
|
| 97 |
+
engine = engine or Engine()
|
| 98 |
+
client = httpx.AsyncClient(transport=httpx.MockTransport(engine))
|
| 99 |
+
backend = SGLangBackend("http://engine", "qwen", client=client, **kwargs)
|
| 100 |
+
await backend.start()
|
| 101 |
+
return backend, engine, client
|
| 102 |
+
|
| 103 |
+
|
| 104 |
+
@pytest.mark.asyncio
|
| 105 |
+
async def test_public_http_contract_and_zero_decode():
|
| 106 |
+
backend, engine, client = await make_backend(api_key="private-key")
|
| 107 |
+
try:
|
| 108 |
+
labels, tokens = backend.labels(3)
|
| 109 |
+
result = await backend.evaluate("pick an answer", [], labels, tokens, None)
|
| 110 |
+
assert result.logprobs == (-0.5, -1.5, -2.5)
|
| 111 |
+
assert result.input_tokens == 71
|
| 112 |
+
path, body, _ = engine.calls[-1]
|
| 113 |
+
assert path == "/generate"
|
| 114 |
+
assert "text" not in body and "image_data" not in body
|
| 115 |
+
assert body["input_ids"] == engine.encode(
|
| 116 |
+
"<user>pick an answer</user><assistant>"
|
| 117 |
+
)
|
| 118 |
+
assert body["sampling_params"] == {
|
| 119 |
+
"max_new_tokens": 0,
|
| 120 |
+
"temperature": 1,
|
| 121 |
+
"top_p": 1,
|
| 122 |
+
"top_k": -1,
|
| 123 |
+
}
|
| 124 |
+
assert body["token_ids_logprob"] == list(tokens)
|
| 125 |
+
assert body["logprob_start_len"] == -1
|
| 126 |
+
assert body["return_logprob"] is True
|
| 127 |
+
assert body["top_logprobs_num"] == 0
|
| 128 |
+
assert body["return_text_in_logprobs"] is False
|
| 129 |
+
assert body["stream"] is False
|
| 130 |
+
assert body["rid"].startswith("jev-")
|
| 131 |
+
assert all(
|
| 132 |
+
request.headers["Authorization"] == "Bearer private-key"
|
| 133 |
+
for _, _, request in engine.calls
|
| 134 |
+
)
|
| 135 |
+
assert engine.calls[-4][1]["chat_template_kwargs"] == {"enable_thinking": False}
|
| 136 |
+
finally:
|
| 137 |
+
await client.aclose()
|
| 138 |
+
|
| 139 |
+
|
| 140 |
+
def test_prompt_wording_rejects_unknown_value():
|
| 141 |
+
with pytest.raises(ValueError):
|
| 142 |
+
SGLangBackend("http://engine", "qwen", prompt_wording="bogus")
|
| 143 |
+
|
| 144 |
+
|
| 145 |
+
@pytest.mark.asyncio
|
| 146 |
+
async def test_native_wording_prepends_system_message_over_http():
|
| 147 |
+
backend, engine, client = await make_backend(
|
| 148 |
+
prompt_wording="native", native_system_prompt="Default system text."
|
| 149 |
+
)
|
| 150 |
+
try:
|
| 151 |
+
labels, tokens = backend.labels(3)
|
| 152 |
+
await backend.evaluate("pick an answer", [], labels, tokens, None)
|
| 153 |
+
# The messages-based /v1/tokenize call, three calls before /generate
|
| 154 |
+
# (detokenize, boundary-check tokenize, generate follow it).
|
| 155 |
+
path, body, _ = engine.calls[-4]
|
| 156 |
+
assert path == "/v1/tokenize"
|
| 157 |
+
assert body["messages"] == [
|
| 158 |
+
{"role": "system", "content": "Default system text."},
|
| 159 |
+
{"role": "user", "content": "pick an answer"},
|
| 160 |
+
]
|
| 161 |
+
finally:
|
| 162 |
+
await client.aclose()
|
| 163 |
+
|
| 164 |
+
|
| 165 |
+
@pytest.mark.asyncio
|
| 166 |
+
async def test_served_wording_never_sends_a_system_message_even_if_configured():
|
| 167 |
+
# native_system_prompt is only honored when prompt_wording == "native";
|
| 168 |
+
# the default ("served") must stay byte-identical to today's behaviour.
|
| 169 |
+
backend, engine, client = await make_backend(
|
| 170 |
+
prompt_wording="served", native_system_prompt="Should be ignored."
|
| 171 |
+
)
|
| 172 |
+
try:
|
| 173 |
+
labels, tokens = backend.labels(2)
|
| 174 |
+
await backend.evaluate("look", [], labels, tokens, None)
|
| 175 |
+
path, body, _ = engine.calls[-4]
|
| 176 |
+
assert path == "/v1/tokenize"
|
| 177 |
+
assert body["messages"] == [{"role": "user", "content": "look"}]
|
| 178 |
+
finally:
|
| 179 |
+
await client.aclose()
|
| 180 |
+
|
| 181 |
+
|
| 182 |
+
@pytest.mark.asyncio
|
| 183 |
+
async def test_native_wording_passes_system_prompt_to_native_tokenizer():
|
| 184 |
+
class RecordingNativeTokenizer:
|
| 185 |
+
def __init__(self):
|
| 186 |
+
self.calls = []
|
| 187 |
+
|
| 188 |
+
def prepare(self, prompt, labels, token_ids, assistant_prefix, system_prompt=None):
|
| 189 |
+
self.calls.append((prompt, labels, token_ids, assistant_prefix, system_prompt))
|
| 190 |
+
return [1, 2, 3]
|
| 191 |
+
|
| 192 |
+
native_tokenizer = RecordingNativeTokenizer()
|
| 193 |
+
engine = Engine()
|
| 194 |
+
client = httpx.AsyncClient(transport=httpx.MockTransport(engine))
|
| 195 |
+
backend = SGLangBackend(
|
| 196 |
+
"http://engine",
|
| 197 |
+
"qwen",
|
| 198 |
+
client=client,
|
| 199 |
+
native_tokenizer=native_tokenizer,
|
| 200 |
+
prompt_wording="native",
|
| 201 |
+
native_system_prompt="Default system text.",
|
| 202 |
+
)
|
| 203 |
+
try:
|
| 204 |
+
await backend.start()
|
| 205 |
+
labels, tokens = backend.labels(2)
|
| 206 |
+
await backend.evaluate("pick", [], labels, tokens, "Answer:")
|
| 207 |
+
assert native_tokenizer.calls == [
|
| 208 |
+
("pick", labels, tokens, "Answer:", "Default system text.")
|
| 209 |
+
]
|
| 210 |
+
finally:
|
| 211 |
+
await client.aclose()
|
| 212 |
+
|
| 213 |
+
|
| 214 |
+
@pytest.mark.asyncio
|
| 215 |
+
async def test_images_remain_in_native_request_with_template_and_prefix():
|
| 216 |
+
backend, engine, client = await make_backend()
|
| 217 |
+
try:
|
| 218 |
+
images = ["data:image/png;base64,YQ==", "https://example.com/image.png"]
|
| 219 |
+
labels, tokens = backend.labels(2)
|
| 220 |
+
await backend.evaluate("look", images, labels, tokens, "Answer: ")
|
| 221 |
+
body = engine.calls[-1][1]
|
| 222 |
+
assert body["image_data"] == images
|
| 223 |
+
assert body["text"] == "<user><image><image>look</user><assistant>Answer: "
|
| 224 |
+
assert "input_ids" not in body
|
| 225 |
+
content = engine.calls[-4][1]["messages"][0]["content"]
|
| 226 |
+
assert content == [
|
| 227 |
+
{"type": "image_url", "image_url": {"url": images[0]}},
|
| 228 |
+
{"type": "image_url", "image_url": {"url": images[1]}},
|
| 229 |
+
{"type": "text", "text": "look"},
|
| 230 |
+
]
|
| 231 |
+
finally:
|
| 232 |
+
await client.aclose()
|
| 233 |
+
|
| 234 |
+
|
| 235 |
+
@pytest.mark.asyncio
|
| 236 |
+
@pytest.mark.parametrize(
|
| 237 |
+
"field,value,code",
|
| 238 |
+
[
|
| 239 |
+
("has_image_understanding", False, "images_not_supported"),
|
| 240 |
+
("model_type", "kimi_k3", "image_model_not_supported"),
|
| 241 |
+
],
|
| 242 |
+
)
|
| 243 |
+
async def test_image_capability_rejections(field, value, code):
|
| 244 |
+
engine = Engine()
|
| 245 |
+
engine.info[field] = value
|
| 246 |
+
engine.info["architectures"] = []
|
| 247 |
+
backend, engine, client = await make_backend(engine)
|
| 248 |
+
try:
|
| 249 |
+
labels, tokens = backend.labels(2)
|
| 250 |
+
with pytest.raises(AdapterError, match="image|Image") as error:
|
| 251 |
+
await backend.evaluate(
|
| 252 |
+
"look", ["https://example.com/a.png"], labels, tokens, None
|
| 253 |
+
)
|
| 254 |
+
assert error.value.code == code
|
| 255 |
+
assert not any(path == "/generate" for path, _, _ in engine.calls)
|
| 256 |
+
finally:
|
| 257 |
+
await client.aclose()
|
| 258 |
+
|
| 259 |
+
|
| 260 |
+
@pytest.mark.asyncio
|
| 261 |
+
@pytest.mark.parametrize(
|
| 262 |
+
"flag,code",
|
| 263 |
+
[
|
| 264 |
+
("invalid_boundary", "invalid_label_boundary"),
|
| 265 |
+
("lossy_roundtrip", "unsupported_tokenizer_roundtrip"),
|
| 266 |
+
],
|
| 267 |
+
)
|
| 268 |
+
async def test_rejects_unsafe_tokenization_before_inference(flag, code):
|
| 269 |
+
backend, engine, client = await make_backend()
|
| 270 |
+
try:
|
| 271 |
+
setattr(engine, flag, True)
|
| 272 |
+
labels, tokens = backend.labels(2)
|
| 273 |
+
with pytest.raises(AdapterError) as error:
|
| 274 |
+
await backend.evaluate("look", [], labels, tokens, None)
|
| 275 |
+
assert error.value.code == code
|
| 276 |
+
assert not any(path == "/generate" for path, _, _ in engine.calls)
|
| 277 |
+
finally:
|
| 278 |
+
await client.aclose()
|
| 279 |
+
|
| 280 |
+
|
| 281 |
+
@pytest.mark.asyncio
|
| 282 |
+
@pytest.mark.parametrize(
|
| 283 |
+
"mutation",
|
| 284 |
+
[
|
| 285 |
+
"decoded",
|
| 286 |
+
"missing_label",
|
| 287 |
+
"duplicate_label",
|
| 288 |
+
"nan",
|
| 289 |
+
"positive",
|
| 290 |
+
"missing_usage",
|
| 291 |
+
"bad_usage",
|
| 292 |
+
"empty_positions",
|
| 293 |
+
"multiple_positions",
|
| 294 |
+
],
|
| 295 |
+
)
|
| 296 |
+
async def test_rejects_unusable_engine_results(mutation):
|
| 297 |
+
backend, engine, client = await make_backend()
|
| 298 |
+
|
| 299 |
+
def mutate(original):
|
| 300 |
+
result = copy.deepcopy(original)
|
| 301 |
+
meta = result["meta_info"]
|
| 302 |
+
entries = meta["output_token_ids_logprobs"][0]
|
| 303 |
+
if mutation == "decoded":
|
| 304 |
+
meta["completion_tokens"] = 1
|
| 305 |
+
elif mutation == "missing_label":
|
| 306 |
+
entries.pop()
|
| 307 |
+
elif mutation == "duplicate_label":
|
| 308 |
+
entries.append(entries[0])
|
| 309 |
+
elif mutation == "nan":
|
| 310 |
+
# JSON null is also rejected; transport JSON forbids NaN values.
|
| 311 |
+
entries[0][0] = None
|
| 312 |
+
elif mutation == "positive":
|
| 313 |
+
entries[0][0] = 1.0
|
| 314 |
+
elif mutation == "missing_usage":
|
| 315 |
+
del meta["completion_tokens"]
|
| 316 |
+
elif mutation == "bad_usage":
|
| 317 |
+
meta["prompt_tokens"] = True
|
| 318 |
+
elif mutation == "empty_positions":
|
| 319 |
+
meta["output_token_ids_logprobs"] = []
|
| 320 |
+
elif mutation == "multiple_positions":
|
| 321 |
+
meta["output_token_ids_logprobs"].append(entries)
|
| 322 |
+
return result
|
| 323 |
+
|
| 324 |
+
engine.mutate_result = mutate
|
| 325 |
+
try:
|
| 326 |
+
labels, tokens = backend.labels(2)
|
| 327 |
+
with pytest.raises(AdapterError) as error:
|
| 328 |
+
await backend.evaluate("look", [], labels, tokens, None)
|
| 329 |
+
assert error.value.status == 502
|
| 330 |
+
assert error.value.code == "invalid_engine_response"
|
| 331 |
+
finally:
|
| 332 |
+
await client.aclose()
|
| 333 |
+
|
| 334 |
+
|
| 335 |
+
@pytest.mark.asyncio
|
| 336 |
+
@pytest.mark.parametrize(
|
| 337 |
+
"failure,status,code",
|
| 338 |
+
[
|
| 339 |
+
(httpx.ReadTimeout("secret upstream location"), 504, "engine_timeout"),
|
| 340 |
+
(httpx.ConnectError("secret upstream location"), 502, "engine_unavailable"),
|
| 341 |
+
(httpx.Response(401, text="secret-key"), 502, "engine_http_error"),
|
| 342 |
+
(httpx.Response(422, text="sensitive prompt"), 422, "engine_rejected_request"),
|
| 343 |
+
(httpx.Response(200, text="not-json secret"), 502, "invalid_engine_response"),
|
| 344 |
+
],
|
| 345 |
+
)
|
| 346 |
+
async def test_transport_failures_are_sanitized(failure, status, code):
|
| 347 |
+
backend, engine, client = await make_backend()
|
| 348 |
+
engine.fail_path, engine.failure = "/generate", failure
|
| 349 |
+
try:
|
| 350 |
+
labels, tokens = backend.labels(2)
|
| 351 |
+
with pytest.raises(AdapterError) as error:
|
| 352 |
+
await backend.evaluate("look", [], labels, tokens, None)
|
| 353 |
+
assert error.value.status == status
|
| 354 |
+
assert error.value.code == code
|
| 355 |
+
assert "secret" not in str(error.value) and "sensitive" not in str(error.value)
|
| 356 |
+
finally:
|
| 357 |
+
await client.aclose()
|
| 358 |
+
|
| 359 |
+
|
| 360 |
+
@pytest.mark.asyncio
|
| 361 |
+
async def test_special_labels_removed_and_start_idempotent():
|
| 362 |
+
engine = Engine()
|
| 363 |
+
engine.special_label = "A"
|
| 364 |
+
backend, engine, client = await make_backend(engine)
|
| 365 |
+
try:
|
| 366 |
+
labels, _ = backend.labels(2)
|
| 367 |
+
assert labels == ("B", "C")
|
| 368 |
+
count = len(engine.calls)
|
| 369 |
+
await backend.start()
|
| 370 |
+
assert len(engine.calls) == count
|
| 371 |
+
with pytest.raises(AdapterError):
|
| 372 |
+
backend.labels(255)
|
| 373 |
+
finally:
|
| 374 |
+
await client.aclose()
|
| 375 |
+
|
| 376 |
+
|
| 377 |
+
@pytest.mark.asyncio
|
| 378 |
+
@pytest.mark.parametrize(
|
| 379 |
+
"config,code",
|
| 380 |
+
[
|
| 381 |
+
({"speculative_algorithm": "EAGLE"}, "speculation_not_supported"),
|
| 382 |
+
({"skip_tokenizer_init": True}, "tokenizer_unavailable"),
|
| 383 |
+
],
|
| 384 |
+
)
|
| 385 |
+
async def test_startup_validates_engine_configuration(config, code):
|
| 386 |
+
engine = Engine()
|
| 387 |
+
engine.config.update(config)
|
| 388 |
+
async with httpx.AsyncClient(transport=httpx.MockTransport(engine)) as client:
|
| 389 |
+
backend = SGLangBackend("http://engine", "qwen", client=client)
|
| 390 |
+
with pytest.raises(AdapterError) as error:
|
| 391 |
+
await backend.start()
|
| 392 |
+
assert error.value.code == code
|
| 393 |
+
|
| 394 |
+
|
| 395 |
+
@pytest.mark.asyncio
|
| 396 |
+
async def test_nonfinite_raw_json_logprob_rejected():
|
| 397 |
+
backend, engine, client = await make_backend()
|
| 398 |
+
engine.fail_path = "/generate"
|
| 399 |
+
engine.failure = httpx.Response(
|
| 400 |
+
200,
|
| 401 |
+
text='{"meta_info":{"completion_tokens":0,"prompt_tokens":1,"output_token_ids_logprobs":[[[NaN,1065],[-1.0,1066]]]}}',
|
| 402 |
+
)
|
| 403 |
+
try:
|
| 404 |
+
labels, tokens = backend.labels(2)
|
| 405 |
+
with pytest.raises(AdapterError) as error:
|
| 406 |
+
await backend.evaluate("look", [], labels, tokens, None)
|
| 407 |
+
assert error.value.code == "invalid_engine_response"
|
| 408 |
+
finally:
|
| 409 |
+
await client.aclose()
|
tekken.json
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:600bb27946565481ecf51ba8aee252e49b9a68507866080ac9c30185bb312843
|
| 3 |
+
size 16753784
|
tokenizer.json
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:d5f6046775b112f0e2d456ee9dba450684ab964fe5c4e231599bdc6773028135
|
| 3 |
+
size 17078128
|