{ "bits": 8, "backend": "Linux MLX CUDA 12 on NVIDIA B200", "runtime": "stock mlx_lm.server 0.31.3", "thinking": false, "temperature": 0, "endpoint": "/v1/chat/completions", "scope": "Two authored HTTP smoke checks; Mac Metal pending. No agent or tool-call compatibility claim.", "checks": [ { "id": "arithmetic", "request": { "model": "default_model", "messages": [ { "role": "user", "content": "Compute 2+2. Answer briefly." } ], "temperature": 0, "max_tokens": 128, "stream": false }, "response": { "id": "chatcmpl-ed4928d9-5c3b-436c-a4c2-111ac3b0632d", "system_fingerprint": "0.31.3-0.32.2-Linux-5.10.134-19.1.al8.x86_64-x86_64-with-glibc2.39-sm_100", "object": "chat.completion", "model": "default_model", "created": 1790974928, "choices": [ { "index": 0, "finish_reason": "stop", "message": { "role": "assistant", "content": "4" } } ], "usage": { "prompt_tokens": 21, "completion_tokens": 2, "total_tokens": 23, "prompt_tokens_details": { "cached_tokens": 0 } } }, "passed": true }, { "id": "json", "request": { "model": "default_model", "messages": [ { "role": "user", "content": "Return only valid JSON with no markdown. The object must be exactly {\"city\":\"Paris\",\"count\":3}." } ], "temperature": 0, "max_tokens": 128, "stream": false }, "response": { "id": "chatcmpl-bfe70a66-61b3-471a-ab86-f16196b0c452", "system_fingerprint": "0.31.3-0.32.2-Linux-5.10.134-19.1.al8.x86_64-x86_64-with-glibc2.39-sm_100", "object": "chat.completion", "model": "default_model", "created": 1790974977, "choices": [ { "index": 0, "finish_reason": "stop", "message": { "role": "assistant", "content": "{\"city\":\"Paris\",\"count\":3}" } } ], "usage": { "prompt_tokens": 34, "completion_tokens": 10, "total_tokens": 44, "prompt_tokens_details": { "cached_tokens": 0 } } }, "passed": true } ], "owned_server_stopped": true }