Download tt_kernel_manifest.json from tt-hous/clef: direct link, hf CLI and curl.
- Browser
- Download file 33.9 kB
-
https://huggingface.co/tt-hous/clef/resolve/main/tt_kernel_manifest.json
- Command line
-
hf download hf://tt-hous/clef/tt_kernel_manifest.json
-
curl -L -o tt_kernel_manifest.json https://huggingface.co/tt-hous/clef/resolve/main/tt_kernel_manifest.json
33.9 kB
| { | |
| "schema_version": "5.1", | |
| "name": "clef", | |
| "tt_metal_version": "0.65.2.dev10918", | |
| "arch": "blackhole", | |
| "device_count": 2, | |
| "producer": { | |
| "tt_kernel_version": "0.1.0", | |
| "created_at": "2026-10-06T14:16:58.486489+00:00", | |
| "hostname": "qb2-120-p11t02" | |
| }, | |
| "weights": { | |
| "repo_id": "Cloudflare/clef", | |
| "revision": "2f3de3dd85f379784083b0814d997ab627200f0c", | |
| "allow_patterns": null, | |
| "ignore_patterns": null, | |
| "repo_type": "model" | |
| }, | |
| "mesh": null, | |
| "entrypoint": null, | |
| "resources": null, | |
| "capabilities": null, | |
| "env": {}, | |
| "bundled": null, | |
| "deps": null, | |
| "container": { | |
| "image": { | |
| "registry": "hf", | |
| "repository": "clef", | |
| "tag": "tt-model/clef:2241a811ee93", | |
| "digest": "sha256:2241a811ee93f8543bc77dcdfeb02210bebcb4a112b07bd87bd7a9d2bc9d1310" | |
| }, | |
| "kind": "tt-dit-server", | |
| "runtime": { | |
| "app": "models.autoports.cloudflare_clef.tt.server:app", | |
| "mesh_shape_env": "CLEF_MESH_SHAPE", | |
| "packages": [ | |
| "fastapi==0.142.2", | |
| "uvicorn==0.54.0", | |
| "pydantic==2.9.2", | |
| "pillow==12.3.0", | |
| "numpy==1.26.4", | |
| "transformers==5.12.1", | |
| "tokenizers==0.22.2", | |
| "safetensors==0.8.0", | |
| "huggingface_hub==1.16.1", | |
| "tqdm==4.66.3", | |
| "av==19.0.1", | |
| "torchvision==0.26.0+cpu" | |
| ], | |
| "lock": "requirements.lock" | |
| }, | |
| "serve": { | |
| "hardware": null, | |
| "mesh_device": null, | |
| "port": 8008, | |
| "max_model_len": null, | |
| "max_num_seqs": null, | |
| "block_size": null, | |
| "server_timeout": null, | |
| "capabilities": null, | |
| "additional_config": {}, | |
| "args": [], | |
| "env": { | |
| "CLEF_MAX_STATE": "16384", | |
| "CLEF_PREFIX_CACHE": "4", | |
| "CLEF_TRACED": "0", | |
| "CLEF_PLANNER": "1" | |
| } | |
| }, | |
| "serve_profiles": [ | |
| { | |
| "hardware": "p150x2", | |
| "mesh_device": "P150x2", | |
| "port": null, | |
| "max_model_len": null, | |
| "max_num_seqs": null, | |
| "block_size": null, | |
| "server_timeout": null, | |
| "capabilities": null, | |
| "additional_config": {}, | |
| "args": [], | |
| "env": {}, | |
| "name": "p150x2", | |
| "description": "Two p150-class Blackhole chips (two p150 boards, or one p300 board) opened as a 1x2 mesh, tensor parallel 2 (TP=2), one worker. CLEF_MESH_SHAPE is 1x2." | |
| } | |
| ], | |
| "default_profile": "p150x2", | |
| "code_dir": "code", | |
| "verify": [ | |
| "import models.autoports.cloudflare_clef.tt.server as s; assert s.app", | |
| "from models.autoports.cloudflare_clef.tt.engine import ClefEngine; assert ClefEngine", | |
| "from models.autoports.cloudflare_clef.tt.head import JointSchemaHead, load_head; assert JointSchemaHead and load_head", | |
| "from models.autoports.cloudflare_clef.tt.loader import ClefModelArgs, LmHeadRows; assert ClefModelArgs and LmHeadRows", | |
| "from models.autoports.cloudflare_clef.tt.encode import encode, split_for_cache, cache_key; assert encode and split_for_cache and cache_key", | |
| "from models.autoports.cloudflare_clef.tt.server import parse_mesh_shape; assert parse_mesh_shape('1x2', 'P150x2') == (1, 2)", | |
| "from models.autoports.cloudflare_clef.tt.server import parse_mesh_shape, mesh_plan; assert mesh_plan((1, 2), 2) == {'fabric': 'FABRIC_1D', 'open': (1, 2), 'submeshes': [], 'parent': None}; assert mesh_plan((1, 2), 4)['parent'] == '1x4'", | |
| "from models.autoports.cloudflare_clef.tt import precision_defaults; assert precision_defaults.profile_name() == 'selected'; assert precision_defaults.profile()['QWEN36_MLP_GATE_UP_DTYPE'] == 'bfp8'", | |
| "from models.demos.blackhole.qwen36.tt.vision.model import DropInVisionTransformer; from models.demos.blackhole.qwen36.tt.vision.vision_model_config import VisionModelArgs; assert DropInVisionTransformer and VisionModelArgs", | |
| "from models.tt_dit.utils.tensor import prepare_for_fused_swiglu; assert prepare_for_fused_swiglu", | |
| "from transformers import Qwen3_5ForConditionalGeneration, AutoProcessor; assert Qwen3_5ForConditionalGeneration and AutoProcessor", | |
| "from transformers.models.qwen3_vl.processing_qwen3_vl import Qwen3VLProcessor; from transformers.models.qwen2_vl.image_processing_qwen2_vl import Qwen2VLImageProcessor; from transformers.models.qwen3_vl.video_processing_qwen3_vl import Qwen3VLVideoProcessor; import torchvision, av, PIL; assert Qwen3VLProcessor and Qwen2VLImageProcessor and Qwen3VLVideoProcessor", | |
| "from safetensors.torch import load_file; from huggingface_hub import snapshot_download; assert load_file and snapshot_download" | |
| ], | |
| "built": { | |
| "image": "tt-model/clef:2241a811ee93", | |
| "repo": "tt-hous/clef", | |
| "tt_model_version": "0.1.0", | |
| "created_at": "2026-10-06T14:13:56+00:00", | |
| "tt_metal": { | |
| "sha": "c72c144e6e4d4cb1a29ee827cfb98d66b9883c30", | |
| "describe": "v0.79.0-dev20260905-1214-gc72c144e6e", | |
| "dirty": true, | |
| "scm_version": "0.65.2.dev10918", | |
| "mode": "local", | |
| "remote": "git@github.com:housTT/tt-metal.git", | |
| "branch": "hous/clef-bringup", | |
| "pushed": true | |
| }, | |
| "code_sha256": "f2f49abe472ac6d36f700fd779daab586ff01b9e979651917a57989025090a59", | |
| "image_digest": "sha256:2241a811ee93f8543bc77dcdfeb02210bebcb4a112b07bd87bd7a9d2bc9d1310" | |
| }, | |
| "card": { | |
| "description": "Clef is a 27B multimodal decision model from Cloudflare, post-trained from Qwen/Qwen3.8-27B. It reads one state (text, JSON, images or video) and a schema of typed questions (choice, score, noul) and returns a probability for every allowed option of every question in one forward pass, without generating text. This package serves it on two Tenstorrent Blackhole chips (tensor parallel 2) through the model's own FastAPI server, which implements the Jev/SystemOne API (POST /v1/systemone), so SystemOne clients work against it unchanged. The weights pointer above covers everything the server loads: the backbone shards, the vision encoder, the joint schema head (joint_head.safetensors), the tokenizer and processor configuration, and the author's joint_schema_model.py all live in the Cloudflare/clef repo at the pinned revision.\n", | |
| "quickstart": "### Request shape\n\nThe server implements the Jev/SystemOne API. A request has `model`, `state` (any string or\nJSON value), `questions` (question id to question), and optional `images` and `videos`.\nA question has `type` (`noul`, `choice` or `score`), optional `instructions`, and\n`criteria` (a mapping of option id to description for `choice`, a list of ordered option\ndescriptions for `score`, optional `true` and `false` descriptions for `noul`). The\nserver accepts each image as base64-encoded image bytes (a bare string or a\n`data:image/...;base64,` URL) or as an http(s) URL, and each video as a list of frames in\nthe same encodings. No API key is checked unless the server was started with `CLEF_API_KEY`\nset. The port is 8008 (`tt-model serve ... --port 8008`).\n\n### Text example (the Jev/SystemOne example of the upstream model card), with curl\n\n```bash\ncurl -s http://127.0.0.1:8008/v1/systemone -H 'content-type: application/json' -d '{\n \"model\": \"clef\",\n \"state\": \"Our checkout started returning errors and orders are blocked.\",\n \"questions\": {\n \"department\": {\"type\": \"choice\", \"instructions\": \"Which team should handle the message?\",\n \"criteria\": {\"billing\": \"Payments or invoices\", \"technical\": \"Bugs or outages\"}},\n \"urgency\": {\"type\": \"score\", \"criteria\": [\"Can wait\", \"This week\", \"Today\"]},\n \"outage\": {\"type\": \"noul\", \"instructions\": \"Is a service down?\"}\n }\n}'\n```\n\n### Image example (the receipt example of the upstream model card), with curl\n\n```bash\nIMG=$(base64 -w0 receipt.jpg)\ncurl -s http://127.0.0.1:8008/v1/systemone -H 'content-type: application/json' -d \"{\n \\\"model\\\": \\\"clef\\\",\n \\\"state\\\": {\\\"task\\\": \\\"Review the attached receipt.\\\"},\n \\\"images\\\": [\\\"$IMG\\\"],\n \\\"questions\\\": {\\\"legible\\\": {\\\"type\\\": \\\"noul\\\", \\\"instructions\\\": \\\"Is the receipt total legible?\\\"}}\n}\"\n```\n\n### The same two requests from Python\n\n```python\nimport base64\nimport requests\n\nURL = \"http://127.0.0.1:8008/v1/systemone\"\n\ntext = requests.post(URL, timeout=300, json={\n \"model\": \"clef\",\n \"state\": \"Our checkout started returning errors and orders are blocked.\",\n \"questions\": {\n \"department\": {\"type\": \"choice\", \"instructions\": \"Which team should handle the message?\",\n \"criteria\": {\"billing\": \"Payments or invoices\", \"technical\": \"Bugs or outages\"}},\n \"urgency\": {\"type\": \"score\", \"criteria\": [\"Can wait\", \"This week\", \"Today\"]},\n \"outage\": {\"type\": \"noul\", \"instructions\": \"Is a service down?\"},\n },\n}).json()\nprint(text[\"answers\"][\"department\"][\"choice\"], text[\"answers\"][\"urgency\"][\"score\"], text[\"answers\"][\"outage\"][\"noul\"])\n\nwith open(\"receipt.jpg\", \"rb\") as f:\n image_b64 = base64.b64encode(f.read()).decode()\nimage = requests.post(URL, timeout=300, json={\n \"model\": \"clef\",\n \"state\": {\"task\": \"Review the attached receipt.\"},\n \"images\": [image_b64],\n \"questions\": {\"legible\": {\"type\": \"noul\", \"instructions\": \"Is the receipt total legible?\"}},\n}).json()\nprint(image[\"answers\"][\"legible\"][\"noul\"], image[\"usage\"][\"input_tokens\"], image[\"latency_ms\"])\n```\n\n### Reference behaviour\n\nThe response body is the one the author's `joint_schema_model.systemone()` returns\n(`model`, `answers`, `usage`) plus `latency_ms`. The author's code on CPU (bf16) answers the\ntext example above with `department` = technical (probability 0.9155), `urgency` score 1.82\n(`Today` at 0.8651) and `outage` noul 0.8995, with 300 input tokens. This package answers it\nwith `department` = technical (0.9138), `urgency` score 1.83 (`Today` at 0.8706) and `outage`\nnoul 0.9011, the same 300 input tokens, in 270 ms of model time (`latency_ms` 269.7, eager engine, measured from the pushed image); the largest\nprobability difference over the three questions is 0.0055. The receipt image of the upstream\ncard is not public; the image path was checked with a New Yorker cartoon and five candidate\ncaptions (330 input tokens): caption D at 0.6402 against 0.6475 on the CPU, 498 ms on a new\nstate and 237 ms when the state is in the prefix cache. The full parity numbers are in\nPerformance.\n", | |
| "architecture": "Qwen3_5ForConditionalGeneration backbone, 64 layers (48 Gated DeltaNet and 16 full-attention), hidden 5120, with a 27-block Qwen3.5 vision tower on device and a 65M-parameter joint schema head on the host CPU; prefill only, no decode", | |
| "status": "Experimental community bring-up", | |
| "intended_use": "Typed decisions over a state of up to 16,384 tokens, with or without images: classification, routing, triage, intent detection, extraction choices, policy and eligibility checks, multiple-choice reading comprehension, judging a proposed answer against stated criteria, and matching a caption or a label to an image. Workflows that act on confidence, with thresholds frozen on a labelled sample of the user's own workload.\n", | |
| "out_of_scope_use": "Text generation, chat, summarisation or open-ended question answering; the model only scores the options it is given. Fully automated decisions with legal, medical, financial, employment or similar consequences for people, without human review. The SystemOne permute and separate routes are not served. Languages other than English are not evaluated.\n", | |
| "usage": "### Routes\n\n| Route | Body |\n|---|---|\n| `POST /v1/systemone` | `{\"model\", \"answers\", \"usage\": {\"input_tokens\", \"output_tokens\"}, \"latency_ms\"}`. Answer shapes follow the upstream `systemone()`: `choice` (`choice`, `confidence`, `probabilities`), `noul` (`noul`, the probability of true), `score` (`score`, `confidence`, `legend`, `probabilities`). |\n| `GET /v1/models` | One card per served name (`clef`) with the weights revision, device, mesh plan, engine mode (`eager`), `prefix_planner`, the precision knobs, the limits and the prefix-cache counters. |\n| `GET /health`, `GET /v1/health` | `{\"status\": \"ok\", \"workers\", \"queued\"}`. Never behind the API key. |\n\nA state over `CLEF_MAX_STATE` tokens, or a schema over `CLEF_MAX_TAIL` tokens, returns 422\nwith the token count. A request whose questions have no criteria (other than `noul`) returns\n422, as upstream `systemone()` raises. In the traced opt-in mode (below) an image grid outside\nthe warm list, or a request with more than one image or video, returns 422.\n\n### Server environment\n\nSet these through the serve profile env of the manifest (`tt-model serve tt-hous/clef --profile p150x2`;\n`tt-model serve` has no `--env` flag) or a bare `docker run --env`. The defaults below are the\nshipped configuration (eager engine, prefix planner on), the one every number in Performance\nwas measured with.\n\n| Variable | Default | Meaning |\n|---|---|---|\n| `HF_MODEL` | `Cloudflare/clef` (set by tt-model) | Hub id or snapshot directory of the weights. `CLEF_MODEL`, when set, is read first; the server writes the resolved snapshot directory back into `CLEF_MODEL`. |\n| `CLEF_REVISION` | `2f3de3dd85f379784083b0814d997ab627200f0c` | Revision used when the model id has no `@rev`. |\n| `HF_HUB_OFFLINE` | unset | `1` resolves Hub ids from the local cache only. |\n| `CLEF_MESH_SHAPE` | `1x2` (set by tt-model from the profile) | Mesh the server opens. `1x2` is one TP=2 worker. With exactly two visible chips the server opens the 1x2 mesh directly under `FABRIC_1D`; with more it opens a 1x4 parent and takes the 1x2 submesh at `CLEF_SUBMESH_OFFSET`. `1x4` and `2x2` (two TP=2 workers) are the planned `p150x4` profile, not in this build. |\n| `MESH_DEVICE` | `P150x2` (set by tt-model) | The mesh SKU; used only when `CLEF_MESH_SHAPE` is unset. |\n| `CLEF_PARENT_MESH` | unset (auto) | `1x4` (`FABRIC_1D`) or `2x2` (`FABRIC_2D`): forces the parent-plus-submesh open for a `1x2` target. Needs four visible chips. |\n| `CLEF_SUBMESH_OFFSET` | `0,0` | Offset of the 1x2 submesh inside the parent. |\n| `CLEF_MAX_STATE` | `16384` | Engine `max_state_len` and the 422 limit for the state piece (system prompt and media placeholder tokens included). The upstream `encode_record` limit. |\n| `CLEF_MAX_TAIL` | `4096` | Engine `max_tail_len` and the 422 limit for the schema plus the suffix. |\n| `CLEF_TRUNCATE_STATES` | `0` | `1` keeps the first `CLEF_MAX_STATE` tokens instead of refusing and marks the response. |\n| `CLEF_PREFIX_CACHE` | `4` | States kept in the prefix cache (KV pages plus the Gated DeltaNet state snapshot per slot), clamped by the engine's snapshot slots; `0` disables the cache. |\n| `CLEF_PLANNER` | `1` | The prefix planner picks the cached prefix length so a cache miss takes the fewest prefill passes. `0` restores the fixed 128-aligned rule. Reported as `prefix_planner` in `/v1/models`. |\n| `CLEF_TRACED` | `0` | `0` is the shipped eager engine: it accepts any image grid and any number of images or videos per request. `1` is the traced opt-in for a deployment with a fixed set of image grids: it needs `CLEF_VISION_WARM_GRID`, refuses an image or video whose grid is not in that list and a request with more than one image or video (422), and reserves `CLEF_TRACE_REGION` of DRAM. It gains 0 to 3 percent of latency at this depth. |\n| `CLEF_VISION_WARM_GRID` | unset (required when `CLEF_TRACED=1`) | `;`-separated `t,h,w` patch grids the traced server warms before capture and then accepts, for example `1,16,20;1,22,38;1,26,36;1,28,36;1,28,38;1,40,50;2,16,20`. Ignored in eager mode. |\n| `CLEF_TRACE_REGION` | `1073741824` when `CLEF_TRACED=1`, else `0` | `trace_region_size` for the mesh open. |\n| `QWEN_GDN_CONV` | `fir` when `CLEF_TRACED=1`; not read in eager mode | Gated DeltaNet causal conv kernel of the traced body. `kda` changes the numerics (one argmax flip on the 16 reference records). |\n| `CLEF_API_KEY` | unset | Bearer key for `/v1/*` (health routes stay open). |\n| `CLEF_ALLOW_REMOTE_IMAGES` | `1` | `0` refuses http(s) image URLs with 422. Fetches use a 10 s timeout and a 20 MiB cap. |\n| `CLEF_MODEL_NAMES` | `clef` | Comma-separated names listed by `/v1/models` and accepted in `model`. |\n| `CLEF_WARMUP` | `1` | `0` skips the warmup request per worker at startup. |\n| `CLEF_PRECISION` | unset (`selected`) | Precision profile of `tt/precision_defaults.py`: `selected` (the shipped knobs, equal to the stage 1 configuration) or `stage1`. The `QWEN36_*` knobs it sets can be overridden one by one. |\n| `CLEF_FAKE_ENGINE` | `0` | `1` opens no device and serves deterministic pseudo-random hidden rows through the real head, tokenizer and processor (API tests on a CPU). |\n### Browser demo and CLI\n\nThe package ships a static demo page and a command-line client in\n`code/models/autoports/cloudflare_clef/demo/` of this repo (inside the image at\n`/opt/tt-metal/models/autoports/cloudflare_clef/demo`). The page sends `POST /v1/systemone`\nfrom the browser and draws every option probability: a support-triage preset from the\nannouncement, the invoice JSON state from the upstream card, a repeated-state preset that shows\nthe prefix cache, two New Yorker cartoons and a synthetic receipt as image requests, and a\nreplay of 24 labelled records (ARC-Challenge, BANKING77, New Yorker) with running accuracy.\nAgainst this image on two chips every preset answers (8 of 8) and the replay scores 22 of 24.\n\nWith the server up on port 8008:\n\n```bash\nhf download tt-hous/clef --include \"code/models/autoports/cloudflare_clef/demo/*\" --local-dir clef-pkg\ncd clef-pkg/code/models/autoports/cloudflare_clef/demo\nDEMO_BIND=0.0.0.0 ./run.sh http://<server-host>:8008\n```\n\nOpen `http://<demo-host>:8080/#server=http://<server-host>:8008`. The page only needs plain\nHTML, CSS and JavaScript; fonts come from Google Fonts with system fallbacks. The same\ndirectory is inside the image, so it can be served from there without a download:\n\n```bash\ndocker run --rm -p 8080:8080 --entrypoint python3 $(docker images -q tt-model/clef | head -1) \\\n -m http.server 8080 --bind 0.0.0.0 --directory /opt/tt-metal/models/autoports/cloudflare_clef/demo\n```\n\nThe CLI needs Python 3 and nothing else: `python3 quickstart.py --all` sends every preset,\n`python3 quickstart.py --replay` runs the labelled replay, `--preset \"<title>\"` sends one, and\n`--show-request` prints the request bodies without sending them. `--server` points it at\nanother host. Presets live in `presets.json`, shared by the page and the CLI; the\ndemo README in the same directory documents every panel and the measured numbers.\n", | |
| "performance": "Model time per request (new / repeated state) and requests per second at 64 concurrent\nclients, in the format of the kev model card, measured on 2026 Oct 05 with the host-side\n`scripts/serving_bench_remote.py` against the shipped server configuration (eager engine,\nprefix planner on) on a p300c box: two Blackhole chips of one p300 board as the 1x2 mesh.\n\n| Device | 6 questions, short state | 5 questions, 2,200-token state | Requests/s, 64 clients |\n|---|---|---|---|\n| P150x2 (2 chips, TP=2), eager engine with the prefix planner [1] | 412.6 / 410.4 ms | 1,231.2 / 533.8 ms | 2.36 |\n\nStartup: 106 s from the server's first log line to `Application startup complete` (mesh open\n4 s, weights converted from the safetensors and the model built in 100.6 s, one warmup\nrequest); no tensor cache is written, so every start pays it.\n\n- New / repeated state: the first number of each pair is a request whose state the server has not seen (prefix-cache miss: the state prefix is prefilled, then the schema tail runs); the second is a request whose state is in the prefix cache (hit: the KV pages and the Gated DeltaNet state of the slot are restored, then the schema tail runs). The worker holds `CLEF_PREFIX_CACHE` = 4 states. A state under 128 tokens has no cached prefix, so new and repeated are equal by construction in the short-state column (that state is 122 tokens with the system prompt).\n- Latency is the server's `latency_ms` field: the wall time of the model section of the request on its worker thread (state prefill or cache restore, schema continuation, vision tower when images are present, joint schema head on the host in fp32), without queue wait and HTTP. Clef decides every question of a request jointly in one pass, so one request runs on one TP group and there is no per-question fan-out.\n- Precision: the `selected` profile of `tt/precision_defaults.py`, which the datatype sweep set equal to the stage 1 configuration: MLP gate / up weights bfp8, MLP down bf16, attention and Gated DeltaNet projections bfp8, MLP matmuls HiFi2 with fp32 accumulation, Gated DeltaNet decay gate fp32, fp32 Gated DeltaNet recurrent state, bf16 KV cache, bf16 embeddings, vision tower bf16 weights and activations with fp32 accumulation, joint schema head fp32 on the host.\n- Code: tt-metal commit `c72c144e6e4` (branch `hous/clef-bringup`, the tree the image was built from) on top of `d76d41fb52c` (the Kev bring-up head) over `7eac776e926` (origin/main 2026 Oct 01). Every code path shipped in this package is byte-identical to the stage 3 serving commit `e75b184fc69`, which produced every number above; the commits between them add only documentation, evaluation scripts and the demo directory (shipped under `code/` but not imported by the server).\n- Method: port of kev's `scripts/serving_bench.py` over HTTP. Latency columns are the median of 20 requests after 2 warm-up requests. Throughput is 256 requests of the 6-question short-state case over 64 distinct states per client level (1, 8, 32, 64 clients), two passes, the second timed; it is flat in the client count (2.35 to 2.36 req/s) because one TP group serves one request at a time, and the queue wait grows linearly with the clients. 64 distinct 2,200-token states give 0.81 req/s.\n- [1] One worker on the 1x2 mesh, tensor parallel 2. Latency detail (new / repeated, p50 ms): 3 questions, 346-token blog triage state 268.2 / 265.0; 2 questions, short state 267.9 / 257.8; 5 questions, 370-token state 426.0 / 423.1. A request costs about 220 ms of fixed work plus about 0.35 s per 1,024-token chunk of prefill; a 16,384-token state (the limit) takes 7,471.8 ms on a miss and 1,600.0 ms on a hit.\n\n### Image requests\n\nOne image record (a New Yorker cartoon, 320 patches, with five candidate captions, 330 input\ntokens): 497.7 ms on a new state and 237.1 ms on a repeated state; median 551 ms over the 8\nreference image records (320 to 2,000 patches). Vision tower parity against the HF\n`model.visual` on CPU on the 4 reference images: merged-output PCC 0.99880, 0.99822, 0.99909,\n0.99898 (bar 0.97); blocks 0, 12, 23 at 0.99856 or better (bar 0.99) and block 26 at 0.99694\nor better (bar 0.85); the 4-frame video record 0.99839. The largest image run end to end is\n57,344 patches ((1, 224, 256), 3584 x 4096 pixels, 14,336 image tokens): 33.2 s per request\nwarm; a 2048 x 2048 image (16,384 patches, 4,096 tokens) takes 5.7 s. Above 4,096 patches\nthe tower time grows about quadratically. The first image of a new patch grid compiles its\nprograms once per box: 1 to 2 s at 2,048 padded rows, 6 to 9 s at 4,096, up to 26 s at\n57,344.\n\n### Evaluations\n\nPublic benchmarks from the upstream model card, rendered once to SystemOne requests (no\nprompt tuning, no option dropping) and run on this server; a CPU bf16 control (the author's\nown `joint_schema_model.py`, same rendering) on a stratified 100-item sample of each\nbenchmark separates the rendering from the device precision. The Decision Index request data\nbehind the card numbers is not public, so the card numbers are indicative; this package does\nnot reproduce the Decision Index and does not claim to.\n\n| Benchmark (test split) | n | Metric | This package, full test set | 100-item sample: this package / CPU bf16 control | Model card |\n|---|---|---|---|---|---|\n| ARC-Challenge | 1,172 | accuracy | 97.8 | 96.0 / 95.0 | 97.7 |\n| BANKING77 | 3,080 | macro-F1 over 77 intents | 94.4 | 91.1 / 91.1 | 94.2 |\n| New Yorker caption matching (image, 5 captions) | 528 | accuracy | 58.3 | 60.0 / 60.0 | 69.5 |\n\nCoverage is complete on every file (0 rejected, 0 truncated, 0 unanswered). The 100-item\nsamples carry a binomial 95 percent interval of about plus or minus 4 points on ARC and plus\nor minus 10 on New Yorker, so the sample column compares the device to the CPU control, not\nto the card. On New Yorker the device and the CPU control agree on the sample (60.0 / 60.0),\nso the 11-point gap to the card is the rendering of the public dataset to a SystemOne\nrequest against the Decision Index rendering, which is not public, and not the device. On\nARC the device is one item above the control (+1.0 pp, one argmax flip at reference margin\n0.37 out of 100); on BANKING77 the two agree (0 flips at margin, 2 near ties).\n\nKev's labelled suites (`hard-v1`, `devtools-v1`, `documents-v1` test splits, SystemOne\nrequests, `kev.benchmark --remote` unchanged, `clean` block): accuracy, Brier and ECE\n(expected calibration error, 10 bins on the top probability). Coverage complete (every\nrecord evaluated, 0 rejected, 0 truncated). These suites are Kev-9B's own development\ndistribution; the Kev column is `jaredpalmer/kev-9b` served by its own port on one P150 of\nthe same box (`tt-hous/kev-9b`), two decision models on the same silicon and the same\nlabelled requests, not a ranking of the ports.\n\n| Suite (test split) | questions | accuracy | Brier | ECE | Kev-9B on one P150: accuracy / ECE |\n|---|---|---|---|---|---|\n| hard-v1 | 1,088 | 0.774 | 0.321 | 0.044 | 0.826 / 0.053 |\n| devtools-v1 | 1,073 | 0.741 | 0.399 | 0.143 | 0.787 / 0.101 |\n| documents-v1 | 936 | 0.893 | 0.169 | 0.044 | 0.896 / 0.015 |\n\n### Parity against the author's CPU reference\n\nOver HTTP against the served package (max |dp| is the largest absolute difference of an\noption probability; flips are argmax changes at a reference top-2 margin of at least 0.05):\n\n| Set | questions | max dp | flips at margin | reference |\n|---|---|---|---|---|\n| 16 reference text records | 27 | 0.072 (mean 0.011) | 0 | CPU bf16 |\n| same 16 records | 27 | 0.076 (mean 0.010) | 0 | CPU fp32 |\n| 8 reference image records | 8 | 0.046 (mean 0.020) | 0 | CPU bf16 (0.038 against fp32) |\n| 100 BANKING77 rows (1,824 to 1,871 tokens) | 100 | 0.110 | 0 (2 near ties) | CPU bf16 |\n| 8 long records, 2,637 to 16,384 tokens | 32 | 0.056 | 0 | CPU bf16 |\n| 64 development text records (dev64) | 95 | 0.192 | 1 (margin 0.17, dp 0.10) | CPU bf16 |\n| 16 development image records (dev16) | 16 | 0.175 | 0 | CPU bf16 |\n\nThe eager engine's probabilities equal the traced engine's bit for bit; the served\nprobabilities equal the engine's to the 4-decimal rounding of the API. The 16 reference\nrecords, the 8 image records and the 8 long records meet the bring-up bars (max dp at most\n0.10, 0 flips at margin); dev64 and dev16 do not on the max-dp bar, see Limitations.\n", | |
| "limitations": "- Hidden-state agreement with the HF reference is gated per layer, not per row: the teacher-forced per-layer PCC of every Gated DeltaNet layer is in the band of the attention layers (means 0.99972 to 0.99979 at 300 and 8,192 tokens, every layer except the last above 0.999). The whole-row PCC of the final hidden state against HF bf16 at 1,500 to 8,192 tokens is reported, not gated (all-rows mean 0.977 to 0.990; worst sampled row 0.76 at 1,500 tokens); HF bf16 itself fails a whole-row 0.99 bar against HF fp32 on 3.9 to 7.4 percent of rows, and the device is 3.5 to 5.9 times more often below that bar. The decision-level parity above is the gate that passes.\n- Two precision-sensitive near-decision records: on a 200-record development set the device agrees with the CPU bf16 reference on 274 of 278 argmaxes and is within 0.36 pp in accuracy, but two questions (`hard-v1/multi_hop/development/00017 paged`, dp 0.33, device wrong; `hard-v1/probability/development/00084 value`, dp 0.47, both wrong) move by up to 0.3 in probability while the reference is confident. No precision knob of the datatype sweep brings both into the normal band; a per-layer probe classifies them as precision sensitivity of a near-decision record (every layer inside the band above), not an op bug. The dev64 flip above (margin 0.17) is of the same kind. Expect a few such records per thousand on confident decisions.\n- Image evaluation: New Yorker caption matching is 58.3 against 69.5 on the card with the device equal to the CPU control on the sample, so the gap is the rendering of the public dataset, not the device. One development image record (`7b7ef383`, dev16) is at max dp 0.175 with the argmax kept (reference margin 0.36); the other 15 are at 0.052 or below. The vision tower runs in bf16 with fp32 accumulation inside the `qwen36` port; its upstream bfp8 configuration puts the 8 reference records at max dp 0.22.\n- One TP group: requests are served one at a time, throughput is 2.36 req/s for short states and 0.81 req/s for new 2,200-token states at any client count, and the queue wait grows linearly with the clients (27 s at 64 clients). A second TP=2 worker on four chips (profile `p150x4`) is planned and not in this build.\n- Startup is 106 s on every start: the weights convert from the safetensors each time (no tensor cache).\n- States are limited to `CLEF_MAX_STATE` tokens (16,384 by default, the upstream `encode_record` limit) and schemas to `CLEF_MAX_TAIL` (4,096); longer requests return 422 instead of the upstream silent truncation (`CLEF_TRUNCATE_STATES=1` opts into truncation). The context the HF config advertises (262,144) is not served.\n- Images run through the on-device vision tower; the largest image validated end to end is 57,344 patches (3584 x 4096 pixels, 14,336 image tokens, 33 s per request) and the limit above it is the 16,384-token request budget, not the device. A request with many or large images is slow (about quadratic in patches above 4,096). Videos and images in one request are not supported together (one media kind per request).\n- Per-grid first-request compile in eager mode: the first image of a patch grid the box has not seen compiles its programs once (1 to 2 s at up to 2,048 padded rows, 6 to 9 s at 4,096, 26 s at 57,344); the kernel cache under `~/.cache/tt-model/clef/cache` persists across container restarts, so this is a per-box cost.\n- Tracing is opt-in and constrained: `CLEF_TRACED=1` needs `CLEF_VISION_WARM_GRID`, refuses image grids outside that list and more than one image or video per request (422), reserves a 1 GiB trace region and gains 0 to 3 percent.\n- The prefix cache holds `CLEF_PREFIX_CACHE` states (4 at the 16,384-token limit); only the 128-aligned state prefix is cached, the schema always runs, and a state under 128 tokens is never cached. The joint schema head runs on the host CPU in fp32 over every row of the record, so the host part grows with the state length (a 16,384-token cache hit is 1,600 ms).\n- The direct 1x2 mesh open with exactly two visible chips is the path this profile takes inside its container; the bring-up box ran every measurement through the four-chip parent-plus-submesh path (the same chips, the same TP=2 worker), so the direct open is exercised for the first time by the package validation.\n- English only (the card's benchmarks are English; other languages are not evaluated here). Knowledge is set by the base model. Option order can change an answer.\n", | |
| "risks": "Calibrated probabilities can create unwarranted trust; measure accuracy and calibration on a labelled sample of your own data before setting thresholds, and monitor production error rates. The model card's benchmark numbers come from non-public Decision Index renderings and are indicative only; this package does not reproduce them. A small fraction of confident decisions moves by up to 0.3 in probability between the device and the CPU reference (see Limitations). The server is open unless `CLEF_API_KEY` is set and accepts http(s) image URLs unless `CLEF_ALLOW_REMOTE_IMAGES=0`; states, images and videos may contain personal or confidential data, so apply your own access control and do not expose the server to untrusted networks without a proxy.\n", | |
| "licensing": "Cloudflare/clef (backbone, vision encoder, joint schema head and joint_schema_model.py) is released under Apache-2.0; its model card states the base model Qwen/Qwen3.8-27B is Apache-2.0. The serving code under `code/` is tt-metal (Apache-2.0) plus the Clef port under `models/autoports/cloudflare_clef` (Apache-2.0, see its NOTICE), which vendors the joint schema head classes from joint_schema_model.py and reuses the serving design of the Apache-2.0 kev project. The port's browser demo (in `code/models/autoports/cloudflare_clef/demo`, also inside the image) ships ten New Yorker caption contest cartoons from the jmhessel/newyorker_caption_contest dataset under CC BY 4.0, as its NOTICE states.\n", | |
| "related": "Upstream model card: https://huggingface.co/Cloudflare/clef. Announcement: https://blog.cloudflare.com/clef-decision-models. Decision Index leaderboard: https://clef-evals.workers-ai-mle.workers.dev. The smaller variant Cloudflare/clef-flash is not packaged here. The same serving pattern on one Blackhole chip: tt-hous/kev-9b.\n", | |
| "license": { | |
| "id": "apache-2.0", | |
| "name": null, | |
| "link": null | |
| }, | |
| "pipeline_tag": "image-text-to-text", | |
| "base_model": [ | |
| "Qwen/Qwen3.8-27B", | |
| "Cloudflare/clef" | |
| ] | |
| } | |
| } | |
| } |