Download config.json from zerodegress/NanoJev-bf16: direct link, hf CLI and curl.
- Browser
- Download file 2.33 kB
-
https://huggingface.co/zerodegress/NanoJev-bf16/resolve/main/config.json
- Command line
-
hf download hf://zerodegress/NanoJev-bf16/config.json
-
curl -L -o config.json https://huggingface.co/zerodegress/NanoJev-bf16/resolve/main/config.json
2.33 kB
| { | |
| "input": "/home/rwang/openjev_codex_20260917/research/private_navigation_v3/views/coords_multi.jsonl", | |
| "output_dir": "/home/rwang/openjev_codex_20260917/runs/v3_teacher_coords_multi_seed17", | |
| "model": "Qwen/Qwen3-0.6B", | |
| "revision": "c1899de289a04d12100db370d81485cdf75e47ca", | |
| "init_checkpoint": "/home/rwang/openjev_codex_20260917/runs/v2_teacher_seed17", | |
| "objective": "teacher", | |
| "set_head": "attention", | |
| "steps": 1200, | |
| "head_steps": 0, | |
| "batch_questions": 12, | |
| "microbatch_questions": 4, | |
| "max_microbatch_tokens": 6000, | |
| "eval_every": 50, | |
| "max_length": 512, | |
| "seed": 17, | |
| "backbone_lr": 2e-05, | |
| "head_lr": 0.0002, | |
| "head_warmup_lr": 0.001, | |
| "precision": "bf16", | |
| "disable_native_triton": true, | |
| "validate_only": false, | |
| "self_check": false, | |
| "schema_version": "openjev-decision-pipeline-v1", | |
| "resolved_model_revision": "c1899de289a04d12100db370d81485cdf75e47ca", | |
| "initialization": "local DecisionModel warm start, fresh optimizer", | |
| "data_sha256": { | |
| "/home/rwang/openjev_codex_20260917/research/private_navigation_v3/views/coords_multi.jsonl": "3a638666206e8ffacadd8beeaf0246687e546716deb8b06adf5e23a228890cdc" | |
| }, | |
| "deps": { | |
| "torch": "2.14.0", | |
| "transformers": "5.17.0", | |
| "safetensors": "0.8.0" | |
| }, | |
| "gpu": "NVIDIA A100-SXM4-80GB", | |
| "parameter_storage": "float32", | |
| "forward_autocast": "bfloat16", | |
| "parameter_count": 596250498, | |
| "train_questions": 10105, | |
| "all_train_questions": 10152, | |
| "schema_counts": { | |
| "records": 3984, | |
| "questions_by_split": { | |
| "train": 10152, | |
| "dev": 360, | |
| "calibration": 360, | |
| "test": 720, | |
| "ood": 360 | |
| }, | |
| "eligible_by_split_objective": { | |
| "calibration/gold_distribution": 360, | |
| "calibration/teacher": 359, | |
| "dev/gold_distribution": 360, | |
| "dev/teacher": 357, | |
| "ood/gold_distribution": 360, | |
| "ood/teacher": 357, | |
| "test/gold_distribution": 720, | |
| "test/teacher": 715, | |
| "train/gold_distribution": 10152, | |
| "train/teacher": 10105 | |
| } | |
| }, | |
| "loss": "per-question forward soft CE, equal question weight; complete candidate normalization", | |
| "selection": "minimum dev target CE; held-out test first evaluated after training and checkpoint selection", | |
| "parallelism": "complete candidate paths batched in one backbone call per microbatch; no prefix sharing" | |
| } | |