{ "sysone_version": "0.3.0", "backend": "sysone", "name": "diffcider-browser", "encoder": { "id": "dllm-hub/Qwen3-0.6B-diffusion-mdlm-v0.1", "revision": "c8d24a3f4adaeef46881b450e1bf7d1005203bd7", "source": "transformers", "hidden_size": 1024, "family": "a2d-qwen3", "modality": "text", "dtype": "float32", "weights": "adapter" }, "template": { "start": null, "sep": null, "end": null, "mask": "<|mask|>", "image": null, "row": "<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n<|im_start|>user\nState:\n{state}\n\nQuestion:\n{instructions}\n\n{options}<|im_end|>\n", "option": "{text}: {mask}", "max_len": 4096, "head_max_len": 2048, "option_tokens": 512, "truncate": "right", "option_sep": "\n", "kinds": { "choice": { "header": "For each option, mark Yes if it answers the question, otherwise No.\n", "option": "{text}: {mask}", "sep": "\n" }, "score": { "header": "For each option, mark Yes if it answers the question, otherwise No.\n", "option": "{text}: {mask}", "sep": "\n" }, "noul": { "header": "Answer Yes or No.\nAnswer:", "option": "{mask}", "sep": "" } }, "name": "jev" }, "tfms": [], "head": { "width": 1024, "layers": 0, "scorer": "yesno", "dropout": 0.1, "nhead": 16, "standardize": false, "readout": "anchor", "query": null, "query_width": null, "pair_dim": 256, "act": false, "regime": "lora" }, "temperature": { "choice": 1.2964, "choice:11+": 1.269, "choice:2": 5.0, "choice:6-10": 1.3646 }, "act_threshold": null, "act_score": null, "question_types": [ "choice", "score", "noul" ], "metrics": null, "training": { "recorded": "2026-10-06T09:41:23+00:00", "code": { "git": null, "sysone_git": { "repo": "sysonelib", "commit": "123ad8bac647e2a14555881ea19e4a6a159b27e6", "branch": "main", "remote": "https://github.com/sgaseretto/sysonelib.git", "dirty": true, "changed": [ "bs/01_template.ipynb", "nbs/02_data.ipynb", "nbs/03_models.ipynb", "nbs/04_losses.ipynb", "nbs/05_learner.ipynb", "nbs/06_inference.ipynb", "nbs/21_screens.ipynb", "nbs/40_cloud.ipynb", "nbs/42_hub.ipynb", "nbs/46_kaggle.ipynb", "nbs/50_cli.ipynb", "nbs/fixtures/make_fixtures.py", "nbs/index.ipynb", "sysone/__init__.py", "sysone/_modidx.py", "sysone/cli.py", "sysone/cloud.py", "sysone/data.py", "sysone/datasets.py", "sysone/hub.py", "sysone/inference.py", "sysone/kaggle.py", "sysone/learner.py", "sysone/losses.py", "sysone/models.py", "sysone/template.py", "nbs/24_mdlm.ipynb", "nbs/fixtures/tiny-a2d-qwen3/", "nbs/tutorials/11_mdlm_browser.ipynb", "sysone/mdlm.py" ], "n_changed": 30, "diff_sha256": "b81003e7b8e888ae895f543def8198f05d13617ca1c19610b4e24bfb36d9fca2" }, "entry": { "name": "run_f.py", "sha256": "c8ac51c45209cf768dbd228cccfc15b57a387283bf8b2681fd76b43191314289" } }, "environment": { "python": "3.13.15", "implementation": "CPython", "os": "Linux 6.18.48+", "packages": { "sysone": "0.3.0", "torch": "2.11.0+cu128", "transformers": "5.16.1", "peft": "0.20.0", "accelerate": "1.14.0", "datasets": "4.8.5", "huggingface_hub": "1.29.0", "tokenizers": "0.23.1", "safetensors": "0.8.0", "numpy": "2.1.3", "fastcore": "2.2.32", "plum-dispatch": "2.10.1" }, "cuda": "12.8", "cudnn": 91900 }, "hardware": { "accelerator": { "kind": "cuda", "devices": [ "Tesla T4", "Tesla T4" ], "memory_gb": [ 14.6, 14.6 ], "capability": "7.5" }, "host": { "cpu": "Intel(R) Xeon(R) CPU @ 2.00GHz", "cores": 4, "memory_gb": 31.3, "system": "Linux", "machine": "x86_64" }, "run": { "backend": "kaggle", "job": "mdlm-browser-e4", "machine": "NvidiaTeslaT4" } }, "data": { "source": { "kind": "hub", "id": "osunlp/Multimodal-Mind2Web", "config": null, "split": "train", "revision": null, "sha": "1b4c6a8cf9f77b7a5e0d641959935c80c4a05889", "note": "the first 1,500 usable steps of the train split, read without the screenshots by sysone.datasets.from_mm_mind2web with every action offered on every step (operations='all'), the operations oversampled to equal shares (RowSampler('oversample')); validation and calibration steps held out by website" }, "splits": { "train": { "cases": 1255, "fingerprint": "19308fb8aeafe3a8" }, "valid": { "cases": 89, "fingerprint": "c3fdcaffa9eecd95" }, "calib": { "cases": 156, "fingerprint": "a61f278fca9aebfa" } }, "template": "jev", "transforms": [], "sampler": "RowSampler(oversample, negatives=None, variants=1, seed=0)", "seed": 0 }, "model": { "encoder": "dllm-hub/Qwen3-0.6B-diffusion-mdlm-v0.1", "regime": "lora", "lora": { "peft_type": "LORA", "r": 16, "lora_alpha": 32, "lora_dropout": 0.05, "target_modules": [ "down_proj", "gate_proj", "k_proj", "o_proj", "q_proj", "up_proj", "v_proj" ], "bias": "none", "use_rslora": false, "use_dora": false, "modules_to_save": null }, "params": { "encoder": 606142464, "head": 4097, "trainable": 10096641 } }, "training": { "preset": "t4", "loss": "soft_ce_rps=0.25", "lr": [ 0.0002, 0.001 ], "seed": 0, "args": { "per_device_train_batch_size": 4, "lr_scheduler_type": "cosine", "warmup_steps": 0.1, "weight_decay": 0.01, "gradient_accumulation_steps": 4, "fp16": true, "gradient_checkpointing": true, "logging_steps": 10, "disable_tqdm": false, "report_to": [], "per_device_eval_batch_size": 16, "save_strategy": "no", "seed": 0, "accelerator_config": "AcceleratorConfig(split_batches=False, dispatch_batches=None, even_batches=True, use_seedable_sampler=True, non_blocking=False, gradient_accumulation_kwargs=None, use_configured_state=False)", "remove_unused_columns": false, "label_names": [ "target" ], "train_sampling_strategy": "group_by_length", "length_column_name": "n_tokens", "debug": [] }, "fits": [ { "started": "2026-10-06T07:39:20+00:00", "regime": "lora", "epochs": 1, "max_steps": 150, "lr_head": 0.001, "lr_encoder": 0.0002, "schedule": "cosine", "warmup": 0.25, "batch_size": 4, "grad_accum": 4, "precision": "fp16", "device": "cuda", "steps": 150, "seconds": 5362.5, "final_loss": 1.1286651611328125, "peak_memory_gb": 7.8 } ], "history": { "loss": [ 3.470833969116211, 1.9087512969970704, 1.0811288833618165, 0.9566798210144043, 1.996589469909668, 1.868391227722168, 0.8938999176025391, 0.922907543182373, 1.6565101623535157, 1.7856624603271485, 1.146382713317871, 0.9506760597229004, 1.1785648345947266, 1.6015304565429687, 1.1286651611328125 ], "eval": [] } }, "tracking": null }, "evaluations": [ { "split": "test_website", "dataset": "osunlp/Multimodal-Mind2Web", "config": null, "n": 600, "metrics": { "accuracy_operation": 0.8533, "accuracy_element": 0.6467, "ece_operation": 0.0542, "ece_element": 0.0826, "accuracy_step": 0.47, "accuracy_operation_every_action_offered": 0.5733 }, "note": "its first 300 steps, from 9 websites none of the training steps came from, two decisions each, each step offering the actions its candidate elements allow (the last metric: every action offered); calibrated, on a Kaggle T4" } ] }