diffcider-browser / sysone.json
sgaseretto's picture
diffcider-browser: retrained with every action offered and the operations balanced · operation 0.853, element 0.647, step 0.470 (test_website)
75242c1 verified
Raw History Blame Contribute Delete
8.97 kB
{
"sysone_version": "0.3.0",
"backend": "sysone",
"name": "diffcider-browser",
"encoder": {
"id": "dllm-hub/Qwen3-0.6B-diffusion-mdlm-v0.1",
"revision": "c8d24a3f4adaeef46881b450e1bf7d1005203bd7",
"source": "transformers",
"hidden_size": 1024,
"family": "a2d-qwen3",
"modality": "text",
"dtype": "float32",
"weights": "adapter"
},
"template": {
"start": null,
"sep": null,
"end": null,
"mask": "<|mask|>",
"image": null,
"row": "<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n<|im_start|>user\nState:\n{state}\n\nQuestion:\n{instructions}\n\n{options}<|im_end|>\n",
"option": "{text}: {mask}",
"max_len": 4096,
"head_max_len": 2048,
"option_tokens": 512,
"truncate": "right",
"option_sep": "\n",
"kinds": {
"choice": {
"header": "For each option, mark Yes if it answers the question, otherwise No.\n",
"option": "{text}: {mask}",
"sep": "\n"
},
"score": {
"header": "For each option, mark Yes if it answers the question, otherwise No.\n",
"option": "{text}: {mask}",
"sep": "\n"
},
"noul": {
"header": "Answer Yes or No.\nAnswer:",
"option": "{mask}",
"sep": ""
}
},
"name": "jev"
},
"tfms": [],
"head": {
"width": 1024,
"layers": 0,
"scorer": "yesno",
"dropout": 0.1,
"nhead": 16,
"standardize": false,
"readout": "anchor",
"query": null,
"query_width": null,
"pair_dim": 256,
"act": false,
"regime": "lora"
},
"temperature": {
"choice": 1.2964,
"choice:11+": 1.269,
"choice:2": 5.0,
"choice:6-10": 1.3646
},
"act_threshold": null,
"act_score": null,
"question_types": [
"choice",
"score",
"noul"
],
"metrics": null,
"training": {
"recorded": "2026-10-06T09:41:23+00:00",
"code": {
"git": null,
"sysone_git": {
"repo": "sysonelib",
"commit": "123ad8bac647e2a14555881ea19e4a6a159b27e6",
"branch": "main",
"remote": "https://github.com/sgaseretto/sysonelib.git",
"dirty": true,
"changed": [
"bs/01_template.ipynb",
"nbs/02_data.ipynb",
"nbs/03_models.ipynb",
"nbs/04_losses.ipynb",
"nbs/05_learner.ipynb",
"nbs/06_inference.ipynb",
"nbs/21_screens.ipynb",
"nbs/40_cloud.ipynb",
"nbs/42_hub.ipynb",
"nbs/46_kaggle.ipynb",
"nbs/50_cli.ipynb",
"nbs/fixtures/make_fixtures.py",
"nbs/index.ipynb",
"sysone/__init__.py",
"sysone/_modidx.py",
"sysone/cli.py",
"sysone/cloud.py",
"sysone/data.py",
"sysone/datasets.py",
"sysone/hub.py",
"sysone/inference.py",
"sysone/kaggle.py",
"sysone/learner.py",
"sysone/losses.py",
"sysone/models.py",
"sysone/template.py",
"nbs/24_mdlm.ipynb",
"nbs/fixtures/tiny-a2d-qwen3/",
"nbs/tutorials/11_mdlm_browser.ipynb",
"sysone/mdlm.py"
],
"n_changed": 30,
"diff_sha256": "b81003e7b8e888ae895f543def8198f05d13617ca1c19610b4e24bfb36d9fca2"
},
"entry": {
"name": "run_f.py",
"sha256": "c8ac51c45209cf768dbd228cccfc15b57a387283bf8b2681fd76b43191314289"
}
},
"environment": {
"python": "3.13.15",
"implementation": "CPython",
"os": "Linux 6.18.48+",
"packages": {
"sysone": "0.3.0",
"torch": "2.11.0+cu128",
"transformers": "5.16.1",
"peft": "0.20.0",
"accelerate": "1.14.0",
"datasets": "4.8.5",
"huggingface_hub": "1.29.0",
"tokenizers": "0.23.1",
"safetensors": "0.8.0",
"numpy": "2.1.3",
"fastcore": "2.2.32",
"plum-dispatch": "2.10.1"
},
"cuda": "12.8",
"cudnn": 91900
},
"hardware": {
"accelerator": {
"kind": "cuda",
"devices": [
"Tesla T4",
"Tesla T4"
],
"memory_gb": [
14.6,
14.6
],
"capability": "7.5"
},
"host": {
"cpu": "Intel(R) Xeon(R) CPU @ 2.00GHz",
"cores": 4,
"memory_gb": 31.3,
"system": "Linux",
"machine": "x86_64"
},
"run": {
"backend": "kaggle",
"job": "mdlm-browser-e4",
"machine": "NvidiaTeslaT4"
}
},
"data": {
"source": {
"kind": "hub",
"id": "osunlp/Multimodal-Mind2Web",
"config": null,
"split": "train",
"revision": null,
"sha": "1b4c6a8cf9f77b7a5e0d641959935c80c4a05889",
"note": "the first 1,500 usable steps of the train split, read without the screenshots by sysone.datasets.from_mm_mind2web with every action offered on every step (operations='all'), the operations oversampled to equal shares (RowSampler('oversample')); validation and calibration steps held out by website"
},
"splits": {
"train": {
"cases": 1255,
"fingerprint": "19308fb8aeafe3a8"
},
"valid": {
"cases": 89,
"fingerprint": "c3fdcaffa9eecd95"
},
"calib": {
"cases": 156,
"fingerprint": "a61f278fca9aebfa"
}
},
"template": "jev",
"transforms": [],
"sampler": "RowSampler(oversample, negatives=None, variants=1, seed=0)",
"seed": 0
},
"model": {
"encoder": "dllm-hub/Qwen3-0.6B-diffusion-mdlm-v0.1",
"regime": "lora",
"lora": {
"peft_type": "LORA",
"r": 16,
"lora_alpha": 32,
"lora_dropout": 0.05,
"target_modules": [
"down_proj",
"gate_proj",
"k_proj",
"o_proj",
"q_proj",
"up_proj",
"v_proj"
],
"bias": "none",
"use_rslora": false,
"use_dora": false,
"modules_to_save": null
},
"params": {
"encoder": 606142464,
"head": 4097,
"trainable": 10096641
}
},
"training": {
"preset": "t4",
"loss": "soft_ce_rps=0.25",
"lr": [
0.0002,
0.001
],
"seed": 0,
"args": {
"per_device_train_batch_size": 4,
"lr_scheduler_type": "cosine",
"warmup_steps": 0.1,
"weight_decay": 0.01,
"gradient_accumulation_steps": 4,
"fp16": true,
"gradient_checkpointing": true,
"logging_steps": 10,
"disable_tqdm": false,
"report_to": [],
"per_device_eval_batch_size": 16,
"save_strategy": "no",
"seed": 0,
"accelerator_config": "AcceleratorConfig(split_batches=False, dispatch_batches=None, even_batches=True, use_seedable_sampler=True, non_blocking=False, gradient_accumulation_kwargs=None, use_configured_state=False)",
"remove_unused_columns": false,
"label_names": [
"target"
],
"train_sampling_strategy": "group_by_length",
"length_column_name": "n_tokens",
"debug": []
},
"fits": [
{
"started": "2026-10-06T07:39:20+00:00",
"regime": "lora",
"epochs": 1,
"max_steps": 150,
"lr_head": 0.001,
"lr_encoder": 0.0002,
"schedule": "cosine",
"warmup": 0.25,
"batch_size": 4,
"grad_accum": 4,
"precision": "fp16",
"device": "cuda",
"steps": 150,
"seconds": 5362.5,
"final_loss": 1.1286651611328125,
"peak_memory_gb": 7.8
}
],
"history": {
"loss": [
3.470833969116211,
1.9087512969970704,
1.0811288833618165,
0.9566798210144043,
1.996589469909668,
1.868391227722168,
0.8938999176025391,
0.922907543182373,
1.6565101623535157,
1.7856624603271485,
1.146382713317871,
0.9506760597229004,
1.1785648345947266,
1.6015304565429687,
1.1286651611328125
],
"eval": []
}
},
"tracking": null
},
"evaluations": [
{
"split": "test_website",
"dataset": "osunlp/Multimodal-Mind2Web",
"config": null,
"n": 600,
"metrics": {
"accuracy_operation": 0.8533,
"accuracy_element": 0.6467,
"ece_operation": 0.0542,
"ece_element": 0.0826,
"accuracy_step": 0.47,
"accuracy_operation_every_action_offered": 0.5733
},
"note": "its first 300 steps, from 9 websites none of the training steps came from, two decisions each, each step offering the actions its candidate elements allow (the last metric: every action offered); calibrated, on a Kaggle T4"
}
]
}