{ "architecture": { "data": "datasets/synthetic-v1", "output": "models/v002-gru", "encoder": "gru", "no_lexical": false, "epochs": 16, "width": 64, "seed": 1729 }, "parameter_count": 153860, "threads": 2, "device": "cpu", "training_seconds": 149.73377439996693, "dataset_manifest": { "train": { "count": 2400, "seed": 101, "templates": [ "fieldset", "grid", "stack" ], "sha256": "3f24899383d914465232cfaac27f654eae7217ed3fb37894cf07e551120c56e5" }, "validation": { "count": 480, "seed": 202, "templates": [ "fieldset", "grid", "stack" ], "sha256": "1cf777e5703257e3471c3f67b2e421fd3de505de429fbf57223f452648c477a8" }, "test": { "count": 480, "seed": 303, "templates": [ "nested", "table" ], "sha256": "849ccce76126296a01ee5ef756f2790470dea0e87fad94802d66c6e43fee20ac" }, "novel_wording": { "count": 480, "seed": 404, "templates": [ "nested", "table" ], "sha256": "8b467567370321ad5fbe604b282b89de6ceaaa4b679621fa0f3685708fa6e1c2" }, "limitations": "Single-step generated tasks; test layouts held out but vocabulary shared. No arbitrary-site claim." }, "history": [ { "epoch": 1, "loss": 2.050257802401718, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.01648493856191635, "target_ece": 0.18910010438412428, "candidate_recall": 1.0 } }, { "epoch": 2, "loss": 0.05505605274811387, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.002022981643676758, "target_ece": 0.0106055504293181, "candidate_recall": 1.0 } }, { "epoch": 3, "loss": 0.007284726947546005, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.0011551976203918457, "target_ece": 0.0059924333181697875, "candidate_recall": 1.0 } }, { "epoch": 4, "loss": 0.0035073312311923424, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.0006573200225830078, "target_ece": 0.0038807697128504515, "candidate_recall": 1.0 } }, { "epoch": 5, "loss": 0.0021497866898579033, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.00046181678771972656, "target_ece": 0.0029688288923352957, "candidate_recall": 1.0 } }, { "epoch": 6, "loss": 0.001442565715957531, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.00035125017166137695, "target_ece": 0.0023638184648007154, "candidate_recall": 1.0 } }, { "epoch": 7, "loss": 0.0010281267061241362, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.00027316808700561523, "target_ece": 0.001909083453938365, "candidate_recall": 1.0 } }, { "epoch": 8, "loss": 0.0007711147826691894, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.00021517276763916016, "target_ece": 0.0016594392945989966, "candidate_recall": 1.0 } }, { "epoch": 9, "loss": 0.0006033984092554372, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.00017768144607543945, "target_ece": 0.001447484944947064, "candidate_recall": 1.0 } }, { "epoch": 10, "loss": 0.00048064832088513007, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.00015157461166381836, "target_ece": 0.0012724856787826866, "candidate_recall": 1.0 } }, { "epoch": 11, "loss": 0.0003860391577626088, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.000125885009765625, "target_ece": 0.0011438861110946164, "candidate_recall": 1.0 } }, { "epoch": 12, "loss": 0.000320646290339554, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.00011038780212402344, "target_ece": 0.00103258244052995, "candidate_recall": 1.0 } }, { "epoch": 13, "loss": 0.00027222086837833847, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 9.60230827331543e-05, "target_ece": 0.0009633943409426138, "candidate_recall": 1.0 } }, { "epoch": 14, "loss": 0.0002315649050471716, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 8.404254913330078e-05, "target_ece": 0.0008865594863891602, "candidate_recall": 1.0 } }, { "epoch": 15, "loss": 0.00020226637302824346, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 7.56382942199707e-05, "target_ece": 0.0008291006088256836, "candidate_recall": 1.0 } }, { "epoch": 16, "loss": 0.00017891612419350023, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 6.74128532409668e-05, "target_ece": 0.0007793307304382324, "candidate_recall": 1.0 } } ], "evaluation": { "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.0, "target_ece": 0.0002561807632446289, "candidate_recall": 1.0 }, "test": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.0, "target_ece": 0.0005216506760916673, "candidate_recall": 1.0 }, "novel_wording": { "samples": 480, "action_accuracy": 0.3812499940395355, "target_accuracy": 0.4312500059604645, "joint_step_accuracy": 0.3583333194255829, "action_ece": 0.6094635868794285, "target_ece": 0.5180023244465701, "candidate_recall": 1.0 } }, "limitations": "Synthetic single-step action/target prediction; not arbitrary-site task success.", "production_promoted": false }