devils-agent / models /v002-gru /training-report.json
devildasdf's picture
Upload experimental BAIM code, research checkpoints and measured evaluations
795f737 verified
Raw History Blame Contribute Delete
8.08 kB
{
"architecture": {
"data": "datasets/synthetic-v1",
"output": "models/v002-gru",
"encoder": "gru",
"no_lexical": false,
"epochs": 16,
"width": 64,
"seed": 1729
},
"parameter_count": 153860,
"threads": 2,
"device": "cpu",
"training_seconds": 149.73377439996693,
"dataset_manifest": {
"train": {
"count": 2400,
"seed": 101,
"templates": [
"fieldset",
"grid",
"stack"
],
"sha256": "3f24899383d914465232cfaac27f654eae7217ed3fb37894cf07e551120c56e5"
},
"validation": {
"count": 480,
"seed": 202,
"templates": [
"fieldset",
"grid",
"stack"
],
"sha256": "1cf777e5703257e3471c3f67b2e421fd3de505de429fbf57223f452648c477a8"
},
"test": {
"count": 480,
"seed": 303,
"templates": [
"nested",
"table"
],
"sha256": "849ccce76126296a01ee5ef756f2790470dea0e87fad94802d66c6e43fee20ac"
},
"novel_wording": {
"count": 480,
"seed": 404,
"templates": [
"nested",
"table"
],
"sha256": "8b467567370321ad5fbe604b282b89de6ceaaa4b679621fa0f3685708fa6e1c2"
},
"limitations": "Single-step generated tasks; test layouts held out but vocabulary shared. No arbitrary-site claim."
},
"history": [
{
"epoch": 1,
"loss": 2.050257802401718,
"validation": {
"samples": 480,
"action_accuracy": 1.0,
"target_accuracy": 1.0,
"joint_step_accuracy": 1.0,
"action_ece": 0.01648493856191635,
"target_ece": 0.18910010438412428,
"candidate_recall": 1.0
}
},
{
"epoch": 2,
"loss": 0.05505605274811387,
"validation": {
"samples": 480,
"action_accuracy": 1.0,
"target_accuracy": 1.0,
"joint_step_accuracy": 1.0,
"action_ece": 0.002022981643676758,
"target_ece": 0.0106055504293181,
"candidate_recall": 1.0
}
},
{
"epoch": 3,
"loss": 0.007284726947546005,
"validation": {
"samples": 480,
"action_accuracy": 1.0,
"target_accuracy": 1.0,
"joint_step_accuracy": 1.0,
"action_ece": 0.0011551976203918457,
"target_ece": 0.0059924333181697875,
"candidate_recall": 1.0
}
},
{
"epoch": 4,
"loss": 0.0035073312311923424,
"validation": {
"samples": 480,
"action_accuracy": 1.0,
"target_accuracy": 1.0,
"joint_step_accuracy": 1.0,
"action_ece": 0.0006573200225830078,
"target_ece": 0.0038807697128504515,
"candidate_recall": 1.0
}
},
{
"epoch": 5,
"loss": 0.0021497866898579033,
"validation": {
"samples": 480,
"action_accuracy": 1.0,
"target_accuracy": 1.0,
"joint_step_accuracy": 1.0,
"action_ece": 0.00046181678771972656,
"target_ece": 0.0029688288923352957,
"candidate_recall": 1.0
}
},
{
"epoch": 6,
"loss": 0.001442565715957531,
"validation": {
"samples": 480,
"action_accuracy": 1.0,
"target_accuracy": 1.0,
"joint_step_accuracy": 1.0,
"action_ece": 0.00035125017166137695,
"target_ece": 0.0023638184648007154,
"candidate_recall": 1.0
}
},
{
"epoch": 7,
"loss": 0.0010281267061241362,
"validation": {
"samples": 480,
"action_accuracy": 1.0,
"target_accuracy": 1.0,
"joint_step_accuracy": 1.0,
"action_ece": 0.00027316808700561523,
"target_ece": 0.001909083453938365,
"candidate_recall": 1.0
}
},
{
"epoch": 8,
"loss": 0.0007711147826691894,
"validation": {
"samples": 480,
"action_accuracy": 1.0,
"target_accuracy": 1.0,
"joint_step_accuracy": 1.0,
"action_ece": 0.00021517276763916016,
"target_ece": 0.0016594392945989966,
"candidate_recall": 1.0
}
},
{
"epoch": 9,
"loss": 0.0006033984092554372,
"validation": {
"samples": 480,
"action_accuracy": 1.0,
"target_accuracy": 1.0,
"joint_step_accuracy": 1.0,
"action_ece": 0.00017768144607543945,
"target_ece": 0.001447484944947064,
"candidate_recall": 1.0
}
},
{
"epoch": 10,
"loss": 0.00048064832088513007,
"validation": {
"samples": 480,
"action_accuracy": 1.0,
"target_accuracy": 1.0,
"joint_step_accuracy": 1.0,
"action_ece": 0.00015157461166381836,
"target_ece": 0.0012724856787826866,
"candidate_recall": 1.0
}
},
{
"epoch": 11,
"loss": 0.0003860391577626088,
"validation": {
"samples": 480,
"action_accuracy": 1.0,
"target_accuracy": 1.0,
"joint_step_accuracy": 1.0,
"action_ece": 0.000125885009765625,
"target_ece": 0.0011438861110946164,
"candidate_recall": 1.0
}
},
{
"epoch": 12,
"loss": 0.000320646290339554,
"validation": {
"samples": 480,
"action_accuracy": 1.0,
"target_accuracy": 1.0,
"joint_step_accuracy": 1.0,
"action_ece": 0.00011038780212402344,
"target_ece": 0.00103258244052995,
"candidate_recall": 1.0
}
},
{
"epoch": 13,
"loss": 0.00027222086837833847,
"validation": {
"samples": 480,
"action_accuracy": 1.0,
"target_accuracy": 1.0,
"joint_step_accuracy": 1.0,
"action_ece": 9.60230827331543e-05,
"target_ece": 0.0009633943409426138,
"candidate_recall": 1.0
}
},
{
"epoch": 14,
"loss": 0.0002315649050471716,
"validation": {
"samples": 480,
"action_accuracy": 1.0,
"target_accuracy": 1.0,
"joint_step_accuracy": 1.0,
"action_ece": 8.404254913330078e-05,
"target_ece": 0.0008865594863891602,
"candidate_recall": 1.0
}
},
{
"epoch": 15,
"loss": 0.00020226637302824346,
"validation": {
"samples": 480,
"action_accuracy": 1.0,
"target_accuracy": 1.0,
"joint_step_accuracy": 1.0,
"action_ece": 7.56382942199707e-05,
"target_ece": 0.0008291006088256836,
"candidate_recall": 1.0
}
},
{
"epoch": 16,
"loss": 0.00017891612419350023,
"validation": {
"samples": 480,
"action_accuracy": 1.0,
"target_accuracy": 1.0,
"joint_step_accuracy": 1.0,
"action_ece": 6.74128532409668e-05,
"target_ece": 0.0007793307304382324,
"candidate_recall": 1.0
}
}
],
"evaluation": {
"validation": {
"samples": 480,
"action_accuracy": 1.0,
"target_accuracy": 1.0,
"joint_step_accuracy": 1.0,
"action_ece": 0.0,
"target_ece": 0.0002561807632446289,
"candidate_recall": 1.0
},
"test": {
"samples": 480,
"action_accuracy": 1.0,
"target_accuracy": 1.0,
"joint_step_accuracy": 1.0,
"action_ece": 0.0,
"target_ece": 0.0005216506760916673,
"candidate_recall": 1.0
},
"novel_wording": {
"samples": 480,
"action_accuracy": 0.3812499940395355,
"target_accuracy": 0.4312500059604645,
"joint_step_accuracy": 0.3583333194255829,
"action_ece": 0.6094635868794285,
"target_ece": 0.5180023244465701,
"candidate_recall": 1.0
}
},
"limitations": "Synthetic single-step action/target prediction; not arbitrary-site task success.",
"production_promoted": false
}