{ "architecture": { "data": "datasets/synthetic-v1", "output": "models/v000-mean", "encoder": "mean", "no_lexical": false, "epochs": 16, "width": 64, "seed": 1729 }, "parameter_count": 128900, "threads": 2, "device": "cpu", "training_seconds": 12.39197339990642, "dataset_manifest": { "train": { "count": 2400, "seed": 101, "templates": [ "fieldset", "grid", "stack" ], "sha256": "3f24899383d914465232cfaac27f654eae7217ed3fb37894cf07e551120c56e5" }, "validation": { "count": 480, "seed": 202, "templates": [ "fieldset", "grid", "stack" ], "sha256": "1cf777e5703257e3471c3f67b2e421fd3de505de429fbf57223f452648c477a8" }, "test": { "count": 480, "seed": 303, "templates": [ "nested", "table" ], "sha256": "849ccce76126296a01ee5ef756f2790470dea0e87fad94802d66c6e43fee20ac" }, "novel_wording": { "count": 480, "seed": 404, "templates": [ "nested", "table" ], "sha256": "8b467567370321ad5fbe604b282b89de6ceaaa4b679621fa0f3685708fa6e1c2" }, "limitations": "Single-step generated tasks; test layouts held out but vocabulary shared. No arbitrary-site claim." }, "history": [ { "epoch": 1, "loss": 1.7651557216518803, "validation": { "samples": 480, "action_accuracy": 0.9937499761581421, "target_accuracy": 0.9979166388511658, "joint_step_accuracy": 0.9916666746139526, "action_ece": 0.2162406847346574, "target_ece": 0.05524349887855351, "candidate_recall": 1.0 } }, { "epoch": 2, "loss": 0.10179991137824561, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.021068825386464596, "target_ece": 0.014485297608189285, "candidate_recall": 1.0 } }, { "epoch": 3, "loss": 0.018946192423371894, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.009046673774719238, "target_ece": 0.009317956049926579, "candidate_recall": 1.0 } }, { "epoch": 4, "loss": 0.010617268835439495, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.0054672956466674805, "target_ece": 0.006611444754526019, "candidate_recall": 1.0 } }, { "epoch": 5, "loss": 0.007092831405124774, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.0039865970611572266, "target_ece": 0.0050632000202313066, "candidate_recall": 1.0 } }, { "epoch": 6, "loss": 0.005264362638914271, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.0029689669609069824, "target_ece": 0.00398828461766243, "candidate_recall": 1.0 } }, { "epoch": 7, "loss": 0.004050923848377639, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.002372443675994873, "target_ece": 0.0031936721643432975, "candidate_recall": 1.0 } }, { "epoch": 8, "loss": 0.003259014102360724, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.0018841028213500977, "target_ece": 0.002662156126461923, "candidate_recall": 1.0 } }, { "epoch": 9, "loss": 0.002767322266376332, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.001614987850189209, "target_ece": 0.002215902553871274, "candidate_recall": 1.0 } }, { "epoch": 10, "loss": 0.002234648154913693, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.001338183879852295, "target_ece": 0.001958910725079477, "candidate_recall": 1.0 } }, { "epoch": 11, "loss": 0.00194322145743124, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.0011820197105407715, "target_ece": 0.0016488697146996856, "candidate_recall": 1.0 } }, { "epoch": 12, "loss": 0.0016326695990037958, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.0010241270065307617, "target_ece": 0.0014443512918660417, "candidate_recall": 1.0 } }, { "epoch": 13, "loss": 0.001446679870193628, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.0009126663208007812, "target_ece": 0.0012873411178588867, "candidate_recall": 1.0 } }, { "epoch": 14, "loss": 0.001253041722137775, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.000838935375213623, "target_ece": 0.0011112093925476074, "candidate_recall": 1.0 } }, { "epoch": 15, "loss": 0.001146693759150558, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.0007374286651611328, "target_ece": 0.0010060667991638184, "candidate_recall": 1.0 } }, { "epoch": 16, "loss": 0.000992622820807523, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.0006838440895080566, "target_ece": 0.0008993744850158691, "candidate_recall": 1.0 } } ], "evaluation": { "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.0, "target_ece": 2.86102294921875e-06, "candidate_recall": 1.0 }, "test": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.0, "target_ece": 6.4373016357421875e-06, "candidate_recall": 1.0 }, "novel_wording": { "samples": 480, "action_accuracy": 0.574999988079071, "target_accuracy": 0.9208333492279053, "joint_step_accuracy": 0.5583333373069763, "action_ece": 0.418779332539998, "target_ece": 0.06966802896931767, "candidate_recall": 1.0 } }, "limitations": "Synthetic single-step action/target prediction; not arbitrary-site task success.", "production_promoted": false }