{ "architecture": { "data": "datasets/synthetic-v1", "output": "models/v004-contextual", "encoder": "mean", "no_lexical": false, "contextual_action": true, "epochs": 16, "width": 64, "seed": 1729 }, "parameter_count": 132996, "threads": 2, "device": "cpu", "training_seconds": 15.113418400054798, "dataset_manifest": { "train": { "count": 2400, "seed": 101, "templates": [ "fieldset", "grid", "stack" ], "sha256": "3f24899383d914465232cfaac27f654eae7217ed3fb37894cf07e551120c56e5" }, "validation": { "count": 480, "seed": 202, "templates": [ "fieldset", "grid", "stack" ], "sha256": "1cf777e5703257e3471c3f67b2e421fd3de505de429fbf57223f452648c477a8" }, "test": { "count": 480, "seed": 303, "templates": [ "nested", "table" ], "sha256": "849ccce76126296a01ee5ef756f2790470dea0e87fad94802d66c6e43fee20ac" }, "novel_wording": { "count": 480, "seed": 404, "templates": [ "nested", "table" ], "sha256": "8b467567370321ad5fbe604b282b89de6ceaaa4b679621fa0f3685708fa6e1c2" }, "limitations": "Single-step generated tasks; test layouts held out but vocabulary shared. No arbitrary-site claim." }, "history": [ { "epoch": 1, "loss": 1.5690920323525603, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.02158182067796588, "target_ece": 0.05565767455846071, "candidate_recall": 1.0 } }, { "epoch": 2, "loss": 0.028715768277547078, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.0017445683479309082, "target_ece": 0.012464901199564338, "candidate_recall": 1.0 } }, { "epoch": 3, "loss": 0.010360222947048513, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.0009952187538146973, "target_ece": 0.008543595438823104, "candidate_recall": 1.0 } }, { "epoch": 4, "loss": 0.007286314448145659, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.0007201433181762695, "target_ece": 0.006550255697220564, "candidate_recall": 1.0 } }, { "epoch": 5, "loss": 0.005476677220461792, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.0005319714546203613, "target_ece": 0.0051317751640453935, "candidate_recall": 1.0 } }, { "epoch": 6, "loss": 0.004250973279244806, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.00040030479431152344, "target_ece": 0.004249494522809982, "candidate_recall": 1.0 } }, { "epoch": 7, "loss": 0.0033946395438090946, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.0003128647804260254, "target_ece": 0.003526005893945694, "candidate_recall": 1.0 } }, { "epoch": 8, "loss": 0.0028304545006616728, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.0002523064613342285, "target_ece": 0.00299576623365283, "candidate_recall": 1.0 } }, { "epoch": 9, "loss": 0.002464234003319258, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.00020706653594970703, "target_ece": 0.0025957798352465034, "candidate_recall": 1.0 } }, { "epoch": 10, "loss": 0.0021179942942628834, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.00017279386520385742, "target_ece": 0.002270008670166135, "candidate_recall": 1.0 } }, { "epoch": 11, "loss": 0.001785455371933303, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.00014674663543701172, "target_ece": 0.0019821266178041697, "candidate_recall": 1.0 } }, { "epoch": 12, "loss": 0.001567806809601423, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.0001264810562133789, "target_ece": 0.0017305880901403725, "candidate_recall": 1.0 } }, { "epoch": 13, "loss": 0.0013826081929820295, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.00010889768600463867, "target_ece": 0.0015634410083293915, "candidate_recall": 1.0 } }, { "epoch": 14, "loss": 0.0012740621605189517, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 9.626150131225586e-05, "target_ece": 0.0013960599899291992, "candidate_recall": 1.0 } }, { "epoch": 15, "loss": 0.001100011857679898, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 8.493661880493164e-05, "target_ece": 0.0012488961219787598, "candidate_recall": 1.0 } }, { "epoch": 16, "loss": 0.000987965263208791, "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 7.56978988647461e-05, "target_ece": 0.0011227130889892578, "candidate_recall": 1.0 } } ], "evaluation": { "validation": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 2.777576446533203e-05, "target_ece": 0.0003707166906679049, "candidate_recall": 1.0 }, "test": { "samples": 480, "action_accuracy": 1.0, "target_accuracy": 1.0, "joint_step_accuracy": 1.0, "action_ece": 0.0, "target_ece": 0.0005633262626361102, "candidate_recall": 1.0 }, "novel_wording": { "samples": 480, "action_accuracy": 0.8520833253860474, "target_accuracy": 0.9458333253860474, "joint_step_accuracy": 0.8395833373069763, "action_ece": 0.13770944606221747, "target_ece": 0.046679925406351686, "candidate_recall": 1.0 } }, "limitations": "Synthetic single-step action/target prediction; not arbitrary-site task success.", "production_promoted": false }