Add trained contextual action candidate with browser evaluation and explicit real-web limits
c8e5620 verified Download models/v004-contextual/training-report.json from devildasdf/devils-agent: direct link, hf CLI and curl.
- Browser
- Download file 8.13 kB
-
https://huggingface.co/devildasdf/devils-agent/resolve/main/models/v004-contextual/training-report.json
- Command line
-
hf download hf://devildasdf/devils-agent/models/v004-contextual/training-report.json
-
curl -L -o training-report.json https://huggingface.co/devildasdf/devils-agent/resolve/main/models/v004-contextual/training-report.json
8.13 kB
| { | |
| "architecture": { | |
| "data": "datasets/synthetic-v1", | |
| "output": "models/v004-contextual", | |
| "encoder": "mean", | |
| "no_lexical": false, | |
| "contextual_action": true, | |
| "epochs": 16, | |
| "width": 64, | |
| "seed": 1729 | |
| }, | |
| "parameter_count": 132996, | |
| "threads": 2, | |
| "device": "cpu", | |
| "training_seconds": 15.113418400054798, | |
| "dataset_manifest": { | |
| "train": { | |
| "count": 2400, | |
| "seed": 101, | |
| "templates": [ | |
| "fieldset", | |
| "grid", | |
| "stack" | |
| ], | |
| "sha256": "3f24899383d914465232cfaac27f654eae7217ed3fb37894cf07e551120c56e5" | |
| }, | |
| "validation": { | |
| "count": 480, | |
| "seed": 202, | |
| "templates": [ | |
| "fieldset", | |
| "grid", | |
| "stack" | |
| ], | |
| "sha256": "1cf777e5703257e3471c3f67b2e421fd3de505de429fbf57223f452648c477a8" | |
| }, | |
| "test": { | |
| "count": 480, | |
| "seed": 303, | |
| "templates": [ | |
| "nested", | |
| "table" | |
| ], | |
| "sha256": "849ccce76126296a01ee5ef756f2790470dea0e87fad94802d66c6e43fee20ac" | |
| }, | |
| "novel_wording": { | |
| "count": 480, | |
| "seed": 404, | |
| "templates": [ | |
| "nested", | |
| "table" | |
| ], | |
| "sha256": "8b467567370321ad5fbe604b282b89de6ceaaa4b679621fa0f3685708fa6e1c2" | |
| }, | |
| "limitations": "Single-step generated tasks; test layouts held out but vocabulary shared. No arbitrary-site claim." | |
| }, | |
| "history": [ | |
| { | |
| "epoch": 1, | |
| "loss": 1.5690920323525603, | |
| "validation": { | |
| "samples": 480, | |
| "action_accuracy": 1.0, | |
| "target_accuracy": 1.0, | |
| "joint_step_accuracy": 1.0, | |
| "action_ece": 0.02158182067796588, | |
| "target_ece": 0.05565767455846071, | |
| "candidate_recall": 1.0 | |
| } | |
| }, | |
| { | |
| "epoch": 2, | |
| "loss": 0.028715768277547078, | |
| "validation": { | |
| "samples": 480, | |
| "action_accuracy": 1.0, | |
| "target_accuracy": 1.0, | |
| "joint_step_accuracy": 1.0, | |
| "action_ece": 0.0017445683479309082, | |
| "target_ece": 0.012464901199564338, | |
| "candidate_recall": 1.0 | |
| } | |
| }, | |
| { | |
| "epoch": 3, | |
| "loss": 0.010360222947048513, | |
| "validation": { | |
| "samples": 480, | |
| "action_accuracy": 1.0, | |
| "target_accuracy": 1.0, | |
| "joint_step_accuracy": 1.0, | |
| "action_ece": 0.0009952187538146973, | |
| "target_ece": 0.008543595438823104, | |
| "candidate_recall": 1.0 | |
| } | |
| }, | |
| { | |
| "epoch": 4, | |
| "loss": 0.007286314448145659, | |
| "validation": { | |
| "samples": 480, | |
| "action_accuracy": 1.0, | |
| "target_accuracy": 1.0, | |
| "joint_step_accuracy": 1.0, | |
| "action_ece": 0.0007201433181762695, | |
| "target_ece": 0.006550255697220564, | |
| "candidate_recall": 1.0 | |
| } | |
| }, | |
| { | |
| "epoch": 5, | |
| "loss": 0.005476677220461792, | |
| "validation": { | |
| "samples": 480, | |
| "action_accuracy": 1.0, | |
| "target_accuracy": 1.0, | |
| "joint_step_accuracy": 1.0, | |
| "action_ece": 0.0005319714546203613, | |
| "target_ece": 0.0051317751640453935, | |
| "candidate_recall": 1.0 | |
| } | |
| }, | |
| { | |
| "epoch": 6, | |
| "loss": 0.004250973279244806, | |
| "validation": { | |
| "samples": 480, | |
| "action_accuracy": 1.0, | |
| "target_accuracy": 1.0, | |
| "joint_step_accuracy": 1.0, | |
| "action_ece": 0.00040030479431152344, | |
| "target_ece": 0.004249494522809982, | |
| "candidate_recall": 1.0 | |
| } | |
| }, | |
| { | |
| "epoch": 7, | |
| "loss": 0.0033946395438090946, | |
| "validation": { | |
| "samples": 480, | |
| "action_accuracy": 1.0, | |
| "target_accuracy": 1.0, | |
| "joint_step_accuracy": 1.0, | |
| "action_ece": 0.0003128647804260254, | |
| "target_ece": 0.003526005893945694, | |
| "candidate_recall": 1.0 | |
| } | |
| }, | |
| { | |
| "epoch": 8, | |
| "loss": 0.0028304545006616728, | |
| "validation": { | |
| "samples": 480, | |
| "action_accuracy": 1.0, | |
| "target_accuracy": 1.0, | |
| "joint_step_accuracy": 1.0, | |
| "action_ece": 0.0002523064613342285, | |
| "target_ece": 0.00299576623365283, | |
| "candidate_recall": 1.0 | |
| } | |
| }, | |
| { | |
| "epoch": 9, | |
| "loss": 0.002464234003319258, | |
| "validation": { | |
| "samples": 480, | |
| "action_accuracy": 1.0, | |
| "target_accuracy": 1.0, | |
| "joint_step_accuracy": 1.0, | |
| "action_ece": 0.00020706653594970703, | |
| "target_ece": 0.0025957798352465034, | |
| "candidate_recall": 1.0 | |
| } | |
| }, | |
| { | |
| "epoch": 10, | |
| "loss": 0.0021179942942628834, | |
| "validation": { | |
| "samples": 480, | |
| "action_accuracy": 1.0, | |
| "target_accuracy": 1.0, | |
| "joint_step_accuracy": 1.0, | |
| "action_ece": 0.00017279386520385742, | |
| "target_ece": 0.002270008670166135, | |
| "candidate_recall": 1.0 | |
| } | |
| }, | |
| { | |
| "epoch": 11, | |
| "loss": 0.001785455371933303, | |
| "validation": { | |
| "samples": 480, | |
| "action_accuracy": 1.0, | |
| "target_accuracy": 1.0, | |
| "joint_step_accuracy": 1.0, | |
| "action_ece": 0.00014674663543701172, | |
| "target_ece": 0.0019821266178041697, | |
| "candidate_recall": 1.0 | |
| } | |
| }, | |
| { | |
| "epoch": 12, | |
| "loss": 0.001567806809601423, | |
| "validation": { | |
| "samples": 480, | |
| "action_accuracy": 1.0, | |
| "target_accuracy": 1.0, | |
| "joint_step_accuracy": 1.0, | |
| "action_ece": 0.0001264810562133789, | |
| "target_ece": 0.0017305880901403725, | |
| "candidate_recall": 1.0 | |
| } | |
| }, | |
| { | |
| "epoch": 13, | |
| "loss": 0.0013826081929820295, | |
| "validation": { | |
| "samples": 480, | |
| "action_accuracy": 1.0, | |
| "target_accuracy": 1.0, | |
| "joint_step_accuracy": 1.0, | |
| "action_ece": 0.00010889768600463867, | |
| "target_ece": 0.0015634410083293915, | |
| "candidate_recall": 1.0 | |
| } | |
| }, | |
| { | |
| "epoch": 14, | |
| "loss": 0.0012740621605189517, | |
| "validation": { | |
| "samples": 480, | |
| "action_accuracy": 1.0, | |
| "target_accuracy": 1.0, | |
| "joint_step_accuracy": 1.0, | |
| "action_ece": 9.626150131225586e-05, | |
| "target_ece": 0.0013960599899291992, | |
| "candidate_recall": 1.0 | |
| } | |
| }, | |
| { | |
| "epoch": 15, | |
| "loss": 0.001100011857679898, | |
| "validation": { | |
| "samples": 480, | |
| "action_accuracy": 1.0, | |
| "target_accuracy": 1.0, | |
| "joint_step_accuracy": 1.0, | |
| "action_ece": 8.493661880493164e-05, | |
| "target_ece": 0.0012488961219787598, | |
| "candidate_recall": 1.0 | |
| } | |
| }, | |
| { | |
| "epoch": 16, | |
| "loss": 0.000987965263208791, | |
| "validation": { | |
| "samples": 480, | |
| "action_accuracy": 1.0, | |
| "target_accuracy": 1.0, | |
| "joint_step_accuracy": 1.0, | |
| "action_ece": 7.56978988647461e-05, | |
| "target_ece": 0.0011227130889892578, | |
| "candidate_recall": 1.0 | |
| } | |
| } | |
| ], | |
| "evaluation": { | |
| "validation": { | |
| "samples": 480, | |
| "action_accuracy": 1.0, | |
| "target_accuracy": 1.0, | |
| "joint_step_accuracy": 1.0, | |
| "action_ece": 2.777576446533203e-05, | |
| "target_ece": 0.0003707166906679049, | |
| "candidate_recall": 1.0 | |
| }, | |
| "test": { | |
| "samples": 480, | |
| "action_accuracy": 1.0, | |
| "target_accuracy": 1.0, | |
| "joint_step_accuracy": 1.0, | |
| "action_ece": 0.0, | |
| "target_ece": 0.0005633262626361102, | |
| "candidate_recall": 1.0 | |
| }, | |
| "novel_wording": { | |
| "samples": 480, | |
| "action_accuracy": 0.8520833253860474, | |
| "target_accuracy": 0.9458333253860474, | |
| "joint_step_accuracy": 0.8395833373069763, | |
| "action_ece": 0.13770944606221747, | |
| "target_ece": 0.046679925406351686, | |
| "candidate_recall": 1.0 | |
| } | |
| }, | |
| "limitations": "Synthetic single-step action/target prediction; not arbitrary-site task success.", | |
| "production_promoted": false | |
| } |