devils-agent / reports /mind2web-v004.json
devildasdf's picture
Add trained contextual action candidate with browser evaluation and explicit real-web limits
c8e5620 verified
Raw History Blame Contribute Delete
993 Bytes
{
"source": "osunlp/Mind2Web",
"revision": "6314166657eec4aa0e22c00f8d801e609ce8e80f",
"file": "train_10.json",
"source_sha256": "182542d7947b3fa9e90fc57a3d82d4d8f2997ca5a06664217720d7a78a956e33",
"license": "CC-BY-4.0",
"attribution": "Deng et al., Mind2Web: Towards a Generalist Agent for the Web, 2023, arXiv:2306.06070",
"checkpoint": "models/v004-contextual",
"tasks": 9,
"websites": 3,
"total_steps": 49,
"scorable_steps": 46,
"unscorable_steps": 3,
"operations": {
"C": 42,
"T": 3,
"O": 1
},
"candidate_recall": 0.47826087474823,
"action_accuracy": 0.021739130839705467,
"target_accuracy": 0.021739130839705467,
"joint_accuracy": 0.021739130839705467,
"browser_execution": false,
"training_on_source": false,
"limitations": "Small training-shard diagnostic, NOT official held-out benchmark. Approximate HTML names/roles; no history; 24-token goal truncation. No raw page or goal text persisted in report."
}