ProCreations's picture
Publish DiscoGen exact native reproduction
6fd091d verified
Raw
History Blame Contribute Delete
56.6 kB
{
"all_gates_pass": true,
"claim_results": [
{
"all_formula_rows_match": true,
"all_small_enumerations_match": true,
"assessment": "verified",
"claim": 1,
"destructive_control": {
"b": 2,
"correct": 4200,
"d": 4,
"m": 3,
"without_nonempty_train_test_exclusion": 6804
},
"destructive_control_detected": true,
"literal_claim": "DiscoGen procedurally generates over 400 million distinct algorithm discovery tasks via a combinatorial formula N_tasks = 2*3*b*(2^m-1)*(3^d-2^(d+1)+1) depending on the number of modules m, datasets d, and backends b (Section 4.2, Equation 1).",
"main_domain_rows": [
{
"b": 1,
"d": 11,
"domain": "Bayesian Optimisation",
"m": 6,
"matches": true,
"recomputed_tasks": 65413656,
"reported_tasks": 65413656
},
{
"b": 1,
"d": 7,
"domain": "Brain Speech Detection",
"m": 3,
"matches": true,
"recomputed_tasks": 81144,
"reported_tasks": 81144
},
{
"b": 1,
"d": 9,
"domain": "Computer Vision Classification",
"m": 4,
"matches": true,
"recomputed_tasks": 1679400,
"reported_tasks": 1679400
},
{
"b": 3,
"d": 3,
"domain": "Continual Learning",
"m": 5,
"matches": true,
"recomputed_tasks": 6696,
"reported_tasks": 6696
},
{
"b": 1,
"d": 4,
"domain": "Greenhouse Gas Prediction",
"m": 2,
"matches": true,
"recomputed_tasks": 900,
"reported_tasks": 900
},
{
"b": 2,
"d": 4,
"domain": "Language Modelling",
"m": 3,
"matches": true,
"recomputed_tasks": 4200,
"reported_tasks": 4200
},
{
"b": 1,
"d": 3,
"domain": "Model Unlearning",
"m": 1,
"matches": true,
"recomputed_tasks": 85176,
"reported_tasks": 85176
},
{
"b": 1,
"d": 4,
"domain": "Off-Policy RL",
"m": 7,
"matches": true,
"recomputed_tasks": 38100,
"reported_tasks": 38100
},
{
"b": 3,
"d": 13,
"domain": "On-Policy RL",
"m": 4,
"matches": true,
"recomputed_tasks": 426043800,
"reported_tasks": 426043800
},
{
"b": 1,
"d": 4,
"domain": "Unsupervised Environment Design",
"m": 3,
"matches": true,
"recomputed_tasks": 2100,
"reported_tasks": 2100
}
],
"main_total_from_rows": 493355172,
"official_model_choices": 13,
"over_400_million": true,
"reported_main_total": 493355172,
"small_exact_enumerations": [
{
"b": 1,
"d": 2,
"direct": 12,
"formula": 12,
"m": 1,
"matches": true
},
{
"b": 1,
"d": 3,
"direct": 216,
"formula": 216,
"m": 2,
"matches": true
},
{
"b": 2,
"d": 4,
"direct": 4200,
"formula": 4200,
"m": 3,
"matches": true
},
{
"b": 3,
"d": 4,
"direct": 13500,
"formula": 13500,
"m": 4,
"matches": true
}
]
},
{
"approximately_99_billion": true,
"assessment": "verified",
"claim": 2,
"destructive_control_detected": true,
"destructive_control_without_on_policy_marl": 1867332264,
"expanded_domain_rows": [
{
"b": 1,
"d": 11,
"domain": "Bayesian Optimisation",
"m": 6,
"reported_tasks": 65413656
},
{
"b": 1,
"d": 7,
"domain": "Brain Speech Detection",
"m": 3,
"reported_tasks": 81144
},
{
"b": 1,
"d": 9,
"domain": "Computer Vision Classification",
"m": 4,
"reported_tasks": 1679400
},
{
"b": 3,
"d": 3,
"domain": "Continual Learning",
"m": 5,
"reported_tasks": 6696
},
{
"b": 1,
"d": 4,
"domain": "Greenhouse Gas Prediction",
"m": 2,
"reported_tasks": 900
},
{
"b": 2,
"d": 4,
"domain": "Language Modelling",
"m": 3,
"reported_tasks": 4200
},
{
"b": 1,
"d": 3,
"domain": "Model Unlearning",
"m": 1,
"reported_tasks": 85176
},
{
"b": 1,
"d": 5,
"domain": "Neural Cellular Automata",
"m": 5,
"reported_tasks": 33480
},
{
"b": 1,
"d": 4,
"domain": "Off-Policy RL",
"m": 7,
"reported_tasks": 38100
},
{
"b": 1,
"d": 10,
"domain": "Offline RL",
"m": 5,
"reported_tasks": 10602372
},
{
"b": 2,
"d": 17,
"domain": "On-Policy MARL",
"m": 6,
"reported_tasks": 97431783120
},
{
"b": 3,
"d": 13,
"domain": "On-Policy RL",
"m": 6,
"reported_tasks": 1789383960
},
{
"b": 3,
"d": 3,
"domain": "Trajectory Prediction",
"m": 4,
"reported_tasks": 1080
},
{
"b": 1,
"d": 4,
"domain": "Unsupervised Environment Design",
"m": 3,
"reported_tasks": 2100
}
],
"expanded_total_from_rows": 99299115384,
"literal_claim": "Including additional domains beyond the main evaluation set, DiscoGen's total task space reaches approximately 99 billion tasks (Appendix C).",
"official_repository_domains": [
{
"backend_count": 1,
"backends": [
"default"
],
"dataset_count": 11,
"datasets": [
"Ackley1d",
"Ackley2d",
"Branin2d",
"Bukin2d",
"Cosine8d",
"DropWave2d",
"EggHolder2d",
"Griewank5d",
"Hartmann6d",
"HolderTable2d",
"Levy6d"
],
"domain": "BayesianOptimisation",
"model_count": 0,
"module_count": 6,
"modules": [
"acq_fn",
"acq_optimizer",
"next_queries",
"sampler",
"surrogate",
"surrogate_optimizer"
]
},
{
"backend_count": 1,
"backends": [
"default"
],
"dataset_count": 7,
"datasets": [
"LibriBrainSherlock1",
"LibriBrainSherlock2",
"LibriBrainSherlock3",
"LibriBrainSherlock4",
"LibriBrainSherlock5",
"LibriBrainSherlock6",
"LibriBrainSherlock7"
],
"domain": "BrainSpeechDetection",
"model_count": 0,
"module_count": 3,
"modules": [
"loss",
"networks",
"optim"
]
},
{
"backend_count": 1,
"backends": [
"default"
],
"dataset_count": 9,
"datasets": [
"CIFAR10C",
"CIFAR10",
"CIFAR10LT",
"CIFAR100",
"FashionMNIST",
"MNIST",
"OxfordFlowers",
"StanfordCars",
"TinyImageNet"
],
"domain": "ComputerVisionClassification",
"model_count": 0,
"module_count": 4,
"modules": [
"loss",
"networks",
"optim",
"preprocess"
]
},
{
"backend_count": 3,
"backends": [
"default",
"parameter_isolation",
"transformer"
],
"dataset_count": 3,
"datasets": [
"PermutedMNIST",
"SplitCIFAR100",
"TinyImageNetSplit"
],
"domain": "ContinualLearning",
"model_count": 0,
"module_count": 5,
"modules": [
"optim",
"regularizer",
"replay",
"sampler",
"scheduler"
]
},
{
"backend_count": 1,
"backends": [
"default"
],
"dataset_count": 4,
"datasets": [
"CH4",
"CO2",
"N2O",
"SF6"
],
"domain": "GreenhouseGasPrediction",
"model_count": 0,
"module_count": 2,
"modules": [
"data_processing",
"model"
]
},
{
"backend_count": 2,
"backends": [
"default",
"ssm"
],
"dataset_count": 4,
"datasets": [
"OPCFineWebCode",
"OPCFineWebMath",
"LMFineWeb",
"TinyStories"
],
"domain": "LanguageModelling",
"model_count": 0,
"module_count": 3,
"modules": [
"loss",
"networks",
"optim"
]
},
{
"backend_count": 1,
"backends": [
"default"
],
"dataset_count": 3,
"datasets": [
"muse",
"tofu",
"wmdp_cyber"
],
"domain": "ModelUnlearning",
"model_count": 13,
"module_count": 1,
"modules": [
"loss"
]
},
{
"backend_count": 1,
"backends": [
"default"
],
"dataset_count": 5,
"datasets": [
"GrowingLizard",
"GrowingButterfly",
"SelfClassifyingMNIST",
"MatrixOperations",
"MNISTInpainting"
],
"domain": "NeuralCellularAutomata",
"model_count": 0,
"module_count": 5,
"modules": [
"loss",
"optimiser",
"perceive",
"train",
"update"
]
},
{
"backend_count": 1,
"backends": [
"default"
],
"dataset_count": 4,
"datasets": [
"MinAtar/Asterix",
"MinAtar/Breakout",
"MinAtar/Freeway",
"MinAtar/SpaceInvaders"
],
"domain": "OffPolicyRL",
"model_count": 0,
"module_count": 7,
"modules": [
"config",
"networks",
"optim",
"policy",
"q_update",
"rb",
"train"
]
},
{
"backend_count": 1,
"backends": [
"default"
],
"dataset_count": 10,
"datasets": [
"OGBench/antmaze-giant-navigate",
"OGBench/antmaze-large-navigate",
"OGBench/antsoccer-arena-navigate",
"OGBench/cube-double-play",
"OGBench/cube-single-play",
"OGBench/humanoidmaze-large-navigate",
"OGBench/humanoidmaze-medium-navigate",
"OGBench/puzzle-3x3-play",
"OGBench/puzzle-4x4-play",
"OGBench/scene-play"
],
"domain": "OfflineRL",
"model_count": 0,
"module_count": 5,
"modules": [
"actor_loss",
"critic_loss",
"networks",
"optim",
"train"
]
},
{
"backend_count": 2,
"backends": [
"default",
"recurrent"
],
"dataset_count": 17,
"datasets": [
"MABrax/Ant",
"MABrax/HalfCheetah",
"MABrax/Hopper",
"MABrax/Walker",
"MABrax/Humanoid",
"MPE/Spread",
"SMAX/2s3z",
"SMAX/3s_vs_5z",
"SMAX/3s5z",
"SMAX/3s5z_vs_3s6z",
"SMAX/5m_vs_6m",
"SMAX/6h_vs_8z",
"SMAX/10m_vs_11m",
"SMAX/27m_vs_30m",
"SMAX/smacv2_5_units",
"SMAX/smacv2_10_units",
"SMAX/smacv2_20_units"
],
"domain": "OnPolicyMARL",
"model_count": 0,
"module_count": 6,
"modules": [
"activation",
"loss",
"networks",
"optim",
"targets",
"train"
]
},
{
"backend_count": 3,
"backends": [
"default",
"recurrent",
"transformer"
],
"dataset_count": 13,
"datasets": [
"MinAtar/Asterix",
"MinAtar/Breakout",
"MinAtar/Freeway",
"MinAtar/SpaceInvaders",
"Brax/Ant",
"Brax/HalfCheetah",
"Brax/Hopper",
"Brax/Humanoid",
"Brax/Pusher",
"Brax/Reacher",
"Brax/Walker2D",
"Craftax/Craftax",
"Craftax/Craftax-Classic"
],
"domain": "OnPolicyRL",
"model_count": 0,
"module_count": 6,
"modules": [
"activation",
"loss",
"networks",
"optim",
"targets",
"train"
]
},
{
"backend_count": 1,
"backends": [
"default"
],
"dataset_count": 3,
"datasets": [
"Argoverse2",
"nuScenes",
"Waymo"
],
"domain": "TrajectoryPrediction",
"model_count": 0,
"module_count": 4,
"modules": [
"loss",
"networks",
"optim",
"train"
]
},
{
"backend_count": 1,
"backends": [
"default"
],
"dataset_count": 4,
"datasets": [
"Kinetix/Small",
"Kinetix/Medium",
"Kinetix/Large",
"Minigrid"
],
"domain": "UnsupervisedEnvironmentDesign",
"model_count": 0,
"module_count": 3,
"modules": [
"sample_levels",
"train_step",
"variable_config"
]
}
],
"reported_expanded_total": 99299115384,
"repository_domain_count": 14
},
{
"all_registered_statistics_match": true,
"assessment": "verified",
"claim": 3,
"computed_median": 59622,
"destructive_control_detected": true,
"destructive_control_drop_one_domain_median": 81144.0,
"domain_count": 10,
"literal_claim": "Across the 10 domains used in the main evaluation, per-domain task counts range from 900 (Greenhouse Gas Prediction) to 426,043,800 (On-Policy RL), with a median of 59,622 tasks per domain (Table 1).",
"maximum": {
"b": 3,
"d": 13,
"domain": "On-Policy RL",
"m": 4,
"reported_tasks": 426043800
},
"minimum": {
"b": 1,
"d": 4,
"domain": "Greenhouse Gas Prediction",
"m": 2,
"reported_tasks": 900
},
"reported_median": 59622
},
{
"actual_generated_files": 82,
"assessment": "verified",
"audited_config_count": 74,
"claim": 4,
"configuration_failures": 0,
"configuration_rows": [
{
"changed_modules": [
"acq_fn"
],
"failures": [],
"file": "BayesianOptimisation_acq_fn.yaml"
},
{
"changed_modules": [
"acq_optimizer"
],
"failures": [],
"file": "BayesianOptimisation_acq_optimizer.yaml"
},
{
"changed_modules": [
"next_queries"
],
"failures": [],
"file": "BayesianOptimisation_next_queries.yaml"
},
{
"changed_modules": [
"sampler"
],
"failures": [],
"file": "BayesianOptimisation_sampler.yaml"
},
{
"changed_modules": [
"surrogate"
],
"failures": [],
"file": "BayesianOptimisation_surrogate.yaml"
},
{
"changed_modules": [
"surrogate_optimizer"
],
"failures": [],
"file": "BayesianOptimisation_surrogate_optimizer.yaml"
},
{
"changed_modules": [
"acq_fn",
"acq_optimizer",
"next_queries",
"sampler",
"surrogate_optimizer",
"surrogate"
],
"failures": [],
"file": "BayesianOptimisation_all.yaml"
},
{
"changed_modules": [
"loss"
],
"failures": [],
"file": "BrainSpeechDetection_loss.yaml"
},
{
"changed_modules": [
"networks"
],
"failures": [],
"file": "BrainSpeechDetection_networks.yaml"
},
{
"changed_modules": [
"optim"
],
"failures": [],
"file": "BrainSpeechDetection_optim.yaml"
},
{
"changed_modules": [
"optim",
"loss",
"networks"
],
"failures": [],
"file": "BrainSpeechDetection_all.yaml"
},
{
"changed_modules": [
"loss"
],
"failures": [],
"file": "ComputerVisionClassification_loss.yaml"
},
{
"changed_modules": [
"networks"
],
"failures": [],
"file": "ComputerVisionClassification_networks.yaml"
},
{
"changed_modules": [
"optim"
],
"failures": [],
"file": "ComputerVisionClassification_optim.yaml"
},
{
"changed_modules": [
"preprocess"
],
"failures": [],
"file": "ComputerVisionClassification_preprocess.yaml"
},
{
"changed_modules": [
"optim",
"loss",
"networks",
"preprocess"
],
"failures": [],
"file": "ComputerVisionClassification_all.yaml"
},
{
"changed_modules": [
"optim"
],
"failures": [],
"file": "ContinualLearning_optim.yaml"
},
{
"changed_modules": [
"regularizer"
],
"failures": [],
"file": "ContinualLearning_regularizer.yaml"
},
{
"changed_modules": [
"replay"
],
"failures": [],
"file": "ContinualLearning_replay.yaml"
},
{
"changed_modules": [
"sampler"
],
"failures": [],
"file": "ContinualLearning_sampler.yaml"
},
{
"changed_modules": [
"scheduler"
],
"failures": [],
"file": "ContinualLearning_scheduler.yaml"
},
{
"changed_modules": [
"regularizer",
"replay",
"sampler",
"scheduler",
"optim"
],
"failures": [],
"file": "ContinualLearning_all.yaml"
},
{
"changed_modules": [
"data_processing"
],
"failures": [],
"file": "GreenhouseGasPrediction_data_processing.yaml"
},
{
"changed_modules": [
"model"
],
"failures": [],
"file": "GreenhouseGasPrediction_model.yaml"
},
{
"changed_modules": [
"data_processing",
"model"
],
"failures": [],
"file": "GreenhouseGasPrediction_all.yaml"
},
{
"changed_modules": [
"loss"
],
"failures": [],
"file": "LanguageModelling_loss.yaml"
},
{
"changed_modules": [
"networks"
],
"failures": [],
"file": "LanguageModelling_networks.yaml"
},
{
"changed_modules": [
"optim"
],
"failures": [],
"file": "LanguageModelling_optim.yaml"
},
{
"changed_modules": [
"optim",
"loss",
"networks"
],
"failures": [],
"file": "LanguageModelling_all.yaml"
},
{
"changed_modules": [
"loss"
],
"failures": [],
"file": "ModelUnlearning_loss.yaml"
},
{
"changed_modules": [
"loss"
],
"failures": [],
"file": "ModelUnlearning_all.yaml"
},
{
"changed_modules": [
"loss"
],
"failures": [],
"file": "NeuralCellularAutomata_loss.yaml"
},
{
"changed_modules": [
"optimiser"
],
"failures": [],
"file": "NeuralCellularAutomata_optimiser.yaml"
},
{
"changed_modules": [
"perceive"
],
"failures": [],
"file": "NeuralCellularAutomata_perceive.yaml"
},
{
"changed_modules": [
"train"
],
"failures": [],
"file": "NeuralCellularAutomata_train.yaml"
},
{
"changed_modules": [
"update"
],
"failures": [],
"file": "NeuralCellularAutomata_update.yaml"
},
{
"changed_modules": [
"perceive",
"update",
"loss",
"train",
"optimiser"
],
"failures": [],
"file": "NeuralCellularAutomata_all.yaml"
},
{
"changed_modules": [
"config"
],
"failures": [],
"file": "OffPolicyRL_config.yaml"
},
{
"changed_modules": [
"networks"
],
"failures": [],
"file": "OffPolicyRL_networks.yaml"
},
{
"changed_modules": [
"optim"
],
"failures": [],
"file": "OffPolicyRL_optim.yaml"
},
{
"changed_modules": [
"policy"
],
"failures": [],
"file": "OffPolicyRL_policy.yaml"
},
{
"changed_modules": [
"q_update"
],
"failures": [],
"file": "OffPolicyRL_q_update.yaml"
},
{
"changed_modules": [
"rb"
],
"failures": [],
"file": "OffPolicyRL_rb.yaml"
},
{
"changed_modules": [
"train"
],
"failures": [],
"file": "OffPolicyRL_train.yaml"
},
{
"changed_modules": [
"config",
"networks",
"optim",
"policy",
"q_update",
"rb",
"train"
],
"failures": [],
"file": "OffPolicyRL_all.yaml"
},
{
"changed_modules": [
"actor_loss"
],
"failures": [],
"file": "OfflineRL_actor_loss.yaml"
},
{
"changed_modules": [
"critic_loss"
],
"failures": [],
"file": "OfflineRL_critic_loss.yaml"
},
{
"changed_modules": [
"networks"
],
"failures": [],
"file": "OfflineRL_networks.yaml"
},
{
"changed_modules": [
"optim"
],
"failures": [],
"file": "OfflineRL_optim.yaml"
},
{
"changed_modules": [
"train"
],
"failures": [],
"file": "OfflineRL_train.yaml"
},
{
"changed_modules": [
"optim",
"actor_loss",
"critic_loss",
"networks",
"train"
],
"failures": [],
"file": "OfflineRL_all.yaml"
},
{
"changed_modules": [
"activation"
],
"failures": [],
"file": "OnPolicyMARL_activation.yaml"
},
{
"changed_modules": [
"loss"
],
"failures": [],
"file": "OnPolicyMARL_loss.yaml"
},
{
"changed_modules": [
"networks"
],
"failures": [],
"file": "OnPolicyMARL_networks.yaml"
},
{
"changed_modules": [
"optim"
],
"failures": [],
"file": "OnPolicyMARL_optim.yaml"
},
{
"changed_modules": [
"targets"
],
"failures": [],
"file": "OnPolicyMARL_targets.yaml"
},
{
"changed_modules": [
"train"
],
"failures": [],
"file": "OnPolicyMARL_train.yaml"
},
{
"changed_modules": [
"optim",
"loss",
"networks",
"train",
"activation",
"targets"
],
"failures": [],
"file": "OnPolicyMARL_all.yaml"
},
{
"changed_modules": [
"activation"
],
"failures": [],
"file": "OnPolicyRL_activation.yaml"
},
{
"changed_modules": [
"loss"
],
"failures": [],
"file": "OnPolicyRL_loss.yaml"
},
{
"changed_modules": [
"networks"
],
"failures": [],
"file": "OnPolicyRL_networks.yaml"
},
{
"changed_modules": [
"optim"
],
"failures": [],
"file": "OnPolicyRL_optim.yaml"
},
{
"changed_modules": [
"targets"
],
"failures": [],
"file": "OnPolicyRL_targets.yaml"
},
{
"changed_modules": [
"train"
],
"failures": [],
"file": "OnPolicyRL_train.yaml"
},
{
"changed_modules": [
"optim",
"loss",
"networks",
"train",
"activation",
"targets"
],
"failures": [],
"file": "OnPolicyRL_all.yaml"
},
{
"changed_modules": [
"loss"
],
"failures": [],
"file": "TrajectoryPrediction_loss.yaml"
},
{
"changed_modules": [
"networks"
],
"failures": [],
"file": "TrajectoryPrediction_networks.yaml"
},
{
"changed_modules": [
"optim"
],
"failures": [],
"file": "TrajectoryPrediction_optim.yaml"
},
{
"changed_modules": [
"train"
],
"failures": [],
"file": "TrajectoryPrediction_train.yaml"
},
{
"changed_modules": [
"optim",
"loss",
"networks",
"train"
],
"failures": [],
"file": "TrajectoryPrediction_all.yaml"
},
{
"changed_modules": [
"sample_levels"
],
"failures": [],
"file": "UnsupervisedEnvironmentDesign_sample_levels.yaml"
},
{
"changed_modules": [
"train_step"
],
"failures": [],
"file": "UnsupervisedEnvironmentDesign_train_step.yaml"
},
{
"changed_modules": [
"variable_config"
],
"failures": [],
"file": "UnsupervisedEnvironmentDesign_variable_config.yaml"
},
{
"changed_modules": [
"sample_levels",
"train_step",
"variable_config"
],
"failures": [],
"file": "UnsupervisedEnvironmentDesign_all.yaml"
}
],
"destructive_control_active_modules": [
"change_data_processing",
"change_model"
],
"destructive_control_detected": true,
"expected_m_plus_one_configs": 74,
"literal_claim": "DiscoBench provides a fixed evaluation subset built from DiscoGen, comprising, for each domain, m single-module tasks (DiscoBench Single) plus one comprehensive all-modules-active task (DiscoBench All) (Section 4.4).",
"official_builder_execution": {
"canonicalized_description_files": 4,
"executed_configs": [
"GreenhouseGasPrediction_all",
"GreenhouseGasPrediction_data_processing",
"OnPolicyRL_loss",
"OnPolicyRL_all"
],
"file_count": 82,
"files": [
{
"bytes": 300,
"path": "GreenhouseGasPrediction_all/CH4/data_processing.py",
"sha256": "e3a4d5be0be7e7d5129181846167a2fd76fc1e9b2b99049f8406ecdc055bac4e"
},
{
"bytes": 1372,
"path": "GreenhouseGasPrediction_all/CH4/main.py",
"sha256": "094787e39823668d2cbd472af1f00eee3f49ca26fd1c2d5f179aee9aafbe7adf"
},
{
"bytes": 597,
"path": "GreenhouseGasPrediction_all/CH4/model.py",
"sha256": "dd94a5687d62d89a8d510ce2c3b7f059de2ff0bc4b7fd7a719cef6ceb5126549"
},
{
"bytes": 300,
"path": "GreenhouseGasPrediction_all/SF6/data_processing.py",
"sha256": "e3a4d5be0be7e7d5129181846167a2fd76fc1e9b2b99049f8406ecdc055bac4e"
},
{
"bytes": 1372,
"path": "GreenhouseGasPrediction_all/SF6/main.py",
"sha256": "094787e39823668d2cbd472af1f00eee3f49ca26fd1c2d5f179aee9aafbe7adf"
},
{
"bytes": 597,
"path": "GreenhouseGasPrediction_all/SF6/model.py",
"sha256": "dd94a5687d62d89a8d510ce2c3b7f059de2ff0bc4b7fd7a719cef6ceb5126549"
},
{
"bytes": 8817,
"path": "GreenhouseGasPrediction_all/description.md",
"sha256": "1b052d3d51ed063eb562bf5a22a3918c358efd215d807e73cf6a20993c0258e1"
},
{
"bytes": 300,
"path": "GreenhouseGasPrediction_all/discovered/data_processing.py",
"sha256": "e3a4d5be0be7e7d5129181846167a2fd76fc1e9b2b99049f8406ecdc055bac4e"
},
{
"bytes": 597,
"path": "GreenhouseGasPrediction_all/discovered/model.py",
"sha256": "dd94a5687d62d89a8d510ce2c3b7f059de2ff0bc4b7fd7a719cef6ceb5126549"
},
{
"bytes": 53,
"path": "GreenhouseGasPrediction_all/install.sh",
"sha256": "6964f075ffc565dd9847b7203a28d7368b0825333435a756b48ec7da08757af7"
},
{
"bytes": 155,
"path": "GreenhouseGasPrediction_all/requirements.txt",
"sha256": "ce1ca66a4ff2d670272eb600e8cd5fd3c530c637c86d7477110015d2d43ca8fa"
},
{
"bytes": 1725,
"path": "GreenhouseGasPrediction_all/run_main.py",
"sha256": "88db3fe20483a035b63bf5cb9c415493ae52cfa09e944fcb5ea461aa66407d90"
},
{
"bytes": 300,
"path": "GreenhouseGasPrediction_data_processing/CH4/data_processing.py",
"sha256": "e3a4d5be0be7e7d5129181846167a2fd76fc1e9b2b99049f8406ecdc055bac4e"
},
{
"bytes": 1372,
"path": "GreenhouseGasPrediction_data_processing/CH4/main.py",
"sha256": "094787e39823668d2cbd472af1f00eee3f49ca26fd1c2d5f179aee9aafbe7adf"
},
{
"bytes": 458,
"path": "GreenhouseGasPrediction_data_processing/CH4/model.py",
"sha256": "57e94b384ad5ccf7a6a803568f8b7a7350db7786b0308375370e7195ed93e172"
},
{
"bytes": 300,
"path": "GreenhouseGasPrediction_data_processing/SF6/data_processing.py",
"sha256": "e3a4d5be0be7e7d5129181846167a2fd76fc1e9b2b99049f8406ecdc055bac4e"
},
{
"bytes": 1372,
"path": "GreenhouseGasPrediction_data_processing/SF6/main.py",
"sha256": "094787e39823668d2cbd472af1f00eee3f49ca26fd1c2d5f179aee9aafbe7adf"
},
{
"bytes": 458,
"path": "GreenhouseGasPrediction_data_processing/SF6/model.py",
"sha256": "57e94b384ad5ccf7a6a803568f8b7a7350db7786b0308375370e7195ed93e172"
},
{
"bytes": 8691,
"path": "GreenhouseGasPrediction_data_processing/description.md",
"sha256": "c65914e7460ad4529dcd70b4788a8ced437909123cc3ea19599a39615db0e2dd"
},
{
"bytes": 300,
"path": "GreenhouseGasPrediction_data_processing/discovered/data_processing.py",
"sha256": "e3a4d5be0be7e7d5129181846167a2fd76fc1e9b2b99049f8406ecdc055bac4e"
},
{
"bytes": 53,
"path": "GreenhouseGasPrediction_data_processing/install.sh",
"sha256": "6964f075ffc565dd9847b7203a28d7368b0825333435a756b48ec7da08757af7"
},
{
"bytes": 155,
"path": "GreenhouseGasPrediction_data_processing/requirements.txt",
"sha256": "ce1ca66a4ff2d670272eb600e8cd5fd3c530c637c86d7477110015d2d43ca8fa"
},
{
"bytes": 1725,
"path": "GreenhouseGasPrediction_data_processing/run_main.py",
"sha256": "88db3fe20483a035b63bf5cb9c415493ae52cfa09e944fcb5ea461aa66407d90"
},
{
"bytes": 0,
"path": "OnPolicyRL_all/MinAtar/Breakout/__init__.py",
"sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"
},
{
"bytes": 277,
"path": "OnPolicyRL_all/MinAtar/Breakout/activation.py",
"sha256": "1830544b21f2ed84ac4e5ac94ad33047ebfa178394a1bf7e3b3567ad2f497e76"
},
{
"bytes": 346,
"path": "OnPolicyRL_all/MinAtar/Breakout/config.py",
"sha256": "8f6949d7df8358ff71e3006ac89118c445aa74c72c099e3e58058689c9b5050c"
},
{
"bytes": 850,
"path": "OnPolicyRL_all/MinAtar/Breakout/loss.py",
"sha256": "7866cfc22268e39cb4e0564893a7afc7284088041278d2f04aab38f7c76f595c"
},
{
"bytes": 8942,
"path": "OnPolicyRL_all/MinAtar/Breakout/main.py",
"sha256": "ad901568e2f90a79ea2c570e2482e3f89189a48e9e62663b2f46965a6744c488"
},
{
"bytes": 245,
"path": "OnPolicyRL_all/MinAtar/Breakout/make_env.py",
"sha256": "c24169e8c5f04fa5f3af519e9f98c466f8c7a6233edc7b4e9085a25562c94602"
},
{
"bytes": 818,
"path": "OnPolicyRL_all/MinAtar/Breakout/networks.py",
"sha256": "8710a87b23befb124a88ca9676133e0dfd336835f61880afe7ab0a3bc911561d"
},
{
"bytes": 1670,
"path": "OnPolicyRL_all/MinAtar/Breakout/optim.py",
"sha256": "9344db01c0d8465c28ed142ca18a7d0acf4e4f8ee5d59f440a16ec6cca076a9b"
},
{
"bytes": 261,
"path": "OnPolicyRL_all/MinAtar/Breakout/targets.py",
"sha256": "d5ee1bddf2ccfe8c4613dfdf5623743d788f269cd546006aa7a18e433d194f0d"
},
{
"bytes": 3648,
"path": "OnPolicyRL_all/MinAtar/Breakout/train.py",
"sha256": "4c9515a812deea8226e22652defc9c9e326d1954216dcc5a859673b80c97be67"
},
{
"bytes": 10698,
"path": "OnPolicyRL_all/MinAtar/Breakout/wrappers.py",
"sha256": "e611bbacd53433415d12fb709da822a755df3770fbf7f51c6d789a485415ac68"
},
{
"bytes": 0,
"path": "OnPolicyRL_all/MinAtar/Freeway/__init__.py",
"sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"
},
{
"bytes": 277,
"path": "OnPolicyRL_all/MinAtar/Freeway/activation.py",
"sha256": "1830544b21f2ed84ac4e5ac94ad33047ebfa178394a1bf7e3b3567ad2f497e76"
},
{
"bytes": 346,
"path": "OnPolicyRL_all/MinAtar/Freeway/config.py",
"sha256": "8f6949d7df8358ff71e3006ac89118c445aa74c72c099e3e58058689c9b5050c"
},
{
"bytes": 850,
"path": "OnPolicyRL_all/MinAtar/Freeway/loss.py",
"sha256": "7866cfc22268e39cb4e0564893a7afc7284088041278d2f04aab38f7c76f595c"
},
{
"bytes": 8942,
"path": "OnPolicyRL_all/MinAtar/Freeway/main.py",
"sha256": "ad901568e2f90a79ea2c570e2482e3f89189a48e9e62663b2f46965a6744c488"
},
{
"bytes": 244,
"path": "OnPolicyRL_all/MinAtar/Freeway/make_env.py",
"sha256": "c18fedfd39747012af573a9f81055144e3bfdc90940a2666bbdbaa53d85bb583"
},
{
"bytes": 818,
"path": "OnPolicyRL_all/MinAtar/Freeway/networks.py",
"sha256": "8710a87b23befb124a88ca9676133e0dfd336835f61880afe7ab0a3bc911561d"
},
{
"bytes": 1670,
"path": "OnPolicyRL_all/MinAtar/Freeway/optim.py",
"sha256": "9344db01c0d8465c28ed142ca18a7d0acf4e4f8ee5d59f440a16ec6cca076a9b"
},
{
"bytes": 261,
"path": "OnPolicyRL_all/MinAtar/Freeway/targets.py",
"sha256": "d5ee1bddf2ccfe8c4613dfdf5623743d788f269cd546006aa7a18e433d194f0d"
},
{
"bytes": 3648,
"path": "OnPolicyRL_all/MinAtar/Freeway/train.py",
"sha256": "4c9515a812deea8226e22652defc9c9e326d1954216dcc5a859673b80c97be67"
},
{
"bytes": 10698,
"path": "OnPolicyRL_all/MinAtar/Freeway/wrappers.py",
"sha256": "e611bbacd53433415d12fb709da822a755df3770fbf7f51c6d789a485415ac68"
},
{
"bytes": 10734,
"path": "OnPolicyRL_all/description.md",
"sha256": "db1ed45e365da592846c9a1acb9f1460a25e39ccc3bb4c4c425ec0b9f71eb11c"
},
{
"bytes": 277,
"path": "OnPolicyRL_all/discovered/activation.py",
"sha256": "1830544b21f2ed84ac4e5ac94ad33047ebfa178394a1bf7e3b3567ad2f497e76"
},
{
"bytes": 850,
"path": "OnPolicyRL_all/discovered/loss.py",
"sha256": "7866cfc22268e39cb4e0564893a7afc7284088041278d2f04aab38f7c76f595c"
},
{
"bytes": 818,
"path": "OnPolicyRL_all/discovered/networks.py",
"sha256": "8710a87b23befb124a88ca9676133e0dfd336835f61880afe7ab0a3bc911561d"
},
{
"bytes": 1670,
"path": "OnPolicyRL_all/discovered/optim.py",
"sha256": "9344db01c0d8465c28ed142ca18a7d0acf4e4f8ee5d59f440a16ec6cca076a9b"
},
{
"bytes": 261,
"path": "OnPolicyRL_all/discovered/targets.py",
"sha256": "d5ee1bddf2ccfe8c4613dfdf5623743d788f269cd546006aa7a18e433d194f0d"
},
{
"bytes": 3648,
"path": "OnPolicyRL_all/discovered/train.py",
"sha256": "4c9515a812deea8226e22652defc9c9e326d1954216dcc5a859673b80c97be67"
},
{
"bytes": 53,
"path": "OnPolicyRL_all/install.sh",
"sha256": "6964f075ffc565dd9847b7203a28d7368b0825333435a756b48ec7da08757af7"
},
{
"bytes": 272,
"path": "OnPolicyRL_all/requirements.txt",
"sha256": "f052fa4717c1411280ec50ba717865be0fe1390c60e68da176b4986fe8f551aa"
},
{
"bytes": 1725,
"path": "OnPolicyRL_all/run_main.py",
"sha256": "88db3fe20483a035b63bf5cb9c415493ae52cfa09e944fcb5ea461aa66407d90"
},
{
"bytes": 0,
"path": "OnPolicyRL_loss/MinAtar/Breakout/__init__.py",
"sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"
},
{
"bytes": 182,
"path": "OnPolicyRL_loss/MinAtar/Breakout/activation.py",
"sha256": "429973432f0977f59dc029c529c9b8989618cc7699c72a1d9fdeea9f9f827e1c"
},
{
"bytes": 346,
"path": "OnPolicyRL_loss/MinAtar/Breakout/config.py",
"sha256": "8f6949d7df8358ff71e3006ac89118c445aa74c72c099e3e58058689c9b5050c"
},
{
"bytes": 850,
"path": "OnPolicyRL_loss/MinAtar/Breakout/loss.py",
"sha256": "7866cfc22268e39cb4e0564893a7afc7284088041278d2f04aab38f7c76f595c"
},
{
"bytes": 8942,
"path": "OnPolicyRL_loss/MinAtar/Breakout/main.py",
"sha256": "ad901568e2f90a79ea2c570e2482e3f89189a48e9e62663b2f46965a6744c488"
},
{
"bytes": 245,
"path": "OnPolicyRL_loss/MinAtar/Breakout/make_env.py",
"sha256": "c24169e8c5f04fa5f3af519e9f98c466f8c7a6233edc7b4e9085a25562c94602"
},
{
"bytes": 1680,
"path": "OnPolicyRL_loss/MinAtar/Breakout/networks.py",
"sha256": "456176a5ce041e85b85cf57b8829265af1431cc48cbfd017dac4dfd0a8689602"
},
{
"bytes": 156,
"path": "OnPolicyRL_loss/MinAtar/Breakout/optim.py",
"sha256": "e88de32252fed900d835598dc968e9d67e134dd5a0d7a3e29b62533e3e5fdaaf"
},
{
"bytes": 767,
"path": "OnPolicyRL_loss/MinAtar/Breakout/targets.py",
"sha256": "a36111a2aebe9c98cf7d3dc7e121bf8b2e2b846d7bc962ef2ba032712bc644d9"
},
{
"bytes": 8102,
"path": "OnPolicyRL_loss/MinAtar/Breakout/train.py",
"sha256": "0013f408cec5cefb3b1426c22e6b57405ca36ad8b72e61ce24a4a27d127f67fe"
},
{
"bytes": 10698,
"path": "OnPolicyRL_loss/MinAtar/Breakout/wrappers.py",
"sha256": "e611bbacd53433415d12fb709da822a755df3770fbf7f51c6d789a485415ac68"
},
{
"bytes": 0,
"path": "OnPolicyRL_loss/MinAtar/Freeway/__init__.py",
"sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"
},
{
"bytes": 182,
"path": "OnPolicyRL_loss/MinAtar/Freeway/activation.py",
"sha256": "429973432f0977f59dc029c529c9b8989618cc7699c72a1d9fdeea9f9f827e1c"
},
{
"bytes": 346,
"path": "OnPolicyRL_loss/MinAtar/Freeway/config.py",
"sha256": "8f6949d7df8358ff71e3006ac89118c445aa74c72c099e3e58058689c9b5050c"
},
{
"bytes": 850,
"path": "OnPolicyRL_loss/MinAtar/Freeway/loss.py",
"sha256": "7866cfc22268e39cb4e0564893a7afc7284088041278d2f04aab38f7c76f595c"
},
{
"bytes": 8942,
"path": "OnPolicyRL_loss/MinAtar/Freeway/main.py",
"sha256": "ad901568e2f90a79ea2c570e2482e3f89189a48e9e62663b2f46965a6744c488"
},
{
"bytes": 244,
"path": "OnPolicyRL_loss/MinAtar/Freeway/make_env.py",
"sha256": "c18fedfd39747012af573a9f81055144e3bfdc90940a2666bbdbaa53d85bb583"
},
{
"bytes": 1680,
"path": "OnPolicyRL_loss/MinAtar/Freeway/networks.py",
"sha256": "456176a5ce041e85b85cf57b8829265af1431cc48cbfd017dac4dfd0a8689602"
},
{
"bytes": 156,
"path": "OnPolicyRL_loss/MinAtar/Freeway/optim.py",
"sha256": "e88de32252fed900d835598dc968e9d67e134dd5a0d7a3e29b62533e3e5fdaaf"
},
{
"bytes": 767,
"path": "OnPolicyRL_loss/MinAtar/Freeway/targets.py",
"sha256": "a36111a2aebe9c98cf7d3dc7e121bf8b2e2b846d7bc962ef2ba032712bc644d9"
},
{
"bytes": 8102,
"path": "OnPolicyRL_loss/MinAtar/Freeway/train.py",
"sha256": "0013f408cec5cefb3b1426c22e6b57405ca36ad8b72e61ce24a4a27d127f67fe"
},
{
"bytes": 10698,
"path": "OnPolicyRL_loss/MinAtar/Freeway/wrappers.py",
"sha256": "e611bbacd53433415d12fb709da822a755df3770fbf7f51c6d789a485415ac68"
},
{
"bytes": 8306,
"path": "OnPolicyRL_loss/description.md",
"sha256": "28f98c846f8542e35ce544aa4c4d5d5387805cd32c70739971e24841fc4914b4"
},
{
"bytes": 850,
"path": "OnPolicyRL_loss/discovered/loss.py",
"sha256": "7866cfc22268e39cb4e0564893a7afc7284088041278d2f04aab38f7c76f595c"
},
{
"bytes": 53,
"path": "OnPolicyRL_loss/install.sh",
"sha256": "6964f075ffc565dd9847b7203a28d7368b0825333435a756b48ec7da08757af7"
},
{
"bytes": 272,
"path": "OnPolicyRL_loss/requirements.txt",
"sha256": "f052fa4717c1411280ec50ba717865be0fe1390c60e68da176b4986fe8f551aa"
},
{
"bytes": 1725,
"path": "OnPolicyRL_loss/run_main.py",
"sha256": "88db3fe20483a035b63bf5cb9c415493ae52cfa09e944fcb5ea461aa66407d90"
}
],
"tree_sha256": "039f3fbad73fac4c9459b3eed28a53f60c98da12bfc07b4478237805ab32af16"
},
"official_domain_count": 14
},
{
"all_three_models_nonincreasing": true,
"assessment": "verified",
"claim": 5,
"combination_count": 15,
"complete_four_module_combinations": [
[
"loss"
],
[
"networks"
],
[
"optim"
],
[
"train"
],
[
"loss",
"networks"
],
[
"loss",
"optim"
],
[
"loss",
"train"
],
[
"networks",
"optim"
],
[
"networks",
"train"
],
[
"optim",
"train"
],
[
"loss",
"networks",
"optim"
],
[
"loss",
"networks",
"train"
],
[
"loss",
"optim",
"train"
],
[
"networks",
"optim",
"train"
],
[
"loss",
"networks",
"optim",
"train"
]
],
"destructive_control_detected": true,
"destructive_control_reversed_success_rates": {
"Deepseek-v3.2": [
0.0,
8.3,
47.2,
75.0
],
"Devstral2": [
0.0,
0.0,
27.8,
29.2
],
"GPT-OSS-120b": [
0.0,
8.3,
11.1,
50.0
]
},
"environments_with_higher_two_module_ceiling": 3,
"literal_claim": "As the number of editable modules increases in DiscoBench tasks, agent success rates consistently decline while the achievable performance ceiling rises (Appendix G).",
"maximum_return_by_module_count": {
"1": [
99.97,
68.07,
39.6,
189.75
],
"2": [
106.12,
66.16,
69.42,
191.62
],
"3": [
88.58,
65.2,
64.85,
186.75
],
"4": [
null,
null,
null,
null
]
},
"mean_ceiling_increase": 8.9825,
"monotonicity_by_model": {
"Deepseek-v3.2": true,
"Devstral2": true,
"GPT-OSS-120b": true
},
"single_module_mean_environment_ceiling": 99.3475,
"source_configuration_rows": [
{
"configuration": "Optimiser",
"module_count": 1,
"returns": [
74.97,
62.86,
18.11,
181.25
],
"success_rate": 83.33
},
{
"configuration": "Loss",
"module_count": 1,
"returns": [
83.91,
62.58,
39.6,
179.5
],
"success_rate": 77.78
},
{
"configuration": "Network",
"module_count": 1,
"returns": [
99.97,
68.07,
14.2,
189.75
],
"success_rate": 33.33
},
{
"configuration": "Train",
"module_count": 1,
"returns": [
8.41,
8.41,
3.73,
177.12
],
"success_rate": 11.11
},
{
"configuration": "Loss + Optimiser",
"module_count": 2,
"returns": [
84.47,
62.89,
20.5,
181.38
],
"success_rate": 61.11
},
{
"configuration": "Network + Optimiser",
"module_count": 2,
"returns": [
91.44,
65.25,
19.77,
191.62
],
"success_rate": 44.44
},
{
"configuration": "Loss + Network",
"module_count": 2,
"returns": [
106.12,
66.16,
69.42,
184.0
],
"success_rate": 38.89
},
{
"configuration": "Loss + Train",
"module_count": 2,
"returns": [
8.56,
61.91,
3.91,
169.38
],
"success_rate": 22.22
},
{
"configuration": "Optimiser + Train",
"module_count": 2,
"returns": [
34.61,
29.45,
null,
null
],
"success_rate": 5.56
},
{
"configuration": "Network + Train",
"module_count": 2,
"returns": [
null,
null,
null,
null
],
"success_rate": 0.0
},
{
"configuration": "Loss + Network + Optimiser",
"module_count": 3,
"returns": [
88.58,
65.2,
64.85,
186.75
],
"success_rate": 11.11
},
{
"configuration": "Loss + Optimiser + Train",
"module_count": 3,
"returns": [
0.3,
2.95,
null,
null
],
"success_rate": 11.11
},
{
"configuration": "Loss + Network + Train",
"module_count": 3,
"returns": [
null,
null,
null,
null
],
"success_rate": 0.0
},
{
"configuration": "Network + Optimiser + Train",
"module_count": 3,
"returns": [
null,
null,
null,
null
],
"success_rate": 0.0
},
{
"configuration": "Loss + Network + Optimiser + Train",
"module_count": 4,
"returns": [
null,
null,
null,
null
],
"success_rate": 0.0
}
],
"success_rate_by_editable_module_count": {
"Deepseek-v3.2": [
75.0,
47.2,
8.3,
0.0
],
"Devstral2": [
29.2,
27.8,
0.0,
0.0
],
"GPT-OSS-120b": [
50.0,
11.1,
8.3,
0.0
]
},
"two_module_mean_environment_ceiling": 108.33
}
],
"claims": [
{
"claim": 1,
"literal_claim": "DiscoGen procedurally generates over 400 million distinct algorithm discovery tasks via a combinatorial formula N_tasks = 2*3*b*(2^m-1)*(3^d-2^(d+1)+1) depending on the number of modules m, datasets d, and backends b (Section 4.2, Equation 1)."
},
{
"claim": 2,
"literal_claim": "Including additional domains beyond the main evaluation set, DiscoGen's total task space reaches approximately 99 billion tasks (Appendix C)."
},
{
"claim": 3,
"literal_claim": "Across the 10 domains used in the main evaluation, per-domain task counts range from 900 (Greenhouse Gas Prediction) to 426,043,800 (On-Policy RL), with a median of 59,622 tasks per domain (Table 1)."
},
{
"claim": 4,
"literal_claim": "DiscoBench provides a fixed evaluation subset built from DiscoGen, comprising, for each domain, m single-module tasks (DiscoBench Single) plus one comprehensive all-modules-active task (DiscoBench All) (Section 4.4)."
},
{
"claim": 5,
"literal_claim": "As the number of editable modules increases in DiscoBench tasks, agent success rates consistently decline while the achievable performance ceiling rises (Appendix G)."
}
],
"gates": [
{
"name": "claim1_all_formula_rows",
"passed": true
},
{
"name": "claim1_small_exhaustive_enumerations",
"passed": true
},
{
"name": "claim1_over_400m",
"passed": true
},
{
"name": "claim1_control",
"passed": true
},
{
"name": "claim2_99b_total",
"passed": true
},
{
"name": "claim2_14_repository_domains",
"passed": true
},
{
"name": "claim2_control",
"passed": true
},
{
"name": "claim3_exact_statistics",
"passed": true
},
{
"name": "claim3_control",
"passed": true
},
{
"name": "claim4_all_m_plus_one_configs",
"passed": true
},
{
"name": "claim4_official_builder_execution",
"passed": true
},
{
"name": "claim4_control",
"passed": true
},
{
"name": "claim5_success_monotone",
"passed": true
},
{
"name": "claim5_ceiling_rises",
"passed": true
},
{
"name": "claim5_all_combinations_and_control",
"passed": true
}
],
"paper_id": "0Mvm3lqLjF",
"provenance": {
"arxiv": "2603.17863v1",
"official_code_archive_sha256": "64f4bef7a116be32df28c1bd7c10098573161d8b910b4dcef26ff1b87edf4da0",
"official_code_commit": "4ad81e3fee8b5d8b8fd76827142e107546f47769",
"python": "3.13.3",
"randomness": "none",
"source_sha256": "63a6cac8554672460ceb2a42f045bb3cb6eecea7f48b12537b81848ed47821d0"
},
"summary": {
"passed": 15,
"total": 15
}
}