{ "all_gates_pass": true, "claim_results": [ { "all_formula_rows_match": true, "all_small_enumerations_match": true, "assessment": "verified", "claim": 1, "destructive_control": { "b": 2, "correct": 4200, "d": 4, "m": 3, "without_nonempty_train_test_exclusion": 6804 }, "destructive_control_detected": true, "literal_claim": "DiscoGen procedurally generates over 400 million distinct algorithm discovery tasks via a combinatorial formula N_tasks = 2*3*b*(2^m-1)*(3^d-2^(d+1)+1) depending on the number of modules m, datasets d, and backends b (Section 4.2, Equation 1).", "main_domain_rows": [ { "b": 1, "d": 11, "domain": "Bayesian Optimisation", "m": 6, "matches": true, "recomputed_tasks": 65413656, "reported_tasks": 65413656 }, { "b": 1, "d": 7, "domain": "Brain Speech Detection", "m": 3, "matches": true, "recomputed_tasks": 81144, "reported_tasks": 81144 }, { "b": 1, "d": 9, "domain": "Computer Vision Classification", "m": 4, "matches": true, "recomputed_tasks": 1679400, "reported_tasks": 1679400 }, { "b": 3, "d": 3, "domain": "Continual Learning", "m": 5, "matches": true, "recomputed_tasks": 6696, "reported_tasks": 6696 }, { "b": 1, "d": 4, "domain": "Greenhouse Gas Prediction", "m": 2, "matches": true, "recomputed_tasks": 900, "reported_tasks": 900 }, { "b": 2, "d": 4, "domain": "Language Modelling", "m": 3, "matches": true, "recomputed_tasks": 4200, "reported_tasks": 4200 }, { "b": 1, "d": 3, "domain": "Model Unlearning", "m": 1, "matches": true, "recomputed_tasks": 85176, "reported_tasks": 85176 }, { "b": 1, "d": 4, "domain": "Off-Policy RL", "m": 7, "matches": true, "recomputed_tasks": 38100, "reported_tasks": 38100 }, { "b": 3, "d": 13, "domain": "On-Policy RL", "m": 4, "matches": true, "recomputed_tasks": 426043800, "reported_tasks": 426043800 }, { "b": 1, "d": 4, "domain": "Unsupervised Environment Design", "m": 3, "matches": true, "recomputed_tasks": 2100, "reported_tasks": 2100 } ], "main_total_from_rows": 493355172, "official_model_choices": 13, "over_400_million": true, "reported_main_total": 493355172, "small_exact_enumerations": [ { "b": 1, "d": 2, "direct": 12, "formula": 12, "m": 1, "matches": true }, { "b": 1, "d": 3, "direct": 216, "formula": 216, "m": 2, "matches": true }, { "b": 2, "d": 4, "direct": 4200, "formula": 4200, "m": 3, "matches": true }, { "b": 3, "d": 4, "direct": 13500, "formula": 13500, "m": 4, "matches": true } ] }, { "approximately_99_billion": true, "assessment": "verified", "claim": 2, "destructive_control_detected": true, "destructive_control_without_on_policy_marl": 1867332264, "expanded_domain_rows": [ { "b": 1, "d": 11, "domain": "Bayesian Optimisation", "m": 6, "reported_tasks": 65413656 }, { "b": 1, "d": 7, "domain": "Brain Speech Detection", "m": 3, "reported_tasks": 81144 }, { "b": 1, "d": 9, "domain": "Computer Vision Classification", "m": 4, "reported_tasks": 1679400 }, { "b": 3, "d": 3, "domain": "Continual Learning", "m": 5, "reported_tasks": 6696 }, { "b": 1, "d": 4, "domain": "Greenhouse Gas Prediction", "m": 2, "reported_tasks": 900 }, { "b": 2, "d": 4, "domain": "Language Modelling", "m": 3, "reported_tasks": 4200 }, { "b": 1, "d": 3, "domain": "Model Unlearning", "m": 1, "reported_tasks": 85176 }, { "b": 1, "d": 5, "domain": "Neural Cellular Automata", "m": 5, "reported_tasks": 33480 }, { "b": 1, "d": 4, "domain": "Off-Policy RL", "m": 7, "reported_tasks": 38100 }, { "b": 1, "d": 10, "domain": "Offline RL", "m": 5, "reported_tasks": 10602372 }, { "b": 2, "d": 17, "domain": "On-Policy MARL", "m": 6, "reported_tasks": 97431783120 }, { "b": 3, "d": 13, "domain": "On-Policy RL", "m": 6, "reported_tasks": 1789383960 }, { "b": 3, "d": 3, "domain": "Trajectory Prediction", "m": 4, "reported_tasks": 1080 }, { "b": 1, "d": 4, "domain": "Unsupervised Environment Design", "m": 3, "reported_tasks": 2100 } ], "expanded_total_from_rows": 99299115384, "literal_claim": "Including additional domains beyond the main evaluation set, DiscoGen's total task space reaches approximately 99 billion tasks (Appendix C).", "official_repository_domains": [ { "backend_count": 1, "backends": [ "default" ], "dataset_count": 11, "datasets": [ "Ackley1d", "Ackley2d", "Branin2d", "Bukin2d", "Cosine8d", "DropWave2d", "EggHolder2d", "Griewank5d", "Hartmann6d", "HolderTable2d", "Levy6d" ], "domain": "BayesianOptimisation", "model_count": 0, "module_count": 6, "modules": [ "acq_fn", "acq_optimizer", "next_queries", "sampler", "surrogate", "surrogate_optimizer" ] }, { "backend_count": 1, "backends": [ "default" ], "dataset_count": 7, "datasets": [ "LibriBrainSherlock1", "LibriBrainSherlock2", "LibriBrainSherlock3", "LibriBrainSherlock4", "LibriBrainSherlock5", "LibriBrainSherlock6", "LibriBrainSherlock7" ], "domain": "BrainSpeechDetection", "model_count": 0, "module_count": 3, "modules": [ "loss", "networks", "optim" ] }, { "backend_count": 1, "backends": [ "default" ], "dataset_count": 9, "datasets": [ "CIFAR10C", "CIFAR10", "CIFAR10LT", "CIFAR100", "FashionMNIST", "MNIST", "OxfordFlowers", "StanfordCars", "TinyImageNet" ], "domain": "ComputerVisionClassification", "model_count": 0, "module_count": 4, "modules": [ "loss", "networks", "optim", "preprocess" ] }, { "backend_count": 3, "backends": [ "default", "parameter_isolation", "transformer" ], "dataset_count": 3, "datasets": [ "PermutedMNIST", "SplitCIFAR100", "TinyImageNetSplit" ], "domain": "ContinualLearning", "model_count": 0, "module_count": 5, "modules": [ "optim", "regularizer", "replay", "sampler", "scheduler" ] }, { "backend_count": 1, "backends": [ "default" ], "dataset_count": 4, "datasets": [ "CH4", "CO2", "N2O", "SF6" ], "domain": "GreenhouseGasPrediction", "model_count": 0, "module_count": 2, "modules": [ "data_processing", "model" ] }, { "backend_count": 2, "backends": [ "default", "ssm" ], "dataset_count": 4, "datasets": [ "OPCFineWebCode", "OPCFineWebMath", "LMFineWeb", "TinyStories" ], "domain": "LanguageModelling", "model_count": 0, "module_count": 3, "modules": [ "loss", "networks", "optim" ] }, { "backend_count": 1, "backends": [ "default" ], "dataset_count": 3, "datasets": [ "muse", "tofu", "wmdp_cyber" ], "domain": "ModelUnlearning", "model_count": 13, "module_count": 1, "modules": [ "loss" ] }, { "backend_count": 1, "backends": [ "default" ], "dataset_count": 5, "datasets": [ "GrowingLizard", "GrowingButterfly", "SelfClassifyingMNIST", "MatrixOperations", "MNISTInpainting" ], "domain": "NeuralCellularAutomata", "model_count": 0, "module_count": 5, "modules": [ "loss", "optimiser", "perceive", "train", "update" ] }, { "backend_count": 1, "backends": [ "default" ], "dataset_count": 4, "datasets": [ "MinAtar/Asterix", "MinAtar/Breakout", "MinAtar/Freeway", "MinAtar/SpaceInvaders" ], "domain": "OffPolicyRL", "model_count": 0, "module_count": 7, "modules": [ "config", "networks", "optim", "policy", "q_update", "rb", "train" ] }, { "backend_count": 1, "backends": [ "default" ], "dataset_count": 10, "datasets": [ "OGBench/antmaze-giant-navigate", "OGBench/antmaze-large-navigate", "OGBench/antsoccer-arena-navigate", "OGBench/cube-double-play", "OGBench/cube-single-play", "OGBench/humanoidmaze-large-navigate", "OGBench/humanoidmaze-medium-navigate", "OGBench/puzzle-3x3-play", "OGBench/puzzle-4x4-play", "OGBench/scene-play" ], "domain": "OfflineRL", "model_count": 0, "module_count": 5, "modules": [ "actor_loss", "critic_loss", "networks", "optim", "train" ] }, { "backend_count": 2, "backends": [ "default", "recurrent" ], "dataset_count": 17, "datasets": [ "MABrax/Ant", "MABrax/HalfCheetah", "MABrax/Hopper", "MABrax/Walker", "MABrax/Humanoid", "MPE/Spread", "SMAX/2s3z", "SMAX/3s_vs_5z", "SMAX/3s5z", "SMAX/3s5z_vs_3s6z", "SMAX/5m_vs_6m", "SMAX/6h_vs_8z", "SMAX/10m_vs_11m", "SMAX/27m_vs_30m", "SMAX/smacv2_5_units", "SMAX/smacv2_10_units", "SMAX/smacv2_20_units" ], "domain": "OnPolicyMARL", "model_count": 0, "module_count": 6, "modules": [ "activation", "loss", "networks", "optim", "targets", "train" ] }, { "backend_count": 3, "backends": [ "default", "recurrent", "transformer" ], "dataset_count": 13, "datasets": [ "MinAtar/Asterix", "MinAtar/Breakout", "MinAtar/Freeway", "MinAtar/SpaceInvaders", "Brax/Ant", "Brax/HalfCheetah", "Brax/Hopper", "Brax/Humanoid", "Brax/Pusher", "Brax/Reacher", "Brax/Walker2D", "Craftax/Craftax", "Craftax/Craftax-Classic" ], "domain": "OnPolicyRL", "model_count": 0, "module_count": 6, "modules": [ "activation", "loss", "networks", "optim", "targets", "train" ] }, { "backend_count": 1, "backends": [ "default" ], "dataset_count": 3, "datasets": [ "Argoverse2", "nuScenes", "Waymo" ], "domain": "TrajectoryPrediction", "model_count": 0, "module_count": 4, "modules": [ "loss", "networks", "optim", "train" ] }, { "backend_count": 1, "backends": [ "default" ], "dataset_count": 4, "datasets": [ "Kinetix/Small", "Kinetix/Medium", "Kinetix/Large", "Minigrid" ], "domain": "UnsupervisedEnvironmentDesign", "model_count": 0, "module_count": 3, "modules": [ "sample_levels", "train_step", "variable_config" ] } ], "reported_expanded_total": 99299115384, "repository_domain_count": 14 }, { "all_registered_statistics_match": true, "assessment": "verified", "claim": 3, "computed_median": 59622, "destructive_control_detected": true, "destructive_control_drop_one_domain_median": 81144.0, "domain_count": 10, "literal_claim": "Across the 10 domains used in the main evaluation, per-domain task counts range from 900 (Greenhouse Gas Prediction) to 426,043,800 (On-Policy RL), with a median of 59,622 tasks per domain (Table 1).", "maximum": { "b": 3, "d": 13, "domain": "On-Policy RL", "m": 4, "reported_tasks": 426043800 }, "minimum": { "b": 1, "d": 4, "domain": "Greenhouse Gas Prediction", "m": 2, "reported_tasks": 900 }, "reported_median": 59622 }, { "actual_generated_files": 82, "assessment": "verified", "audited_config_count": 74, "claim": 4, "configuration_failures": 0, "configuration_rows": [ { "changed_modules": [ "acq_fn" ], "failures": [], "file": "BayesianOptimisation_acq_fn.yaml" }, { "changed_modules": [ "acq_optimizer" ], "failures": [], "file": "BayesianOptimisation_acq_optimizer.yaml" }, { "changed_modules": [ "next_queries" ], "failures": [], "file": "BayesianOptimisation_next_queries.yaml" }, { "changed_modules": [ "sampler" ], "failures": [], "file": "BayesianOptimisation_sampler.yaml" }, { "changed_modules": [ "surrogate" ], "failures": [], "file": "BayesianOptimisation_surrogate.yaml" }, { "changed_modules": [ "surrogate_optimizer" ], "failures": [], "file": "BayesianOptimisation_surrogate_optimizer.yaml" }, { "changed_modules": [ "acq_fn", "acq_optimizer", "next_queries", "sampler", "surrogate_optimizer", "surrogate" ], "failures": [], "file": "BayesianOptimisation_all.yaml" }, { "changed_modules": [ "loss" ], "failures": [], "file": "BrainSpeechDetection_loss.yaml" }, { "changed_modules": [ "networks" ], "failures": [], "file": "BrainSpeechDetection_networks.yaml" }, { "changed_modules": [ "optim" ], "failures": [], "file": "BrainSpeechDetection_optim.yaml" }, { "changed_modules": [ "optim", "loss", "networks" ], "failures": [], "file": "BrainSpeechDetection_all.yaml" }, { "changed_modules": [ "loss" ], "failures": [], "file": "ComputerVisionClassification_loss.yaml" }, { "changed_modules": [ "networks" ], "failures": [], "file": "ComputerVisionClassification_networks.yaml" }, { "changed_modules": [ "optim" ], "failures": [], "file": "ComputerVisionClassification_optim.yaml" }, { "changed_modules": [ "preprocess" ], "failures": [], "file": "ComputerVisionClassification_preprocess.yaml" }, { "changed_modules": [ "optim", "loss", "networks", "preprocess" ], "failures": [], "file": "ComputerVisionClassification_all.yaml" }, { "changed_modules": [ "optim" ], "failures": [], "file": "ContinualLearning_optim.yaml" }, { "changed_modules": [ "regularizer" ], "failures": [], "file": "ContinualLearning_regularizer.yaml" }, { "changed_modules": [ "replay" ], "failures": [], "file": "ContinualLearning_replay.yaml" }, { "changed_modules": [ "sampler" ], "failures": [], "file": "ContinualLearning_sampler.yaml" }, { "changed_modules": [ "scheduler" ], "failures": [], "file": "ContinualLearning_scheduler.yaml" }, { "changed_modules": [ "regularizer", "replay", "sampler", "scheduler", "optim" ], "failures": [], "file": "ContinualLearning_all.yaml" }, { "changed_modules": [ "data_processing" ], "failures": [], "file": "GreenhouseGasPrediction_data_processing.yaml" }, { "changed_modules": [ "model" ], "failures": [], "file": "GreenhouseGasPrediction_model.yaml" }, { "changed_modules": [ "data_processing", "model" ], "failures": [], "file": "GreenhouseGasPrediction_all.yaml" }, { "changed_modules": [ "loss" ], "failures": [], "file": "LanguageModelling_loss.yaml" }, { "changed_modules": [ "networks" ], "failures": [], "file": "LanguageModelling_networks.yaml" }, { "changed_modules": [ "optim" ], "failures": [], "file": "LanguageModelling_optim.yaml" }, { "changed_modules": [ "optim", "loss", "networks" ], "failures": [], "file": "LanguageModelling_all.yaml" }, { "changed_modules": [ "loss" ], "failures": [], "file": "ModelUnlearning_loss.yaml" }, { "changed_modules": [ "loss" ], "failures": [], "file": "ModelUnlearning_all.yaml" }, { "changed_modules": [ "loss" ], "failures": [], "file": "NeuralCellularAutomata_loss.yaml" }, { "changed_modules": [ "optimiser" ], "failures": [], "file": "NeuralCellularAutomata_optimiser.yaml" }, { "changed_modules": [ "perceive" ], "failures": [], "file": "NeuralCellularAutomata_perceive.yaml" }, { "changed_modules": [ "train" ], "failures": [], "file": "NeuralCellularAutomata_train.yaml" }, { "changed_modules": [ "update" ], "failures": [], "file": "NeuralCellularAutomata_update.yaml" }, { "changed_modules": [ "perceive", "update", "loss", "train", "optimiser" ], "failures": [], "file": "NeuralCellularAutomata_all.yaml" }, { "changed_modules": [ "config" ], "failures": [], "file": "OffPolicyRL_config.yaml" }, { "changed_modules": [ "networks" ], "failures": [], "file": "OffPolicyRL_networks.yaml" }, { "changed_modules": [ "optim" ], "failures": [], "file": "OffPolicyRL_optim.yaml" }, { "changed_modules": [ "policy" ], "failures": [], "file": "OffPolicyRL_policy.yaml" }, { "changed_modules": [ "q_update" ], "failures": [], "file": "OffPolicyRL_q_update.yaml" }, { "changed_modules": [ "rb" ], "failures": [], "file": "OffPolicyRL_rb.yaml" }, { "changed_modules": [ "train" ], "failures": [], "file": "OffPolicyRL_train.yaml" }, { "changed_modules": [ "config", "networks", "optim", "policy", "q_update", "rb", "train" ], "failures": [], "file": "OffPolicyRL_all.yaml" }, { "changed_modules": [ "actor_loss" ], "failures": [], "file": "OfflineRL_actor_loss.yaml" }, { "changed_modules": [ "critic_loss" ], "failures": [], "file": "OfflineRL_critic_loss.yaml" }, { "changed_modules": [ "networks" ], "failures": [], "file": "OfflineRL_networks.yaml" }, { "changed_modules": [ "optim" ], "failures": [], "file": "OfflineRL_optim.yaml" }, { "changed_modules": [ "train" ], "failures": [], "file": "OfflineRL_train.yaml" }, { "changed_modules": [ "optim", "actor_loss", "critic_loss", "networks", "train" ], "failures": [], "file": "OfflineRL_all.yaml" }, { "changed_modules": [ "activation" ], "failures": [], "file": "OnPolicyMARL_activation.yaml" }, { "changed_modules": [ "loss" ], "failures": [], "file": "OnPolicyMARL_loss.yaml" }, { "changed_modules": [ "networks" ], "failures": [], "file": "OnPolicyMARL_networks.yaml" }, { "changed_modules": [ "optim" ], "failures": [], "file": "OnPolicyMARL_optim.yaml" }, { "changed_modules": [ "targets" ], "failures": [], "file": "OnPolicyMARL_targets.yaml" }, { "changed_modules": [ "train" ], "failures": [], "file": "OnPolicyMARL_train.yaml" }, { "changed_modules": [ "optim", "loss", "networks", "train", "activation", "targets" ], "failures": [], "file": "OnPolicyMARL_all.yaml" }, { "changed_modules": [ "activation" ], "failures": [], "file": "OnPolicyRL_activation.yaml" }, { "changed_modules": [ "loss" ], "failures": [], "file": "OnPolicyRL_loss.yaml" }, { "changed_modules": [ "networks" ], "failures": [], "file": "OnPolicyRL_networks.yaml" }, { "changed_modules": [ "optim" ], "failures": [], "file": "OnPolicyRL_optim.yaml" }, { "changed_modules": [ "targets" ], "failures": [], "file": "OnPolicyRL_targets.yaml" }, { "changed_modules": [ "train" ], "failures": [], "file": "OnPolicyRL_train.yaml" }, { "changed_modules": [ "optim", "loss", "networks", "train", "activation", "targets" ], "failures": [], "file": "OnPolicyRL_all.yaml" }, { "changed_modules": [ "loss" ], "failures": [], "file": "TrajectoryPrediction_loss.yaml" }, { "changed_modules": [ "networks" ], "failures": [], "file": "TrajectoryPrediction_networks.yaml" }, { "changed_modules": [ "optim" ], "failures": [], "file": "TrajectoryPrediction_optim.yaml" }, { "changed_modules": [ "train" ], "failures": [], "file": "TrajectoryPrediction_train.yaml" }, { "changed_modules": [ "optim", "loss", "networks", "train" ], "failures": [], "file": "TrajectoryPrediction_all.yaml" }, { "changed_modules": [ "sample_levels" ], "failures": [], "file": "UnsupervisedEnvironmentDesign_sample_levels.yaml" }, { "changed_modules": [ "train_step" ], "failures": [], "file": "UnsupervisedEnvironmentDesign_train_step.yaml" }, { "changed_modules": [ "variable_config" ], "failures": [], "file": "UnsupervisedEnvironmentDesign_variable_config.yaml" }, { "changed_modules": [ "sample_levels", "train_step", "variable_config" ], "failures": [], "file": "UnsupervisedEnvironmentDesign_all.yaml" } ], "destructive_control_active_modules": [ "change_data_processing", "change_model" ], "destructive_control_detected": true, "expected_m_plus_one_configs": 74, "literal_claim": "DiscoBench provides a fixed evaluation subset built from DiscoGen, comprising, for each domain, m single-module tasks (DiscoBench Single) plus one comprehensive all-modules-active task (DiscoBench All) (Section 4.4).", "official_builder_execution": { "canonicalized_description_files": 4, "executed_configs": [ "GreenhouseGasPrediction_all", "GreenhouseGasPrediction_data_processing", "OnPolicyRL_loss", "OnPolicyRL_all" ], "file_count": 82, "files": [ { "bytes": 300, "path": "GreenhouseGasPrediction_all/CH4/data_processing.py", "sha256": "e3a4d5be0be7e7d5129181846167a2fd76fc1e9b2b99049f8406ecdc055bac4e" }, { "bytes": 1372, "path": "GreenhouseGasPrediction_all/CH4/main.py", "sha256": "094787e39823668d2cbd472af1f00eee3f49ca26fd1c2d5f179aee9aafbe7adf" }, { "bytes": 597, "path": "GreenhouseGasPrediction_all/CH4/model.py", "sha256": "dd94a5687d62d89a8d510ce2c3b7f059de2ff0bc4b7fd7a719cef6ceb5126549" }, { "bytes": 300, "path": "GreenhouseGasPrediction_all/SF6/data_processing.py", "sha256": "e3a4d5be0be7e7d5129181846167a2fd76fc1e9b2b99049f8406ecdc055bac4e" }, { "bytes": 1372, "path": "GreenhouseGasPrediction_all/SF6/main.py", "sha256": "094787e39823668d2cbd472af1f00eee3f49ca26fd1c2d5f179aee9aafbe7adf" }, { "bytes": 597, "path": "GreenhouseGasPrediction_all/SF6/model.py", "sha256": "dd94a5687d62d89a8d510ce2c3b7f059de2ff0bc4b7fd7a719cef6ceb5126549" }, { "bytes": 8817, "path": "GreenhouseGasPrediction_all/description.md", "sha256": "1b052d3d51ed063eb562bf5a22a3918c358efd215d807e73cf6a20993c0258e1" }, { "bytes": 300, "path": "GreenhouseGasPrediction_all/discovered/data_processing.py", "sha256": "e3a4d5be0be7e7d5129181846167a2fd76fc1e9b2b99049f8406ecdc055bac4e" }, { "bytes": 597, "path": "GreenhouseGasPrediction_all/discovered/model.py", "sha256": "dd94a5687d62d89a8d510ce2c3b7f059de2ff0bc4b7fd7a719cef6ceb5126549" }, { "bytes": 53, "path": "GreenhouseGasPrediction_all/install.sh", "sha256": "6964f075ffc565dd9847b7203a28d7368b0825333435a756b48ec7da08757af7" }, { "bytes": 155, "path": "GreenhouseGasPrediction_all/requirements.txt", "sha256": "ce1ca66a4ff2d670272eb600e8cd5fd3c530c637c86d7477110015d2d43ca8fa" }, { "bytes": 1725, "path": "GreenhouseGasPrediction_all/run_main.py", "sha256": "88db3fe20483a035b63bf5cb9c415493ae52cfa09e944fcb5ea461aa66407d90" }, { "bytes": 300, "path": "GreenhouseGasPrediction_data_processing/CH4/data_processing.py", "sha256": "e3a4d5be0be7e7d5129181846167a2fd76fc1e9b2b99049f8406ecdc055bac4e" }, { "bytes": 1372, "path": "GreenhouseGasPrediction_data_processing/CH4/main.py", "sha256": "094787e39823668d2cbd472af1f00eee3f49ca26fd1c2d5f179aee9aafbe7adf" }, { "bytes": 458, "path": "GreenhouseGasPrediction_data_processing/CH4/model.py", "sha256": "57e94b384ad5ccf7a6a803568f8b7a7350db7786b0308375370e7195ed93e172" }, { "bytes": 300, "path": "GreenhouseGasPrediction_data_processing/SF6/data_processing.py", "sha256": "e3a4d5be0be7e7d5129181846167a2fd76fc1e9b2b99049f8406ecdc055bac4e" }, { "bytes": 1372, "path": "GreenhouseGasPrediction_data_processing/SF6/main.py", "sha256": "094787e39823668d2cbd472af1f00eee3f49ca26fd1c2d5f179aee9aafbe7adf" }, { "bytes": 458, "path": "GreenhouseGasPrediction_data_processing/SF6/model.py", "sha256": "57e94b384ad5ccf7a6a803568f8b7a7350db7786b0308375370e7195ed93e172" }, { "bytes": 8691, "path": "GreenhouseGasPrediction_data_processing/description.md", "sha256": "c65914e7460ad4529dcd70b4788a8ced437909123cc3ea19599a39615db0e2dd" }, { "bytes": 300, "path": "GreenhouseGasPrediction_data_processing/discovered/data_processing.py", "sha256": "e3a4d5be0be7e7d5129181846167a2fd76fc1e9b2b99049f8406ecdc055bac4e" }, { "bytes": 53, "path": "GreenhouseGasPrediction_data_processing/install.sh", "sha256": "6964f075ffc565dd9847b7203a28d7368b0825333435a756b48ec7da08757af7" }, { "bytes": 155, "path": "GreenhouseGasPrediction_data_processing/requirements.txt", "sha256": "ce1ca66a4ff2d670272eb600e8cd5fd3c530c637c86d7477110015d2d43ca8fa" }, { "bytes": 1725, "path": "GreenhouseGasPrediction_data_processing/run_main.py", "sha256": "88db3fe20483a035b63bf5cb9c415493ae52cfa09e944fcb5ea461aa66407d90" }, { "bytes": 0, "path": "OnPolicyRL_all/MinAtar/Breakout/__init__.py", "sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" }, { "bytes": 277, "path": "OnPolicyRL_all/MinAtar/Breakout/activation.py", "sha256": "1830544b21f2ed84ac4e5ac94ad33047ebfa178394a1bf7e3b3567ad2f497e76" }, { "bytes": 346, "path": "OnPolicyRL_all/MinAtar/Breakout/config.py", "sha256": "8f6949d7df8358ff71e3006ac89118c445aa74c72c099e3e58058689c9b5050c" }, { "bytes": 850, "path": "OnPolicyRL_all/MinAtar/Breakout/loss.py", "sha256": "7866cfc22268e39cb4e0564893a7afc7284088041278d2f04aab38f7c76f595c" }, { "bytes": 8942, "path": "OnPolicyRL_all/MinAtar/Breakout/main.py", "sha256": "ad901568e2f90a79ea2c570e2482e3f89189a48e9e62663b2f46965a6744c488" }, { "bytes": 245, "path": "OnPolicyRL_all/MinAtar/Breakout/make_env.py", "sha256": "c24169e8c5f04fa5f3af519e9f98c466f8c7a6233edc7b4e9085a25562c94602" }, { "bytes": 818, "path": "OnPolicyRL_all/MinAtar/Breakout/networks.py", "sha256": "8710a87b23befb124a88ca9676133e0dfd336835f61880afe7ab0a3bc911561d" }, { "bytes": 1670, "path": "OnPolicyRL_all/MinAtar/Breakout/optim.py", "sha256": "9344db01c0d8465c28ed142ca18a7d0acf4e4f8ee5d59f440a16ec6cca076a9b" }, { "bytes": 261, "path": "OnPolicyRL_all/MinAtar/Breakout/targets.py", "sha256": "d5ee1bddf2ccfe8c4613dfdf5623743d788f269cd546006aa7a18e433d194f0d" }, { "bytes": 3648, "path": "OnPolicyRL_all/MinAtar/Breakout/train.py", "sha256": "4c9515a812deea8226e22652defc9c9e326d1954216dcc5a859673b80c97be67" }, { "bytes": 10698, "path": "OnPolicyRL_all/MinAtar/Breakout/wrappers.py", "sha256": "e611bbacd53433415d12fb709da822a755df3770fbf7f51c6d789a485415ac68" }, { "bytes": 0, "path": "OnPolicyRL_all/MinAtar/Freeway/__init__.py", "sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" }, { "bytes": 277, "path": "OnPolicyRL_all/MinAtar/Freeway/activation.py", "sha256": "1830544b21f2ed84ac4e5ac94ad33047ebfa178394a1bf7e3b3567ad2f497e76" }, { "bytes": 346, "path": "OnPolicyRL_all/MinAtar/Freeway/config.py", "sha256": "8f6949d7df8358ff71e3006ac89118c445aa74c72c099e3e58058689c9b5050c" }, { "bytes": 850, "path": "OnPolicyRL_all/MinAtar/Freeway/loss.py", "sha256": "7866cfc22268e39cb4e0564893a7afc7284088041278d2f04aab38f7c76f595c" }, { "bytes": 8942, "path": "OnPolicyRL_all/MinAtar/Freeway/main.py", "sha256": "ad901568e2f90a79ea2c570e2482e3f89189a48e9e62663b2f46965a6744c488" }, { "bytes": 244, "path": "OnPolicyRL_all/MinAtar/Freeway/make_env.py", "sha256": "c18fedfd39747012af573a9f81055144e3bfdc90940a2666bbdbaa53d85bb583" }, { "bytes": 818, "path": "OnPolicyRL_all/MinAtar/Freeway/networks.py", "sha256": "8710a87b23befb124a88ca9676133e0dfd336835f61880afe7ab0a3bc911561d" }, { "bytes": 1670, "path": "OnPolicyRL_all/MinAtar/Freeway/optim.py", "sha256": "9344db01c0d8465c28ed142ca18a7d0acf4e4f8ee5d59f440a16ec6cca076a9b" }, { "bytes": 261, "path": "OnPolicyRL_all/MinAtar/Freeway/targets.py", "sha256": "d5ee1bddf2ccfe8c4613dfdf5623743d788f269cd546006aa7a18e433d194f0d" }, { "bytes": 3648, "path": "OnPolicyRL_all/MinAtar/Freeway/train.py", "sha256": "4c9515a812deea8226e22652defc9c9e326d1954216dcc5a859673b80c97be67" }, { "bytes": 10698, "path": "OnPolicyRL_all/MinAtar/Freeway/wrappers.py", "sha256": "e611bbacd53433415d12fb709da822a755df3770fbf7f51c6d789a485415ac68" }, { "bytes": 10734, "path": "OnPolicyRL_all/description.md", "sha256": "db1ed45e365da592846c9a1acb9f1460a25e39ccc3bb4c4c425ec0b9f71eb11c" }, { "bytes": 277, "path": "OnPolicyRL_all/discovered/activation.py", "sha256": "1830544b21f2ed84ac4e5ac94ad33047ebfa178394a1bf7e3b3567ad2f497e76" }, { "bytes": 850, "path": "OnPolicyRL_all/discovered/loss.py", "sha256": "7866cfc22268e39cb4e0564893a7afc7284088041278d2f04aab38f7c76f595c" }, { "bytes": 818, "path": "OnPolicyRL_all/discovered/networks.py", "sha256": "8710a87b23befb124a88ca9676133e0dfd336835f61880afe7ab0a3bc911561d" }, { "bytes": 1670, "path": "OnPolicyRL_all/discovered/optim.py", "sha256": "9344db01c0d8465c28ed142ca18a7d0acf4e4f8ee5d59f440a16ec6cca076a9b" }, { "bytes": 261, "path": "OnPolicyRL_all/discovered/targets.py", "sha256": "d5ee1bddf2ccfe8c4613dfdf5623743d788f269cd546006aa7a18e433d194f0d" }, { "bytes": 3648, "path": "OnPolicyRL_all/discovered/train.py", "sha256": "4c9515a812deea8226e22652defc9c9e326d1954216dcc5a859673b80c97be67" }, { "bytes": 53, "path": "OnPolicyRL_all/install.sh", "sha256": "6964f075ffc565dd9847b7203a28d7368b0825333435a756b48ec7da08757af7" }, { "bytes": 272, "path": "OnPolicyRL_all/requirements.txt", "sha256": "f052fa4717c1411280ec50ba717865be0fe1390c60e68da176b4986fe8f551aa" }, { "bytes": 1725, "path": "OnPolicyRL_all/run_main.py", "sha256": "88db3fe20483a035b63bf5cb9c415493ae52cfa09e944fcb5ea461aa66407d90" }, { "bytes": 0, "path": "OnPolicyRL_loss/MinAtar/Breakout/__init__.py", "sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" }, { "bytes": 182, "path": "OnPolicyRL_loss/MinAtar/Breakout/activation.py", "sha256": "429973432f0977f59dc029c529c9b8989618cc7699c72a1d9fdeea9f9f827e1c" }, { "bytes": 346, "path": "OnPolicyRL_loss/MinAtar/Breakout/config.py", "sha256": "8f6949d7df8358ff71e3006ac89118c445aa74c72c099e3e58058689c9b5050c" }, { "bytes": 850, "path": "OnPolicyRL_loss/MinAtar/Breakout/loss.py", "sha256": "7866cfc22268e39cb4e0564893a7afc7284088041278d2f04aab38f7c76f595c" }, { "bytes": 8942, "path": "OnPolicyRL_loss/MinAtar/Breakout/main.py", "sha256": "ad901568e2f90a79ea2c570e2482e3f89189a48e9e62663b2f46965a6744c488" }, { "bytes": 245, "path": "OnPolicyRL_loss/MinAtar/Breakout/make_env.py", "sha256": "c24169e8c5f04fa5f3af519e9f98c466f8c7a6233edc7b4e9085a25562c94602" }, { "bytes": 1680, "path": "OnPolicyRL_loss/MinAtar/Breakout/networks.py", "sha256": "456176a5ce041e85b85cf57b8829265af1431cc48cbfd017dac4dfd0a8689602" }, { "bytes": 156, "path": "OnPolicyRL_loss/MinAtar/Breakout/optim.py", "sha256": "e88de32252fed900d835598dc968e9d67e134dd5a0d7a3e29b62533e3e5fdaaf" }, { "bytes": 767, "path": "OnPolicyRL_loss/MinAtar/Breakout/targets.py", "sha256": "a36111a2aebe9c98cf7d3dc7e121bf8b2e2b846d7bc962ef2ba032712bc644d9" }, { "bytes": 8102, "path": "OnPolicyRL_loss/MinAtar/Breakout/train.py", "sha256": "0013f408cec5cefb3b1426c22e6b57405ca36ad8b72e61ce24a4a27d127f67fe" }, { "bytes": 10698, "path": "OnPolicyRL_loss/MinAtar/Breakout/wrappers.py", "sha256": "e611bbacd53433415d12fb709da822a755df3770fbf7f51c6d789a485415ac68" }, { "bytes": 0, "path": "OnPolicyRL_loss/MinAtar/Freeway/__init__.py", "sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" }, { "bytes": 182, "path": "OnPolicyRL_loss/MinAtar/Freeway/activation.py", "sha256": "429973432f0977f59dc029c529c9b8989618cc7699c72a1d9fdeea9f9f827e1c" }, { "bytes": 346, "path": "OnPolicyRL_loss/MinAtar/Freeway/config.py", "sha256": "8f6949d7df8358ff71e3006ac89118c445aa74c72c099e3e58058689c9b5050c" }, { "bytes": 850, "path": "OnPolicyRL_loss/MinAtar/Freeway/loss.py", "sha256": "7866cfc22268e39cb4e0564893a7afc7284088041278d2f04aab38f7c76f595c" }, { "bytes": 8942, "path": "OnPolicyRL_loss/MinAtar/Freeway/main.py", "sha256": "ad901568e2f90a79ea2c570e2482e3f89189a48e9e62663b2f46965a6744c488" }, { "bytes": 244, "path": "OnPolicyRL_loss/MinAtar/Freeway/make_env.py", "sha256": "c18fedfd39747012af573a9f81055144e3bfdc90940a2666bbdbaa53d85bb583" }, { "bytes": 1680, "path": "OnPolicyRL_loss/MinAtar/Freeway/networks.py", "sha256": "456176a5ce041e85b85cf57b8829265af1431cc48cbfd017dac4dfd0a8689602" }, { "bytes": 156, "path": "OnPolicyRL_loss/MinAtar/Freeway/optim.py", "sha256": "e88de32252fed900d835598dc968e9d67e134dd5a0d7a3e29b62533e3e5fdaaf" }, { "bytes": 767, "path": "OnPolicyRL_loss/MinAtar/Freeway/targets.py", "sha256": "a36111a2aebe9c98cf7d3dc7e121bf8b2e2b846d7bc962ef2ba032712bc644d9" }, { "bytes": 8102, "path": "OnPolicyRL_loss/MinAtar/Freeway/train.py", "sha256": "0013f408cec5cefb3b1426c22e6b57405ca36ad8b72e61ce24a4a27d127f67fe" }, { "bytes": 10698, "path": "OnPolicyRL_loss/MinAtar/Freeway/wrappers.py", "sha256": "e611bbacd53433415d12fb709da822a755df3770fbf7f51c6d789a485415ac68" }, { "bytes": 8306, "path": "OnPolicyRL_loss/description.md", "sha256": "28f98c846f8542e35ce544aa4c4d5d5387805cd32c70739971e24841fc4914b4" }, { "bytes": 850, "path": "OnPolicyRL_loss/discovered/loss.py", "sha256": "7866cfc22268e39cb4e0564893a7afc7284088041278d2f04aab38f7c76f595c" }, { "bytes": 53, "path": "OnPolicyRL_loss/install.sh", "sha256": "6964f075ffc565dd9847b7203a28d7368b0825333435a756b48ec7da08757af7" }, { "bytes": 272, "path": "OnPolicyRL_loss/requirements.txt", "sha256": "f052fa4717c1411280ec50ba717865be0fe1390c60e68da176b4986fe8f551aa" }, { "bytes": 1725, "path": "OnPolicyRL_loss/run_main.py", "sha256": "88db3fe20483a035b63bf5cb9c415493ae52cfa09e944fcb5ea461aa66407d90" } ], "tree_sha256": "039f3fbad73fac4c9459b3eed28a53f60c98da12bfc07b4478237805ab32af16" }, "official_domain_count": 14 }, { "all_three_models_nonincreasing": true, "assessment": "verified", "claim": 5, "combination_count": 15, "complete_four_module_combinations": [ [ "loss" ], [ "networks" ], [ "optim" ], [ "train" ], [ "loss", "networks" ], [ "loss", "optim" ], [ "loss", "train" ], [ "networks", "optim" ], [ "networks", "train" ], [ "optim", "train" ], [ "loss", "networks", "optim" ], [ "loss", "networks", "train" ], [ "loss", "optim", "train" ], [ "networks", "optim", "train" ], [ "loss", "networks", "optim", "train" ] ], "destructive_control_detected": true, "destructive_control_reversed_success_rates": { "Deepseek-v3.2": [ 0.0, 8.3, 47.2, 75.0 ], "Devstral2": [ 0.0, 0.0, 27.8, 29.2 ], "GPT-OSS-120b": [ 0.0, 8.3, 11.1, 50.0 ] }, "environments_with_higher_two_module_ceiling": 3, "literal_claim": "As the number of editable modules increases in DiscoBench tasks, agent success rates consistently decline while the achievable performance ceiling rises (Appendix G).", "maximum_return_by_module_count": { "1": [ 99.97, 68.07, 39.6, 189.75 ], "2": [ 106.12, 66.16, 69.42, 191.62 ], "3": [ 88.58, 65.2, 64.85, 186.75 ], "4": [ null, null, null, null ] }, "mean_ceiling_increase": 8.9825, "monotonicity_by_model": { "Deepseek-v3.2": true, "Devstral2": true, "GPT-OSS-120b": true }, "single_module_mean_environment_ceiling": 99.3475, "source_configuration_rows": [ { "configuration": "Optimiser", "module_count": 1, "returns": [ 74.97, 62.86, 18.11, 181.25 ], "success_rate": 83.33 }, { "configuration": "Loss", "module_count": 1, "returns": [ 83.91, 62.58, 39.6, 179.5 ], "success_rate": 77.78 }, { "configuration": "Network", "module_count": 1, "returns": [ 99.97, 68.07, 14.2, 189.75 ], "success_rate": 33.33 }, { "configuration": "Train", "module_count": 1, "returns": [ 8.41, 8.41, 3.73, 177.12 ], "success_rate": 11.11 }, { "configuration": "Loss + Optimiser", "module_count": 2, "returns": [ 84.47, 62.89, 20.5, 181.38 ], "success_rate": 61.11 }, { "configuration": "Network + Optimiser", "module_count": 2, "returns": [ 91.44, 65.25, 19.77, 191.62 ], "success_rate": 44.44 }, { "configuration": "Loss + Network", "module_count": 2, "returns": [ 106.12, 66.16, 69.42, 184.0 ], "success_rate": 38.89 }, { "configuration": "Loss + Train", "module_count": 2, "returns": [ 8.56, 61.91, 3.91, 169.38 ], "success_rate": 22.22 }, { "configuration": "Optimiser + Train", "module_count": 2, "returns": [ 34.61, 29.45, null, null ], "success_rate": 5.56 }, { "configuration": "Network + Train", "module_count": 2, "returns": [ null, null, null, null ], "success_rate": 0.0 }, { "configuration": "Loss + Network + Optimiser", "module_count": 3, "returns": [ 88.58, 65.2, 64.85, 186.75 ], "success_rate": 11.11 }, { "configuration": "Loss + Optimiser + Train", "module_count": 3, "returns": [ 0.3, 2.95, null, null ], "success_rate": 11.11 }, { "configuration": "Loss + Network + Train", "module_count": 3, "returns": [ null, null, null, null ], "success_rate": 0.0 }, { "configuration": "Network + Optimiser + Train", "module_count": 3, "returns": [ null, null, null, null ], "success_rate": 0.0 }, { "configuration": "Loss + Network + Optimiser + Train", "module_count": 4, "returns": [ null, null, null, null ], "success_rate": 0.0 } ], "success_rate_by_editable_module_count": { "Deepseek-v3.2": [ 75.0, 47.2, 8.3, 0.0 ], "Devstral2": [ 29.2, 27.8, 0.0, 0.0 ], "GPT-OSS-120b": [ 50.0, 11.1, 8.3, 0.0 ] }, "two_module_mean_environment_ceiling": 108.33 } ], "claims": [ { "claim": 1, "literal_claim": "DiscoGen procedurally generates over 400 million distinct algorithm discovery tasks via a combinatorial formula N_tasks = 2*3*b*(2^m-1)*(3^d-2^(d+1)+1) depending on the number of modules m, datasets d, and backends b (Section 4.2, Equation 1)." }, { "claim": 2, "literal_claim": "Including additional domains beyond the main evaluation set, DiscoGen's total task space reaches approximately 99 billion tasks (Appendix C)." }, { "claim": 3, "literal_claim": "Across the 10 domains used in the main evaluation, per-domain task counts range from 900 (Greenhouse Gas Prediction) to 426,043,800 (On-Policy RL), with a median of 59,622 tasks per domain (Table 1)." }, { "claim": 4, "literal_claim": "DiscoBench provides a fixed evaluation subset built from DiscoGen, comprising, for each domain, m single-module tasks (DiscoBench Single) plus one comprehensive all-modules-active task (DiscoBench All) (Section 4.4)." }, { "claim": 5, "literal_claim": "As the number of editable modules increases in DiscoBench tasks, agent success rates consistently decline while the achievable performance ceiling rises (Appendix G)." } ], "gates": [ { "name": "claim1_all_formula_rows", "passed": true }, { "name": "claim1_small_exhaustive_enumerations", "passed": true }, { "name": "claim1_over_400m", "passed": true }, { "name": "claim1_control", "passed": true }, { "name": "claim2_99b_total", "passed": true }, { "name": "claim2_14_repository_domains", "passed": true }, { "name": "claim2_control", "passed": true }, { "name": "claim3_exact_statistics", "passed": true }, { "name": "claim3_control", "passed": true }, { "name": "claim4_all_m_plus_one_configs", "passed": true }, { "name": "claim4_official_builder_execution", "passed": true }, { "name": "claim4_control", "passed": true }, { "name": "claim5_success_monotone", "passed": true }, { "name": "claim5_ceiling_rises", "passed": true }, { "name": "claim5_all_combinations_and_control", "passed": true } ], "paper_id": "0Mvm3lqLjF", "provenance": { "arxiv": "2603.17863v1", "official_code_archive_sha256": "64f4bef7a116be32df28c1bd7c10098573161d8b910b4dcef26ff1b87edf4da0", "official_code_commit": "4ad81e3fee8b5d8b8fd76827142e107546f47769", "python": "3.13.3", "randomness": "none", "source_sha256": "63a6cac8554672460ceb2a42f045bb3cb6eecea7f48b12537b81848ed47821d0" }, "summary": { "passed": 15, "total": 15 } }