| { |
| "all_gates_pass": true, |
| "claim_results": [ |
| { |
| "all_formula_rows_match": true, |
| "all_small_enumerations_match": true, |
| "assessment": "verified", |
| "claim": 1, |
| "destructive_control": { |
| "b": 2, |
| "correct": 4200, |
| "d": 4, |
| "m": 3, |
| "without_nonempty_train_test_exclusion": 6804 |
| }, |
| "destructive_control_detected": true, |
| "literal_claim": "DiscoGen procedurally generates over 400 million distinct algorithm discovery tasks via a combinatorial formula N_tasks = 2*3*b*(2^m-1)*(3^d-2^(d+1)+1) depending on the number of modules m, datasets d, and backends b (Section 4.2, Equation 1).", |
| "main_domain_rows": [ |
| { |
| "b": 1, |
| "d": 11, |
| "domain": "Bayesian Optimisation", |
| "m": 6, |
| "matches": true, |
| "recomputed_tasks": 65413656, |
| "reported_tasks": 65413656 |
| }, |
| { |
| "b": 1, |
| "d": 7, |
| "domain": "Brain Speech Detection", |
| "m": 3, |
| "matches": true, |
| "recomputed_tasks": 81144, |
| "reported_tasks": 81144 |
| }, |
| { |
| "b": 1, |
| "d": 9, |
| "domain": "Computer Vision Classification", |
| "m": 4, |
| "matches": true, |
| "recomputed_tasks": 1679400, |
| "reported_tasks": 1679400 |
| }, |
| { |
| "b": 3, |
| "d": 3, |
| "domain": "Continual Learning", |
| "m": 5, |
| "matches": true, |
| "recomputed_tasks": 6696, |
| "reported_tasks": 6696 |
| }, |
| { |
| "b": 1, |
| "d": 4, |
| "domain": "Greenhouse Gas Prediction", |
| "m": 2, |
| "matches": true, |
| "recomputed_tasks": 900, |
| "reported_tasks": 900 |
| }, |
| { |
| "b": 2, |
| "d": 4, |
| "domain": "Language Modelling", |
| "m": 3, |
| "matches": true, |
| "recomputed_tasks": 4200, |
| "reported_tasks": 4200 |
| }, |
| { |
| "b": 1, |
| "d": 3, |
| "domain": "Model Unlearning", |
| "m": 1, |
| "matches": true, |
| "recomputed_tasks": 85176, |
| "reported_tasks": 85176 |
| }, |
| { |
| "b": 1, |
| "d": 4, |
| "domain": "Off-Policy RL", |
| "m": 7, |
| "matches": true, |
| "recomputed_tasks": 38100, |
| "reported_tasks": 38100 |
| }, |
| { |
| "b": 3, |
| "d": 13, |
| "domain": "On-Policy RL", |
| "m": 4, |
| "matches": true, |
| "recomputed_tasks": 426043800, |
| "reported_tasks": 426043800 |
| }, |
| { |
| "b": 1, |
| "d": 4, |
| "domain": "Unsupervised Environment Design", |
| "m": 3, |
| "matches": true, |
| "recomputed_tasks": 2100, |
| "reported_tasks": 2100 |
| } |
| ], |
| "main_total_from_rows": 493355172, |
| "official_model_choices": 13, |
| "over_400_million": true, |
| "reported_main_total": 493355172, |
| "small_exact_enumerations": [ |
| { |
| "b": 1, |
| "d": 2, |
| "direct": 12, |
| "formula": 12, |
| "m": 1, |
| "matches": true |
| }, |
| { |
| "b": 1, |
| "d": 3, |
| "direct": 216, |
| "formula": 216, |
| "m": 2, |
| "matches": true |
| }, |
| { |
| "b": 2, |
| "d": 4, |
| "direct": 4200, |
| "formula": 4200, |
| "m": 3, |
| "matches": true |
| }, |
| { |
| "b": 3, |
| "d": 4, |
| "direct": 13500, |
| "formula": 13500, |
| "m": 4, |
| "matches": true |
| } |
| ] |
| }, |
| { |
| "approximately_99_billion": true, |
| "assessment": "verified", |
| "claim": 2, |
| "destructive_control_detected": true, |
| "destructive_control_without_on_policy_marl": 1867332264, |
| "expanded_domain_rows": [ |
| { |
| "b": 1, |
| "d": 11, |
| "domain": "Bayesian Optimisation", |
| "m": 6, |
| "reported_tasks": 65413656 |
| }, |
| { |
| "b": 1, |
| "d": 7, |
| "domain": "Brain Speech Detection", |
| "m": 3, |
| "reported_tasks": 81144 |
| }, |
| { |
| "b": 1, |
| "d": 9, |
| "domain": "Computer Vision Classification", |
| "m": 4, |
| "reported_tasks": 1679400 |
| }, |
| { |
| "b": 3, |
| "d": 3, |
| "domain": "Continual Learning", |
| "m": 5, |
| "reported_tasks": 6696 |
| }, |
| { |
| "b": 1, |
| "d": 4, |
| "domain": "Greenhouse Gas Prediction", |
| "m": 2, |
| "reported_tasks": 900 |
| }, |
| { |
| "b": 2, |
| "d": 4, |
| "domain": "Language Modelling", |
| "m": 3, |
| "reported_tasks": 4200 |
| }, |
| { |
| "b": 1, |
| "d": 3, |
| "domain": "Model Unlearning", |
| "m": 1, |
| "reported_tasks": 85176 |
| }, |
| { |
| "b": 1, |
| "d": 5, |
| "domain": "Neural Cellular Automata", |
| "m": 5, |
| "reported_tasks": 33480 |
| }, |
| { |
| "b": 1, |
| "d": 4, |
| "domain": "Off-Policy RL", |
| "m": 7, |
| "reported_tasks": 38100 |
| }, |
| { |
| "b": 1, |
| "d": 10, |
| "domain": "Offline RL", |
| "m": 5, |
| "reported_tasks": 10602372 |
| }, |
| { |
| "b": 2, |
| "d": 17, |
| "domain": "On-Policy MARL", |
| "m": 6, |
| "reported_tasks": 97431783120 |
| }, |
| { |
| "b": 3, |
| "d": 13, |
| "domain": "On-Policy RL", |
| "m": 6, |
| "reported_tasks": 1789383960 |
| }, |
| { |
| "b": 3, |
| "d": 3, |
| "domain": "Trajectory Prediction", |
| "m": 4, |
| "reported_tasks": 1080 |
| }, |
| { |
| "b": 1, |
| "d": 4, |
| "domain": "Unsupervised Environment Design", |
| "m": 3, |
| "reported_tasks": 2100 |
| } |
| ], |
| "expanded_total_from_rows": 99299115384, |
| "literal_claim": "Including additional domains beyond the main evaluation set, DiscoGen's total task space reaches approximately 99 billion tasks (Appendix C).", |
| "official_repository_domains": [ |
| { |
| "backend_count": 1, |
| "backends": [ |
| "default" |
| ], |
| "dataset_count": 11, |
| "datasets": [ |
| "Ackley1d", |
| "Ackley2d", |
| "Branin2d", |
| "Bukin2d", |
| "Cosine8d", |
| "DropWave2d", |
| "EggHolder2d", |
| "Griewank5d", |
| "Hartmann6d", |
| "HolderTable2d", |
| "Levy6d" |
| ], |
| "domain": "BayesianOptimisation", |
| "model_count": 0, |
| "module_count": 6, |
| "modules": [ |
| "acq_fn", |
| "acq_optimizer", |
| "next_queries", |
| "sampler", |
| "surrogate", |
| "surrogate_optimizer" |
| ] |
| }, |
| { |
| "backend_count": 1, |
| "backends": [ |
| "default" |
| ], |
| "dataset_count": 7, |
| "datasets": [ |
| "LibriBrainSherlock1", |
| "LibriBrainSherlock2", |
| "LibriBrainSherlock3", |
| "LibriBrainSherlock4", |
| "LibriBrainSherlock5", |
| "LibriBrainSherlock6", |
| "LibriBrainSherlock7" |
| ], |
| "domain": "BrainSpeechDetection", |
| "model_count": 0, |
| "module_count": 3, |
| "modules": [ |
| "loss", |
| "networks", |
| "optim" |
| ] |
| }, |
| { |
| "backend_count": 1, |
| "backends": [ |
| "default" |
| ], |
| "dataset_count": 9, |
| "datasets": [ |
| "CIFAR10C", |
| "CIFAR10", |
| "CIFAR10LT", |
| "CIFAR100", |
| "FashionMNIST", |
| "MNIST", |
| "OxfordFlowers", |
| "StanfordCars", |
| "TinyImageNet" |
| ], |
| "domain": "ComputerVisionClassification", |
| "model_count": 0, |
| "module_count": 4, |
| "modules": [ |
| "loss", |
| "networks", |
| "optim", |
| "preprocess" |
| ] |
| }, |
| { |
| "backend_count": 3, |
| "backends": [ |
| "default", |
| "parameter_isolation", |
| "transformer" |
| ], |
| "dataset_count": 3, |
| "datasets": [ |
| "PermutedMNIST", |
| "SplitCIFAR100", |
| "TinyImageNetSplit" |
| ], |
| "domain": "ContinualLearning", |
| "model_count": 0, |
| "module_count": 5, |
| "modules": [ |
| "optim", |
| "regularizer", |
| "replay", |
| "sampler", |
| "scheduler" |
| ] |
| }, |
| { |
| "backend_count": 1, |
| "backends": [ |
| "default" |
| ], |
| "dataset_count": 4, |
| "datasets": [ |
| "CH4", |
| "CO2", |
| "N2O", |
| "SF6" |
| ], |
| "domain": "GreenhouseGasPrediction", |
| "model_count": 0, |
| "module_count": 2, |
| "modules": [ |
| "data_processing", |
| "model" |
| ] |
| }, |
| { |
| "backend_count": 2, |
| "backends": [ |
| "default", |
| "ssm" |
| ], |
| "dataset_count": 4, |
| "datasets": [ |
| "OPCFineWebCode", |
| "OPCFineWebMath", |
| "LMFineWeb", |
| "TinyStories" |
| ], |
| "domain": "LanguageModelling", |
| "model_count": 0, |
| "module_count": 3, |
| "modules": [ |
| "loss", |
| "networks", |
| "optim" |
| ] |
| }, |
| { |
| "backend_count": 1, |
| "backends": [ |
| "default" |
| ], |
| "dataset_count": 3, |
| "datasets": [ |
| "muse", |
| "tofu", |
| "wmdp_cyber" |
| ], |
| "domain": "ModelUnlearning", |
| "model_count": 13, |
| "module_count": 1, |
| "modules": [ |
| "loss" |
| ] |
| }, |
| { |
| "backend_count": 1, |
| "backends": [ |
| "default" |
| ], |
| "dataset_count": 5, |
| "datasets": [ |
| "GrowingLizard", |
| "GrowingButterfly", |
| "SelfClassifyingMNIST", |
| "MatrixOperations", |
| "MNISTInpainting" |
| ], |
| "domain": "NeuralCellularAutomata", |
| "model_count": 0, |
| "module_count": 5, |
| "modules": [ |
| "loss", |
| "optimiser", |
| "perceive", |
| "train", |
| "update" |
| ] |
| }, |
| { |
| "backend_count": 1, |
| "backends": [ |
| "default" |
| ], |
| "dataset_count": 4, |
| "datasets": [ |
| "MinAtar/Asterix", |
| "MinAtar/Breakout", |
| "MinAtar/Freeway", |
| "MinAtar/SpaceInvaders" |
| ], |
| "domain": "OffPolicyRL", |
| "model_count": 0, |
| "module_count": 7, |
| "modules": [ |
| "config", |
| "networks", |
| "optim", |
| "policy", |
| "q_update", |
| "rb", |
| "train" |
| ] |
| }, |
| { |
| "backend_count": 1, |
| "backends": [ |
| "default" |
| ], |
| "dataset_count": 10, |
| "datasets": [ |
| "OGBench/antmaze-giant-navigate", |
| "OGBench/antmaze-large-navigate", |
| "OGBench/antsoccer-arena-navigate", |
| "OGBench/cube-double-play", |
| "OGBench/cube-single-play", |
| "OGBench/humanoidmaze-large-navigate", |
| "OGBench/humanoidmaze-medium-navigate", |
| "OGBench/puzzle-3x3-play", |
| "OGBench/puzzle-4x4-play", |
| "OGBench/scene-play" |
| ], |
| "domain": "OfflineRL", |
| "model_count": 0, |
| "module_count": 5, |
| "modules": [ |
| "actor_loss", |
| "critic_loss", |
| "networks", |
| "optim", |
| "train" |
| ] |
| }, |
| { |
| "backend_count": 2, |
| "backends": [ |
| "default", |
| "recurrent" |
| ], |
| "dataset_count": 17, |
| "datasets": [ |
| "MABrax/Ant", |
| "MABrax/HalfCheetah", |
| "MABrax/Hopper", |
| "MABrax/Walker", |
| "MABrax/Humanoid", |
| "MPE/Spread", |
| "SMAX/2s3z", |
| "SMAX/3s_vs_5z", |
| "SMAX/3s5z", |
| "SMAX/3s5z_vs_3s6z", |
| "SMAX/5m_vs_6m", |
| "SMAX/6h_vs_8z", |
| "SMAX/10m_vs_11m", |
| "SMAX/27m_vs_30m", |
| "SMAX/smacv2_5_units", |
| "SMAX/smacv2_10_units", |
| "SMAX/smacv2_20_units" |
| ], |
| "domain": "OnPolicyMARL", |
| "model_count": 0, |
| "module_count": 6, |
| "modules": [ |
| "activation", |
| "loss", |
| "networks", |
| "optim", |
| "targets", |
| "train" |
| ] |
| }, |
| { |
| "backend_count": 3, |
| "backends": [ |
| "default", |
| "recurrent", |
| "transformer" |
| ], |
| "dataset_count": 13, |
| "datasets": [ |
| "MinAtar/Asterix", |
| "MinAtar/Breakout", |
| "MinAtar/Freeway", |
| "MinAtar/SpaceInvaders", |
| "Brax/Ant", |
| "Brax/HalfCheetah", |
| "Brax/Hopper", |
| "Brax/Humanoid", |
| "Brax/Pusher", |
| "Brax/Reacher", |
| "Brax/Walker2D", |
| "Craftax/Craftax", |
| "Craftax/Craftax-Classic" |
| ], |
| "domain": "OnPolicyRL", |
| "model_count": 0, |
| "module_count": 6, |
| "modules": [ |
| "activation", |
| "loss", |
| "networks", |
| "optim", |
| "targets", |
| "train" |
| ] |
| }, |
| { |
| "backend_count": 1, |
| "backends": [ |
| "default" |
| ], |
| "dataset_count": 3, |
| "datasets": [ |
| "Argoverse2", |
| "nuScenes", |
| "Waymo" |
| ], |
| "domain": "TrajectoryPrediction", |
| "model_count": 0, |
| "module_count": 4, |
| "modules": [ |
| "loss", |
| "networks", |
| "optim", |
| "train" |
| ] |
| }, |
| { |
| "backend_count": 1, |
| "backends": [ |
| "default" |
| ], |
| "dataset_count": 4, |
| "datasets": [ |
| "Kinetix/Small", |
| "Kinetix/Medium", |
| "Kinetix/Large", |
| "Minigrid" |
| ], |
| "domain": "UnsupervisedEnvironmentDesign", |
| "model_count": 0, |
| "module_count": 3, |
| "modules": [ |
| "sample_levels", |
| "train_step", |
| "variable_config" |
| ] |
| } |
| ], |
| "reported_expanded_total": 99299115384, |
| "repository_domain_count": 14 |
| }, |
| { |
| "all_registered_statistics_match": true, |
| "assessment": "verified", |
| "claim": 3, |
| "computed_median": 59622, |
| "destructive_control_detected": true, |
| "destructive_control_drop_one_domain_median": 81144.0, |
| "domain_count": 10, |
| "literal_claim": "Across the 10 domains used in the main evaluation, per-domain task counts range from 900 (Greenhouse Gas Prediction) to 426,043,800 (On-Policy RL), with a median of 59,622 tasks per domain (Table 1).", |
| "maximum": { |
| "b": 3, |
| "d": 13, |
| "domain": "On-Policy RL", |
| "m": 4, |
| "reported_tasks": 426043800 |
| }, |
| "minimum": { |
| "b": 1, |
| "d": 4, |
| "domain": "Greenhouse Gas Prediction", |
| "m": 2, |
| "reported_tasks": 900 |
| }, |
| "reported_median": 59622 |
| }, |
| { |
| "actual_generated_files": 82, |
| "assessment": "verified", |
| "audited_config_count": 74, |
| "claim": 4, |
| "configuration_failures": 0, |
| "configuration_rows": [ |
| { |
| "changed_modules": [ |
| "acq_fn" |
| ], |
| "failures": [], |
| "file": "BayesianOptimisation_acq_fn.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "acq_optimizer" |
| ], |
| "failures": [], |
| "file": "BayesianOptimisation_acq_optimizer.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "next_queries" |
| ], |
| "failures": [], |
| "file": "BayesianOptimisation_next_queries.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "sampler" |
| ], |
| "failures": [], |
| "file": "BayesianOptimisation_sampler.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "surrogate" |
| ], |
| "failures": [], |
| "file": "BayesianOptimisation_surrogate.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "surrogate_optimizer" |
| ], |
| "failures": [], |
| "file": "BayesianOptimisation_surrogate_optimizer.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "acq_fn", |
| "acq_optimizer", |
| "next_queries", |
| "sampler", |
| "surrogate_optimizer", |
| "surrogate" |
| ], |
| "failures": [], |
| "file": "BayesianOptimisation_all.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "loss" |
| ], |
| "failures": [], |
| "file": "BrainSpeechDetection_loss.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "networks" |
| ], |
| "failures": [], |
| "file": "BrainSpeechDetection_networks.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "optim" |
| ], |
| "failures": [], |
| "file": "BrainSpeechDetection_optim.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "optim", |
| "loss", |
| "networks" |
| ], |
| "failures": [], |
| "file": "BrainSpeechDetection_all.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "loss" |
| ], |
| "failures": [], |
| "file": "ComputerVisionClassification_loss.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "networks" |
| ], |
| "failures": [], |
| "file": "ComputerVisionClassification_networks.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "optim" |
| ], |
| "failures": [], |
| "file": "ComputerVisionClassification_optim.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "preprocess" |
| ], |
| "failures": [], |
| "file": "ComputerVisionClassification_preprocess.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "optim", |
| "loss", |
| "networks", |
| "preprocess" |
| ], |
| "failures": [], |
| "file": "ComputerVisionClassification_all.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "optim" |
| ], |
| "failures": [], |
| "file": "ContinualLearning_optim.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "regularizer" |
| ], |
| "failures": [], |
| "file": "ContinualLearning_regularizer.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "replay" |
| ], |
| "failures": [], |
| "file": "ContinualLearning_replay.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "sampler" |
| ], |
| "failures": [], |
| "file": "ContinualLearning_sampler.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "scheduler" |
| ], |
| "failures": [], |
| "file": "ContinualLearning_scheduler.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "regularizer", |
| "replay", |
| "sampler", |
| "scheduler", |
| "optim" |
| ], |
| "failures": [], |
| "file": "ContinualLearning_all.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "data_processing" |
| ], |
| "failures": [], |
| "file": "GreenhouseGasPrediction_data_processing.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "model" |
| ], |
| "failures": [], |
| "file": "GreenhouseGasPrediction_model.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "data_processing", |
| "model" |
| ], |
| "failures": [], |
| "file": "GreenhouseGasPrediction_all.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "loss" |
| ], |
| "failures": [], |
| "file": "LanguageModelling_loss.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "networks" |
| ], |
| "failures": [], |
| "file": "LanguageModelling_networks.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "optim" |
| ], |
| "failures": [], |
| "file": "LanguageModelling_optim.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "optim", |
| "loss", |
| "networks" |
| ], |
| "failures": [], |
| "file": "LanguageModelling_all.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "loss" |
| ], |
| "failures": [], |
| "file": "ModelUnlearning_loss.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "loss" |
| ], |
| "failures": [], |
| "file": "ModelUnlearning_all.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "loss" |
| ], |
| "failures": [], |
| "file": "NeuralCellularAutomata_loss.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "optimiser" |
| ], |
| "failures": [], |
| "file": "NeuralCellularAutomata_optimiser.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "perceive" |
| ], |
| "failures": [], |
| "file": "NeuralCellularAutomata_perceive.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "train" |
| ], |
| "failures": [], |
| "file": "NeuralCellularAutomata_train.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "update" |
| ], |
| "failures": [], |
| "file": "NeuralCellularAutomata_update.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "perceive", |
| "update", |
| "loss", |
| "train", |
| "optimiser" |
| ], |
| "failures": [], |
| "file": "NeuralCellularAutomata_all.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "config" |
| ], |
| "failures": [], |
| "file": "OffPolicyRL_config.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "networks" |
| ], |
| "failures": [], |
| "file": "OffPolicyRL_networks.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "optim" |
| ], |
| "failures": [], |
| "file": "OffPolicyRL_optim.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "policy" |
| ], |
| "failures": [], |
| "file": "OffPolicyRL_policy.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "q_update" |
| ], |
| "failures": [], |
| "file": "OffPolicyRL_q_update.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "rb" |
| ], |
| "failures": [], |
| "file": "OffPolicyRL_rb.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "train" |
| ], |
| "failures": [], |
| "file": "OffPolicyRL_train.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "config", |
| "networks", |
| "optim", |
| "policy", |
| "q_update", |
| "rb", |
| "train" |
| ], |
| "failures": [], |
| "file": "OffPolicyRL_all.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "actor_loss" |
| ], |
| "failures": [], |
| "file": "OfflineRL_actor_loss.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "critic_loss" |
| ], |
| "failures": [], |
| "file": "OfflineRL_critic_loss.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "networks" |
| ], |
| "failures": [], |
| "file": "OfflineRL_networks.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "optim" |
| ], |
| "failures": [], |
| "file": "OfflineRL_optim.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "train" |
| ], |
| "failures": [], |
| "file": "OfflineRL_train.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "optim", |
| "actor_loss", |
| "critic_loss", |
| "networks", |
| "train" |
| ], |
| "failures": [], |
| "file": "OfflineRL_all.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "activation" |
| ], |
| "failures": [], |
| "file": "OnPolicyMARL_activation.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "loss" |
| ], |
| "failures": [], |
| "file": "OnPolicyMARL_loss.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "networks" |
| ], |
| "failures": [], |
| "file": "OnPolicyMARL_networks.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "optim" |
| ], |
| "failures": [], |
| "file": "OnPolicyMARL_optim.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "targets" |
| ], |
| "failures": [], |
| "file": "OnPolicyMARL_targets.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "train" |
| ], |
| "failures": [], |
| "file": "OnPolicyMARL_train.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "optim", |
| "loss", |
| "networks", |
| "train", |
| "activation", |
| "targets" |
| ], |
| "failures": [], |
| "file": "OnPolicyMARL_all.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "activation" |
| ], |
| "failures": [], |
| "file": "OnPolicyRL_activation.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "loss" |
| ], |
| "failures": [], |
| "file": "OnPolicyRL_loss.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "networks" |
| ], |
| "failures": [], |
| "file": "OnPolicyRL_networks.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "optim" |
| ], |
| "failures": [], |
| "file": "OnPolicyRL_optim.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "targets" |
| ], |
| "failures": [], |
| "file": "OnPolicyRL_targets.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "train" |
| ], |
| "failures": [], |
| "file": "OnPolicyRL_train.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "optim", |
| "loss", |
| "networks", |
| "train", |
| "activation", |
| "targets" |
| ], |
| "failures": [], |
| "file": "OnPolicyRL_all.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "loss" |
| ], |
| "failures": [], |
| "file": "TrajectoryPrediction_loss.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "networks" |
| ], |
| "failures": [], |
| "file": "TrajectoryPrediction_networks.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "optim" |
| ], |
| "failures": [], |
| "file": "TrajectoryPrediction_optim.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "train" |
| ], |
| "failures": [], |
| "file": "TrajectoryPrediction_train.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "optim", |
| "loss", |
| "networks", |
| "train" |
| ], |
| "failures": [], |
| "file": "TrajectoryPrediction_all.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "sample_levels" |
| ], |
| "failures": [], |
| "file": "UnsupervisedEnvironmentDesign_sample_levels.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "train_step" |
| ], |
| "failures": [], |
| "file": "UnsupervisedEnvironmentDesign_train_step.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "variable_config" |
| ], |
| "failures": [], |
| "file": "UnsupervisedEnvironmentDesign_variable_config.yaml" |
| }, |
| { |
| "changed_modules": [ |
| "sample_levels", |
| "train_step", |
| "variable_config" |
| ], |
| "failures": [], |
| "file": "UnsupervisedEnvironmentDesign_all.yaml" |
| } |
| ], |
| "destructive_control_active_modules": [ |
| "change_data_processing", |
| "change_model" |
| ], |
| "destructive_control_detected": true, |
| "expected_m_plus_one_configs": 74, |
| "literal_claim": "DiscoBench provides a fixed evaluation subset built from DiscoGen, comprising, for each domain, m single-module tasks (DiscoBench Single) plus one comprehensive all-modules-active task (DiscoBench All) (Section 4.4).", |
| "official_builder_execution": { |
| "canonicalized_description_files": 4, |
| "executed_configs": [ |
| "GreenhouseGasPrediction_all", |
| "GreenhouseGasPrediction_data_processing", |
| "OnPolicyRL_loss", |
| "OnPolicyRL_all" |
| ], |
| "file_count": 82, |
| "files": [ |
| { |
| "bytes": 300, |
| "path": "GreenhouseGasPrediction_all/CH4/data_processing.py", |
| "sha256": "e3a4d5be0be7e7d5129181846167a2fd76fc1e9b2b99049f8406ecdc055bac4e" |
| }, |
| { |
| "bytes": 1372, |
| "path": "GreenhouseGasPrediction_all/CH4/main.py", |
| "sha256": "094787e39823668d2cbd472af1f00eee3f49ca26fd1c2d5f179aee9aafbe7adf" |
| }, |
| { |
| "bytes": 597, |
| "path": "GreenhouseGasPrediction_all/CH4/model.py", |
| "sha256": "dd94a5687d62d89a8d510ce2c3b7f059de2ff0bc4b7fd7a719cef6ceb5126549" |
| }, |
| { |
| "bytes": 300, |
| "path": "GreenhouseGasPrediction_all/SF6/data_processing.py", |
| "sha256": "e3a4d5be0be7e7d5129181846167a2fd76fc1e9b2b99049f8406ecdc055bac4e" |
| }, |
| { |
| "bytes": 1372, |
| "path": "GreenhouseGasPrediction_all/SF6/main.py", |
| "sha256": "094787e39823668d2cbd472af1f00eee3f49ca26fd1c2d5f179aee9aafbe7adf" |
| }, |
| { |
| "bytes": 597, |
| "path": "GreenhouseGasPrediction_all/SF6/model.py", |
| "sha256": "dd94a5687d62d89a8d510ce2c3b7f059de2ff0bc4b7fd7a719cef6ceb5126549" |
| }, |
| { |
| "bytes": 8817, |
| "path": "GreenhouseGasPrediction_all/description.md", |
| "sha256": "1b052d3d51ed063eb562bf5a22a3918c358efd215d807e73cf6a20993c0258e1" |
| }, |
| { |
| "bytes": 300, |
| "path": "GreenhouseGasPrediction_all/discovered/data_processing.py", |
| "sha256": "e3a4d5be0be7e7d5129181846167a2fd76fc1e9b2b99049f8406ecdc055bac4e" |
| }, |
| { |
| "bytes": 597, |
| "path": "GreenhouseGasPrediction_all/discovered/model.py", |
| "sha256": "dd94a5687d62d89a8d510ce2c3b7f059de2ff0bc4b7fd7a719cef6ceb5126549" |
| }, |
| { |
| "bytes": 53, |
| "path": "GreenhouseGasPrediction_all/install.sh", |
| "sha256": "6964f075ffc565dd9847b7203a28d7368b0825333435a756b48ec7da08757af7" |
| }, |
| { |
| "bytes": 155, |
| "path": "GreenhouseGasPrediction_all/requirements.txt", |
| "sha256": "ce1ca66a4ff2d670272eb600e8cd5fd3c530c637c86d7477110015d2d43ca8fa" |
| }, |
| { |
| "bytes": 1725, |
| "path": "GreenhouseGasPrediction_all/run_main.py", |
| "sha256": "88db3fe20483a035b63bf5cb9c415493ae52cfa09e944fcb5ea461aa66407d90" |
| }, |
| { |
| "bytes": 300, |
| "path": "GreenhouseGasPrediction_data_processing/CH4/data_processing.py", |
| "sha256": "e3a4d5be0be7e7d5129181846167a2fd76fc1e9b2b99049f8406ecdc055bac4e" |
| }, |
| { |
| "bytes": 1372, |
| "path": "GreenhouseGasPrediction_data_processing/CH4/main.py", |
| "sha256": "094787e39823668d2cbd472af1f00eee3f49ca26fd1c2d5f179aee9aafbe7adf" |
| }, |
| { |
| "bytes": 458, |
| "path": "GreenhouseGasPrediction_data_processing/CH4/model.py", |
| "sha256": "57e94b384ad5ccf7a6a803568f8b7a7350db7786b0308375370e7195ed93e172" |
| }, |
| { |
| "bytes": 300, |
| "path": "GreenhouseGasPrediction_data_processing/SF6/data_processing.py", |
| "sha256": "e3a4d5be0be7e7d5129181846167a2fd76fc1e9b2b99049f8406ecdc055bac4e" |
| }, |
| { |
| "bytes": 1372, |
| "path": "GreenhouseGasPrediction_data_processing/SF6/main.py", |
| "sha256": "094787e39823668d2cbd472af1f00eee3f49ca26fd1c2d5f179aee9aafbe7adf" |
| }, |
| { |
| "bytes": 458, |
| "path": "GreenhouseGasPrediction_data_processing/SF6/model.py", |
| "sha256": "57e94b384ad5ccf7a6a803568f8b7a7350db7786b0308375370e7195ed93e172" |
| }, |
| { |
| "bytes": 8691, |
| "path": "GreenhouseGasPrediction_data_processing/description.md", |
| "sha256": "c65914e7460ad4529dcd70b4788a8ced437909123cc3ea19599a39615db0e2dd" |
| }, |
| { |
| "bytes": 300, |
| "path": "GreenhouseGasPrediction_data_processing/discovered/data_processing.py", |
| "sha256": "e3a4d5be0be7e7d5129181846167a2fd76fc1e9b2b99049f8406ecdc055bac4e" |
| }, |
| { |
| "bytes": 53, |
| "path": "GreenhouseGasPrediction_data_processing/install.sh", |
| "sha256": "6964f075ffc565dd9847b7203a28d7368b0825333435a756b48ec7da08757af7" |
| }, |
| { |
| "bytes": 155, |
| "path": "GreenhouseGasPrediction_data_processing/requirements.txt", |
| "sha256": "ce1ca66a4ff2d670272eb600e8cd5fd3c530c637c86d7477110015d2d43ca8fa" |
| }, |
| { |
| "bytes": 1725, |
| "path": "GreenhouseGasPrediction_data_processing/run_main.py", |
| "sha256": "88db3fe20483a035b63bf5cb9c415493ae52cfa09e944fcb5ea461aa66407d90" |
| }, |
| { |
| "bytes": 0, |
| "path": "OnPolicyRL_all/MinAtar/Breakout/__init__.py", |
| "sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" |
| }, |
| { |
| "bytes": 277, |
| "path": "OnPolicyRL_all/MinAtar/Breakout/activation.py", |
| "sha256": "1830544b21f2ed84ac4e5ac94ad33047ebfa178394a1bf7e3b3567ad2f497e76" |
| }, |
| { |
| "bytes": 346, |
| "path": "OnPolicyRL_all/MinAtar/Breakout/config.py", |
| "sha256": "8f6949d7df8358ff71e3006ac89118c445aa74c72c099e3e58058689c9b5050c" |
| }, |
| { |
| "bytes": 850, |
| "path": "OnPolicyRL_all/MinAtar/Breakout/loss.py", |
| "sha256": "7866cfc22268e39cb4e0564893a7afc7284088041278d2f04aab38f7c76f595c" |
| }, |
| { |
| "bytes": 8942, |
| "path": "OnPolicyRL_all/MinAtar/Breakout/main.py", |
| "sha256": "ad901568e2f90a79ea2c570e2482e3f89189a48e9e62663b2f46965a6744c488" |
| }, |
| { |
| "bytes": 245, |
| "path": "OnPolicyRL_all/MinAtar/Breakout/make_env.py", |
| "sha256": "c24169e8c5f04fa5f3af519e9f98c466f8c7a6233edc7b4e9085a25562c94602" |
| }, |
| { |
| "bytes": 818, |
| "path": "OnPolicyRL_all/MinAtar/Breakout/networks.py", |
| "sha256": "8710a87b23befb124a88ca9676133e0dfd336835f61880afe7ab0a3bc911561d" |
| }, |
| { |
| "bytes": 1670, |
| "path": "OnPolicyRL_all/MinAtar/Breakout/optim.py", |
| "sha256": "9344db01c0d8465c28ed142ca18a7d0acf4e4f8ee5d59f440a16ec6cca076a9b" |
| }, |
| { |
| "bytes": 261, |
| "path": "OnPolicyRL_all/MinAtar/Breakout/targets.py", |
| "sha256": "d5ee1bddf2ccfe8c4613dfdf5623743d788f269cd546006aa7a18e433d194f0d" |
| }, |
| { |
| "bytes": 3648, |
| "path": "OnPolicyRL_all/MinAtar/Breakout/train.py", |
| "sha256": "4c9515a812deea8226e22652defc9c9e326d1954216dcc5a859673b80c97be67" |
| }, |
| { |
| "bytes": 10698, |
| "path": "OnPolicyRL_all/MinAtar/Breakout/wrappers.py", |
| "sha256": "e611bbacd53433415d12fb709da822a755df3770fbf7f51c6d789a485415ac68" |
| }, |
| { |
| "bytes": 0, |
| "path": "OnPolicyRL_all/MinAtar/Freeway/__init__.py", |
| "sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" |
| }, |
| { |
| "bytes": 277, |
| "path": "OnPolicyRL_all/MinAtar/Freeway/activation.py", |
| "sha256": "1830544b21f2ed84ac4e5ac94ad33047ebfa178394a1bf7e3b3567ad2f497e76" |
| }, |
| { |
| "bytes": 346, |
| "path": "OnPolicyRL_all/MinAtar/Freeway/config.py", |
| "sha256": "8f6949d7df8358ff71e3006ac89118c445aa74c72c099e3e58058689c9b5050c" |
| }, |
| { |
| "bytes": 850, |
| "path": "OnPolicyRL_all/MinAtar/Freeway/loss.py", |
| "sha256": "7866cfc22268e39cb4e0564893a7afc7284088041278d2f04aab38f7c76f595c" |
| }, |
| { |
| "bytes": 8942, |
| "path": "OnPolicyRL_all/MinAtar/Freeway/main.py", |
| "sha256": "ad901568e2f90a79ea2c570e2482e3f89189a48e9e62663b2f46965a6744c488" |
| }, |
| { |
| "bytes": 244, |
| "path": "OnPolicyRL_all/MinAtar/Freeway/make_env.py", |
| "sha256": "c18fedfd39747012af573a9f81055144e3bfdc90940a2666bbdbaa53d85bb583" |
| }, |
| { |
| "bytes": 818, |
| "path": "OnPolicyRL_all/MinAtar/Freeway/networks.py", |
| "sha256": "8710a87b23befb124a88ca9676133e0dfd336835f61880afe7ab0a3bc911561d" |
| }, |
| { |
| "bytes": 1670, |
| "path": "OnPolicyRL_all/MinAtar/Freeway/optim.py", |
| "sha256": "9344db01c0d8465c28ed142ca18a7d0acf4e4f8ee5d59f440a16ec6cca076a9b" |
| }, |
| { |
| "bytes": 261, |
| "path": "OnPolicyRL_all/MinAtar/Freeway/targets.py", |
| "sha256": "d5ee1bddf2ccfe8c4613dfdf5623743d788f269cd546006aa7a18e433d194f0d" |
| }, |
| { |
| "bytes": 3648, |
| "path": "OnPolicyRL_all/MinAtar/Freeway/train.py", |
| "sha256": "4c9515a812deea8226e22652defc9c9e326d1954216dcc5a859673b80c97be67" |
| }, |
| { |
| "bytes": 10698, |
| "path": "OnPolicyRL_all/MinAtar/Freeway/wrappers.py", |
| "sha256": "e611bbacd53433415d12fb709da822a755df3770fbf7f51c6d789a485415ac68" |
| }, |
| { |
| "bytes": 10734, |
| "path": "OnPolicyRL_all/description.md", |
| "sha256": "db1ed45e365da592846c9a1acb9f1460a25e39ccc3bb4c4c425ec0b9f71eb11c" |
| }, |
| { |
| "bytes": 277, |
| "path": "OnPolicyRL_all/discovered/activation.py", |
| "sha256": "1830544b21f2ed84ac4e5ac94ad33047ebfa178394a1bf7e3b3567ad2f497e76" |
| }, |
| { |
| "bytes": 850, |
| "path": "OnPolicyRL_all/discovered/loss.py", |
| "sha256": "7866cfc22268e39cb4e0564893a7afc7284088041278d2f04aab38f7c76f595c" |
| }, |
| { |
| "bytes": 818, |
| "path": "OnPolicyRL_all/discovered/networks.py", |
| "sha256": "8710a87b23befb124a88ca9676133e0dfd336835f61880afe7ab0a3bc911561d" |
| }, |
| { |
| "bytes": 1670, |
| "path": "OnPolicyRL_all/discovered/optim.py", |
| "sha256": "9344db01c0d8465c28ed142ca18a7d0acf4e4f8ee5d59f440a16ec6cca076a9b" |
| }, |
| { |
| "bytes": 261, |
| "path": "OnPolicyRL_all/discovered/targets.py", |
| "sha256": "d5ee1bddf2ccfe8c4613dfdf5623743d788f269cd546006aa7a18e433d194f0d" |
| }, |
| { |
| "bytes": 3648, |
| "path": "OnPolicyRL_all/discovered/train.py", |
| "sha256": "4c9515a812deea8226e22652defc9c9e326d1954216dcc5a859673b80c97be67" |
| }, |
| { |
| "bytes": 53, |
| "path": "OnPolicyRL_all/install.sh", |
| "sha256": "6964f075ffc565dd9847b7203a28d7368b0825333435a756b48ec7da08757af7" |
| }, |
| { |
| "bytes": 272, |
| "path": "OnPolicyRL_all/requirements.txt", |
| "sha256": "f052fa4717c1411280ec50ba717865be0fe1390c60e68da176b4986fe8f551aa" |
| }, |
| { |
| "bytes": 1725, |
| "path": "OnPolicyRL_all/run_main.py", |
| "sha256": "88db3fe20483a035b63bf5cb9c415493ae52cfa09e944fcb5ea461aa66407d90" |
| }, |
| { |
| "bytes": 0, |
| "path": "OnPolicyRL_loss/MinAtar/Breakout/__init__.py", |
| "sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" |
| }, |
| { |
| "bytes": 182, |
| "path": "OnPolicyRL_loss/MinAtar/Breakout/activation.py", |
| "sha256": "429973432f0977f59dc029c529c9b8989618cc7699c72a1d9fdeea9f9f827e1c" |
| }, |
| { |
| "bytes": 346, |
| "path": "OnPolicyRL_loss/MinAtar/Breakout/config.py", |
| "sha256": "8f6949d7df8358ff71e3006ac89118c445aa74c72c099e3e58058689c9b5050c" |
| }, |
| { |
| "bytes": 850, |
| "path": "OnPolicyRL_loss/MinAtar/Breakout/loss.py", |
| "sha256": "7866cfc22268e39cb4e0564893a7afc7284088041278d2f04aab38f7c76f595c" |
| }, |
| { |
| "bytes": 8942, |
| "path": "OnPolicyRL_loss/MinAtar/Breakout/main.py", |
| "sha256": "ad901568e2f90a79ea2c570e2482e3f89189a48e9e62663b2f46965a6744c488" |
| }, |
| { |
| "bytes": 245, |
| "path": "OnPolicyRL_loss/MinAtar/Breakout/make_env.py", |
| "sha256": "c24169e8c5f04fa5f3af519e9f98c466f8c7a6233edc7b4e9085a25562c94602" |
| }, |
| { |
| "bytes": 1680, |
| "path": "OnPolicyRL_loss/MinAtar/Breakout/networks.py", |
| "sha256": "456176a5ce041e85b85cf57b8829265af1431cc48cbfd017dac4dfd0a8689602" |
| }, |
| { |
| "bytes": 156, |
| "path": "OnPolicyRL_loss/MinAtar/Breakout/optim.py", |
| "sha256": "e88de32252fed900d835598dc968e9d67e134dd5a0d7a3e29b62533e3e5fdaaf" |
| }, |
| { |
| "bytes": 767, |
| "path": "OnPolicyRL_loss/MinAtar/Breakout/targets.py", |
| "sha256": "a36111a2aebe9c98cf7d3dc7e121bf8b2e2b846d7bc962ef2ba032712bc644d9" |
| }, |
| { |
| "bytes": 8102, |
| "path": "OnPolicyRL_loss/MinAtar/Breakout/train.py", |
| "sha256": "0013f408cec5cefb3b1426c22e6b57405ca36ad8b72e61ce24a4a27d127f67fe" |
| }, |
| { |
| "bytes": 10698, |
| "path": "OnPolicyRL_loss/MinAtar/Breakout/wrappers.py", |
| "sha256": "e611bbacd53433415d12fb709da822a755df3770fbf7f51c6d789a485415ac68" |
| }, |
| { |
| "bytes": 0, |
| "path": "OnPolicyRL_loss/MinAtar/Freeway/__init__.py", |
| "sha256": "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" |
| }, |
| { |
| "bytes": 182, |
| "path": "OnPolicyRL_loss/MinAtar/Freeway/activation.py", |
| "sha256": "429973432f0977f59dc029c529c9b8989618cc7699c72a1d9fdeea9f9f827e1c" |
| }, |
| { |
| "bytes": 346, |
| "path": "OnPolicyRL_loss/MinAtar/Freeway/config.py", |
| "sha256": "8f6949d7df8358ff71e3006ac89118c445aa74c72c099e3e58058689c9b5050c" |
| }, |
| { |
| "bytes": 850, |
| "path": "OnPolicyRL_loss/MinAtar/Freeway/loss.py", |
| "sha256": "7866cfc22268e39cb4e0564893a7afc7284088041278d2f04aab38f7c76f595c" |
| }, |
| { |
| "bytes": 8942, |
| "path": "OnPolicyRL_loss/MinAtar/Freeway/main.py", |
| "sha256": "ad901568e2f90a79ea2c570e2482e3f89189a48e9e62663b2f46965a6744c488" |
| }, |
| { |
| "bytes": 244, |
| "path": "OnPolicyRL_loss/MinAtar/Freeway/make_env.py", |
| "sha256": "c18fedfd39747012af573a9f81055144e3bfdc90940a2666bbdbaa53d85bb583" |
| }, |
| { |
| "bytes": 1680, |
| "path": "OnPolicyRL_loss/MinAtar/Freeway/networks.py", |
| "sha256": "456176a5ce041e85b85cf57b8829265af1431cc48cbfd017dac4dfd0a8689602" |
| }, |
| { |
| "bytes": 156, |
| "path": "OnPolicyRL_loss/MinAtar/Freeway/optim.py", |
| "sha256": "e88de32252fed900d835598dc968e9d67e134dd5a0d7a3e29b62533e3e5fdaaf" |
| }, |
| { |
| "bytes": 767, |
| "path": "OnPolicyRL_loss/MinAtar/Freeway/targets.py", |
| "sha256": "a36111a2aebe9c98cf7d3dc7e121bf8b2e2b846d7bc962ef2ba032712bc644d9" |
| }, |
| { |
| "bytes": 8102, |
| "path": "OnPolicyRL_loss/MinAtar/Freeway/train.py", |
| "sha256": "0013f408cec5cefb3b1426c22e6b57405ca36ad8b72e61ce24a4a27d127f67fe" |
| }, |
| { |
| "bytes": 10698, |
| "path": "OnPolicyRL_loss/MinAtar/Freeway/wrappers.py", |
| "sha256": "e611bbacd53433415d12fb709da822a755df3770fbf7f51c6d789a485415ac68" |
| }, |
| { |
| "bytes": 8306, |
| "path": "OnPolicyRL_loss/description.md", |
| "sha256": "28f98c846f8542e35ce544aa4c4d5d5387805cd32c70739971e24841fc4914b4" |
| }, |
| { |
| "bytes": 850, |
| "path": "OnPolicyRL_loss/discovered/loss.py", |
| "sha256": "7866cfc22268e39cb4e0564893a7afc7284088041278d2f04aab38f7c76f595c" |
| }, |
| { |
| "bytes": 53, |
| "path": "OnPolicyRL_loss/install.sh", |
| "sha256": "6964f075ffc565dd9847b7203a28d7368b0825333435a756b48ec7da08757af7" |
| }, |
| { |
| "bytes": 272, |
| "path": "OnPolicyRL_loss/requirements.txt", |
| "sha256": "f052fa4717c1411280ec50ba717865be0fe1390c60e68da176b4986fe8f551aa" |
| }, |
| { |
| "bytes": 1725, |
| "path": "OnPolicyRL_loss/run_main.py", |
| "sha256": "88db3fe20483a035b63bf5cb9c415493ae52cfa09e944fcb5ea461aa66407d90" |
| } |
| ], |
| "tree_sha256": "039f3fbad73fac4c9459b3eed28a53f60c98da12bfc07b4478237805ab32af16" |
| }, |
| "official_domain_count": 14 |
| }, |
| { |
| "all_three_models_nonincreasing": true, |
| "assessment": "verified", |
| "claim": 5, |
| "combination_count": 15, |
| "complete_four_module_combinations": [ |
| [ |
| "loss" |
| ], |
| [ |
| "networks" |
| ], |
| [ |
| "optim" |
| ], |
| [ |
| "train" |
| ], |
| [ |
| "loss", |
| "networks" |
| ], |
| [ |
| "loss", |
| "optim" |
| ], |
| [ |
| "loss", |
| "train" |
| ], |
| [ |
| "networks", |
| "optim" |
| ], |
| [ |
| "networks", |
| "train" |
| ], |
| [ |
| "optim", |
| "train" |
| ], |
| [ |
| "loss", |
| "networks", |
| "optim" |
| ], |
| [ |
| "loss", |
| "networks", |
| "train" |
| ], |
| [ |
| "loss", |
| "optim", |
| "train" |
| ], |
| [ |
| "networks", |
| "optim", |
| "train" |
| ], |
| [ |
| "loss", |
| "networks", |
| "optim", |
| "train" |
| ] |
| ], |
| "destructive_control_detected": true, |
| "destructive_control_reversed_success_rates": { |
| "Deepseek-v3.2": [ |
| 0.0, |
| 8.3, |
| 47.2, |
| 75.0 |
| ], |
| "Devstral2": [ |
| 0.0, |
| 0.0, |
| 27.8, |
| 29.2 |
| ], |
| "GPT-OSS-120b": [ |
| 0.0, |
| 8.3, |
| 11.1, |
| 50.0 |
| ] |
| }, |
| "environments_with_higher_two_module_ceiling": 3, |
| "literal_claim": "As the number of editable modules increases in DiscoBench tasks, agent success rates consistently decline while the achievable performance ceiling rises (Appendix G).", |
| "maximum_return_by_module_count": { |
| "1": [ |
| 99.97, |
| 68.07, |
| 39.6, |
| 189.75 |
| ], |
| "2": [ |
| 106.12, |
| 66.16, |
| 69.42, |
| 191.62 |
| ], |
| "3": [ |
| 88.58, |
| 65.2, |
| 64.85, |
| 186.75 |
| ], |
| "4": [ |
| null, |
| null, |
| null, |
| null |
| ] |
| }, |
| "mean_ceiling_increase": 8.9825, |
| "monotonicity_by_model": { |
| "Deepseek-v3.2": true, |
| "Devstral2": true, |
| "GPT-OSS-120b": true |
| }, |
| "single_module_mean_environment_ceiling": 99.3475, |
| "source_configuration_rows": [ |
| { |
| "configuration": "Optimiser", |
| "module_count": 1, |
| "returns": [ |
| 74.97, |
| 62.86, |
| 18.11, |
| 181.25 |
| ], |
| "success_rate": 83.33 |
| }, |
| { |
| "configuration": "Loss", |
| "module_count": 1, |
| "returns": [ |
| 83.91, |
| 62.58, |
| 39.6, |
| 179.5 |
| ], |
| "success_rate": 77.78 |
| }, |
| { |
| "configuration": "Network", |
| "module_count": 1, |
| "returns": [ |
| 99.97, |
| 68.07, |
| 14.2, |
| 189.75 |
| ], |
| "success_rate": 33.33 |
| }, |
| { |
| "configuration": "Train", |
| "module_count": 1, |
| "returns": [ |
| 8.41, |
| 8.41, |
| 3.73, |
| 177.12 |
| ], |
| "success_rate": 11.11 |
| }, |
| { |
| "configuration": "Loss + Optimiser", |
| "module_count": 2, |
| "returns": [ |
| 84.47, |
| 62.89, |
| 20.5, |
| 181.38 |
| ], |
| "success_rate": 61.11 |
| }, |
| { |
| "configuration": "Network + Optimiser", |
| "module_count": 2, |
| "returns": [ |
| 91.44, |
| 65.25, |
| 19.77, |
| 191.62 |
| ], |
| "success_rate": 44.44 |
| }, |
| { |
| "configuration": "Loss + Network", |
| "module_count": 2, |
| "returns": [ |
| 106.12, |
| 66.16, |
| 69.42, |
| 184.0 |
| ], |
| "success_rate": 38.89 |
| }, |
| { |
| "configuration": "Loss + Train", |
| "module_count": 2, |
| "returns": [ |
| 8.56, |
| 61.91, |
| 3.91, |
| 169.38 |
| ], |
| "success_rate": 22.22 |
| }, |
| { |
| "configuration": "Optimiser + Train", |
| "module_count": 2, |
| "returns": [ |
| 34.61, |
| 29.45, |
| null, |
| null |
| ], |
| "success_rate": 5.56 |
| }, |
| { |
| "configuration": "Network + Train", |
| "module_count": 2, |
| "returns": [ |
| null, |
| null, |
| null, |
| null |
| ], |
| "success_rate": 0.0 |
| }, |
| { |
| "configuration": "Loss + Network + Optimiser", |
| "module_count": 3, |
| "returns": [ |
| 88.58, |
| 65.2, |
| 64.85, |
| 186.75 |
| ], |
| "success_rate": 11.11 |
| }, |
| { |
| "configuration": "Loss + Optimiser + Train", |
| "module_count": 3, |
| "returns": [ |
| 0.3, |
| 2.95, |
| null, |
| null |
| ], |
| "success_rate": 11.11 |
| }, |
| { |
| "configuration": "Loss + Network + Train", |
| "module_count": 3, |
| "returns": [ |
| null, |
| null, |
| null, |
| null |
| ], |
| "success_rate": 0.0 |
| }, |
| { |
| "configuration": "Network + Optimiser + Train", |
| "module_count": 3, |
| "returns": [ |
| null, |
| null, |
| null, |
| null |
| ], |
| "success_rate": 0.0 |
| }, |
| { |
| "configuration": "Loss + Network + Optimiser + Train", |
| "module_count": 4, |
| "returns": [ |
| null, |
| null, |
| null, |
| null |
| ], |
| "success_rate": 0.0 |
| } |
| ], |
| "success_rate_by_editable_module_count": { |
| "Deepseek-v3.2": [ |
| 75.0, |
| 47.2, |
| 8.3, |
| 0.0 |
| ], |
| "Devstral2": [ |
| 29.2, |
| 27.8, |
| 0.0, |
| 0.0 |
| ], |
| "GPT-OSS-120b": [ |
| 50.0, |
| 11.1, |
| 8.3, |
| 0.0 |
| ] |
| }, |
| "two_module_mean_environment_ceiling": 108.33 |
| } |
| ], |
| "claims": [ |
| { |
| "claim": 1, |
| "literal_claim": "DiscoGen procedurally generates over 400 million distinct algorithm discovery tasks via a combinatorial formula N_tasks = 2*3*b*(2^m-1)*(3^d-2^(d+1)+1) depending on the number of modules m, datasets d, and backends b (Section 4.2, Equation 1)." |
| }, |
| { |
| "claim": 2, |
| "literal_claim": "Including additional domains beyond the main evaluation set, DiscoGen's total task space reaches approximately 99 billion tasks (Appendix C)." |
| }, |
| { |
| "claim": 3, |
| "literal_claim": "Across the 10 domains used in the main evaluation, per-domain task counts range from 900 (Greenhouse Gas Prediction) to 426,043,800 (On-Policy RL), with a median of 59,622 tasks per domain (Table 1)." |
| }, |
| { |
| "claim": 4, |
| "literal_claim": "DiscoBench provides a fixed evaluation subset built from DiscoGen, comprising, for each domain, m single-module tasks (DiscoBench Single) plus one comprehensive all-modules-active task (DiscoBench All) (Section 4.4)." |
| }, |
| { |
| "claim": 5, |
| "literal_claim": "As the number of editable modules increases in DiscoBench tasks, agent success rates consistently decline while the achievable performance ceiling rises (Appendix G)." |
| } |
| ], |
| "gates": [ |
| { |
| "name": "claim1_all_formula_rows", |
| "passed": true |
| }, |
| { |
| "name": "claim1_small_exhaustive_enumerations", |
| "passed": true |
| }, |
| { |
| "name": "claim1_over_400m", |
| "passed": true |
| }, |
| { |
| "name": "claim1_control", |
| "passed": true |
| }, |
| { |
| "name": "claim2_99b_total", |
| "passed": true |
| }, |
| { |
| "name": "claim2_14_repository_domains", |
| "passed": true |
| }, |
| { |
| "name": "claim2_control", |
| "passed": true |
| }, |
| { |
| "name": "claim3_exact_statistics", |
| "passed": true |
| }, |
| { |
| "name": "claim3_control", |
| "passed": true |
| }, |
| { |
| "name": "claim4_all_m_plus_one_configs", |
| "passed": true |
| }, |
| { |
| "name": "claim4_official_builder_execution", |
| "passed": true |
| }, |
| { |
| "name": "claim4_control", |
| "passed": true |
| }, |
| { |
| "name": "claim5_success_monotone", |
| "passed": true |
| }, |
| { |
| "name": "claim5_ceiling_rises", |
| "passed": true |
| }, |
| { |
| "name": "claim5_all_combinations_and_control", |
| "passed": true |
| } |
| ], |
| "paper_id": "0Mvm3lqLjF", |
| "provenance": { |
| "arxiv": "2603.17863v1", |
| "official_code_archive_sha256": "64f4bef7a116be32df28c1bd7c10098573161d8b910b4dcef26ff1b87edf4da0", |
| "official_code_commit": "4ad81e3fee8b5d8b8fd76827142e107546f47769", |
| "python": "3.13.3", |
| "randomness": "none", |
| "source_sha256": "63a6cac8554672460ceb2a42f045bb3cb6eecea7f48b12537b81848ed47821d0" |
| }, |
| "summary": { |
| "passed": 15, |
| "total": 15 |
| } |
| } |
|
|