| project: | |
| name: AtmosphericDA-DModel | |
| format_version: "2.0" | |
| seed: 7 | |
| independent_code_license: Apache-2.0 | |
| paper_specification: | |
| title: Using machine learning to correct model error in data assimilation and forecast applications | |
| authors: [Alban Farchi, Patrick Laloyaux, Massimo Bonavita, Marc Bocquet] | |
| arxiv: "2010.12605" | |
| doi: 10.1002/qj.4116 | |
| journal: Quarterly Journal of the Royal Meteorological Society | |
| publication_year: 2021 | |
| verified_facts: | |
| state: two-layer streamfunction psi | |
| grid: {nx: 40, ny: 20, state_size: 1600} | |
| boundary: x periodic and y fixed | |
| time_steps_minutes: {reference: 10, perturbed_model: 20} | |
| observations: {interval_hours: 2, random_locations: 50, batches_per_window: 12, window_start_utc: "01:00", covariance: "0.1 I"} | |
| trajectories: {cycles: 1032, discarded_cycles: 8, usable_samples: 1024} | |
| splits: {training_samples: 1024, validation_samples: 1024, test_sets: 16, samples_per_test_set: 1024} | |
| supervised_input: x_a_k | |
| supervised_target: "x_a_{k+1} - M_o(x_a_k)" | |
| final_correction_model: "D model, one hidden Dense layer with 8 linear nodes" | |
| optimizer: TensorFlow Adam with MSE | |
| schedule: [{epochs: 1000, learning_rate: 0.001}, {epochs: 1000, learning_rate: 0.0001}] | |
| engineering_implementation: | |
| dynamics: stable finite-difference two-layer channel surrogate with perturbed advection, coupling, and damping | |
| analysis: sequential bilinear innovation spreading, an executable simplification of variational analysis | |
| framework_note: PyTorch implements the same Adam/MSE objective and D-model topology for local DDP support | |
| paths: | |
| data: data/qg_analysis_windows.npz | |
| checkpoint: result/checkpoints/d_model.pt | |
| training_metrics: result/training/metrics.json | |
| predictions: result/output/predictions.npz | |
| evaluation_metrics: result/evaluation/metrics.json | |
| comparison_figure: result/evaluation/comparison.png | |
| qg: | |
| nx: 40 | |
| ny: 20 | |
| reference_dt_minutes: 10 | |
| model_dt_minutes: 20 | |
| observation_interval_minutes: 120 | |
| window_batches: 12 | |
| diffusion: 0.002 | |
| truth_advection: 0.18 | |
| model_advection: 0.15 | |
| truth_coupling: 0.025 | |
| model_coupling: 0.018 | |
| truth_damping: 0.006 | |
| model_damping: 0.009 | |
| data: | |
| dataset: structured synthetic two-layer streamfunction analysis windows | |
| protocol: hybrid_qg_analysis_increment_npz_v2 | |
| format: NPZ | |
| format_version: "2.0" | |
| state_layout: NCYX | |
| state_shape: [2, 20, 40] | |
| samples: 4 | |
| observation_batches: 12 | |
| observations_per_batch: 50 | |
| observation_variance: 0.1 | |
| analysis_gain: 0.35 | |
| seed: 7 | |
| model: | |
| architecture: DModel | |
| hidden_size: 8 | |
| activation: linear | |
| training: | |
| stage_epochs: [2, 2] | |
| stage_learning_rates: [0.001, 0.0001] | |
| batch_size: 2 | |
| seed: 19 | |
| paper_model: | |
| hidden_size: 8 | |
| activation: linear | |
| loss: MSE | |
| optimizer: Adam | |
| stage_epochs: [1000, 1000] | |
| stage_learning_rates: [0.001, 0.0001] | |
| training_samples: 1024 | |
| validation_samples: 1024 | |
| test_sets: 16 | |
| samples_per_test_set: 1024 | |
| state_shape: [2, 20, 40] | |
| observation_shape_per_window: [12, 50] | |