| """Tensile2d (PLAID-datasets/Tensile2d, HF, CC-BY-SA-4.0) adapter — SECOND benchmark. |
| |
| Why: unlike Geo-FNO Elasticity (scalar von Mises target -> our tensor is latent), Tensile2d directly |
| supervises the full 2D Cauchy stress tensor (sig11, sig22, sig12) per node on an irregular mesh, so our |
| equilibrium regularizer acts on a SUPERVISED tensor and we can validate the predicted tensor itself. |
| |
| VERIFIED facts (golden rule, by reading real samples): |
| - Each sample is a pickled CGNS tree (no PLAID lib needed): GridCoordinates {CoordinateX, CoordinateY} |
| and PointData {sig11, sig22, sig12, U1, U2, q}; plus input scalars {P, p1..p5} and output scalars. |
| - 702 rows: samples 0-499 are the labeled `train_500` set (full stress); samples 500-701 (test 200 + |
| OOD 2) have stress fields WITHHELD (competition) -> only coords + input scalars. So we use the 500 |
| labeled samples and hold out a local test split (no official test ground truth is public). |
| - Node count varies per sample (6143-11801, mean ~9400). Static, plane-strain, quasistatic. |
| |
| `build_cache()` writes data/tensile2d/samples/*.npz once. `build_tensile_splits()` loads them, fits |
| z-score normalizers on the train split, and returns per-sample lists (variable N -> no stacking). |
| """ |
| from __future__ import annotations |
|
|
| import glob |
| import os |
| from dataclasses import dataclass |
| from typing import List |
|
|
| import numpy as np |
| import torch |
|
|
| INPUT_SCALARS = ["P", "p1", "p2", "p3", "p4", "p5"] |
|
|
|
|
| def _field(tree, name): |
| stack = [tree] |
| while stack: |
| n = stack.pop() |
| if isinstance(n, list) and len(n) == 4: |
| if n[0] == name and isinstance(n[1], np.ndarray): |
| return n[1] |
| stack.extend(n[2] or []) |
| return None |
|
|
|
|
| def build_cache(out_dir: str = "data/tensile2d/samples") -> int: |
| """Download Tensile2d parquet shards and cache the 500 labeled samples as .npz. Returns count.""" |
| from huggingface_hub import hf_hub_download |
| import pyarrow.parquet as pq |
| import pickle |
|
|
| os.makedirs(out_dir, exist_ok=True) |
| shards = [hf_hub_download("PLAID-datasets/Tensile2d", |
| f"data/all_samples-0000{i}-of-00002.parquet", repo_type="dataset") |
| for i in (0, 1)] |
| idx = 0 |
| saved = 0 |
| for sh in shards: |
| col = pq.read_table(sh).column("sample") |
| for i in range(len(col)): |
| try: |
| obj = pickle.loads(col[i].as_py()) |
| tree = list(obj["meshes"].values())[0] |
| x, y = _field(tree, "CoordinateX"), _field(tree, "CoordinateY") |
| s11, s22, s12 = _field(tree, "sig11"), _field(tree, "sig22"), _field(tree, "sig12") |
| if any(v is None for v in (x, y, s11, s22, s12)): |
| idx += 1 |
| continue |
| coords = np.stack([x, y], 1).astype(np.float32) |
| sigma = np.stack([s11, s22, s12], 1).astype(np.float32) |
| vm = np.sqrt(s11 ** 2 - s11 * s22 + s22 ** 2 + 3 * s12 ** 2).astype(np.float32) |
| sc = obj.get("scalars", {}) |
| scin = np.array([float(sc[k]) for k in INPUT_SCALARS], np.float32) |
| np.savez(os.path.join(out_dir, f"{saved:04d}.npz"), |
| coords=coords, sigma=sigma, vm=vm, scalars_in=scin) |
| saved += 1 |
| except Exception: |
| pass |
| idx += 1 |
| return saved |
|
|
|
|
| @dataclass |
| class TensileSplits: |
| train_coords: List[torch.Tensor] |
| train_sigma: List[torch.Tensor] |
| train_scalars: List[torch.Tensor] |
| test_coords: List[torch.Tensor] |
| test_sigma: List[torch.Tensor] |
| test_scalars: List[torch.Tensor] |
| scale_S: float |
| scalar_mean: torch.Tensor |
| scalar_std: torch.Tensor |
|
|
|
|
| def build_tensile_splits(samples_dir: str = "data/tensile2d/samples", |
| ntrain: int = 400, ntest: int = 100, seed: int = 0) -> TensileSplits: |
| """Load cached samples; deterministic train/test split; z-score normalizers fit on TRAIN only. |
| |
| Coords are kept RAW (physical units) so the MLS divergence operator measures true distances; the |
| model conditions on raw coords + normalized input scalars (the encoder MLP handles the coord scale). |
| Stress is z-scored per component; input scalars z-scored per dim. |
| """ |
| files = sorted(glob.glob(os.path.join(samples_dir, "*.npz"))) |
| if not files: |
| raise FileNotFoundError(f"no cached Tensile2d samples in {samples_dir} — run build_cache() first") |
| rng = np.random.default_rng(seed) |
| order = rng.permutation(len(files)) |
| tr_idx, te_idx = order[:ntrain], order[ntrain:ntrain + ntest] |
|
|
| def load(i): |
| d = np.load(files[i]) |
| return d["coords"], d["sigma"], d["scalars_in"] |
|
|
| tr = [load(i) for i in tr_idx] |
| te = [load(i) for i in te_idx] |
|
|
| |
| |
| |
| sig_all = np.concatenate([s for _, s, _ in tr], 0) |
| S = float(np.sqrt((sig_all ** 2).mean())) |
| sc_all = np.stack([sc for _, _, sc in tr], 0) |
| sc_mean, sc_std = sc_all.mean(0), sc_all.std(0) + 1e-8 |
|
|
| def pack(rows): |
| cs, ss, scs = [], [], [] |
| for c, s, sc in rows: |
| cs.append(torch.tensor(c)) |
| ss.append(torch.tensor(s / S)) |
| scs.append(torch.tensor((sc - sc_mean) / sc_std)) |
| return cs, ss, scs |
|
|
| trc, trs, trsc = pack(tr) |
| tec, tes, tesc = pack(te) |
| return TensileSplits( |
| train_coords=trc, train_sigma=trs, train_scalars=trsc, |
| test_coords=tec, test_sigma=tes, test_scalars=tesc, |
| scale_S=S, |
| scalar_mean=torch.tensor(sc_mean), scalar_std=torch.tensor(sc_std), |
| ) |
|
|