File size: 7,461 Bytes
f15d29e | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 | # Copyright (c) Microsoft Corporation.
# Licensed under the MIT License.
import fnmatch
import os
from dataclasses import asdict, dataclass, field
from functools import cached_property
from pathlib import Path
from typing import Any, Literal, Protocol
import numpy as np
from huggingface_hub import hf_hub_download
from hydra import compose, initialize_config_dir
from omegaconf import DictConfig
from omegaconf import ListConfig
PRETRAINED_MODEL_NAME = Literal[
"mattergen_base",
"chemical_system",
"space_group",
"dft_mag_density",
"dft_band_gap",
"ml_bulk_modulus",
"dft_mag_density_hhi_score",
"chemical_system_energy_above_hull",
"mp_20_base",
]
def _rewrite_vendored_targets(value):
"""Rewrite upstream Hydra targets when loading an external checkpoint config."""
if isinstance(value, DictConfig):
for key in value:
value[key] = _rewrite_vendored_targets(value[key])
return value
if isinstance(value, ListConfig):
for index in range(len(value)):
value[index] = _rewrite_vendored_targets(value[index])
return value
if isinstance(value, str):
if value.startswith("onescience.models.mattergen."):
return "model." + value[len("onescience.models.mattergen."):]
if value.startswith("mattergen."):
if value.startswith("mattergen.common.data."):
return "onescience.datapipes.materials.mattergen." + value[len("mattergen.common.data."):]
if value.startswith("mattergen.common.gemnet.layers."):
return "onescience.modules.layer.mattergen." + value[len("mattergen.common.gemnet.layers."):]
if value.startswith("mattergen.property_embeddings."):
return "onescience.modules.embedding.mattergen_property_embeddings." + value[len("mattergen.property_embeddings."):]
return "model." + value[len("mattergen."):]
return value
return value
def find_local_files(local_path: str, glob: str = "*", relative: bool = False) -> list[str]:
"""
Find files in the given directory or blob storage path, and return the list of files
matching the given glob pattern. If relative is True, the returned paths are relative
to the given directory or blob storage path.
Args:
blob_or_local_path: path to the directory or blob storage path
glob: glob pattern to match. By default, all files are returned.
relative: whether to return relative paths. By default, absolute paths are returned.
Returns:
list of paths to files matching the given glob pattern.
"""
# list all files here, filtering happens in the `fnmatch.filter` step
local_files = [x for x in Path(local_path).rglob("*") if os.path.isfile(x)]
files_list = [str(x.relative_to(local_path)) if relative else str(x) for x in local_files]
return fnmatch.filter(files_list, glob)
@dataclass(frozen=True)
class MatterGenCheckpointInfo:
model_path: str
load_epoch: int | Literal["best", "last"] | None = "last"
config_overrides: list[str] = field(default_factory=list)
split: str = "val"
strict_checkpoint_loading: bool = True
@classmethod
def from_hf_hub(
cls,
model_name: PRETRAINED_MODEL_NAME,
repository_name: str = "microsoft/mattergen",
config_overrides: list[str] = None,
):
"""
Instantiate a MatterGenCheckpointInfo object from a model hosted on the Hugging Face Hub.
"""
hf_hub_download(
repo_id=repository_name, filename=f"checkpoints/{model_name}/checkpoints/last.ckpt"
)
config_path = hf_hub_download(
repo_id=repository_name, filename=f"checkpoints/{model_name}/config.yaml"
)
return cls(
model_path=Path(config_path).parent,
config_overrides=config_overrides or [],
load_epoch="last",
)
def as_dict(self) -> dict[str, Any]:
d = asdict(self)
d["model_path"] = str(self.model_path) # we cannot put Path object in mongo DB
return d
@classmethod
def from_dict(cls, d) -> "MatterGenCheckpointInfo":
d = d.copy()
d["model_path"] = Path(d["model_path"])
# no longer used
if "load_data" in d:
del d["load_data"]
return cls(**d)
@property
def config(self) -> DictConfig:
with initialize_config_dir(str(self.model_path)):
cfg = compose(config_name="config", overrides=self.config_overrides)
return _rewrite_vendored_targets(cfg)
@cached_property
def checkpoint_path(self) -> str:
"""
Search for checkpoint files in the given directory, and return the path
to the checkpoint with the given epoch number or the best checkpoint if load_epoch is "best".
"Best" is selected via the lowest validation loss, which is stored in the checkpoint filename.
Assumes that the checkpoint filenames are of the form "epoch=1-val_loss=0.1234.ckpt" or 'last.ckpt'.
Returns:
Path to the checkpoint file to load.
"""
# look for checkpoints recursively in the given directory or blob storage path.
# I.e., if the path is '/path/', we will find .ckpt files in '/path/version_0/checkpoints'
# and '/path/version_1/checkpoints', and so on.
model_path = str(self.model_path)
ckpts = find_local_files(local_path=model_path, glob="*.ckpt")
assert len(ckpts) > 0, f"No checkpoints found at {model_path}"
if self.load_epoch == "last":
assert any(
[x.endswith("last.ckpt") for x in ckpts]
), "No last.ckpt found in checkpoints."
return [x for x in ckpts if x.endswith("last.ckpt")][0]
# Drop last.ckpt to exclude it from the epoch selection
ckpts = [x for x in ckpts if not x.endswith("last.ckpt")]
# Convert strings to Path to be able to use the .parts attribute
ckpt_paths = [Path(x) for x in ckpts]
# Extract the epoch number and validation loss from the checkpoint filenames
ckpt_epochs = np.array(
[
int(ckpt.parts[-1].split(".ckpt")[0].split("-")[0].split("=")[1])
for ckpt in ckpt_paths
]
)
ckpt_val_losses = np.array(
[
(
float(ckpt.parts[-1].replace(".ckpt", "").split("-")[1].split("=")[1])
if "loss_val" in ckpt.parts[-1]
else 99999999.9
)
for ckpt in ckpt_paths
]
)
# Determine the matching checkpoint index.
if self.load_epoch == "best":
ckpt_ix = ckpt_val_losses.argmin()
elif isinstance(self.load_epoch, int):
assert (
self.load_epoch in ckpt_epochs
), f"Epoch {self.load_epoch} not found in checkpoints."
ckpt_ix = (ckpt_epochs == self.load_epoch).nonzero()[0][0].item()
else:
raise ValueError(f"Unrecognized load_epoch {self.load_epoch}")
ckpt = ckpts[ckpt_ix]
return ckpt
class ProgressCallback(Protocol):
def __call__(self, progress: float):
"""Callback which can be used to report progress on long-running inference.
Args:
progress: Float between 0 and 1.
"""
pass
|