| from __future__ import annotations |
|
|
| import json |
| import math |
| import random |
| from copy import deepcopy |
| from pathlib import Path |
|
|
| import numpy as np |
| import torch |
| from torch import Tensor, nn |
|
|
|
|
| def seed_everything(seed: int) -> None: |
| random.seed(seed) |
| np.random.seed(seed) |
| torch.manual_seed(seed) |
| torch.cuda.manual_seed_all(seed) |
|
|
|
|
| def move_targets(targets: list[dict[str, Tensor]], device: torch.device): |
| return [ |
| {key: value.to(device, non_blocking=True) for key, value in target.items()} |
| for target in targets |
| ] |
|
|
|
|
| class ModelEMA: |
| def __init__(self, model: nn.Module, decay: float = 0.9998) -> None: |
| self.model = deepcopy(model).eval() |
| self.decay = decay |
| for parameter in self.model.parameters(): |
| parameter.requires_grad_(False) |
| target = dict(self.model.state_dict()) |
| self._float_names = [name for name, value in target.items() if value.is_floating_point()] |
| self._float_targets = [target[name] for name in self._float_names] |
| self._other = [(name, target[name]) for name in target if not target[name].is_floating_point()] |
| self._float_sources: list[Tensor] | None = None |
| self._other_sources: list[Tensor] | None = None |
|
|
| @torch.no_grad() |
| def update(self, model: nn.Module) -> None: |
| if self._float_sources is None: |
| source = dict(model.state_dict()) |
| self._float_sources = [source[name].detach() for name in self._float_names] |
| self._other_sources = [source[name].detach() for name, _ in self._other] |
| torch._foreach_mul_(self._float_targets, self.decay) |
| torch._foreach_add_(self._float_targets, self._float_sources, alpha=1.0 - self.decay) |
| for (_, target_value), source_value in zip(self._other, self._other_sources): |
| target_value.copy_(source_value) |
|
|
|
|
| def learning_rate_factor(step: int, total_steps: int, warmup_steps: int, min_ratio: float) -> float: |
| if warmup_steps > 0 and step < warmup_steps: |
| return max(step + 1, 1) / warmup_steps |
| progress = (step - warmup_steps) / max(total_steps - warmup_steps, 1) |
| cosine = 0.5 * (1.0 + math.cos(math.pi * min(max(progress, 0.0), 1.0))) |
| return min_ratio + (1.0 - min_ratio) * cosine |
|
|
|
|
| def save_checkpoint(path: str | Path, **state) -> None: |
| path = Path(path) |
| path.parent.mkdir(parents=True, exist_ok=True) |
| temporary = path.with_suffix(path.suffix + ".tmp") |
| torch.save(state, temporary) |
| temporary.replace(path) |
|
|
|
|
| def write_json(path: str | Path, value) -> None: |
| with Path(path).open("w", encoding="utf-8") as handle: |
| json.dump(value, handle, indent=2) |
|
|
|
|
| def trainable_parameter_count(model: nn.Module) -> int: |
| return sum(parameter.numel() for parameter in model.parameters() if parameter.requires_grad) |
|
|