File size: 1,622 Bytes
cd9b2d8
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
"""Pinned experiment paths and atomic, credential-free metadata."""
import hashlib
import json
import os
from pathlib import Path

HERE = Path(__file__).resolve().parent
CODE = HERE.parent
ROOT = Path('/e/scratch/reformo/schuhmann1_moss/whisper_score_regression/embedding_probe_study_20261006')
BENCH = ROOT.parent / 'bench_eval_20261004'
GEMINI = ROOT.parent / 'gemini_finetune_preparation_20261005'
RELEASE = ROOT.parent / 'hf_layered_release/model'
CLAP_CODE = Path('/e/project1/laionize/foerster5_jupiter/clap_v2_refactor/open_clip')
CLAP_BENCH = CLAP_CODE.parent / 'clap_benchmark'
TRANSFORMERS_REVISION = '14e738b5d0cc69aa27a95dde272aea41fde44f2f'
SEED = 20261006


def read(path):
    return json.loads(Path(path).read_text())


def json_number(value):
    # Keep the orchestration process stdlib-only while accepting NumPy metric
    # scalars/arrays produced by the compute jobs. Nonfinite values still fail.
    if type(value).__module__.split('.')[0] == 'numpy' and hasattr(value, 'tolist'):
        return value.tolist()
    raise TypeError('Unsupported metadata value: ' + type(value).__name__)


def write(path, value):
    path = Path(path)
    path.parent.mkdir(parents=True, exist_ok=True)
    temporary = path.with_name(path.name + '.tmp-' + str(os.getpid()))
    temporary.write_text(json.dumps(value, indent=2, ensure_ascii=False, allow_nan=False, default=json_number) + '\n')
    os.replace(temporary, path)


def digest(path):
    h = hashlib.sha256()
    with Path(path).open('rb') as stream:
        while chunk := stream.read(8 * 1024 * 1024):
            h.update(chunk)
    return h.hexdigest()