mof-discovery-platform / utils /data_generator.py
shrut27's picture
Upload utils/data_generator.py with huggingface_hub
b06f94b verified
Raw
History Blame Contribute Delete
6.54 kB
import numpy as np
import pandas as pd
from pathlib import Path
METALS = ["Zn", "Cu", "Fe", "Co", "Ni", "Zr", "Al", "Cr", "Mn", "In", "Cd", "Mg"]
LINKERS = ["BDC", "BTC", "BPDC", "NDC", "TATB", "BTB", "H2BDC", "TPDC", "ADB", "DPA", "BPYDC", "TCPP"]
TOPOLOGIES = ["pcu", "dia", "soc", "rht", "hkust", "uio", "mof-5", "mil-53", "zif-8", "nott-112", "pcn-250", "BUT-66"]
SOLVENTS = ["DMF", "DEF", "DMSO", "EtOH", "MeOH", "Water", "DMA", "THF"]
APPLICATIONS = ["CO2 capture", "H2 storage", "CH4 storage", "Catalysis", "Water harvesting", "Drug delivery", "Battery", "Sensing"]
DATABASES = ["CoRE-MOF", "QMOF", "CSD", "hMOF", "ToBaCCo", "Literature"]
RNG = np.random.default_rng(42)
def generate_mof_dataset(n: int = 8000) -> pd.DataFrame:
metals = RNG.choice(METALS, n)
linkers = RNG.choice(LINKERS, n)
topologies = RNG.choice(TOPOLOGIES, n)
solvents = RNG.choice(SOLVENTS, n)
applications = RNG.choice(APPLICATIONS, n)
databases = RNG.choice(DATABASES, n)
structural_score = np.clip(RNG.beta(6, 1.5, n) * 100, 0, 100)
chemical_score = np.clip(RNG.beta(5, 1.8, n) * 100, 0, 100)
novelty_score = np.clip(RNG.beta(2.5, 2.5, n) * 100, 0, 100)
fto_score = np.clip(RNG.beta(4, 2, n) * 100, 0, 100)
synthesizability = np.clip(RNG.beta(3.5, 2, n) * 100, 0, 100)
surface_area = np.clip(RNG.lognormal(7.0, 0.6, n), 200, 8000)
pore_size = np.clip(RNG.lognormal(1.5, 0.5, n), 2, 40)
void_fraction = np.clip(RNG.beta(3, 3, n), 0.15, 0.90)
co2_uptake = np.clip(RNG.lognormal(4.0, 0.8, n), 20, 500)
h2_uptake = np.clip(RNG.lognormal(2.0, 0.7, n), 1, 80)
ch4_uptake = np.clip(RNG.lognormal(3.5, 0.7, n), 10, 250)
band_gap = np.clip(RNG.normal(2.8, 1.2, n), 0.2, 6.5)
thermal_stability = np.clip(RNG.normal(350, 80, n), 100, 600)
water_stability = RNG.choice(["High", "Medium", "Low"], n, p=[0.3, 0.45, 0.25])
catalytic_activity = np.clip(RNG.beta(2.5, 3, n) * 100, 0, 100)
ion_conductivity = np.clip(RNG.lognormal(-3, 1.5, n), 0.0001, 0.5)
synthesis_steps = RNG.integers(1, 7, n)
synthesis_difficulty = np.clip(RNG.beta(2, 3, n) * 10, 0.5, 9.5)
synthesis_cost = np.clip(RNG.lognormal(4.5, 0.8, n), 50, 5000)
green_score = np.clip(RNG.beta(3, 2, n) * 100, 0, 100)
success_prob = np.clip(RNG.beta(5, 2, n) * 100, 0, 100)
precursor_availability = RNG.choice(["High", "Medium", "Low"], n, p=[0.5, 0.35, 0.15])
similar_patents = RNG.integers(0, 20, n)
patent_similarity = np.clip(RNG.beta(2, 4, n) * 100, 0, 100)
claim_overlap = RNG.choice(["None", "Low", "Medium", "High"], n, p=[0.25, 0.35, 0.3, 0.1])
fto_risk = np.where(fto_score > 70, "Low", np.where(fto_score > 40, "Medium", "High"))
perf_weight = 0.30
synth_weight = 0.20
struct_weight = 0.15
novelty_weight = 0.15
fto_weight = 0.10
cost_weight = 0.05
sustain_weight = 0.05
composite_score = (
np.clip(co2_uptake / 5, 0, 100) * perf_weight
+ synthesizability * synth_weight
+ structural_score * struct_weight
+ novelty_score * novelty_weight
+ fto_score * fto_weight
+ (100 - synthesis_difficulty * 10) * cost_weight
+ green_score * sustain_weight
)
composite_score = np.clip(composite_score, 0, 100)
mof_ids = [f"MOF-{i+1:05d}" for i in range(n)]
years = RNG.integers(2010, 2026, n)
df = pd.DataFrame({
"mof_id": mof_ids,
"metal": metals,
"linker": linkers,
"topology": topologies,
"solvent": solvents,
"application": applications,
"source_database": databases,
"year": years,
"structural_score": structural_score.round(2),
"chemical_score": chemical_score.round(2),
"novelty_score": novelty_score.round(2),
"fto_score": fto_score.round(2),
"synthesizability": synthesizability.round(2),
"surface_area_m2g": surface_area.round(1),
"pore_size_angstrom": pore_size.round(2),
"void_fraction": void_fraction.round(3),
"co2_uptake_cc_g": co2_uptake.round(2),
"h2_uptake_cc_g": h2_uptake.round(2),
"ch4_uptake_cc_g": ch4_uptake.round(2),
"band_gap_eV": band_gap.round(3),
"thermal_stability_C": thermal_stability.round(1),
"water_stability": water_stability,
"catalytic_activity": catalytic_activity.round(2),
"ion_conductivity_S_cm": ion_conductivity.round(6),
"synthesis_steps": synthesis_steps,
"synthesis_difficulty": synthesis_difficulty.round(2),
"synthesis_cost_USD": synthesis_cost.round(0).astype(int),
"green_score": green_score.round(2),
"success_probability": success_prob.round(2),
"precursor_availability": precursor_availability,
"similar_patents": similar_patents,
"patent_similarity_pct": patent_similarity.round(2),
"claim_overlap": claim_overlap,
"fto_risk": fto_risk,
"composite_score": composite_score.round(2),
})
rank = df["composite_score"].rank(ascending=False).astype(int)
df["rank"] = rank
return df.sort_values("rank").reset_index(drop=True)
def generate_pipeline_metrics(n_days: int = 365) -> pd.DataFrame:
dates = pd.date_range("2024-01-01", periods=n_days, freq="D")
ingested = RNG.integers(80, 200, n_days).cumsum()
validated_struct = (ingested * RNG.uniform(0.85, 0.96, n_days)).astype(int)
validated_chem = (validated_struct * RNG.uniform(0.88, 0.97, n_days)).astype(int)
auto_repaired = (ingested - validated_struct) * RNG.integers(3, 8, n_days)
auto_repaired = np.clip(auto_repaired.cumsum(), 0, None)
generated = RNG.integers(200, 1000, n_days).cumsum()
shortlisted = (generated * RNG.uniform(0.05, 0.15, n_days)).astype(int)
return pd.DataFrame({
"date": dates,
"total_ingested": ingested,
"struct_validated": validated_struct,
"chem_validated": validated_chem,
"auto_repaired": auto_repaired,
"generated_candidates": generated,
"shortlisted": shortlisted,
})
def save_datasets(output_dir: str = "data") -> None:
path = Path(output_dir)
path.mkdir(exist_ok=True)
mof_df = generate_mof_dataset(8000)
mof_df.to_csv(path / "mof_candidates.csv", index=False)
pipeline_df = generate_pipeline_metrics(365)
pipeline_df.to_csv(path / "pipeline_metrics.csv", index=False)
print(f"Saved {len(mof_df)} MOF records and {len(pipeline_df)} pipeline daily records.")
if __name__ == "__main__":
save_datasets()