LLPSense / preprocess /add_example.py
Minjun Kang
update feature
62a01bd
Raw
History Blame Contribute Delete
6.23 kB
"""Add example sequence(s) to examples.py from a raw-data Excel file.
The training-data Excel files (e.g. LLPSense/data/raw/nonPSPex/ASCL1.xlsx)
carry a "Protein name" column and a sequence column (spelled either
"Sequence" or "Seqence" across files) alongside condition columns this demo
doesn't need. Some files hold one protein across several condition rows
(e.g. ASCL1.xlsx); others hold several distinct proteins, one per row, each
with its own name/sequence (e.g. UBQLN4_mut.xlsx's UBQLN_mut10/20/30). This
script groups rows by protein name and appends one EXAMPLES entry per
distinct (name, sequence) pair found in the file.
Requires pandas + openpyxl (not part of requirements.txt β€” those are only
needed to run this offline script, not the deployed app):
pip install pandas openpyxl
Usage:
python preprocess/add_example.py /path/to/ASCL1.xlsx
python preprocess/add_example.py /path/to/ASCL1.xlsx --id ascl1 --name "ASCL1"
python preprocess/add_example.py /path/to/UBQLN4_mut.xlsx # adds all 3 mutants
--id/--name only apply when the file contains a single protein.
After adding, run preprocess/extract_example_feat.py to cache the T5 feature(s).
"""
import argparse
import re
import sys
from pathlib import Path
import pandas as pd
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from examples import EXAMPLES # noqa: E402
EXAMPLES_FILE = Path(__file__).resolve().parent.parent / "examples.py"
VALID_AA = set("ACDEFGHIKLMNPQRSTVWY")
ALLOW_AA = VALID_AA | set("XBJOUZ")
def _find_col(df, *keywords):
for col in df.columns:
low = col.lower()
if all(keyword in low for keyword in keywords):
return col
return None
def slugify(name: str) -> str:
slug = re.sub(r"[^a-z0-9]+", "_", name.strip().lower()).strip("_")
if not slug:
raise ValueError(f"Could not derive an id from name: {name!r}")
return slug
def read_proteins(xlsx_path: Path) -> list[tuple[str, str]]:
"""Return one (name, sequence) pair per distinct protein in the file.
Rows are grouped by protein name β€” a file may repeat the same protein
across several condition rows (ASCL1.xlsx: 1 protein, several conditions)
or list several distinct proteins one per row (UBQLN4_mut.xlsx: 3
mutants). Either shape yields one entry per unique protein name here.
"""
df = pd.read_excel(xlsx_path, engine="openpyxl")
name_col = _find_col(df, "protein", "name")
seq_col = _find_col(df, "seq")
if name_col is None or seq_col is None:
raise ValueError(
f"{xlsx_path} is missing a protein-name or sequence column "
f"(found columns: {df.columns.tolist()})"
)
df = df[[name_col, seq_col]].dropna()
if df.empty:
raise ValueError(f"{xlsx_path} has no rows with both a name and a sequence")
proteins = []
for name, group in df.groupby(name_col, sort=False):
seqs = group[seq_col].astype(str).str.strip().str.upper().unique()
if len(seqs) != 1:
raise ValueError(
f"Protein '{name}' in {xlsx_path} has {len(seqs)} distinct "
f"sequences across its rows β€” expected exactly one"
)
proteins.append((str(name).strip(), seqs[0]))
return proteins
def format_seq_literal(seq: str, width: int = 60) -> str:
lines = [seq[i:i + width] for i in range(0, len(seq), width)]
body = "\n".join(f' "{line}"' for line in lines)
return f"(\n{body}\n )"
def append_example(entry: dict) -> None:
text = EXAMPLES_FILE.read_text()
match = re.search(r"(EXAMPLES\s*=\s*\[)(.*?)(\n\])", text, flags=re.DOTALL)
if not match:
raise RuntimeError(f"Could not find EXAMPLES list in {EXAMPLES_FILE}")
# The existing last entry may or may not have a trailing comma (both
# forms occur in practice) β€” normalize to one before appending.
body = match.group(2).rstrip()
if body and not body.endswith(","):
body += ","
new_block = (
"\n {\n"
f' "id": {entry["id"]!r},\n'
f' "name": {entry["name"]!r},\n'
f' "seq": {format_seq_literal(entry["seq"])},\n'
" },"
)
updated = text[:match.start(2)] + body + new_block + text[match.end(2):]
EXAMPLES_FILE.write_text(updated)
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("xlsx_path", type=Path)
parser.add_argument(
"--id", type=str, default=None,
help="override the generated example id (only valid for single-protein files)",
)
parser.add_argument(
"--name", type=str, default=None,
help="override the display name (only valid for single-protein files)",
)
args = parser.parse_args()
proteins = read_proteins(args.xlsx_path)
if (args.id or args.name) and len(proteins) > 1:
parser.error(
f"{args.xlsx_path} contains {len(proteins)} distinct proteins β€” "
"--id/--name can only be used with a single-protein file"
)
# Track ids seen so far in this run too, so two proteins in the same
# file that happen to slugify to the same id are still caught.
known_ids = {example["id"] for example in EXAMPLES}
added = 0
for name, seq in proteins:
example_id = args.id or slugify(name)
display_name = args.name or name
invalid = set(seq) - ALLOW_AA
if invalid:
print(
f"Warning: '{name}' sequence contains unexpected characters: "
f"{''.join(sorted(invalid))}",
file=sys.stderr,
)
if example_id in known_ids:
print(f"'{example_id}' already exists in EXAMPLES β€” skipping.")
continue
append_example({"id": example_id, "name": display_name, "seq": seq})
known_ids.add(example_id)
added += 1
print(f"Added '{example_id}' ({display_name}, {len(seq)} aa) to {EXAMPLES_FILE}")
if added:
print("Run `python preprocess/extract_example_feat.py` to cache the T5 feature(s).")
else:
print("No new examples added.")
if __name__ == "__main__":
main()