masterformat-classifier / ensemble /predict_ensemble.py
constructelligence's picture
Upgrade ensemble to word+char TF-IDF: line-item 0.686->0.702, section 0.613->0.620 (weights chosen on val)
ad1145a verified
Raw History Blame Contribute Delete
6.73 kB
"""Ensemble predictor: the mf-0.2 transformer + a TF-IDF/SGD scorer, blended in log-prob space.
The transformer alone reaches 0.597 top-1 on the 194 hand-labelled estimate line items; the fitted TF-IDF
scorer (word + char n-grams) reaches 0.675; the blend reaches 0.702 (and improves whole-section and
manual-chunk accuracy). This script reproduces the blend. The weight is chosen on the UFGS validation split
(see `blend.json`), not on the line-item test set.
pip install transformers torch scikit-learn joblib
python predict_ensemble.py "EPDM membrane roofing" "4000 psi concrete slab on grade"
python predict_ensemble.py --top-k 5 --divisions "12\" RCP storm drain pipe"
python predict_ensemble.py --json --file items.txt > out.json
# default TF-IDF/results live beside this script; point --model at the Hub or a local run
python predict_ensemble.py --model constructelligence/masterformat-classifier \
--tfidf tfidf.joblib --weight 0.2 "Wet pipe sprinkler system, light hazard"
"""
import argparse
import json
import sys
from pathlib import Path
import numpy as np
HERE = Path(__file__).resolve().parent
MODEL = "constructelligence/masterformat-classifier"
DIVISIONS = {
"01": "General Requirements", "02": "Existing Conditions", "03": "Concrete",
"04": "Masonry", "05": "Metals", "06": "Wood, Plastics, and Composites",
"07": "Thermal and Moisture Protection", "08": "Openings", "09": "Finishes",
"10": "Specialties", "11": "Equipment", "12": "Furnishings",
"13": "Special Construction", "14": "Conveying Equipment", "21": "Fire Suppression",
"22": "Plumbing", "23": "HVAC", "25": "Integrated Automation", "26": "Electrical",
"27": "Communications", "28": "Electronic Safety and Security", "31": "Earthwork",
"32": "Exterior Improvements", "33": "Utilities", "34": "Transportation",
"35": "Waterway and Marine Construction", "40": "Process Interconnections",
"41": "Material Processing and Handling Equipment",
"43": "Process Gas and Liquid Handling, Purification, and Storage Equipment",
"44": "Pollution and Waste Control Equipment", "46": "Water and Wastewater Equipment",
"48": "Electrical Power Generation",
}
def norm(lp):
return lp - np.logaddexp.reduce(lp, axis=1, keepdims=True)
class Transformer:
def __init__(self, model, batch=64):
import torch
from transformers import AutoModelForSequenceClassification, AutoTokenizer
self.torch = torch
self.tok = AutoTokenizer.from_pretrained(model)
self.m = AutoModelForSequenceClassification.from_pretrained(model).eval()
self.labels = {int(k): v for k, v in self.m.config.id2label.items()}
self.batch = batch
def logprobs(self, texts):
out = []
with self.torch.no_grad():
for i in range(0, len(texts), self.batch):
enc = self.tok(texts[i:i + self.batch], truncation=True, max_length=128,
padding=True, return_tensors="pt")
logits = self.m(**enc).logits.float()
out.append(self.torch.log_softmax(logits, -1).numpy())
return norm(np.concatenate(out))
class Tfidf:
def __init__(self, path):
import joblib
d = joblib.load(path)
if isinstance(d, dict): # legacy {vectorizer, classifier}
self.pipe, self.vec, self.clf = None, d["vectorizer"], d["classifier"]
else: # sklearn Pipeline
self.pipe, self.vec, self.clf = d, None, d.named_steps["clf"]
# classifier classes_ are indices into the sorted label list; verify against the transformer later
self.classes = list(self.clf.classes_)
def logprobs(self, texts):
if self.pipe is not None:
return norm(self.pipe.predict_log_proba(texts))
return norm(self.clf.predict_log_proba(self.vec.transform(texts)))
def main():
ap = argparse.ArgumentParser(description="MasterFormat ensemble (mf-0.2 + TF-IDF).")
ap.add_argument("text", nargs="*", help="text to classify; '-' reads stdin")
ap.add_argument("--model", default=MODEL, help="HF repo id or local checkpoint dir")
ap.add_argument("--tfidf", default=str(HERE / "tfidf.joblib"))
ap.add_argument("--weight", type=float, default=None,
help="transformer weight in the blend (default: blend.json beside --tfidf)")
ap.add_argument("--top-k", type=int, default=3)
ap.add_argument("--divisions", action="store_true")
ap.add_argument("--file")
ap.add_argument("--json", action="store_true")
a = ap.parse_args()
if a.weight is None:
bf = Path(a.tfidf).with_name("blend.json")
a.weight = json.loads(bf.read_text())["weight"] if bf.exists() else 0.2
w = a.weight
texts = list(a.text)
if a.file:
texts += [l.rstrip("\n") for l in open(a.file, encoding="utf-8") if l.strip()]
if "-" in texts:
texts = [t for t in texts if t != "-"] + [l.rstrip("\n") for l in sys.stdin if l.strip()]
texts = [t for t in texts if t.strip()]
if not texts:
texts = ["EPDM membrane roofing"]
tr, tf = Transformer(a.model), Tfidf(a.tfidf)
assert tf.classes == list(range(len(tr.labels))), "TF-IDF classes do not align with the transformer labels"
lp = norm(w * tr.logprobs(texts) + (1 - w) * tf.logprobs(texts)) # blend of log-probs, renormalised
results = []
for text, row in zip(texts, lp):
order = np.argsort(-row)[:a.top_k]
preds = [{"code": tr.labels[int(i)][:8].strip(), "label": tr.labels[int(i)],
"name": tr.labels[int(i)][8:].strip(), "score": round(float(np.exp(row[i])), 4)} for i in order]
item = {"text": text, "weight_transformer": w, "predictions": preds}
if a.divisions:
tot = {}
for p in preds:
tot[p["code"][:2]] = tot.get(p["code"][:2], 0.0) + p["score"]
item["divisions"] = [{"code": d, "name": DIVISIONS.get(d, d), "score": round(s, 4)}
for d, s in sorted(tot.items(), key=lambda kv: -kv[1])]
results.append(item)
if a.json:
json.dump(results, sys.stdout, indent=2, ensure_ascii=False)
print()
return
for r in results:
print(f"\n{r['text']} (blend, transformer weight {w:g})")
for p in r["predictions"]:
print(f" {p['code']} {p['name']:<48.48} {p['score']:.3f}")
if a.divisions:
print(" -- divisions --")
for d in r.get("divisions", []):
print(f" {d['code']} {d['name']:<48.48} {d['score']:.3f}")
if __name__ == "__main__":
main()