"""Ensemble predictor: the mf-0.2 transformer + a TF-IDF/SGD scorer, blended in log-prob space. The transformer alone reaches 0.597 top-1 on the 194 hand-labelled estimate line items; the fitted TF-IDF scorer (word + char n-grams) reaches 0.675; the blend reaches 0.702 (and improves whole-section and manual-chunk accuracy). This script reproduces the blend. The weight is chosen on the UFGS validation split (see `blend.json`), not on the line-item test set. pip install transformers torch scikit-learn joblib python predict_ensemble.py "EPDM membrane roofing" "4000 psi concrete slab on grade" python predict_ensemble.py --top-k 5 --divisions "12\" RCP storm drain pipe" python predict_ensemble.py --json --file items.txt > out.json # default TF-IDF/results live beside this script; point --model at the Hub or a local run python predict_ensemble.py --model constructelligence/masterformat-classifier \ --tfidf tfidf.joblib --weight 0.2 "Wet pipe sprinkler system, light hazard" """ import argparse import json import sys from pathlib import Path import numpy as np HERE = Path(__file__).resolve().parent MODEL = "constructelligence/masterformat-classifier" DIVISIONS = { "01": "General Requirements", "02": "Existing Conditions", "03": "Concrete", "04": "Masonry", "05": "Metals", "06": "Wood, Plastics, and Composites", "07": "Thermal and Moisture Protection", "08": "Openings", "09": "Finishes", "10": "Specialties", "11": "Equipment", "12": "Furnishings", "13": "Special Construction", "14": "Conveying Equipment", "21": "Fire Suppression", "22": "Plumbing", "23": "HVAC", "25": "Integrated Automation", "26": "Electrical", "27": "Communications", "28": "Electronic Safety and Security", "31": "Earthwork", "32": "Exterior Improvements", "33": "Utilities", "34": "Transportation", "35": "Waterway and Marine Construction", "40": "Process Interconnections", "41": "Material Processing and Handling Equipment", "43": "Process Gas and Liquid Handling, Purification, and Storage Equipment", "44": "Pollution and Waste Control Equipment", "46": "Water and Wastewater Equipment", "48": "Electrical Power Generation", } def norm(lp): return lp - np.logaddexp.reduce(lp, axis=1, keepdims=True) class Transformer: def __init__(self, model, batch=64): import torch from transformers import AutoModelForSequenceClassification, AutoTokenizer self.torch = torch self.tok = AutoTokenizer.from_pretrained(model) self.m = AutoModelForSequenceClassification.from_pretrained(model).eval() self.labels = {int(k): v for k, v in self.m.config.id2label.items()} self.batch = batch def logprobs(self, texts): out = [] with self.torch.no_grad(): for i in range(0, len(texts), self.batch): enc = self.tok(texts[i:i + self.batch], truncation=True, max_length=128, padding=True, return_tensors="pt") logits = self.m(**enc).logits.float() out.append(self.torch.log_softmax(logits, -1).numpy()) return norm(np.concatenate(out)) class Tfidf: def __init__(self, path): import joblib d = joblib.load(path) if isinstance(d, dict): # legacy {vectorizer, classifier} self.pipe, self.vec, self.clf = None, d["vectorizer"], d["classifier"] else: # sklearn Pipeline self.pipe, self.vec, self.clf = d, None, d.named_steps["clf"] # classifier classes_ are indices into the sorted label list; verify against the transformer later self.classes = list(self.clf.classes_) def logprobs(self, texts): if self.pipe is not None: return norm(self.pipe.predict_log_proba(texts)) return norm(self.clf.predict_log_proba(self.vec.transform(texts))) def main(): ap = argparse.ArgumentParser(description="MasterFormat ensemble (mf-0.2 + TF-IDF).") ap.add_argument("text", nargs="*", help="text to classify; '-' reads stdin") ap.add_argument("--model", default=MODEL, help="HF repo id or local checkpoint dir") ap.add_argument("--tfidf", default=str(HERE / "tfidf.joblib")) ap.add_argument("--weight", type=float, default=None, help="transformer weight in the blend (default: blend.json beside --tfidf)") ap.add_argument("--top-k", type=int, default=3) ap.add_argument("--divisions", action="store_true") ap.add_argument("--file") ap.add_argument("--json", action="store_true") a = ap.parse_args() if a.weight is None: bf = Path(a.tfidf).with_name("blend.json") a.weight = json.loads(bf.read_text())["weight"] if bf.exists() else 0.2 w = a.weight texts = list(a.text) if a.file: texts += [l.rstrip("\n") for l in open(a.file, encoding="utf-8") if l.strip()] if "-" in texts: texts = [t for t in texts if t != "-"] + [l.rstrip("\n") for l in sys.stdin if l.strip()] texts = [t for t in texts if t.strip()] if not texts: texts = ["EPDM membrane roofing"] tr, tf = Transformer(a.model), Tfidf(a.tfidf) assert tf.classes == list(range(len(tr.labels))), "TF-IDF classes do not align with the transformer labels" lp = norm(w * tr.logprobs(texts) + (1 - w) * tf.logprobs(texts)) # blend of log-probs, renormalised results = [] for text, row in zip(texts, lp): order = np.argsort(-row)[:a.top_k] preds = [{"code": tr.labels[int(i)][:8].strip(), "label": tr.labels[int(i)], "name": tr.labels[int(i)][8:].strip(), "score": round(float(np.exp(row[i])), 4)} for i in order] item = {"text": text, "weight_transformer": w, "predictions": preds} if a.divisions: tot = {} for p in preds: tot[p["code"][:2]] = tot.get(p["code"][:2], 0.0) + p["score"] item["divisions"] = [{"code": d, "name": DIVISIONS.get(d, d), "score": round(s, 4)} for d, s in sorted(tot.items(), key=lambda kv: -kv[1])] results.append(item) if a.json: json.dump(results, sys.stdout, indent=2, ensure_ascii=False) print() return for r in results: print(f"\n{r['text']} (blend, transformer weight {w:g})") for p in r["predictions"]: print(f" {p['code']} {p['name']:<48.48} {p['score']:.3f}") if a.divisions: print(" -- divisions --") for d in r.get("divisions", []): print(f" {d['code']} {d['name']:<48.48} {d['score']:.3f}") if __name__ == "__main__": main()