Text Classification
Transformers
Joblib
ONNX
Safetensors
English
bert
masterformat
masterformat-classifier
csi-masterformat
construction
construction-technology
sequence-classification
tfidf
ensemble
specs
spec-writing
specifications
takeoff
estimating
cost-code
ufgs
public-domain
Eval Results (legacy)
text-embeddings-inference
Instructions to use constructelligence/masterformat-classifier with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use constructelligence/masterformat-classifier with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-classification", model="constructelligence/masterformat-classifier")# pip install -U transformers accelerate # Load model directly from transformers import AutoTokenizer, AutoModelForSequenceClassification tokenizer = AutoTokenizer.from_pretrained("constructelligence/masterformat-classifier") model = AutoModelForSequenceClassification.from_pretrained("constructelligence/masterformat-classifier", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Upgrade ensemble to word+char TF-IDF: line-item 0.686->0.702, section 0.613->0.620 (weights chosen on val)
ad1145a verified Download ensemble/predict_ensemble.py from constructelligence/masterformat-classifier: direct link, hf CLI and curl.
- Browser
- Download file 6.73 kB
-
https://huggingface.co/constructelligence/masterformat-classifier/resolve/main/ensemble/predict_ensemble.py
- Command line
-
hf download hf://constructelligence/masterformat-classifier/ensemble/predict_ensemble.py
-
curl -L -o predict_ensemble.py https://huggingface.co/constructelligence/masterformat-classifier/resolve/main/ensemble/predict_ensemble.py
6.73 kB
| """Ensemble predictor: the mf-0.2 transformer + a TF-IDF/SGD scorer, blended in log-prob space. | |
| The transformer alone reaches 0.597 top-1 on the 194 hand-labelled estimate line items; the fitted TF-IDF | |
| scorer (word + char n-grams) reaches 0.675; the blend reaches 0.702 (and improves whole-section and | |
| manual-chunk accuracy). This script reproduces the blend. The weight is chosen on the UFGS validation split | |
| (see `blend.json`), not on the line-item test set. | |
| pip install transformers torch scikit-learn joblib | |
| python predict_ensemble.py "EPDM membrane roofing" "4000 psi concrete slab on grade" | |
| python predict_ensemble.py --top-k 5 --divisions "12\" RCP storm drain pipe" | |
| python predict_ensemble.py --json --file items.txt > out.json | |
| # default TF-IDF/results live beside this script; point --model at the Hub or a local run | |
| python predict_ensemble.py --model constructelligence/masterformat-classifier \ | |
| --tfidf tfidf.joblib --weight 0.2 "Wet pipe sprinkler system, light hazard" | |
| """ | |
| import argparse | |
| import json | |
| import sys | |
| from pathlib import Path | |
| import numpy as np | |
| HERE = Path(__file__).resolve().parent | |
| MODEL = "constructelligence/masterformat-classifier" | |
| DIVISIONS = { | |
| "01": "General Requirements", "02": "Existing Conditions", "03": "Concrete", | |
| "04": "Masonry", "05": "Metals", "06": "Wood, Plastics, and Composites", | |
| "07": "Thermal and Moisture Protection", "08": "Openings", "09": "Finishes", | |
| "10": "Specialties", "11": "Equipment", "12": "Furnishings", | |
| "13": "Special Construction", "14": "Conveying Equipment", "21": "Fire Suppression", | |
| "22": "Plumbing", "23": "HVAC", "25": "Integrated Automation", "26": "Electrical", | |
| "27": "Communications", "28": "Electronic Safety and Security", "31": "Earthwork", | |
| "32": "Exterior Improvements", "33": "Utilities", "34": "Transportation", | |
| "35": "Waterway and Marine Construction", "40": "Process Interconnections", | |
| "41": "Material Processing and Handling Equipment", | |
| "43": "Process Gas and Liquid Handling, Purification, and Storage Equipment", | |
| "44": "Pollution and Waste Control Equipment", "46": "Water and Wastewater Equipment", | |
| "48": "Electrical Power Generation", | |
| } | |
| def norm(lp): | |
| return lp - np.logaddexp.reduce(lp, axis=1, keepdims=True) | |
| class Transformer: | |
| def __init__(self, model, batch=64): | |
| import torch | |
| from transformers import AutoModelForSequenceClassification, AutoTokenizer | |
| self.torch = torch | |
| self.tok = AutoTokenizer.from_pretrained(model) | |
| self.m = AutoModelForSequenceClassification.from_pretrained(model).eval() | |
| self.labels = {int(k): v for k, v in self.m.config.id2label.items()} | |
| self.batch = batch | |
| def logprobs(self, texts): | |
| out = [] | |
| with self.torch.no_grad(): | |
| for i in range(0, len(texts), self.batch): | |
| enc = self.tok(texts[i:i + self.batch], truncation=True, max_length=128, | |
| padding=True, return_tensors="pt") | |
| logits = self.m(**enc).logits.float() | |
| out.append(self.torch.log_softmax(logits, -1).numpy()) | |
| return norm(np.concatenate(out)) | |
| class Tfidf: | |
| def __init__(self, path): | |
| import joblib | |
| d = joblib.load(path) | |
| if isinstance(d, dict): # legacy {vectorizer, classifier} | |
| self.pipe, self.vec, self.clf = None, d["vectorizer"], d["classifier"] | |
| else: # sklearn Pipeline | |
| self.pipe, self.vec, self.clf = d, None, d.named_steps["clf"] | |
| # classifier classes_ are indices into the sorted label list; verify against the transformer later | |
| self.classes = list(self.clf.classes_) | |
| def logprobs(self, texts): | |
| if self.pipe is not None: | |
| return norm(self.pipe.predict_log_proba(texts)) | |
| return norm(self.clf.predict_log_proba(self.vec.transform(texts))) | |
| def main(): | |
| ap = argparse.ArgumentParser(description="MasterFormat ensemble (mf-0.2 + TF-IDF).") | |
| ap.add_argument("text", nargs="*", help="text to classify; '-' reads stdin") | |
| ap.add_argument("--model", default=MODEL, help="HF repo id or local checkpoint dir") | |
| ap.add_argument("--tfidf", default=str(HERE / "tfidf.joblib")) | |
| ap.add_argument("--weight", type=float, default=None, | |
| help="transformer weight in the blend (default: blend.json beside --tfidf)") | |
| ap.add_argument("--top-k", type=int, default=3) | |
| ap.add_argument("--divisions", action="store_true") | |
| ap.add_argument("--file") | |
| ap.add_argument("--json", action="store_true") | |
| a = ap.parse_args() | |
| if a.weight is None: | |
| bf = Path(a.tfidf).with_name("blend.json") | |
| a.weight = json.loads(bf.read_text())["weight"] if bf.exists() else 0.2 | |
| w = a.weight | |
| texts = list(a.text) | |
| if a.file: | |
| texts += [l.rstrip("\n") for l in open(a.file, encoding="utf-8") if l.strip()] | |
| if "-" in texts: | |
| texts = [t for t in texts if t != "-"] + [l.rstrip("\n") for l in sys.stdin if l.strip()] | |
| texts = [t for t in texts if t.strip()] | |
| if not texts: | |
| texts = ["EPDM membrane roofing"] | |
| tr, tf = Transformer(a.model), Tfidf(a.tfidf) | |
| assert tf.classes == list(range(len(tr.labels))), "TF-IDF classes do not align with the transformer labels" | |
| lp = norm(w * tr.logprobs(texts) + (1 - w) * tf.logprobs(texts)) # blend of log-probs, renormalised | |
| results = [] | |
| for text, row in zip(texts, lp): | |
| order = np.argsort(-row)[:a.top_k] | |
| preds = [{"code": tr.labels[int(i)][:8].strip(), "label": tr.labels[int(i)], | |
| "name": tr.labels[int(i)][8:].strip(), "score": round(float(np.exp(row[i])), 4)} for i in order] | |
| item = {"text": text, "weight_transformer": w, "predictions": preds} | |
| if a.divisions: | |
| tot = {} | |
| for p in preds: | |
| tot[p["code"][:2]] = tot.get(p["code"][:2], 0.0) + p["score"] | |
| item["divisions"] = [{"code": d, "name": DIVISIONS.get(d, d), "score": round(s, 4)} | |
| for d, s in sorted(tot.items(), key=lambda kv: -kv[1])] | |
| results.append(item) | |
| if a.json: | |
| json.dump(results, sys.stdout, indent=2, ensure_ascii=False) | |
| print() | |
| return | |
| for r in results: | |
| print(f"\n{r['text']} (blend, transformer weight {w:g})") | |
| for p in r["predictions"]: | |
| print(f" {p['code']} {p['name']:<48.48} {p['score']:.3f}") | |
| if a.divisions: | |
| print(" -- divisions --") | |
| for d in r.get("divisions", []): | |
| print(f" {d['code']} {d['name']:<48.48} {d['score']:.3f}") | |
| if __name__ == "__main__": | |
| main() | |