Text Classification
Transformers
Joblib
ONNX
Safetensors
English
bert
masterformat
masterformat-classifier
csi-masterformat
construction
construction-technology
sequence-classification
tfidf
ensemble
specs
spec-writing
specifications
takeoff
estimating
cost-code
ufgs
public-domain
Eval Results (legacy)
text-embeddings-inference
Instructions to use constructelligence/masterformat-classifier with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use constructelligence/masterformat-classifier with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-classification", model="constructelligence/masterformat-classifier")# pip install -U transformers accelerate # Load model directly from transformers import AutoTokenizer, AutoModelForSequenceClassification tokenizer = AutoTokenizer.from_pretrained("constructelligence/masterformat-classifier") model = AutoModelForSequenceClassification.from_pretrained("constructelligence/masterformat-classifier", device_map="auto") - Notebooks
- Google Colab
- Kaggle
File size: 6,726 Bytes
9d4c04e ad1145a 9d4c04e ad1145a 9d4c04e ad1145a 9d4c04e | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 | """Ensemble predictor: the mf-0.2 transformer + a TF-IDF/SGD scorer, blended in log-prob space.
The transformer alone reaches 0.597 top-1 on the 194 hand-labelled estimate line items; the fitted TF-IDF
scorer (word + char n-grams) reaches 0.675; the blend reaches 0.702 (and improves whole-section and
manual-chunk accuracy). This script reproduces the blend. The weight is chosen on the UFGS validation split
(see `blend.json`), not on the line-item test set.
pip install transformers torch scikit-learn joblib
python predict_ensemble.py "EPDM membrane roofing" "4000 psi concrete slab on grade"
python predict_ensemble.py --top-k 5 --divisions "12\" RCP storm drain pipe"
python predict_ensemble.py --json --file items.txt > out.json
# default TF-IDF/results live beside this script; point --model at the Hub or a local run
python predict_ensemble.py --model constructelligence/masterformat-classifier \
--tfidf tfidf.joblib --weight 0.2 "Wet pipe sprinkler system, light hazard"
"""
import argparse
import json
import sys
from pathlib import Path
import numpy as np
HERE = Path(__file__).resolve().parent
MODEL = "constructelligence/masterformat-classifier"
DIVISIONS = {
"01": "General Requirements", "02": "Existing Conditions", "03": "Concrete",
"04": "Masonry", "05": "Metals", "06": "Wood, Plastics, and Composites",
"07": "Thermal and Moisture Protection", "08": "Openings", "09": "Finishes",
"10": "Specialties", "11": "Equipment", "12": "Furnishings",
"13": "Special Construction", "14": "Conveying Equipment", "21": "Fire Suppression",
"22": "Plumbing", "23": "HVAC", "25": "Integrated Automation", "26": "Electrical",
"27": "Communications", "28": "Electronic Safety and Security", "31": "Earthwork",
"32": "Exterior Improvements", "33": "Utilities", "34": "Transportation",
"35": "Waterway and Marine Construction", "40": "Process Interconnections",
"41": "Material Processing and Handling Equipment",
"43": "Process Gas and Liquid Handling, Purification, and Storage Equipment",
"44": "Pollution and Waste Control Equipment", "46": "Water and Wastewater Equipment",
"48": "Electrical Power Generation",
}
def norm(lp):
return lp - np.logaddexp.reduce(lp, axis=1, keepdims=True)
class Transformer:
def __init__(self, model, batch=64):
import torch
from transformers import AutoModelForSequenceClassification, AutoTokenizer
self.torch = torch
self.tok = AutoTokenizer.from_pretrained(model)
self.m = AutoModelForSequenceClassification.from_pretrained(model).eval()
self.labels = {int(k): v for k, v in self.m.config.id2label.items()}
self.batch = batch
def logprobs(self, texts):
out = []
with self.torch.no_grad():
for i in range(0, len(texts), self.batch):
enc = self.tok(texts[i:i + self.batch], truncation=True, max_length=128,
padding=True, return_tensors="pt")
logits = self.m(**enc).logits.float()
out.append(self.torch.log_softmax(logits, -1).numpy())
return norm(np.concatenate(out))
class Tfidf:
def __init__(self, path):
import joblib
d = joblib.load(path)
if isinstance(d, dict): # legacy {vectorizer, classifier}
self.pipe, self.vec, self.clf = None, d["vectorizer"], d["classifier"]
else: # sklearn Pipeline
self.pipe, self.vec, self.clf = d, None, d.named_steps["clf"]
# classifier classes_ are indices into the sorted label list; verify against the transformer later
self.classes = list(self.clf.classes_)
def logprobs(self, texts):
if self.pipe is not None:
return norm(self.pipe.predict_log_proba(texts))
return norm(self.clf.predict_log_proba(self.vec.transform(texts)))
def main():
ap = argparse.ArgumentParser(description="MasterFormat ensemble (mf-0.2 + TF-IDF).")
ap.add_argument("text", nargs="*", help="text to classify; '-' reads stdin")
ap.add_argument("--model", default=MODEL, help="HF repo id or local checkpoint dir")
ap.add_argument("--tfidf", default=str(HERE / "tfidf.joblib"))
ap.add_argument("--weight", type=float, default=None,
help="transformer weight in the blend (default: blend.json beside --tfidf)")
ap.add_argument("--top-k", type=int, default=3)
ap.add_argument("--divisions", action="store_true")
ap.add_argument("--file")
ap.add_argument("--json", action="store_true")
a = ap.parse_args()
if a.weight is None:
bf = Path(a.tfidf).with_name("blend.json")
a.weight = json.loads(bf.read_text())["weight"] if bf.exists() else 0.2
w = a.weight
texts = list(a.text)
if a.file:
texts += [l.rstrip("\n") for l in open(a.file, encoding="utf-8") if l.strip()]
if "-" in texts:
texts = [t for t in texts if t != "-"] + [l.rstrip("\n") for l in sys.stdin if l.strip()]
texts = [t for t in texts if t.strip()]
if not texts:
texts = ["EPDM membrane roofing"]
tr, tf = Transformer(a.model), Tfidf(a.tfidf)
assert tf.classes == list(range(len(tr.labels))), "TF-IDF classes do not align with the transformer labels"
lp = norm(w * tr.logprobs(texts) + (1 - w) * tf.logprobs(texts)) # blend of log-probs, renormalised
results = []
for text, row in zip(texts, lp):
order = np.argsort(-row)[:a.top_k]
preds = [{"code": tr.labels[int(i)][:8].strip(), "label": tr.labels[int(i)],
"name": tr.labels[int(i)][8:].strip(), "score": round(float(np.exp(row[i])), 4)} for i in order]
item = {"text": text, "weight_transformer": w, "predictions": preds}
if a.divisions:
tot = {}
for p in preds:
tot[p["code"][:2]] = tot.get(p["code"][:2], 0.0) + p["score"]
item["divisions"] = [{"code": d, "name": DIVISIONS.get(d, d), "score": round(s, 4)}
for d, s in sorted(tot.items(), key=lambda kv: -kv[1])]
results.append(item)
if a.json:
json.dump(results, sys.stdout, indent=2, ensure_ascii=False)
print()
return
for r in results:
print(f"\n{r['text']} (blend, transformer weight {w:g})")
for p in r["predictions"]:
print(f" {p['code']} {p['name']:<48.48} {p['score']:.3f}")
if a.divisions:
print(" -- divisions --")
for d in r.get("divisions", []):
print(f" {d['code']} {d['name']:<48.48} {d['score']:.3f}")
if __name__ == "__main__":
main()
|