File size: 6,726 Bytes
9d4c04e
 
 
ad1145a
 
 
9d4c04e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
ad1145a
 
 
 
9d4c04e
 
 
 
ad1145a
 
9d4c04e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
"""Ensemble predictor: the mf-0.2 transformer + a TF-IDF/SGD scorer, blended in log-prob space.

The transformer alone reaches 0.597 top-1 on the 194 hand-labelled estimate line items; the fitted TF-IDF
scorer (word + char n-grams) reaches 0.675; the blend reaches 0.702 (and improves whole-section and
manual-chunk accuracy). This script reproduces the blend. The weight is chosen on the UFGS validation split
(see `blend.json`), not on the line-item test set.

    pip install transformers torch scikit-learn joblib

    python predict_ensemble.py "EPDM membrane roofing" "4000 psi concrete slab on grade"
    python predict_ensemble.py --top-k 5 --divisions "12\" RCP storm drain pipe"
    python predict_ensemble.py --json --file items.txt > out.json

    # default TF-IDF/results live beside this script; point --model at the Hub or a local run
    python predict_ensemble.py --model constructelligence/masterformat-classifier \
        --tfidf tfidf.joblib --weight 0.2 "Wet pipe sprinkler system, light hazard"
"""
import argparse
import json
import sys
from pathlib import Path

import numpy as np

HERE = Path(__file__).resolve().parent
MODEL = "constructelligence/masterformat-classifier"

DIVISIONS = {
    "01": "General Requirements", "02": "Existing Conditions", "03": "Concrete",
    "04": "Masonry", "05": "Metals", "06": "Wood, Plastics, and Composites",
    "07": "Thermal and Moisture Protection", "08": "Openings", "09": "Finishes",
    "10": "Specialties", "11": "Equipment", "12": "Furnishings",
    "13": "Special Construction", "14": "Conveying Equipment", "21": "Fire Suppression",
    "22": "Plumbing", "23": "HVAC", "25": "Integrated Automation", "26": "Electrical",
    "27": "Communications", "28": "Electronic Safety and Security", "31": "Earthwork",
    "32": "Exterior Improvements", "33": "Utilities", "34": "Transportation",
    "35": "Waterway and Marine Construction", "40": "Process Interconnections",
    "41": "Material Processing and Handling Equipment",
    "43": "Process Gas and Liquid Handling, Purification, and Storage Equipment",
    "44": "Pollution and Waste Control Equipment", "46": "Water and Wastewater Equipment",
    "48": "Electrical Power Generation",
}


def norm(lp):
    return lp - np.logaddexp.reduce(lp, axis=1, keepdims=True)


class Transformer:
    def __init__(self, model, batch=64):
        import torch
        from transformers import AutoModelForSequenceClassification, AutoTokenizer
        self.torch = torch
        self.tok = AutoTokenizer.from_pretrained(model)
        self.m = AutoModelForSequenceClassification.from_pretrained(model).eval()
        self.labels = {int(k): v for k, v in self.m.config.id2label.items()}
        self.batch = batch

    def logprobs(self, texts):
        out = []
        with self.torch.no_grad():
            for i in range(0, len(texts), self.batch):
                enc = self.tok(texts[i:i + self.batch], truncation=True, max_length=128,
                               padding=True, return_tensors="pt")
                logits = self.m(**enc).logits.float()
                out.append(self.torch.log_softmax(logits, -1).numpy())
        return norm(np.concatenate(out))


class Tfidf:
    def __init__(self, path):
        import joblib
        d = joblib.load(path)
        if isinstance(d, dict):                 # legacy {vectorizer, classifier}
            self.pipe, self.vec, self.clf = None, d["vectorizer"], d["classifier"]
        else:                                   # sklearn Pipeline
            self.pipe, self.vec, self.clf = d, None, d.named_steps["clf"]
        # classifier classes_ are indices into the sorted label list; verify against the transformer later
        self.classes = list(self.clf.classes_)

    def logprobs(self, texts):
        if self.pipe is not None:
            return norm(self.pipe.predict_log_proba(texts))
        return norm(self.clf.predict_log_proba(self.vec.transform(texts)))


def main():
    ap = argparse.ArgumentParser(description="MasterFormat ensemble (mf-0.2 + TF-IDF).")
    ap.add_argument("text", nargs="*", help="text to classify; '-' reads stdin")
    ap.add_argument("--model", default=MODEL, help="HF repo id or local checkpoint dir")
    ap.add_argument("--tfidf", default=str(HERE / "tfidf.joblib"))
    ap.add_argument("--weight", type=float, default=None,
                    help="transformer weight in the blend (default: blend.json beside --tfidf)")
    ap.add_argument("--top-k", type=int, default=3)
    ap.add_argument("--divisions", action="store_true")
    ap.add_argument("--file")
    ap.add_argument("--json", action="store_true")
    a = ap.parse_args()

    if a.weight is None:
        bf = Path(a.tfidf).with_name("blend.json")
        a.weight = json.loads(bf.read_text())["weight"] if bf.exists() else 0.2
    w = a.weight

    texts = list(a.text)
    if a.file:
        texts += [l.rstrip("\n") for l in open(a.file, encoding="utf-8") if l.strip()]
    if "-" in texts:
        texts = [t for t in texts if t != "-"] + [l.rstrip("\n") for l in sys.stdin if l.strip()]
    texts = [t for t in texts if t.strip()]
    if not texts:
        texts = ["EPDM membrane roofing"]

    tr, tf = Transformer(a.model), Tfidf(a.tfidf)
    assert tf.classes == list(range(len(tr.labels))), "TF-IDF classes do not align with the transformer labels"
    lp = norm(w * tr.logprobs(texts) + (1 - w) * tf.logprobs(texts))   # blend of log-probs, renormalised

    results = []
    for text, row in zip(texts, lp):
        order = np.argsort(-row)[:a.top_k]
        preds = [{"code": tr.labels[int(i)][:8].strip(), "label": tr.labels[int(i)],
                  "name": tr.labels[int(i)][8:].strip(), "score": round(float(np.exp(row[i])), 4)} for i in order]
        item = {"text": text, "weight_transformer": w, "predictions": preds}
        if a.divisions:
            tot = {}
            for p in preds:
                tot[p["code"][:2]] = tot.get(p["code"][:2], 0.0) + p["score"]
            item["divisions"] = [{"code": d, "name": DIVISIONS.get(d, d), "score": round(s, 4)}
                                 for d, s in sorted(tot.items(), key=lambda kv: -kv[1])]
        results.append(item)

    if a.json:
        json.dump(results, sys.stdout, indent=2, ensure_ascii=False)
        print()
        return
    for r in results:
        print(f"\n{r['text']}  (blend, transformer weight {w:g})")
        for p in r["predictions"]:
            print(f"  {p['code']} {p['name']:<48.48} {p['score']:.3f}")
        if a.divisions:
            print("  -- divisions --")
            for d in r.get("divisions", []):
                print(f"  {d['code']}      {d['name']:<48.48} {d['score']:.3f}")


if __name__ == "__main__":
    main()