Sentence Similarity
sentence-transformers
Safetensors
bert
feature-extraction
compliance
nist-800-53
hipaa
cybersecurity
text-embeddings-inference
Instructions to use stetteh/regmap-embedder with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- sentence-transformers
How to use stetteh/regmap-embedder with sentence-transformers:
from sentence_transformers import SentenceTransformer model = SentenceTransformer("stetteh/regmap-embedder") sentences = [ "That is a happy person", "That is a happy dog", "That is a very happy person", "Today is a sunny day" ] embeddings = model.encode(sentences) similarities = model.similarity(embeddings, embeddings) print(similarities.shape) # [4, 4] - Notebooks
- Google Colab
- Kaggle
File size: 2,216 Bytes
1936541 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 | """
RegMap inference wrapper — map a NIST SP 800-53 control to the most relevant HIPAA Security Rule
provisions. Loads the bundled fine-tuned embedder + the bundled HIPAA corpus and returns the top-k
citations by cosine similarity.
Usage:
from regmap_map import map_control
map_control("Enforce multi-factor authentication for remote access.", top_k=5)
# or from the command line:
python regmap_map.py "Employ integrity verification tools to detect unauthorized changes."
This file lives inside the model directory, so the model, this wrapper, and hipaa_corpus.csv all
sit together and the package works out of the box.
"""
import csv
import json
import os
import sys
from functools import lru_cache
_HERE = os.path.dirname(os.path.abspath(__file__))
_CORPUS = os.path.join(_HERE, "hipaa_corpus.csv")
@lru_cache(maxsize=1)
def _load():
from sentence_transformers import SentenceTransformer
model = SentenceTransformer(_HERE) # model files are in this directory
citations, texts = [], []
with open(_CORPUS, encoding="utf-8") as f:
for row in csv.DictReader(f):
if row.get("hipaa_text"):
citations.append(row.get("hipaa_citation", ""))
texts.append(row["hipaa_text"])
emb = model.encode(texts, convert_to_tensor=True, normalize_embeddings=True)
return model, citations, texts, emb
def map_control(control_text: str, top_k: int = 5):
"""Return the top-k HIPAA provisions for a NIST control description:
[{"hipaa_citation", "hipaa_text", "score"}], most similar first."""
from sentence_transformers import util
import torch
model, citations, texts, emb = _load()
q = model.encode(control_text.strip(), convert_to_tensor=True, normalize_embeddings=True)
scores = util.cos_sim(q, emb)[0]
k = min(top_k, len(citations))
idx = torch.topk(scores, k=k).indices.tolist()
return [{"hipaa_citation": citations[i], "hipaa_text": texts[i],
"score": round(float(scores[i]), 4)} for i in idx]
if __name__ == "__main__":
query = " ".join(sys.argv[1:]) or "Enforce multi-factor authentication for remote access."
print(json.dumps(map_control(query), indent=2))
|