smartdoc-ner / modules /analytics.py
Nafees Ahmad
SmartDoc NER Updated
fbd78fc
Raw
History Blame Contribute Delete
2.98 kB
"""
Analytics Module — NEW feature not in original project
- Entity frequency distribution
- Co-occurrence analysis (which persons appear with which orgs)
- Confidence score statistics
- Entity deduplication with canonical form
"""
from collections import Counter, defaultdict
import re
def deduplicate_entities(entities: list[dict]) -> list[dict]:
"""
Remove duplicate mentions of the same entity.
'Barack Obama' and 'Obama' are treated as separate (conservative approach).
Case-insensitive deduplication within exact matches.
"""
seen = {}
result = []
for ent in entities:
key = (ent["word"].lower().strip(), ent["label"])
if key not in seen:
seen[key] = True
result.append(ent)
return result
def entity_frequency(entities: list[dict]) -> dict[str, Counter]:
"""
Returns per-label frequency counters.
e.g. {"Person": Counter({"Nafees Ahmad": 3, "Ali": 1}), ...}
"""
freq = defaultdict(Counter)
for ent in entities:
freq[ent["label"]][ent["word"]] += 1
return dict(freq)
def top_entities(entities: list[dict], top_n: int = 5) -> dict[str, list]:
"""Returns top N entities per label sorted by frequency."""
freq = entity_frequency(entities)
result = {}
for label, counter in freq.items():
result[label] = counter.most_common(top_n)
return result
def co_occurrence(entities: list[dict]) -> list[tuple]:
"""
Simple co-occurrence: find Person-Organization pairs that appear
in the same document. Useful for relationship extraction heuristic.
Returns list of (person, org) tuples.
"""
persons = [e["word"] for e in entities if e["label"] == "Person"]
orgs = [e["word"] for e in entities if e["label"] == "Organization"]
pairs = []
for person in set(persons):
for org in set(orgs):
pairs.append((person, org))
return pairs[:20] # limit output
def confidence_stats(entities: list[dict]) -> dict:
"""Return average, min, max confidence scores."""
if not entities:
return {"avg": 0, "min": 0, "max": 0, "total": 0}
scores = [e["score"] for e in entities]
return {
"avg": round(sum(scores) / len(scores), 1),
"min": round(min(scores), 1),
"max": round(max(scores), 1),
"total": len(entities),
}
def build_summary_table(entities: list[dict]) -> list[list]:
"""
Returns rows for Gradio DataFrame display.
Columns: Entity, Type, Confidence (%)
Sorted by confidence descending.
"""
sorted_ents = sorted(entities, key=lambda e: e["score"], reverse=True)
rows = [
[e["word"], e["label"], f'{e["score"]}%']
for e in sorted_ents
]
return rows
def label_counts(entities: list[dict]) -> dict[str, int]:
"""Count of entities per label."""
counts = defaultdict(int)
for e in entities:
counts[e["label"]] += 1
return dict(counts)