File size: 4,434 Bytes
b611f38
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
"""
Word-Level, Character-Level Diff Engine & Consensus Matrix Generator
Uses difflib.SequenceMatcher to compare OCR outputs across multiple models.
Strictly processes ONLY valid successful OCR outputs.
"""

import difflib
import re
from typing import Dict, Any, List, Tuple, Optional


def tokenize_text(text: Optional[str], mode: str = "word") -> List[str]:
    """
    Tokenizes text into words or characters for diffing.
    """
    if not text or not isinstance(text, str):
        return []
    if mode == "char":
        return list(text)
    # Word tokenization (splits by whitespace while preserving punctuation boundaries)
    return re.findall(r"\S+|\n", text)


def compute_similarity_ratio(text_a: Optional[str], text_b: Optional[str]) -> Optional[float]:
    """
    Computes normalized similarity ratio between two valid texts [0.0 - 1.0].
    Returns None if either text is invalid/empty.
    """
    if text_a is None or text_b is None:
        return None
    if not text_a and not text_b:
        return 1.0
    matcher = difflib.SequenceMatcher(None, text_a, text_b)
    return round(matcher.ratio(), 4)


def compute_diff(text_a: Optional[str], text_b: Optional[str], mode: str = "word") -> List[Dict[str, Any]]:
    """
    Computes detailed word or character level diff between text_a (baseline) and text_b (comparison).
    Returns list of chunks with type: 'equal' | 'delete' | 'insert' | 'replace'.
    """
    if text_a is None and text_b is None:
        return []
    
    tokens_a = tokenize_text(text_a or "", mode=mode)
    tokens_b = tokenize_text(text_b or "", mode=mode)

    matcher = difflib.SequenceMatcher(None, tokens_a, tokens_b)
    diff_chunks = []

    for tag, i1, i2, j1, j2 in matcher.get_opcodes():
        chunk_a = " ".join(tokens_a[i1:i2]) if mode == "word" else "".join(tokens_a[i1:i2])
        chunk_b = " ".join(tokens_b[j1:j2]) if mode == "word" else "".join(tokens_b[j1:j2])

        diff_chunks.append({
            "tag": tag,
            "text_a": chunk_a,
            "text_b": chunk_b,
            "span_a": [i1, i2],
            "span_b": [j1, j2]
        })

    return diff_chunks


def generate_consensus_matrix(model_outputs: Dict[str, str]) -> Dict[str, Any]:
    """
    Builds a pairwise similarity and agreement matrix ONLY across valid successful models.
    Failed/Error models must be filtered out before calling or are skipped.
    """
    # Filter out empty or None texts
    valid_outputs = {k: v for k, v in model_outputs.items() if v and isinstance(v, str) and v.strip()}
    model_names = list(valid_outputs.keys())
    n = len(model_names)

    if n == 0:
        return {
            "models": [],
            "matrix": {},
            "average_agreement_pct": {},
            "consensus_leader": None,
            "disagreements": [],
            "status": "NO_VALID_MODELS"
        }

    matrix = {}
    average_agreement = {}

    for name_a in model_names:
        matrix[name_a] = {}
        total_sim = 0.0
        comparisons = 0

        for name_b in model_names:
            if name_a == name_b:
                sim = 1.0
            else:
                sim = compute_similarity_ratio(valid_outputs[name_a], valid_outputs[name_b]) or 0.0
                total_sim += sim
                comparisons += 1
            matrix[name_a][name_b] = round(sim * 100, 1)

        if comparisons > 0:
            average_agreement[name_a] = round((total_sim / comparisons) * 100, 1)
        else:
            average_agreement[name_a] = 100.0  # Only 1 model exists

    # Determine highest consensus model
    consensus_leader = max(average_agreement.items(), key=lambda x: x[1])[0] if average_agreement else None

    # Flag genuine disagreements among valid models
    disagreements = []
    for i in range(n):
        for j in range(i + 1, n):
            mA = model_names[i]
            mB = model_names[j]
            sim = matrix[mA][mB]
            if sim < 75.0:
                disagreements.append({
                    "model_a": mA,
                    "model_b": mB,
                    "similarity_pct": sim,
                    "severity": "high" if sim < 50.0 else "medium"
                })

    return {
        "models": model_names,
        "matrix": matrix,
        "average_agreement_pct": average_agreement,
        "consensus_leader": consensus_leader,
        "disagreements": disagreements,
        "status": "VALID"
    }