File size: 3,775 Bytes
5fd640d
d1d57e3
5fd640d
 
 
 
 
d1d57e3
5fd640d
d1d57e3
5fd640d
 
d1d57e3
 
5fd640d
 
 
 
 
 
 
d1d57e3
 
 
5fd640d
d1d57e3
5fd640d
d1d57e3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5fd640d
 
d1d57e3
 
5fd640d
d1d57e3
5fd640d
d1d57e3
 
 
 
5fd640d
d1d57e3
 
5fd640d
d1d57e3
5fd640d
d1d57e3
 
 
5fd640d
d1d57e3
 
5fd640d
d1d57e3
 
 
 
 
 
 
5fd640d
d1d57e3
5fd640d
d1d57e3
 
5fd640d
d1d57e3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5fd640d
d1d57e3
 
 
 
5fd640d
 
d1d57e3
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
"""Minimal plotting helpers used by api.py.

Only `plot_label_distribution` and `plot_label_cooccurrence` are exercised in
the Space; the full visualization module from the source repo pulls in many
heavy deps (seaborn, networkx, umap, levenshtein, internal modules) we don't
need here.
"""
import ast
from collections import Counter

import numpy as np
import plotly.graph_objects as go


def _parse_labels(labels_str):
    if labels_str is None or labels_str == "":
        return []
    try:
        return ast.literal_eval(str(labels_str))
    except (ValueError, SyntaxError):
        return []


def plot_label_distribution(df, taxonomy, output_path: str):
    """Stacked bar chart: per-label total count split into 'alone' vs 'with others'."""
    df_parsed = df.copy()
    df_parsed["Labels"] = df_parsed["Labels"].apply(_parse_labels)

    label_counts = Counter([label for labels in df_parsed["Labels"] for label in labels])
    alone_counts = {}
    for label in label_counts.keys():
        alone_counts[label] = df_parsed["Labels"].apply(
            lambda lst: isinstance(lst, list) and len(lst) == 1 and lst[0] == label
        ).sum()

    sorted_labels = [label for label in sorted(label_counts) if label in taxonomy]
    alone = [alone_counts.get(label, 0) for label in sorted_labels]
    with_others = [label_counts[label] - alone_counts.get(label, 0) for label in sorted_labels]
    totals = [label_counts[label] for label in sorted_labels]
    x_labels = [f"{taxonomy[label]['category']} ({label})" for label in sorted_labels]

    fig = go.Figure(data=[
        go.Bar(name="Alone", x=x_labels, y=alone, marker_color="navy"),
        go.Bar(name="With Other Labels", x=x_labels, y=with_others, marker_color="orange"),
    ])

    for x, t, a in zip(x_labels, totals, alone):
        fig.add_annotation(
            x=x, y=t,
            text=f"Total: {t}<br>Alone: {a}",
            showarrow=False,
            yshift=15,
            font=dict(size=13, color="black"),
            align="center",
        )

    y_max = max([a + w for a, w in zip(alone, with_others)]) * 1.2 if alone else 10
    fig.update_layout(
        barmode="stack",
        title="Distribution of Edit Types Across Revisions",
        xaxis_title="Edit Type",
        yaxis_title="Number of Section Revisions",
        yaxis=dict(range=[0, y_max]),
        xaxis=dict(tickmode="linear"),
        font=dict(size=15),
        legend=dict(title="Occurrence Type"),
    )

    fig.write_image(output_path, width=1920, height=1080, scale=2)
    return output_path


def plot_label_cooccurrence(df, taxonomy, output_path: str):
    """Heatmap of label co-occurrence within the same revision."""
    df_parsed = df.copy()
    df_parsed["Labels"] = df_parsed["Labels"].apply(_parse_labels)

    sorted_keys = sorted(taxonomy.keys())
    label_titles = [taxonomy[key]["category"] for key in sorted_keys]
    num_labels = len(sorted_keys)
    co_occurrence_matrix = np.zeros((num_labels, num_labels), dtype=int)

    for labels in df_parsed["Labels"]:
        for i in labels:
            for j in labels:
                if i != j and i in sorted_keys and j in sorted_keys:
                    co_occurrence_matrix[i - 1, j - 1] += 1

    fig = go.Figure(data=go.Heatmap(
        z=co_occurrence_matrix,
        x=label_titles,
        y=label_titles,
        colorscale="Blues",
        text=co_occurrence_matrix,
        texttemplate="%{text}",
    ))

    fig.update_layout(
        title="Co-occurrence of Edit Types in Revisions",
        xaxis=dict(title="Edit Type", tickangle=45, tickfont=dict(size=12)),
        yaxis=dict(title="Co-occurring Edit Type", tickfont=dict(size=12)),
    )

    fig.write_image(output_path, width=1920, height=1080, scale=2)
    return output_path