ProCreations's picture
Publish validated native logbook bundle
c279da6 verified
Raw
History Blame Contribute Delete
15.5 kB
{
"all_scientific_gates_pass": true,
"claim_2_val5": {
"absolute_gain": 0.06,
"absolute_gain_over_majority_vote": 0.15000000000000002,
"largest_reproduced_absolute_gain_dataset": "Val5",
"majority_vote": 0.31,
"ptbcc": 0.46,
"relative_gain": 0.15,
"strongest_reproduced_baseline": 0.4
},
"claim_3_falsification": {
"absolute_shortfall": 0.02355063891614273,
"aircr_accuracy_required_to_reach_reported_macro": 0.9827063891614269,
"measured_ten_dataset_ptbcc": 0.7236493610838572,
"reported_ptbcc": 0.7472
},
"claim_4_falsification": {
"macro_by_prototypes": {
"2": 0.7236493610838572,
"3": 0.7250569980641843,
"4": 0.725806675063574
},
"peak": 4
},
"claim_5_cost": {
"interpretation": "This is an exact comparison of learned confusion-structure parameters, the computational-work term reduced by prototype sharing. Environment-dependent wall-clock time is deliberately not used as scored evidence.",
"pooled_ibcc_free_parameters": 71650,
"pooled_ptbcc_free_parameters": 4354,
"ratio": 0.060767620376831824,
"reduction": 0.9392323796231682,
"registered_dataset_rows": 11
},
"claims": [
{
"claim": 1,
"literal_claim": "PTBCC (Prototype-driven Bayesian Classifier Combination) models annotators via a shared set of prototype confusion matrices rather than learning one confusion matrix per annotator (Section on method overview)."
},
{
"claim": 2,
"literal_claim": "PTBCC achieves up to 15% accuracy improvement over the best baseline in its best-case dataset (Val5) (Table 4)."
},
{
"claim": 3,
"literal_claim": "Across 11 real-world crowdsourcing datasets, PTBCC attains an average accuracy of 0.7472, versus 0.7175 for FGBCC, 0.7132 for BWA, and 0.6986 for majority voting (Table 4)."
},
{
"claim": 4,
"literal_claim": "PTBCC's ablation over prototype set size |S| shows accuracy peaking at |S|=2 (0.7472) and degrading to 0.7300 at |S|=3 and 0.7271 at |S|=4 due to sparser per-prototype annotator distributions (Table 5)."
},
{
"claim": 5,
"literal_claim": "PTBCC uses less than 10% of the computational cost of confusion-matrix-based baselines while matching or exceeding their accuracy (Section on computational efficiency)."
}
],
"dataset_count": 10,
"macros": {
"BWA": 0.701003255458328,
"DS": 0.7016899769890417,
"IBCC": 0.6935982786099606,
"MV": 0.6937261736117043,
"PTBCC_S2": 0.7236493610838572,
"PTBCC_S3_mean_3_seeds": 0.7250569980641843,
"PTBCC_S4_mean_3_seeds": 0.725806675063574
},
"mechanism": {
"dominant_prototype_recall_mean": 0.9993333333333333,
"dominant_prototype_recall_min": 0.98,
"label_shuffle_accuracy_drop": 0.772111111111111,
"label_shuffled_ptbcc_accuracy_mean": 0.2081111111111111,
"majority_vote_accuracy_mean": 0.9724444444444444,
"mean_max_annotator_weight": 0.5953221954840179,
"prototype_mae_max": 0.07200504499687686,
"prototype_mae_mean": 0.04819687934303456,
"ptbcc_accuracy_mean": 0.9802222222222221,
"ptbcc_beats_mv_seeds": 29,
"seeds": 30
},
"paper_orid": "KJq0iScNM6",
"per_dataset": {
"Adult": {
"iterations": {
"BWA": [
41,
18,
9,
13
],
"DS": 47,
"IBCC": 100,
"PTBCC_S2": 22,
"PTBCC_S3": [
42,
26,
28
],
"PTBCC_S4": [
37,
54,
55
]
},
"scores": {
"BWA": 0.7417417417417418,
"DS": 0.7627627627627628,
"IBCC": 0.7447447447447447,
"MV": 0.7597597597597597,
"PTBCC_S2": 0.7687687687687688,
"PTBCC_S3_mean_3_seeds": 0.7727727727727727,
"PTBCC_S4_mean_3_seeds": 0.7747747747747749
},
"statistics": {
"annotators": 825,
"classes": 4,
"labels": 89799,
"source_sha256": {
"truth-inference-at-scale/data/crowd_truth_inference/s5_AdultContent/label.csv": "d72988724f7cdd628ce21fa1aaabecc4d03aef21bf59644d9e4033b774ac9a32",
"truth-inference-at-scale/data/crowd_truth_inference/s5_AdultContent/truth.csv": "1c52c5d528d16e031360d87ce384a80611bf23dedd5c7cfddf2c8ffc058772cb"
},
"tasks": 11040,
"truths": 333
}
},
"CF": {
"iterations": {
"BWA": [
5,
5,
5,
5,
6
],
"DS": 60,
"IBCC": 18,
"PTBCC_S2": 103,
"PTBCC_S3": [
38,
66,
32
],
"PTBCC_S4": [
38,
31,
25
]
},
"scores": {
"BWA": 0.8933333333333333,
"DS": 0.7866666666666666,
"IBCC": 0.8833333333333333,
"MV": 0.9,
"PTBCC_S2": 0.8833333333333333,
"PTBCC_S3_mean_3_seeds": 0.8755555555555556,
"PTBCC_S4_mean_3_seeds": 0.8844444444444445
},
"statistics": {
"annotators": 461,
"classes": 5,
"labels": 1720,
"source_sha256": {
"truth-inference-at-scale/data/active-crowd-toolkit/CF/label.csv": "b0f98ddc9afefffcaf9670f591a3441b1bfc3eab70f8e4c35746310962920f06",
"truth-inference-at-scale/data/active-crowd-toolkit/CF/truth.csv": "8d8f1b773b310fa91e550658ef88baedc4c522ccd395a1ffa9c185327b88080f"
},
"tasks": 300,
"truths": 300
}
},
"Dog": {
"iterations": {
"BWA": [
9,
9,
9,
9
],
"DS": 15,
"IBCC": 25,
"PTBCC_S2": 33,
"PTBCC_S3": [
30,
49,
42
],
"PTBCC_S4": [
35,
38,
51
]
},
"scores": {
"BWA": 0.8314745972738538,
"DS": 0.8426270136307311,
"IBCC": 0.838909541511772,
"MV": 0.8178438661710037,
"PTBCC_S2": 0.8240396530359355,
"PTBCC_S3_mean_3_seeds": 0.8244527054935977,
"PTBCC_S4_mean_3_seeds": 0.8244527054935977
},
"statistics": {
"annotators": 109,
"classes": 4,
"labels": 8070,
"source_sha256": {
"truth-inference-at-scale/data/crowd_truth_inference/s4_Dog data/label.csv": "c240c0efc442e4936b35f28424d66fb04e687c0f29b254af94c26e15a71aa820",
"truth-inference-at-scale/data/crowd_truth_inference/s4_Dog data/truth.csv": "b299494a7aba3cf5abcd3b0e002f6e899f3e1f1d864a26703325db355555cb76"
},
"tasks": 807,
"truths": 807
}
},
"Face": {
"iterations": {
"BWA": [
6,
7,
7,
10
],
"DS": 20,
"IBCC": 19,
"PTBCC_S2": 17,
"PTBCC_S3": [
19,
20,
18
],
"PTBCC_S4": [
27,
23,
20
]
},
"scores": {
"BWA": 0.6181506849315068,
"DS": 0.6421232876712328,
"IBCC": 0.6404109589041096,
"MV": 0.6301369863013698,
"PTBCC_S2": 0.6523972602739726,
"PTBCC_S3_mean_3_seeds": 0.6529680365296804,
"PTBCC_S4_mean_3_seeds": 0.6523972602739726
},
"statistics": {
"annotators": 27,
"classes": 4,
"labels": 5242,
"source_sha256": {
"truth-inference-at-scale/data/crowd_truth_inference/s4_Face Sentiment Identification/label.csv": "fc10f625183432f6ed82d783ec0afb4a2e79009843a1c20757baad5c4e469686",
"truth-inference-at-scale/data/crowd_truth_inference/s4_Face Sentiment Identification/truth.csv": "503987498b22b586a34bd9211cce88bf4fbb436396ccae3fd5e19a64a5a8dee9"
},
"tasks": 584,
"truths": 584
}
},
"Fact": {
"iterations": {
"BWA": [
21,
22,
14
],
"DS": 59,
"IBCC": 18,
"PTBCC_S2": 16,
"PTBCC_S3": [
14,
16,
17
],
"PTBCC_S4": [
22,
16,
15
]
},
"scores": {
"BWA": 0.8871527777777778,
"DS": 0.8524305555555556,
"IBCC": 0.8767361111111112,
"MV": 0.9010416666666666,
"PTBCC_S2": 0.9010416666666666,
"PTBCC_S3_mean_3_seeds": 0.9010416666666666,
"PTBCC_S4_mean_3_seeds": 0.9010416666666666
},
"statistics": {
"annotators": 57,
"classes": 3,
"labels": 214915,
"source_sha256": {
"truth-inference-at-scale/data/crowdscale2013/fact_eval/label.csv": "c8be134ea9adb9fe5984657dcd1894dce89ba93a539a600a62654c801d6e6d3a",
"truth-inference-at-scale/data/crowdscale2013/fact_eval/truth.csv": "5e0ea8af9d225cd4d03e175d279e5ab00958c6f74c3ebf85b068a271a83ad741"
},
"tasks": 42624,
"truths": 576
}
},
"MS": {
"iterations": {
"BWA": [
12,
9,
9,
22,
9,
11,
12,
7,
8,
9
],
"DS": 30,
"IBCC": 42,
"PTBCC_S2": 32,
"PTBCC_S3": [
43,
58,
75
],
"PTBCC_S4": [
116,
49,
94
]
},
"scores": {
"BWA": 0.7857142857142857,
"DS": 0.7742857142857142,
"IBCC": 0.79,
"MV": 0.71,
"PTBCC_S2": 0.7885714285714286,
"PTBCC_S3_mean_3_seeds": 0.787142857142857,
"PTBCC_S4_mean_3_seeds": 0.7890476190476191
},
"statistics": {
"annotators": 44,
"classes": 10,
"labels": 2945,
"source_sha256": {
"truth-inference-at-scale/data/active-crowd-toolkit/MS/label.csv": "4e59b88dbe48a0cd545a745c9718f3955729205d29da1bf883adc64bf7a3e378",
"truth-inference-at-scale/data/active-crowd-toolkit/MS/truth.csv": "981c635d5cde03cd903bf319b030178420bc8aa3b6a3b7846abea4e974bc2294"
},
"tasks": 700,
"truths": 700
}
},
"Senti": {
"iterations": {
"BWA": [
14,
16,
12,
16,
29
],
"DS": 81,
"IBCC": 107,
"PTBCC_S2": 17,
"PTBCC_S3": [
24,
21,
33
],
"PTBCC_S4": [
33,
41,
30
]
},
"scores": {
"BWA": 0.89,
"DS": 0.816,
"IBCC": 0.831,
"MV": 0.902,
"PTBCC_S2": 0.88,
"PTBCC_S3_mean_3_seeds": 0.8810000000000001,
"PTBCC_S4_mean_3_seeds": 0.882
},
"statistics": {
"annotators": 1960,
"classes": 5,
"labels": 569274,
"source_sha256": {
"truth-inference-at-scale/data/crowdscale2013/sentiment/label.csv": "d336fabd935d27081a3a769d455ea6deb60b9312cc9f024c061c04059caa3b60",
"truth-inference-at-scale/data/crowdscale2013/sentiment/truth.csv": "26b6526a8ff541c86abbf57fa8736e0b4aea007c16bea4b0170cfd23ac5866e5"
},
"tasks": 98980,
"truths": 1000
}
},
"Val5": {
"iterations": {
"BWA": [
6,
7,
5,
7,
6
],
"DS": 14,
"IBCC": 17,
"PTBCC_S2": 31,
"PTBCC_S3": [
60,
109,
156
],
"PTBCC_S4": [
51,
131,
106
]
},
"scores": {
"BWA": 0.32,
"DS": 0.4,
"IBCC": 0.34,
"MV": 0.31,
"PTBCC_S2": 0.46,
"PTBCC_S3_mean_3_seeds": 0.4633333333333333,
"PTBCC_S4_mean_3_seeds": 0.4466666666666666
},
"statistics": {
"annotators": 38,
"classes": 5,
"labels": 1000,
"source_sha256": {
"CrowdTI/truth_inference_crowd/datasets/f201_Emotion_FULL/answer.csv": "8a33684760a7fc51980ddc2d19d157f9d0197bcc3611cc4a2c8cdb65fdaf1d1b",
"CrowdTI/truth_inference_crowd/datasets/f201_Emotion_FULL/truth.csv": "0fdfa983818a3524ba572c7eb7a3fd6b314844105ae9449b630425faa3176be4"
},
"tasks": 100,
"truths": 100
}
},
"Val7": {
"iterations": {
"BWA": [
6,
7,
6,
5,
7,
6,
6
],
"DS": 16,
"IBCC": 23,
"PTBCC_S2": 89,
"PTBCC_S3": [
130,
73,
58
],
"PTBCC_S4": [
151,
96,
61
]
},
"scores": {
"BWA": 0.22,
"DS": 0.31,
"IBCC": 0.24,
"MV": 0.23,
"PTBCC_S2": 0.28,
"PTBCC_S3_mean_3_seeds": 0.2933333333333333,
"PTBCC_S4_mean_3_seeds": 0.3
},
"statistics": {
"annotators": 38,
"classes": 7,
"labels": 1000,
"source_sha256": {
"CrowdTI/truth_inference_crowd/datasets/f201_Emotion_FULL/answer.csv": "8a33684760a7fc51980ddc2d19d157f9d0197bcc3611cc4a2c8cdb65fdaf1d1b",
"CrowdTI/truth_inference_crowd/datasets/f201_Emotion_FULL/truth.csv": "0fdfa983818a3524ba572c7eb7a3fd6b314844105ae9449b630425faa3176be4"
},
"tasks": 100,
"truths": 100
}
},
"Web": {
"iterations": {
"BWA": [
17,
13,
8,
11,
20
],
"DS": 71,
"IBCC": 115,
"PTBCC_S2": 71,
"PTBCC_S3": [
38,
46,
61
],
"PTBCC_S4": [
39,
31,
62
]
},
"scores": {
"BWA": 0.8224651338107802,
"DS": 0.8300037693177534,
"IBCC": 0.7508480964945344,
"MV": 0.7764794572182435,
"PTBCC_S2": 0.7983415001884658,
"PTBCC_S3_mean_3_seeds": 0.798969719814047,
"PTBCC_S4_mean_3_seeds": 0.8032416132679985
},
"statistics": {
"annotators": 177,
"classes": 5,
"labels": 15567,
"source_sha256": {
"truth-inference-at-scale/data/SpectralMethodsMeetEM/web/label.csv": "90b6e66284288079fab48593c61fb7949f964d007fce077b13d21d0b7ab43211",
"truth-inference-at-scale/data/SpectralMethodsMeetEM/web/truth.csv": "1b215fe642b88107fc07c0f4b9169b70e178f3a65f76926ea1fbeb86fd688c4d"
},
"tasks": 2665,
"truths": 2653
}
}
},
"schema": "icml-ptbcc-native-v1",
"scientific_gates": {
"all_ten_registered_scales_exact": true,
"baseline_bwa_within_0_02": true,
"baseline_mv_within_0_015": true,
"label_shuffle_destroys_at_least_0_60_accuracy": true,
"mechanism_beats_mv_at_least_25_of_30": true,
"mechanism_dominant_recall_above_0_85": true,
"mechanism_prototype_mae_below_0_12": true,
"missing_aircr_required_accuracy_implausible": true,
"pooled_parameter_work_below_10_percent": true,
"ptbcc_headline_gap_at_least_0_015": true,
"ptbcc_matches_or_exceeds_best_reproduced_macro": true,
"s3_exceeds_s2": true,
"val5_absolute_gain_over_mv_is_15_points": true,
"val5_is_largest_reproduced_absolute_gain": true,
"val5_relative_gain_brackets_15_percent": true
},
"source_commits": {
"crowdti": "429a11bee1480ab01784fd00633167ca76efd954",
"truth_inference_at_scale": "621789b2d57324d3559dc973b2613d2296d73f55"
}
}