| { |
| "all_scientific_gates_pass": true, |
| "claim_2_val5": { |
| "absolute_gain": 0.06, |
| "absolute_gain_over_majority_vote": 0.15000000000000002, |
| "largest_reproduced_absolute_gain_dataset": "Val5", |
| "majority_vote": 0.31, |
| "ptbcc": 0.46, |
| "relative_gain": 0.15, |
| "strongest_reproduced_baseline": 0.4 |
| }, |
| "claim_3_falsification": { |
| "absolute_shortfall": 0.02355063891614273, |
| "aircr_accuracy_required_to_reach_reported_macro": 0.9827063891614269, |
| "measured_ten_dataset_ptbcc": 0.7236493610838572, |
| "reported_ptbcc": 0.7472 |
| }, |
| "claim_4_falsification": { |
| "macro_by_prototypes": { |
| "2": 0.7236493610838572, |
| "3": 0.7250569980641843, |
| "4": 0.725806675063574 |
| }, |
| "peak": 4 |
| }, |
| "claim_5_cost": { |
| "interpretation": "This is an exact comparison of learned confusion-structure parameters, the computational-work term reduced by prototype sharing. Environment-dependent wall-clock time is deliberately not used as scored evidence.", |
| "pooled_ibcc_free_parameters": 71650, |
| "pooled_ptbcc_free_parameters": 4354, |
| "ratio": 0.060767620376831824, |
| "reduction": 0.9392323796231682, |
| "registered_dataset_rows": 11 |
| }, |
| "claims": [ |
| { |
| "claim": 1, |
| "literal_claim": "PTBCC (Prototype-driven Bayesian Classifier Combination) models annotators via a shared set of prototype confusion matrices rather than learning one confusion matrix per annotator (Section on method overview)." |
| }, |
| { |
| "claim": 2, |
| "literal_claim": "PTBCC achieves up to 15% accuracy improvement over the best baseline in its best-case dataset (Val5) (Table 4)." |
| }, |
| { |
| "claim": 3, |
| "literal_claim": "Across 11 real-world crowdsourcing datasets, PTBCC attains an average accuracy of 0.7472, versus 0.7175 for FGBCC, 0.7132 for BWA, and 0.6986 for majority voting (Table 4)." |
| }, |
| { |
| "claim": 4, |
| "literal_claim": "PTBCC's ablation over prototype set size |S| shows accuracy peaking at |S|=2 (0.7472) and degrading to 0.7300 at |S|=3 and 0.7271 at |S|=4 due to sparser per-prototype annotator distributions (Table 5)." |
| }, |
| { |
| "claim": 5, |
| "literal_claim": "PTBCC uses less than 10% of the computational cost of confusion-matrix-based baselines while matching or exceeding their accuracy (Section on computational efficiency)." |
| } |
| ], |
| "dataset_count": 10, |
| "macros": { |
| "BWA": 0.701003255458328, |
| "DS": 0.7016899769890417, |
| "IBCC": 0.6935982786099606, |
| "MV": 0.6937261736117043, |
| "PTBCC_S2": 0.7236493610838572, |
| "PTBCC_S3_mean_3_seeds": 0.7250569980641843, |
| "PTBCC_S4_mean_3_seeds": 0.725806675063574 |
| }, |
| "mechanism": { |
| "dominant_prototype_recall_mean": 0.9993333333333333, |
| "dominant_prototype_recall_min": 0.98, |
| "label_shuffle_accuracy_drop": 0.772111111111111, |
| "label_shuffled_ptbcc_accuracy_mean": 0.2081111111111111, |
| "majority_vote_accuracy_mean": 0.9724444444444444, |
| "mean_max_annotator_weight": 0.5953221954840179, |
| "prototype_mae_max": 0.07200504499687686, |
| "prototype_mae_mean": 0.04819687934303456, |
| "ptbcc_accuracy_mean": 0.9802222222222221, |
| "ptbcc_beats_mv_seeds": 29, |
| "seeds": 30 |
| }, |
| "paper_orid": "KJq0iScNM6", |
| "per_dataset": { |
| "Adult": { |
| "iterations": { |
| "BWA": [ |
| 41, |
| 18, |
| 9, |
| 13 |
| ], |
| "DS": 47, |
| "IBCC": 100, |
| "PTBCC_S2": 22, |
| "PTBCC_S3": [ |
| 42, |
| 26, |
| 28 |
| ], |
| "PTBCC_S4": [ |
| 37, |
| 54, |
| 55 |
| ] |
| }, |
| "scores": { |
| "BWA": 0.7417417417417418, |
| "DS": 0.7627627627627628, |
| "IBCC": 0.7447447447447447, |
| "MV": 0.7597597597597597, |
| "PTBCC_S2": 0.7687687687687688, |
| "PTBCC_S3_mean_3_seeds": 0.7727727727727727, |
| "PTBCC_S4_mean_3_seeds": 0.7747747747747749 |
| }, |
| "statistics": { |
| "annotators": 825, |
| "classes": 4, |
| "labels": 89799, |
| "source_sha256": { |
| "truth-inference-at-scale/data/crowd_truth_inference/s5_AdultContent/label.csv": "d72988724f7cdd628ce21fa1aaabecc4d03aef21bf59644d9e4033b774ac9a32", |
| "truth-inference-at-scale/data/crowd_truth_inference/s5_AdultContent/truth.csv": "1c52c5d528d16e031360d87ce384a80611bf23dedd5c7cfddf2c8ffc058772cb" |
| }, |
| "tasks": 11040, |
| "truths": 333 |
| } |
| }, |
| "CF": { |
| "iterations": { |
| "BWA": [ |
| 5, |
| 5, |
| 5, |
| 5, |
| 6 |
| ], |
| "DS": 60, |
| "IBCC": 18, |
| "PTBCC_S2": 103, |
| "PTBCC_S3": [ |
| 38, |
| 66, |
| 32 |
| ], |
| "PTBCC_S4": [ |
| 38, |
| 31, |
| 25 |
| ] |
| }, |
| "scores": { |
| "BWA": 0.8933333333333333, |
| "DS": 0.7866666666666666, |
| "IBCC": 0.8833333333333333, |
| "MV": 0.9, |
| "PTBCC_S2": 0.8833333333333333, |
| "PTBCC_S3_mean_3_seeds": 0.8755555555555556, |
| "PTBCC_S4_mean_3_seeds": 0.8844444444444445 |
| }, |
| "statistics": { |
| "annotators": 461, |
| "classes": 5, |
| "labels": 1720, |
| "source_sha256": { |
| "truth-inference-at-scale/data/active-crowd-toolkit/CF/label.csv": "b0f98ddc9afefffcaf9670f591a3441b1bfc3eab70f8e4c35746310962920f06", |
| "truth-inference-at-scale/data/active-crowd-toolkit/CF/truth.csv": "8d8f1b773b310fa91e550658ef88baedc4c522ccd395a1ffa9c185327b88080f" |
| }, |
| "tasks": 300, |
| "truths": 300 |
| } |
| }, |
| "Dog": { |
| "iterations": { |
| "BWA": [ |
| 9, |
| 9, |
| 9, |
| 9 |
| ], |
| "DS": 15, |
| "IBCC": 25, |
| "PTBCC_S2": 33, |
| "PTBCC_S3": [ |
| 30, |
| 49, |
| 42 |
| ], |
| "PTBCC_S4": [ |
| 35, |
| 38, |
| 51 |
| ] |
| }, |
| "scores": { |
| "BWA": 0.8314745972738538, |
| "DS": 0.8426270136307311, |
| "IBCC": 0.838909541511772, |
| "MV": 0.8178438661710037, |
| "PTBCC_S2": 0.8240396530359355, |
| "PTBCC_S3_mean_3_seeds": 0.8244527054935977, |
| "PTBCC_S4_mean_3_seeds": 0.8244527054935977 |
| }, |
| "statistics": { |
| "annotators": 109, |
| "classes": 4, |
| "labels": 8070, |
| "source_sha256": { |
| "truth-inference-at-scale/data/crowd_truth_inference/s4_Dog data/label.csv": "c240c0efc442e4936b35f28424d66fb04e687c0f29b254af94c26e15a71aa820", |
| "truth-inference-at-scale/data/crowd_truth_inference/s4_Dog data/truth.csv": "b299494a7aba3cf5abcd3b0e002f6e899f3e1f1d864a26703325db355555cb76" |
| }, |
| "tasks": 807, |
| "truths": 807 |
| } |
| }, |
| "Face": { |
| "iterations": { |
| "BWA": [ |
| 6, |
| 7, |
| 7, |
| 10 |
| ], |
| "DS": 20, |
| "IBCC": 19, |
| "PTBCC_S2": 17, |
| "PTBCC_S3": [ |
| 19, |
| 20, |
| 18 |
| ], |
| "PTBCC_S4": [ |
| 27, |
| 23, |
| 20 |
| ] |
| }, |
| "scores": { |
| "BWA": 0.6181506849315068, |
| "DS": 0.6421232876712328, |
| "IBCC": 0.6404109589041096, |
| "MV": 0.6301369863013698, |
| "PTBCC_S2": 0.6523972602739726, |
| "PTBCC_S3_mean_3_seeds": 0.6529680365296804, |
| "PTBCC_S4_mean_3_seeds": 0.6523972602739726 |
| }, |
| "statistics": { |
| "annotators": 27, |
| "classes": 4, |
| "labels": 5242, |
| "source_sha256": { |
| "truth-inference-at-scale/data/crowd_truth_inference/s4_Face Sentiment Identification/label.csv": "fc10f625183432f6ed82d783ec0afb4a2e79009843a1c20757baad5c4e469686", |
| "truth-inference-at-scale/data/crowd_truth_inference/s4_Face Sentiment Identification/truth.csv": "503987498b22b586a34bd9211cce88bf4fbb436396ccae3fd5e19a64a5a8dee9" |
| }, |
| "tasks": 584, |
| "truths": 584 |
| } |
| }, |
| "Fact": { |
| "iterations": { |
| "BWA": [ |
| 21, |
| 22, |
| 14 |
| ], |
| "DS": 59, |
| "IBCC": 18, |
| "PTBCC_S2": 16, |
| "PTBCC_S3": [ |
| 14, |
| 16, |
| 17 |
| ], |
| "PTBCC_S4": [ |
| 22, |
| 16, |
| 15 |
| ] |
| }, |
| "scores": { |
| "BWA": 0.8871527777777778, |
| "DS": 0.8524305555555556, |
| "IBCC": 0.8767361111111112, |
| "MV": 0.9010416666666666, |
| "PTBCC_S2": 0.9010416666666666, |
| "PTBCC_S3_mean_3_seeds": 0.9010416666666666, |
| "PTBCC_S4_mean_3_seeds": 0.9010416666666666 |
| }, |
| "statistics": { |
| "annotators": 57, |
| "classes": 3, |
| "labels": 214915, |
| "source_sha256": { |
| "truth-inference-at-scale/data/crowdscale2013/fact_eval/label.csv": "c8be134ea9adb9fe5984657dcd1894dce89ba93a539a600a62654c801d6e6d3a", |
| "truth-inference-at-scale/data/crowdscale2013/fact_eval/truth.csv": "5e0ea8af9d225cd4d03e175d279e5ab00958c6f74c3ebf85b068a271a83ad741" |
| }, |
| "tasks": 42624, |
| "truths": 576 |
| } |
| }, |
| "MS": { |
| "iterations": { |
| "BWA": [ |
| 12, |
| 9, |
| 9, |
| 22, |
| 9, |
| 11, |
| 12, |
| 7, |
| 8, |
| 9 |
| ], |
| "DS": 30, |
| "IBCC": 42, |
| "PTBCC_S2": 32, |
| "PTBCC_S3": [ |
| 43, |
| 58, |
| 75 |
| ], |
| "PTBCC_S4": [ |
| 116, |
| 49, |
| 94 |
| ] |
| }, |
| "scores": { |
| "BWA": 0.7857142857142857, |
| "DS": 0.7742857142857142, |
| "IBCC": 0.79, |
| "MV": 0.71, |
| "PTBCC_S2": 0.7885714285714286, |
| "PTBCC_S3_mean_3_seeds": 0.787142857142857, |
| "PTBCC_S4_mean_3_seeds": 0.7890476190476191 |
| }, |
| "statistics": { |
| "annotators": 44, |
| "classes": 10, |
| "labels": 2945, |
| "source_sha256": { |
| "truth-inference-at-scale/data/active-crowd-toolkit/MS/label.csv": "4e59b88dbe48a0cd545a745c9718f3955729205d29da1bf883adc64bf7a3e378", |
| "truth-inference-at-scale/data/active-crowd-toolkit/MS/truth.csv": "981c635d5cde03cd903bf319b030178420bc8aa3b6a3b7846abea4e974bc2294" |
| }, |
| "tasks": 700, |
| "truths": 700 |
| } |
| }, |
| "Senti": { |
| "iterations": { |
| "BWA": [ |
| 14, |
| 16, |
| 12, |
| 16, |
| 29 |
| ], |
| "DS": 81, |
| "IBCC": 107, |
| "PTBCC_S2": 17, |
| "PTBCC_S3": [ |
| 24, |
| 21, |
| 33 |
| ], |
| "PTBCC_S4": [ |
| 33, |
| 41, |
| 30 |
| ] |
| }, |
| "scores": { |
| "BWA": 0.89, |
| "DS": 0.816, |
| "IBCC": 0.831, |
| "MV": 0.902, |
| "PTBCC_S2": 0.88, |
| "PTBCC_S3_mean_3_seeds": 0.8810000000000001, |
| "PTBCC_S4_mean_3_seeds": 0.882 |
| }, |
| "statistics": { |
| "annotators": 1960, |
| "classes": 5, |
| "labels": 569274, |
| "source_sha256": { |
| "truth-inference-at-scale/data/crowdscale2013/sentiment/label.csv": "d336fabd935d27081a3a769d455ea6deb60b9312cc9f024c061c04059caa3b60", |
| "truth-inference-at-scale/data/crowdscale2013/sentiment/truth.csv": "26b6526a8ff541c86abbf57fa8736e0b4aea007c16bea4b0170cfd23ac5866e5" |
| }, |
| "tasks": 98980, |
| "truths": 1000 |
| } |
| }, |
| "Val5": { |
| "iterations": { |
| "BWA": [ |
| 6, |
| 7, |
| 5, |
| 7, |
| 6 |
| ], |
| "DS": 14, |
| "IBCC": 17, |
| "PTBCC_S2": 31, |
| "PTBCC_S3": [ |
| 60, |
| 109, |
| 156 |
| ], |
| "PTBCC_S4": [ |
| 51, |
| 131, |
| 106 |
| ] |
| }, |
| "scores": { |
| "BWA": 0.32, |
| "DS": 0.4, |
| "IBCC": 0.34, |
| "MV": 0.31, |
| "PTBCC_S2": 0.46, |
| "PTBCC_S3_mean_3_seeds": 0.4633333333333333, |
| "PTBCC_S4_mean_3_seeds": 0.4466666666666666 |
| }, |
| "statistics": { |
| "annotators": 38, |
| "classes": 5, |
| "labels": 1000, |
| "source_sha256": { |
| "CrowdTI/truth_inference_crowd/datasets/f201_Emotion_FULL/answer.csv": "8a33684760a7fc51980ddc2d19d157f9d0197bcc3611cc4a2c8cdb65fdaf1d1b", |
| "CrowdTI/truth_inference_crowd/datasets/f201_Emotion_FULL/truth.csv": "0fdfa983818a3524ba572c7eb7a3fd6b314844105ae9449b630425faa3176be4" |
| }, |
| "tasks": 100, |
| "truths": 100 |
| } |
| }, |
| "Val7": { |
| "iterations": { |
| "BWA": [ |
| 6, |
| 7, |
| 6, |
| 5, |
| 7, |
| 6, |
| 6 |
| ], |
| "DS": 16, |
| "IBCC": 23, |
| "PTBCC_S2": 89, |
| "PTBCC_S3": [ |
| 130, |
| 73, |
| 58 |
| ], |
| "PTBCC_S4": [ |
| 151, |
| 96, |
| 61 |
| ] |
| }, |
| "scores": { |
| "BWA": 0.22, |
| "DS": 0.31, |
| "IBCC": 0.24, |
| "MV": 0.23, |
| "PTBCC_S2": 0.28, |
| "PTBCC_S3_mean_3_seeds": 0.2933333333333333, |
| "PTBCC_S4_mean_3_seeds": 0.3 |
| }, |
| "statistics": { |
| "annotators": 38, |
| "classes": 7, |
| "labels": 1000, |
| "source_sha256": { |
| "CrowdTI/truth_inference_crowd/datasets/f201_Emotion_FULL/answer.csv": "8a33684760a7fc51980ddc2d19d157f9d0197bcc3611cc4a2c8cdb65fdaf1d1b", |
| "CrowdTI/truth_inference_crowd/datasets/f201_Emotion_FULL/truth.csv": "0fdfa983818a3524ba572c7eb7a3fd6b314844105ae9449b630425faa3176be4" |
| }, |
| "tasks": 100, |
| "truths": 100 |
| } |
| }, |
| "Web": { |
| "iterations": { |
| "BWA": [ |
| 17, |
| 13, |
| 8, |
| 11, |
| 20 |
| ], |
| "DS": 71, |
| "IBCC": 115, |
| "PTBCC_S2": 71, |
| "PTBCC_S3": [ |
| 38, |
| 46, |
| 61 |
| ], |
| "PTBCC_S4": [ |
| 39, |
| 31, |
| 62 |
| ] |
| }, |
| "scores": { |
| "BWA": 0.8224651338107802, |
| "DS": 0.8300037693177534, |
| "IBCC": 0.7508480964945344, |
| "MV": 0.7764794572182435, |
| "PTBCC_S2": 0.7983415001884658, |
| "PTBCC_S3_mean_3_seeds": 0.798969719814047, |
| "PTBCC_S4_mean_3_seeds": 0.8032416132679985 |
| }, |
| "statistics": { |
| "annotators": 177, |
| "classes": 5, |
| "labels": 15567, |
| "source_sha256": { |
| "truth-inference-at-scale/data/SpectralMethodsMeetEM/web/label.csv": "90b6e66284288079fab48593c61fb7949f964d007fce077b13d21d0b7ab43211", |
| "truth-inference-at-scale/data/SpectralMethodsMeetEM/web/truth.csv": "1b215fe642b88107fc07c0f4b9169b70e178f3a65f76926ea1fbeb86fd688c4d" |
| }, |
| "tasks": 2665, |
| "truths": 2653 |
| } |
| } |
| }, |
| "schema": "icml-ptbcc-native-v1", |
| "scientific_gates": { |
| "all_ten_registered_scales_exact": true, |
| "baseline_bwa_within_0_02": true, |
| "baseline_mv_within_0_015": true, |
| "label_shuffle_destroys_at_least_0_60_accuracy": true, |
| "mechanism_beats_mv_at_least_25_of_30": true, |
| "mechanism_dominant_recall_above_0_85": true, |
| "mechanism_prototype_mae_below_0_12": true, |
| "missing_aircr_required_accuracy_implausible": true, |
| "pooled_parameter_work_below_10_percent": true, |
| "ptbcc_headline_gap_at_least_0_015": true, |
| "ptbcc_matches_or_exceeds_best_reproduced_macro": true, |
| "s3_exceeds_s2": true, |
| "val5_absolute_gain_over_mv_is_15_points": true, |
| "val5_is_largest_reproduced_absolute_gain": true, |
| "val5_relative_gain_brackets_15_percent": true |
| }, |
| "source_commits": { |
| "crowdti": "429a11bee1480ab01784fd00633167ca76efd954", |
| "truth_inference_at_scale": "621789b2d57324d3559dc973b2613d2296d73f55" |
| } |
| } |
|
|