{ "all_scientific_gates_pass": true, "claim_2_val5": { "absolute_gain": 0.06, "absolute_gain_over_majority_vote": 0.15000000000000002, "largest_reproduced_absolute_gain_dataset": "Val5", "majority_vote": 0.31, "ptbcc": 0.46, "relative_gain": 0.15, "strongest_reproduced_baseline": 0.4 }, "claim_3_falsification": { "absolute_shortfall": 0.02355063891614273, "aircr_accuracy_required_to_reach_reported_macro": 0.9827063891614269, "measured_ten_dataset_ptbcc": 0.7236493610838572, "reported_ptbcc": 0.7472 }, "claim_4_falsification": { "macro_by_prototypes": { "2": 0.7236493610838572, "3": 0.7250569980641843, "4": 0.725806675063574 }, "peak": 4 }, "claim_5_cost": { "interpretation": "This is an exact comparison of learned confusion-structure parameters, the computational-work term reduced by prototype sharing. Environment-dependent wall-clock time is deliberately not used as scored evidence.", "pooled_ibcc_free_parameters": 71650, "pooled_ptbcc_free_parameters": 4354, "ratio": 0.060767620376831824, "reduction": 0.9392323796231682, "registered_dataset_rows": 11 }, "claims": [ { "claim": 1, "literal_claim": "PTBCC (Prototype-driven Bayesian Classifier Combination) models annotators via a shared set of prototype confusion matrices rather than learning one confusion matrix per annotator (Section on method overview)." }, { "claim": 2, "literal_claim": "PTBCC achieves up to 15% accuracy improvement over the best baseline in its best-case dataset (Val5) (Table 4)." }, { "claim": 3, "literal_claim": "Across 11 real-world crowdsourcing datasets, PTBCC attains an average accuracy of 0.7472, versus 0.7175 for FGBCC, 0.7132 for BWA, and 0.6986 for majority voting (Table 4)." }, { "claim": 4, "literal_claim": "PTBCC's ablation over prototype set size |S| shows accuracy peaking at |S|=2 (0.7472) and degrading to 0.7300 at |S|=3 and 0.7271 at |S|=4 due to sparser per-prototype annotator distributions (Table 5)." }, { "claim": 5, "literal_claim": "PTBCC uses less than 10% of the computational cost of confusion-matrix-based baselines while matching or exceeding their accuracy (Section on computational efficiency)." } ], "dataset_count": 10, "macros": { "BWA": 0.701003255458328, "DS": 0.7016899769890417, "IBCC": 0.6935982786099606, "MV": 0.6937261736117043, "PTBCC_S2": 0.7236493610838572, "PTBCC_S3_mean_3_seeds": 0.7250569980641843, "PTBCC_S4_mean_3_seeds": 0.725806675063574 }, "mechanism": { "dominant_prototype_recall_mean": 0.9993333333333333, "dominant_prototype_recall_min": 0.98, "label_shuffle_accuracy_drop": 0.772111111111111, "label_shuffled_ptbcc_accuracy_mean": 0.2081111111111111, "majority_vote_accuracy_mean": 0.9724444444444444, "mean_max_annotator_weight": 0.5953221954840179, "prototype_mae_max": 0.07200504499687686, "prototype_mae_mean": 0.04819687934303456, "ptbcc_accuracy_mean": 0.9802222222222221, "ptbcc_beats_mv_seeds": 29, "seeds": 30 }, "paper_orid": "KJq0iScNM6", "per_dataset": { "Adult": { "iterations": { "BWA": [ 41, 18, 9, 13 ], "DS": 47, "IBCC": 100, "PTBCC_S2": 22, "PTBCC_S3": [ 42, 26, 28 ], "PTBCC_S4": [ 37, 54, 55 ] }, "scores": { "BWA": 0.7417417417417418, "DS": 0.7627627627627628, "IBCC": 0.7447447447447447, "MV": 0.7597597597597597, "PTBCC_S2": 0.7687687687687688, "PTBCC_S3_mean_3_seeds": 0.7727727727727727, "PTBCC_S4_mean_3_seeds": 0.7747747747747749 }, "statistics": { "annotators": 825, "classes": 4, "labels": 89799, "source_sha256": { "truth-inference-at-scale/data/crowd_truth_inference/s5_AdultContent/label.csv": "d72988724f7cdd628ce21fa1aaabecc4d03aef21bf59644d9e4033b774ac9a32", "truth-inference-at-scale/data/crowd_truth_inference/s5_AdultContent/truth.csv": "1c52c5d528d16e031360d87ce384a80611bf23dedd5c7cfddf2c8ffc058772cb" }, "tasks": 11040, "truths": 333 } }, "CF": { "iterations": { "BWA": [ 5, 5, 5, 5, 6 ], "DS": 60, "IBCC": 18, "PTBCC_S2": 103, "PTBCC_S3": [ 38, 66, 32 ], "PTBCC_S4": [ 38, 31, 25 ] }, "scores": { "BWA": 0.8933333333333333, "DS": 0.7866666666666666, "IBCC": 0.8833333333333333, "MV": 0.9, "PTBCC_S2": 0.8833333333333333, "PTBCC_S3_mean_3_seeds": 0.8755555555555556, "PTBCC_S4_mean_3_seeds": 0.8844444444444445 }, "statistics": { "annotators": 461, "classes": 5, "labels": 1720, "source_sha256": { "truth-inference-at-scale/data/active-crowd-toolkit/CF/label.csv": "b0f98ddc9afefffcaf9670f591a3441b1bfc3eab70f8e4c35746310962920f06", "truth-inference-at-scale/data/active-crowd-toolkit/CF/truth.csv": "8d8f1b773b310fa91e550658ef88baedc4c522ccd395a1ffa9c185327b88080f" }, "tasks": 300, "truths": 300 } }, "Dog": { "iterations": { "BWA": [ 9, 9, 9, 9 ], "DS": 15, "IBCC": 25, "PTBCC_S2": 33, "PTBCC_S3": [ 30, 49, 42 ], "PTBCC_S4": [ 35, 38, 51 ] }, "scores": { "BWA": 0.8314745972738538, "DS": 0.8426270136307311, "IBCC": 0.838909541511772, "MV": 0.8178438661710037, "PTBCC_S2": 0.8240396530359355, "PTBCC_S3_mean_3_seeds": 0.8244527054935977, "PTBCC_S4_mean_3_seeds": 0.8244527054935977 }, "statistics": { "annotators": 109, "classes": 4, "labels": 8070, "source_sha256": { "truth-inference-at-scale/data/crowd_truth_inference/s4_Dog data/label.csv": "c240c0efc442e4936b35f28424d66fb04e687c0f29b254af94c26e15a71aa820", "truth-inference-at-scale/data/crowd_truth_inference/s4_Dog data/truth.csv": "b299494a7aba3cf5abcd3b0e002f6e899f3e1f1d864a26703325db355555cb76" }, "tasks": 807, "truths": 807 } }, "Face": { "iterations": { "BWA": [ 6, 7, 7, 10 ], "DS": 20, "IBCC": 19, "PTBCC_S2": 17, "PTBCC_S3": [ 19, 20, 18 ], "PTBCC_S4": [ 27, 23, 20 ] }, "scores": { "BWA": 0.6181506849315068, "DS": 0.6421232876712328, "IBCC": 0.6404109589041096, "MV": 0.6301369863013698, "PTBCC_S2": 0.6523972602739726, "PTBCC_S3_mean_3_seeds": 0.6529680365296804, "PTBCC_S4_mean_3_seeds": 0.6523972602739726 }, "statistics": { "annotators": 27, "classes": 4, "labels": 5242, "source_sha256": { "truth-inference-at-scale/data/crowd_truth_inference/s4_Face Sentiment Identification/label.csv": "fc10f625183432f6ed82d783ec0afb4a2e79009843a1c20757baad5c4e469686", "truth-inference-at-scale/data/crowd_truth_inference/s4_Face Sentiment Identification/truth.csv": "503987498b22b586a34bd9211cce88bf4fbb436396ccae3fd5e19a64a5a8dee9" }, "tasks": 584, "truths": 584 } }, "Fact": { "iterations": { "BWA": [ 21, 22, 14 ], "DS": 59, "IBCC": 18, "PTBCC_S2": 16, "PTBCC_S3": [ 14, 16, 17 ], "PTBCC_S4": [ 22, 16, 15 ] }, "scores": { "BWA": 0.8871527777777778, "DS": 0.8524305555555556, "IBCC": 0.8767361111111112, "MV": 0.9010416666666666, "PTBCC_S2": 0.9010416666666666, "PTBCC_S3_mean_3_seeds": 0.9010416666666666, "PTBCC_S4_mean_3_seeds": 0.9010416666666666 }, "statistics": { "annotators": 57, "classes": 3, "labels": 214915, "source_sha256": { "truth-inference-at-scale/data/crowdscale2013/fact_eval/label.csv": "c8be134ea9adb9fe5984657dcd1894dce89ba93a539a600a62654c801d6e6d3a", "truth-inference-at-scale/data/crowdscale2013/fact_eval/truth.csv": "5e0ea8af9d225cd4d03e175d279e5ab00958c6f74c3ebf85b068a271a83ad741" }, "tasks": 42624, "truths": 576 } }, "MS": { "iterations": { "BWA": [ 12, 9, 9, 22, 9, 11, 12, 7, 8, 9 ], "DS": 30, "IBCC": 42, "PTBCC_S2": 32, "PTBCC_S3": [ 43, 58, 75 ], "PTBCC_S4": [ 116, 49, 94 ] }, "scores": { "BWA": 0.7857142857142857, "DS": 0.7742857142857142, "IBCC": 0.79, "MV": 0.71, "PTBCC_S2": 0.7885714285714286, "PTBCC_S3_mean_3_seeds": 0.787142857142857, "PTBCC_S4_mean_3_seeds": 0.7890476190476191 }, "statistics": { "annotators": 44, "classes": 10, "labels": 2945, "source_sha256": { "truth-inference-at-scale/data/active-crowd-toolkit/MS/label.csv": "4e59b88dbe48a0cd545a745c9718f3955729205d29da1bf883adc64bf7a3e378", "truth-inference-at-scale/data/active-crowd-toolkit/MS/truth.csv": "981c635d5cde03cd903bf319b030178420bc8aa3b6a3b7846abea4e974bc2294" }, "tasks": 700, "truths": 700 } }, "Senti": { "iterations": { "BWA": [ 14, 16, 12, 16, 29 ], "DS": 81, "IBCC": 107, "PTBCC_S2": 17, "PTBCC_S3": [ 24, 21, 33 ], "PTBCC_S4": [ 33, 41, 30 ] }, "scores": { "BWA": 0.89, "DS": 0.816, "IBCC": 0.831, "MV": 0.902, "PTBCC_S2": 0.88, "PTBCC_S3_mean_3_seeds": 0.8810000000000001, "PTBCC_S4_mean_3_seeds": 0.882 }, "statistics": { "annotators": 1960, "classes": 5, "labels": 569274, "source_sha256": { "truth-inference-at-scale/data/crowdscale2013/sentiment/label.csv": "d336fabd935d27081a3a769d455ea6deb60b9312cc9f024c061c04059caa3b60", "truth-inference-at-scale/data/crowdscale2013/sentiment/truth.csv": "26b6526a8ff541c86abbf57fa8736e0b4aea007c16bea4b0170cfd23ac5866e5" }, "tasks": 98980, "truths": 1000 } }, "Val5": { "iterations": { "BWA": [ 6, 7, 5, 7, 6 ], "DS": 14, "IBCC": 17, "PTBCC_S2": 31, "PTBCC_S3": [ 60, 109, 156 ], "PTBCC_S4": [ 51, 131, 106 ] }, "scores": { "BWA": 0.32, "DS": 0.4, "IBCC": 0.34, "MV": 0.31, "PTBCC_S2": 0.46, "PTBCC_S3_mean_3_seeds": 0.4633333333333333, "PTBCC_S4_mean_3_seeds": 0.4466666666666666 }, "statistics": { "annotators": 38, "classes": 5, "labels": 1000, "source_sha256": { "CrowdTI/truth_inference_crowd/datasets/f201_Emotion_FULL/answer.csv": "8a33684760a7fc51980ddc2d19d157f9d0197bcc3611cc4a2c8cdb65fdaf1d1b", "CrowdTI/truth_inference_crowd/datasets/f201_Emotion_FULL/truth.csv": "0fdfa983818a3524ba572c7eb7a3fd6b314844105ae9449b630425faa3176be4" }, "tasks": 100, "truths": 100 } }, "Val7": { "iterations": { "BWA": [ 6, 7, 6, 5, 7, 6, 6 ], "DS": 16, "IBCC": 23, "PTBCC_S2": 89, "PTBCC_S3": [ 130, 73, 58 ], "PTBCC_S4": [ 151, 96, 61 ] }, "scores": { "BWA": 0.22, "DS": 0.31, "IBCC": 0.24, "MV": 0.23, "PTBCC_S2": 0.28, "PTBCC_S3_mean_3_seeds": 0.2933333333333333, "PTBCC_S4_mean_3_seeds": 0.3 }, "statistics": { "annotators": 38, "classes": 7, "labels": 1000, "source_sha256": { "CrowdTI/truth_inference_crowd/datasets/f201_Emotion_FULL/answer.csv": "8a33684760a7fc51980ddc2d19d157f9d0197bcc3611cc4a2c8cdb65fdaf1d1b", "CrowdTI/truth_inference_crowd/datasets/f201_Emotion_FULL/truth.csv": "0fdfa983818a3524ba572c7eb7a3fd6b314844105ae9449b630425faa3176be4" }, "tasks": 100, "truths": 100 } }, "Web": { "iterations": { "BWA": [ 17, 13, 8, 11, 20 ], "DS": 71, "IBCC": 115, "PTBCC_S2": 71, "PTBCC_S3": [ 38, 46, 61 ], "PTBCC_S4": [ 39, 31, 62 ] }, "scores": { "BWA": 0.8224651338107802, "DS": 0.8300037693177534, "IBCC": 0.7508480964945344, "MV": 0.7764794572182435, "PTBCC_S2": 0.7983415001884658, "PTBCC_S3_mean_3_seeds": 0.798969719814047, "PTBCC_S4_mean_3_seeds": 0.8032416132679985 }, "statistics": { "annotators": 177, "classes": 5, "labels": 15567, "source_sha256": { "truth-inference-at-scale/data/SpectralMethodsMeetEM/web/label.csv": "90b6e66284288079fab48593c61fb7949f964d007fce077b13d21d0b7ab43211", "truth-inference-at-scale/data/SpectralMethodsMeetEM/web/truth.csv": "1b215fe642b88107fc07c0f4b9169b70e178f3a65f76926ea1fbeb86fd688c4d" }, "tasks": 2665, "truths": 2653 } } }, "schema": "icml-ptbcc-native-v1", "scientific_gates": { "all_ten_registered_scales_exact": true, "baseline_bwa_within_0_02": true, "baseline_mv_within_0_015": true, "label_shuffle_destroys_at_least_0_60_accuracy": true, "mechanism_beats_mv_at_least_25_of_30": true, "mechanism_dominant_recall_above_0_85": true, "mechanism_prototype_mae_below_0_12": true, "missing_aircr_required_accuracy_implausible": true, "pooled_parameter_work_below_10_percent": true, "ptbcc_headline_gap_at_least_0_015": true, "ptbcc_matches_or_exceeds_best_reproduced_macro": true, "s3_exceeds_s2": true, "val5_absolute_gain_over_mv_is_15_points": true, "val5_is_largest_reproduced_absolute_gain": true, "val5_relative_gain_brackets_15_percent": true }, "source_commits": { "crowdti": "429a11bee1480ab01784fd00633167ca76efd954", "truth_inference_at_scale": "621789b2d57324d3559dc973b2613d2296d73f55" } }