{"id":"W4408595536","doi":"10.5705/ss.202023.0202","title":"Addressing Label Noise in Causation Classification via Kernel Embeddings","year":2025,"lang":"en","type":"article","venue":"Statistica Sinica","topic":"Artificial Intelligence in Law","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Kernel (algebra); Noise (video); Causation; Computer science; Pattern recognition (psychology); Artificial intelligence; Mathematics; Machine learning; Natural language processing; Pure mathematics; Epistemology; Philosophy","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02483737,0.001070553,0.00211183,0.002625231,0.001693313,0.003269167,0.003071644,0.003774107,0.001982393],"category_scores_gemma":[0.1301802,0.0007158503,0.001310724,0.002520178,0.004252666,0.008224242,0.005004723,0.004676297,0.0004070322],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002057841,"about_ca_system_score_gemma":0.00218059,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002231048,"about_ca_topic_score_gemma":0.002001823,"domain_scores_codex":[0.9880514,0.006920937,0.0007894055,0.00204768,0.001736266,0.0004542516],"domain_scores_gemma":[0.8903726,0.08694579,0.006682693,0.01033614,0.004963956,0.0006987887],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004298748,0.0004308241,0.02133774,0.0007253878,0.0003753105,0.0003107824,0.001764979,0.2120951,0.002330533,0.4148234,0.005427414,0.3399487],"study_design_scores_gemma":[0.00003517034,0.00006238287,0.001275869,0.00009200189,0.00004196824,0.00009113656,0.0001386442,0.6435049,0.0009820715,0.3522608,0.00147901,0.00003611208],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02645254,0.0004925622,0.9710432,0.001041145,0.00006195206,0.00006909453,0.00009061289,0.0001867851,0.0005620611],"genre_scores_gemma":[0.6453138,0.0005900582,0.3507099,0.0005234436,0.0002601558,0.0003696669,0.0005214765,0.0001145373,0.001596968],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02483737,"threshold_uncertainty_score":0.1313542,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1676556962956567,"score_gpt":0.4664966625380305,"score_spread":0.2988409662423739,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}