{"id":"W6891607533","doi":"10.48448/qcph-r127","title":"Why Don’t Prompt-Based Fairness Metrics Correlate?","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Metric (unit); Correlation; Fairness measure; Reliability (semiconductor); Code (set theory); Scale (ratio); Pearson product-moment correlation coefficient","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03208468,0.001489896,0.001226286,0.00234377,0.001308845,0.003475533,0.001490331,0.001457555,0.002818249],"category_scores_gemma":[0.2181529,0.0005846432,0.0005148599,0.002212881,0.001875996,0.005763032,0.003783207,0.002563764,0.002690875],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001515634,"about_ca_system_score_gemma":0.002117775,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003674886,"about_ca_topic_score_gemma":0.003688109,"domain_scores_codex":[0.9633893,0.02105131,0.002032416,0.004815942,0.007505129,0.001205971],"domain_scores_gemma":[0.861878,0.08287449,0.01122939,0.02166281,0.01971756,0.002637755],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001755492,0.0004016844,0.340511,0.001143905,0.0008113301,0.0004006323,0.006071881,0.03888969,0.01224523,0.03578371,0.0399877,0.5219977],"study_design_scores_gemma":[0.0002624149,0.001274903,0.1772665,0.000878942,0.0003865552,0.001272023,0.0043586,0.3846129,0.06138228,0.3011556,0.06647724,0.0006720953],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3426018,0.004219475,0.6043295,0.005546004,0.001446309,0.0005861279,0.004255835,0.01194916,0.02506567],"genre_scores_gemma":[0.9218576,0.000287416,0.06996828,0.001015571,0.0001983629,0.0003648302,0.001835937,0.001603331,0.002868827],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9679153,"threshold_uncertainty_score":0.1696821,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02668869585672316,"score_gpt":0.3009667431686595,"score_spread":0.2742780473119363,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}