{"id":"W6891607533","doi":"10.48448/qcph-r127","title":"Why Don’t Prompt-Based Fairness Metrics Correlate?","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Metric (unit); Correlation; Fairness measure; Reliability (semiconductor); Code (set theory); Scale (ratio); Pearson product-moment correlation coefficient","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.002135552,0.0008710718,0.0007434257,0.006782306,0.0002321584,0.0007718389,0.002581489,0.0005937482,0.004813867],"category_scores_gemma":[0.0006734873,0.0007444172,0.0001875632,0.01293982,0.002954579,0.0003113161,0.0005403328,0.001125987,0.04581996],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001009092,"about_ca_system_score_gemma":0.003036414,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001457048,"about_ca_topic_score_gemma":0.001359505,"domain_scores_codex":[0.9927348,0.0001024604,0.0006172726,0.002132597,0.003059368,0.00135354],"domain_scores_gemma":[0.9968333,0.0001264844,0.0005200004,0.001647863,0.0003580908,0.0005142838],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000008471066,0.0001868162,0.00008738303,0.0002334927,0.00004577293,0.00009896843,0.00004297575,0.0002055154,0.0004359134,0.00388673,0.9928465,0.00192145],"study_design_scores_gemma":[0.0004965824,0.0001229118,0.00001301831,0.0006909628,0.000158316,0.00002295435,0.00009401275,0.04352127,0.000307095,0.001113661,0.9524269,0.001032334],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.00003827515,0.005091059,0.01675435,0.001066888,0.007293826,0.001878856,0.0009406984,0.004653894,0.9622822],"genre_scores_gemma":[0.01508091,0.0000392379,0.02156077,0.001905566,0.001157175,0.0001427618,0.0003473378,0.004748134,0.9550181],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.04331575,"threshold_uncertainty_score":0.9997588,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02668869585672316,"score_gpt":0.3009667431686595,"score_spread":0.2742780473119363,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}