{"id":"W4415746035","doi":"10.1109/icsme64153.2025.00012","title":"Software Fairness Testing in Practice","year":2025,"lang":"","type":"article","venue":"","topic":"Ethics and Social Impacts of AI","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Oracle; Key (lock); Quality (philosophy); Software; Bridge (graph theory); Test (biology); Component (thermodynamics); System integration testing","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.2069744,0.00129404,0.001369237,0.00533704,0.007173886,0.01437279,0.005258697,0.007742074,0.01130772],"category_scores_gemma":[0.4029518,0.001018499,0.001048085,0.003390381,0.03667801,0.01902368,0.01472223,0.009716975,0.003432615],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01087868,"about_ca_system_score_gemma":0.02946783,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005149915,"about_ca_topic_score_gemma":0.003600546,"domain_scores_codex":[0.7235063,0.2051776,0.01204,0.0157854,0.03751963,0.005971121],"domain_scores_gemma":[0.5328819,0.323624,0.01642089,0.05948225,0.05789448,0.009696548],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","study_design_scores_codex":[0.000140814,0.0002372672,0.0104101,0.0007483652,0.00007881287,0.0005862617,0.01717465,0.004698808,0.0007959779,0.753917,0.02700506,0.1842068],"study_design_scores_gemma":[0.00009429765,0.000238507,0.002102024,0.002471791,0.00003505676,0.0004929308,0.005680044,0.007005682,0.001750328,0.811953,0.1680937,0.00008263666],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03156985,0.005543374,0.6029376,0.1397173,0.00271794,0.001276543,0.0002230264,0.001995901,0.2140184],"genre_scores_gemma":[0.7125514,0.002461239,0.2426848,0.02431524,0.001036553,0.002844073,0.0002837361,0.00112828,0.01269463],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.2069744,"threshold_uncertainty_score":0.9779418,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06368786895981896,"score_gpt":0.4286340845875976,"score_spread":0.3649462156277786,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}