{"id":"W4391925938","doi":"10.35542/osf.io/tqkv8","title":"Lessons Learned About Transparency, Fairness, and Explainability from Two Automated Scoring Challenges","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Ethics and Social Impacts of AI","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Impact","funders":"","keywords":"Transparency (behavior); Computer science; Accounting; Business; Psychology; Computer security","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.2375311,0.001327467,0.001578188,0.002770177,0.004444692,0.01273183,0.005532958,0.00395432,0.004542207],"category_scores_gemma":[0.4464585,0.001124508,0.001263215,0.001741799,0.01258017,0.02206292,0.01031092,0.01038395,0.0008732569],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005539726,"about_ca_system_score_gemma":0.01101956,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01016905,"about_ca_topic_score_gemma":0.01140431,"domain_scores_codex":[0.7498415,0.196753,0.009893481,0.009978163,0.03022345,0.003310429],"domain_scores_gemma":[0.3103807,0.5744219,0.01255355,0.04361621,0.05360807,0.00541957],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.0006682659,0.001004798,0.06041564,0.001828013,0.0003011688,0.001714969,0.1595571,0.009007839,0.004827804,0.1410652,0.03538235,0.5842268],"study_design_scores_gemma":[0.0003141201,0.001268576,0.0411467,0.004243044,0.0002318716,0.002959646,0.0752951,0.07329408,0.01287048,0.6565461,0.1310756,0.0007546626],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2301128,0.002987032,0.5145933,0.2054325,0.00197023,0.001411436,0.00036902,0.0014818,0.04164179],"genre_scores_gemma":[0.719284,0.000922,0.2667345,0.007264738,0.0008180551,0.0008655629,0.0002169602,0.0005748794,0.003319276],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.2375311,"threshold_uncertainty_score":0.9402599,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2283291913173935,"score_gpt":0.4696973614995253,"score_spread":0.2413681701821318,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}