{"id":"W4404781072","doi":"10.18653/v1/2024.wmt-1.25","title":"WMT24 Test Suite: Gender Resolution in Speaker-Listener Dialogue Roles","year":2024,"lang":"en","type":"article","venue":"","topic":"Language, Discourse, Communication Strategies","field":"Arts and Humanities","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Suite; Computer science; Test (biology); Test suite; Resolution (logic); Speech recognition; Artificial intelligence; Test case; Political science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007882168,0.001301032,0.0007226546,0.001284339,0.001220818,0.002336178,0.00178416,0.001685913,0.01206408],"category_scores_gemma":[0.07233495,0.0005360449,0.0006948148,0.0007894281,0.001160451,0.002462906,0.003324298,0.00180052,0.007955411],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005404073,"about_ca_system_score_gemma":0.0005836336,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002219502,"about_ca_topic_score_gemma":0.003001741,"domain_scores_codex":[0.9874591,0.007523479,0.0009188233,0.001166086,0.002415547,0.0005169744],"domain_scores_gemma":[0.9255207,0.05655484,0.002724487,0.007244701,0.005734252,0.002220993],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.01196911,0.01121376,0.1506076,0.004965829,0.0007316231,0.003686436,0.06204615,0.01382229,0.1039696,0.008587678,0.2392171,0.3891828],"study_design_scores_gemma":[0.002610863,0.01440746,0.3641596,0.001340614,0.0006400269,0.009081666,0.03795348,0.1146872,0.1809487,0.0187194,0.2541605,0.001290631],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"dataset","genre_scores_codex":[0.9340202,0.0002793105,0.0230401,0.000556063,0.0004765567,0.001104143,0.00832337,0.007348833,0.02485154],"genre_scores_gemma":[0.9355188,0.0001010737,0.02763237,0.0005003408,0.0001870011,0.003142332,0.02114589,0.00315601,0.00861613],"genre_candidate":"dataset","genre_consensus":null,"teacher_disagreement_score":0.01206408,"threshold_uncertainty_score":0.04168534,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06776917273351057,"score_gpt":0.292143731668265,"score_spread":0.2243745589347544,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}