{"id":"W7126460285","doi":"10.18653/v1/2022.wmt-1.8","title":"Test Set Sampling Affects System Rankings: Expanded Human Evaluation of WMT20 English-Inuktitut Systems","year":2022,"lang":"en","type":"article","venue":"","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Set (abstract data type); Sampling (signal processing); Test (biology); Data set; Test set","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.05598568,0.0001866236,0.0006133663,0.0009417773,0.0006837462,0.0002601488,0.00107652,0.00006782817,0.0005595408],"category_scores_gemma":[0.1072232,0.0001410147,0.0001540589,0.003234621,0.00005408419,0.0002249914,0.000423871,0.0002307314,0.00001243897],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003003697,"about_ca_system_score_gemma":0.0001313019,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002121721,"about_ca_topic_score_gemma":0.000007867199,"domain_scores_codex":[0.9904903,0.002679319,0.001111681,0.0006852768,0.004682224,0.0003512054],"domain_scores_gemma":[0.9613973,0.03514149,0.001074944,0.0009896714,0.001295248,0.0001012811],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001950838,0.0008362883,0.486095,0.0009480664,0.0003394218,0.00004189287,0.01629245,0.2310283,0.04103941,0.05341887,0.04290837,0.1268568],"study_design_scores_gemma":[0.01386065,0.002780395,0.1173484,0.0007515063,0.0007067552,0.0003505921,0.3186948,0.4971745,0.005214332,0.0174744,0.02311965,0.00252404],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9223695,0.0008882277,0.01461655,0.00002394314,0.003865074,0.001078714,0.0000446767,0.0002846901,0.05682866],"genre_scores_gemma":[0.9968523,0.000001048796,0.00237102,0.0000195353,0.0002052446,0.0001557218,0.00001046211,0.00001503999,0.0003696419],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3687466,"threshold_uncertainty_score":0.9720614,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6765852235668929,"score_gpt":0.5162689098489014,"score_spread":0.1603163137179915,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}