{"id":"W4399298003","doi":"10.1017/nlp.2024.5","title":"Calibration and context in human evaluation of machine translation","year":2024,"lang":"en","type":"article","venue":"Natural language processing.","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Calibration; Context (archaeology); Translation (biology); Machine translation; Computer science; Artificial intelligence; Natural language processing; Machine learning; Chemistry; Biology; Mathematics; Statistics; Biochemistry","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1233818,0.001277344,0.00113178,0.003155762,0.002381086,0.005163773,0.002104234,0.002147941,0.002098529],"category_scores_gemma":[0.3890582,0.0009923097,0.0006853218,0.002324636,0.004251792,0.004760565,0.00757032,0.00229917,0.0006238022],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002286055,"about_ca_system_score_gemma":0.001582256,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001379956,"about_ca_topic_score_gemma":0.002022447,"domain_scores_codex":[0.6229384,0.334097,0.00805994,0.01305972,0.02059283,0.001252177],"domain_scores_gemma":[0.602674,0.2989887,0.03130128,0.02853955,0.03582019,0.002676276],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.01326505,0.001083081,0.121547,0.003837435,0.002758399,0.0007053195,0.03593083,0.0489224,0.06267169,0.03181332,0.007383508,0.670082],"study_design_scores_gemma":[0.001503211,0.008361804,0.2977031,0.005963982,0.002537868,0.002746942,0.01399983,0.2096984,0.1756246,0.2177701,0.06204737,0.002042646],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6657237,0.01135859,0.2795195,0.002429877,0.000580689,0.001490472,0.0004726325,0.001385321,0.0370393],"genre_scores_gemma":[0.9514693,0.000307143,0.04635189,0.0003473379,0.0001205802,0.0004972448,0.0001580964,0.0001955351,0.000552889],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8766181,"threshold_uncertainty_score":0.6525133,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01995254238234069,"score_gpt":0.3295239246854237,"score_spread":0.3095713823030831,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}