{"id":"W2608615602","doi":"10.18653/v1/d17-1263","title":"A Challenge Set Approach to Evaluating Machine Translation","year":2017,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Machine translation; Computer science; Phrase; Set (abstract data type); Translation (biology); Natural language processing; Artificial intelligence; Divergence (linguistics); Bridge (graph theory); Strengths and weaknesses; Quality (philosophy); Neural system; Machine learning; Linguistics; Programming language; Psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.001075322,0.0003019169,0.0003203137,0.0002012124,0.0001924014,0.0007213819,0.003234999,0.0003006051,0.000007930456],"category_scores_gemma":[0.0001230259,0.0002536818,0.0001071369,0.00009310073,0.00002056238,0.0003364434,0.001909862,0.0006988117,0.00001788402],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006853652,"about_ca_system_score_gemma":0.0001301894,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002156397,"about_ca_topic_score_gemma":0.00002861279,"domain_scores_codex":[0.9978541,0.00009968698,0.0002841023,0.0009359451,0.0005467361,0.0002794493],"domain_scores_gemma":[0.9976804,0.00004185722,0.0002364838,0.001780779,0.0001542554,0.0001061845],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000007061055,0.00006745233,0.000006806745,0.0003365429,0.00002557853,0.000005459538,0.003636964,0.0002017308,0.0002847032,0.02723743,0.0004829284,0.9677073],"study_design_scores_gemma":[0.0001606072,0.0000808423,0.0000158756,0.0002394651,0.00001752665,0.00001189163,0.000008515603,0.840867,0.001278868,0.1562633,0.0004690189,0.000587163],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00008609759,0.00570196,0.9590028,0.002273042,0.0002413824,0.0007192947,0.000009212725,0.001343886,0.03062234],"genre_scores_gemma":[0.2008784,0.00001885891,0.798313,0.0001814055,0.000100225,0.0001249344,0.00003452795,0.00001804633,0.0003305809],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9671202,"threshold_uncertainty_score":0.9999915,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1379582053225443,"score_gpt":0.3916650449870761,"score_spread":0.2537068396645318,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}