{"id":"W2251810465","doi":"","title":"Improving AMBER, an MT Evaluation Metric","year":2012,"lang":"en","type":"article","venue":"NPARC","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Metric (unit); Computer science; Task (project management); Machine translation; Translation (biology); Simplex; Artificial intelligence; Algorithm; Pattern recognition (psychology); Mathematics; Chemistry; Combinatorics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009864446,0.002066126,0.002239336,0.006411909,0.001550929,0.002860276,0.001642147,0.002014757,0.003937898],"category_scores_gemma":[0.0417777,0.0004566869,0.0008384898,0.003759219,0.0009199787,0.005800717,0.002834549,0.002214738,0.002929746],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00170927,"about_ca_system_score_gemma":0.001610288,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003915523,"about_ca_topic_score_gemma":0.007519195,"domain_scores_codex":[0.9810184,0.008748253,0.001900551,0.001991057,0.00595033,0.0003913885],"domain_scores_gemma":[0.9773731,0.009764036,0.001317444,0.003492531,0.007545879,0.0005069328],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000563381,0.0002565363,0.004127717,0.001010344,0.0004166359,0.0001557339,0.0003712438,0.03391292,0.01736244,0.01251107,0.04336391,0.8859481],"study_design_scores_gemma":[0.0003475827,0.002605431,0.01687449,0.0003293628,0.0004763561,0.002341787,0.0003326242,0.7248832,0.07495703,0.04440894,0.1319234,0.0005198008],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.09915528,0.01917047,0.8069521,0.001992525,0.00209786,0.0008950278,0.005259316,0.02403344,0.04044397],"genre_scores_gemma":[0.3186669,0.001876484,0.6525931,0.0008352454,0.0007461153,0.0006766206,0.008290177,0.002927083,0.01338833],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.009864446,"threshold_uncertainty_score":0.05216885,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02691899758865522,"score_gpt":0.3156656344312011,"score_spread":0.2887466368425459,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}