{"id":"W3021107458","doi":"10.18653/v1/2020.eval4nlp-1.5","title":"BLEU Neighbors: A Reference-less Approach to Automatic Evaluation","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"BLEU; Computer science; Machine translation; Artificial intelligence; Natural language processing; Natural language generation; Evaluation of machine translation; Language model; Bottleneck; Lexical diversity; Ground truth; Sentence; Machine learning; Natural language; Vocabulary; Linguistics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02004761,0.002489554,0.002102803,0.008808377,0.001281152,0.003809465,0.002718662,0.002687473,0.005046223],"category_scores_gemma":[0.0826691,0.0007344388,0.001165931,0.005257822,0.00111479,0.003989181,0.002809067,0.002578403,0.004413232],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001862263,"about_ca_system_score_gemma":0.001913587,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004832213,"about_ca_topic_score_gemma":0.009948242,"domain_scores_codex":[0.970064,0.01756143,0.001720907,0.004405268,0.00580342,0.0004449144],"domain_scores_gemma":[0.9498284,0.02611296,0.002418618,0.01079923,0.01019955,0.000641265],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001091643,0.0005022953,0.01895314,0.001170918,0.001292612,0.0001865651,0.001057666,0.09228644,0.01166986,0.0195004,0.06885578,0.7834326],"study_design_scores_gemma":[0.000197502,0.0009230273,0.01193869,0.0003578076,0.0002986787,0.0004704965,0.0003818313,0.8466069,0.03356596,0.05634622,0.04857139,0.0003415344],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02482071,0.003050721,0.9283112,0.0003488011,0.0004185712,0.0006355269,0.004812135,0.02603351,0.01156893],"genre_scores_gemma":[0.3791505,0.0008229235,0.5865026,0.0004359081,0.0003075224,0.001553048,0.01560765,0.006404129,0.009215697],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02004761,"threshold_uncertainty_score":0.1060231,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.20922197770817,"score_gpt":0.3298302363485546,"score_spread":0.1206082586403846,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}