{"id":"W4392637304","doi":"10.18653/v1/2023.eval4nlp-1.3","title":"Delving into Evaluation Metrics for Generation: A Thorough Assessment of How Metrics Generalize to Rephrasing Across Languages","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"York University","keywords":"Computer science; Generative grammar; Heuristics; Natural language processing; Artificial intelligence; Variety (cybernetics); Natural language generation; Phrase; Robustness (evolution); Natural language","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.08420991,0.002582379,0.002062141,0.008602167,0.001312156,0.006432684,0.002484706,0.002377706,0.001156761],"category_scores_gemma":[0.344037,0.0005881226,0.001660977,0.007994606,0.002672229,0.009500186,0.004595623,0.002811615,0.0006685765],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002826445,"about_ca_system_score_gemma":0.002163364,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004220499,"about_ca_topic_score_gemma":0.003487182,"domain_scores_codex":[0.9043027,0.0545509,0.01030259,0.005944133,0.02314535,0.001754388],"domain_scores_gemma":[0.6107544,0.287583,0.02284655,0.03518556,0.0410734,0.002557246],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001552265,0.0008483257,0.1140841,0.004013508,0.001978109,0.0003136345,0.007209231,0.09262799,0.02109626,0.03151451,0.01293644,0.7118257],"study_design_scores_gemma":[0.000248159,0.008218689,0.1433496,0.003765655,0.001471292,0.002084997,0.008977581,0.5916653,0.07199585,0.1046256,0.0622385,0.001358764],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3044323,0.01207201,0.6542767,0.002685898,0.0007971685,0.001877171,0.003293332,0.005899897,0.01466559],"genre_scores_gemma":[0.7143228,0.001543479,0.2747703,0.0004855354,0.0001756911,0.001259423,0.003969816,0.002155888,0.001317044],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9157901,"threshold_uncertainty_score":0.4453499,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1734054124733606,"score_gpt":0.4540214667908573,"score_spread":0.2806160543174967,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}