{"id":"W4404782882","doi":"10.18653/v1/2024.emnlp-main.764","title":"A Systematic Survey and Critical Review on Evaluating Large Language Models: Challenges, Limitations, and Recommendations","year":2024,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":50,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Compute Canada","keywords":"Computer science; Data science; Management science; Engineering","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.219875,0.002579168,0.009298088,0.02384304,0.001672935,0.007573993,0.005649747,0.002803611,0.00564239],"category_scores_gemma":[0.5153186,0.002374473,0.01128793,0.01506678,0.003700833,0.01041222,0.005156078,0.00403965,0.00115557],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01043624,"about_ca_system_score_gemma":0.03010575,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009251032,"about_ca_topic_score_gemma":0.02753834,"domain_scores_codex":[0.8623264,0.07470629,0.04032498,0.004779056,0.01688961,0.0009735853],"domain_scores_gemma":[0.3180421,0.5748593,0.03696271,0.01151582,0.05612775,0.00249235],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","study_design_scores_codex":[0.0007943772,0.00009936394,0.003409369,0.6172892,0.01571294,0.0001166166,0.001352536,0.000610511,0.000349457,0.00232013,0.02600463,0.3319409],"study_design_scores_gemma":[0.0004453468,0.0005008483,0.003436281,0.8648286,0.03829052,0.0002058518,0.00116625,0.0004922865,0.0004260384,0.004296226,0.08578683,0.0001248819],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.001117195,0.9852811,0.002291583,0.007995433,0.0008642494,0.0008104494,0.0008211418,0.00006358552,0.0007552786],"genre_scores_gemma":[0.0212534,0.9534339,0.01092212,0.008512832,0.001079078,0.0032674,0.001189077,0.0001004309,0.0002416953],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.219875,"threshold_uncertainty_score":0.962033,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3056448804553619,"score_gpt":0.4092377600908385,"score_spread":0.1035928796354765,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}