{"id":"W6966597627","doi":"10.48448/m47m-ta93","title":"A Systematic Survey and Critical Review on Evaluating Large Language Models: Challenges, Limitations, and Recommendations","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"","keywords":"Process (computing); Systematic review; Evaluation methods; Reliability (semiconductor); Comparability","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.10823,0.002058716,0.005460727,0.01900448,0.001487144,0.006129244,0.004483517,0.00302845,0.008791836],"category_scores_gemma":[0.4133912,0.001652902,0.007002088,0.01205098,0.003339282,0.007371434,0.004128485,0.003209694,0.002071766],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00740919,"about_ca_system_score_gemma":0.03576935,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009447671,"about_ca_topic_score_gemma":0.02697802,"domain_scores_codex":[0.9162041,0.03947535,0.02746111,0.003327976,0.01256702,0.0009645389],"domain_scores_gemma":[0.5008061,0.390485,0.03577306,0.01004294,0.06066428,0.002228555],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","study_design_scores_codex":[0.0002474223,0.00004477449,0.001384547,0.668209,0.003170327,0.0001283899,0.001291616,0.0002999217,0.0003009053,0.002866067,0.03680879,0.2852482],"study_design_scores_gemma":[0.00008220586,0.0001137625,0.001224328,0.8772755,0.006843182,0.0001425959,0.0006141198,0.0001037301,0.0002432181,0.002285837,0.1110136,0.00005795318],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.0006656721,0.9848043,0.001996659,0.008174628,0.001174235,0.0007530238,0.0009200896,0.00007493466,0.001436499],"genre_scores_gemma":[0.008994035,0.9720269,0.006710628,0.007739175,0.0006817375,0.002301915,0.0009919156,0.00009487874,0.0004587721],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.8917701,"threshold_uncertainty_score":0.5723815,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2749499417627526,"score_gpt":0.4398776242197662,"score_spread":0.1649276824570136,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}