{"id":"W4389518784","doi":"10.18653/v1/2023.emnlp-main.397","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","year":2023,"lang":"en","type":"article","venue":"","topic":"Mental Health via Writing","field":"Psychology","cited_by":234,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"Renmin University of China; National Natural Science Foundation of China","keywords":"Hallucinating; Benchmark (surveying); Computer science; Sampling (signal processing); Artificial intelligence; Face (sociological concept); Natural language processing; Machine learning; Computer vision; Sociology; Social science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009496585,0.00322221,0.00121719,0.002197349,0.001108934,0.002148816,0.003569513,0.002672075,0.003680005],"category_scores_gemma":[0.04359212,0.0006377797,0.001684405,0.001463708,0.001082913,0.003897064,0.003430018,0.002710236,0.002481343],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001425474,"about_ca_system_score_gemma":0.00186984,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01044937,"about_ca_topic_score_gemma":0.01324946,"domain_scores_codex":[0.9884391,0.006867899,0.001265393,0.001578217,0.001485569,0.000363814],"domain_scores_gemma":[0.9700078,0.02201626,0.0007540021,0.003155979,0.003037307,0.001028615],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.004103078,0.003233439,0.0377981,0.008036894,0.002325485,0.002604063,0.003084238,0.2229781,0.02465108,0.006408961,0.2156788,0.4690978],"study_design_scores_gemma":[0.0008563689,0.001607862,0.008495584,0.0001983942,0.000275052,0.001228601,0.001424977,0.9319236,0.02043675,0.008050794,0.025287,0.0002150511],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"dataset","genre_scores_codex":[0.5804108,0.01088986,0.2501847,0.003291231,0.001658323,0.002950478,0.04980402,0.08448964,0.01632085],"genre_scores_gemma":[0.6983452,0.001204079,0.1765283,0.001273165,0.0002132634,0.001562476,0.113465,0.00310862,0.004299879],"genre_candidate":"dataset","genre_consensus":null,"teacher_disagreement_score":0.01044937,"threshold_uncertainty_score":0.05022335,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08507588964639624,"score_gpt":0.4452306925761189,"score_spread":0.3601548029297226,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}