{"id":"W6891812606","doi":"10.48448/6y6k-7w61","title":"TofuEval: Evaluating Hallucinations of LLMs on Topic-Focused Dialogue Summarization","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Automatic summarization; Consistency (knowledge bases); Benchmark (surveying); Hallucinating; Binary number","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.002045099,0.0003690846,0.0004102898,0.002181245,0.000146521,0.0001736541,0.000885589,0.0002477608,0.001198062],"category_scores_gemma":[0.001568079,0.0003338517,0.00009474229,0.002715088,0.000863155,0.0001935594,0.0002370857,0.0003661125,0.004085903],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000428276,"about_ca_system_score_gemma":0.0009553512,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005305765,"about_ca_topic_score_gemma":0.001420356,"domain_scores_codex":[0.9958977,0.0001353228,0.0005543776,0.0009708112,0.001963513,0.000478341],"domain_scores_gemma":[0.9979433,0.0001485805,0.0005166427,0.0008708601,0.0003886168,0.0001320305],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005244213,0.001172507,0.0004927286,0.001081337,0.0002841532,0.00003199859,0.002544362,0.005164196,0.07342944,0.3013457,0.4888743,0.1255268],"study_design_scores_gemma":[0.004747614,0.003262405,0.001780423,0.009132931,0.00147351,0.0000249155,0.0008638244,0.4618681,0.01753288,0.06888535,0.4255647,0.004863341],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.0007697298,0.0004709851,0.003293647,0.0005518929,0.002551594,0.00133425,0.0006541473,0.001011688,0.9893621],"genre_scores_gemma":[0.197387,0.00004995398,0.02883159,0.0002555544,0.001471008,0.0001134054,0.0008076303,0.001997211,0.7690867],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.4567039,"threshold_uncertainty_score":0.9999114,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07145306131015673,"score_gpt":0.3807486708137582,"score_spread":0.3092956095036015,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}