{"id":"W7077905675","doi":"10.48448/geb8-4d78","title":"FaithBench: A Diverse Hallucination Benchmark for Summarization by Modern LLMs","year":2025,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Credit Risk and Financial Regulations","field":"Economics, Econometrics and Finance","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Automatic summarization; Benchmark (surveying); Ground truth; Grammaticality; Diversity (politics); Meaning (existential)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004910268,0.002274565,0.0008714031,0.003156655,0.0008604464,0.002020405,0.002020508,0.002373755,0.006558602],"category_scores_gemma":[0.02936589,0.0003571734,0.001164968,0.001765779,0.0008923511,0.003055333,0.002514632,0.001513774,0.004043138],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001041396,"about_ca_system_score_gemma":0.0009362672,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004837479,"about_ca_topic_score_gemma":0.00694447,"domain_scores_codex":[0.9950836,0.002289543,0.0006377367,0.0008273301,0.0009611821,0.0002006578],"domain_scores_gemma":[0.9874369,0.007347842,0.0007708268,0.002052559,0.001924944,0.0004669829],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002854088,0.0009432694,0.00841771,0.006131927,0.0008438462,0.001681239,0.002270182,0.1093239,0.03016873,0.005725643,0.2033389,0.6283005],"study_design_scores_gemma":[0.001066685,0.003450393,0.01204955,0.0006116897,0.0003984211,0.002035233,0.00276891,0.7486383,0.07110903,0.0214758,0.1359928,0.0004030922],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.3506648,0.01996857,0.3704454,0.003628561,0.002431888,0.002998259,0.08122277,0.13916,0.02947972],"genre_scores_gemma":[0.520843,0.002048915,0.2779788,0.001195085,0.0003848446,0.00120007,0.181669,0.004004891,0.0106753],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006558602,"threshold_uncertainty_score":0.02596831,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02244945441742962,"score_gpt":0.2514437560826722,"score_spread":0.2289943016652425,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}