{"id":"W4417510529","doi":"10.1109/eecsi67060.2025.11290439","title":"Faithful by Design: Improving Large Language Model Rationales through Counterfactual Consistency Verification","year":2025,"lang":"","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Marriott International (Canada)","funders":"","keywords":"Counterfactual thinking; Consistency (knowledge bases); Benchmark (surveying); Generalizability theory; Language model; Natural language understanding; Limiting; Inference; Ranging","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01812678,0.001613375,0.00104936,0.002410435,0.0007418628,0.004210353,0.003826363,0.002009586,0.004670716],"category_scores_gemma":[0.09539346,0.0009713429,0.003258767,0.0008650487,0.002704899,0.006079695,0.004662537,0.00320629,0.001294497],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001694839,"about_ca_system_score_gemma":0.004815905,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003719589,"about_ca_topic_score_gemma":0.006239452,"domain_scores_codex":[0.9869376,0.007061531,0.0007644892,0.001504612,0.003308001,0.0004237713],"domain_scores_gemma":[0.9457532,0.0334866,0.002967827,0.0135632,0.00374135,0.0004878281],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009262576,0.0006117479,0.01707438,0.001435345,0.0004594228,0.00116111,0.002426847,0.278203,0.02649205,0.1375086,0.01283444,0.5208668],"study_design_scores_gemma":[0.0001456477,0.0001545552,0.000621436,0.000166999,0.0001269664,0.0002306143,0.0001778289,0.8784953,0.01253376,0.09854202,0.008748104,0.00005673154],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02737835,0.0003563714,0.9557527,0.001221377,0.00008880058,0.0003860289,0.000575518,0.01241379,0.001827138],"genre_scores_gemma":[0.2911853,0.0002046568,0.7026049,0.0005057549,0.00005612488,0.0003677779,0.002246324,0.001748183,0.001080785],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01812678,"threshold_uncertainty_score":0.09586465,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02994314451629832,"score_gpt":0.2802219874567194,"score_spread":0.2502788429404211,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}