{"id":"W4390962525","doi":"10.48550/arxiv.2401.07927","title":"Are self-explanations from Large Language Models faithful?","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Counterfactual thinking; Counterfactual conditional; Interpretability; Consistency (knowledge bases); Measure (data warehouse); Set (abstract data type); Task (project management); Computer science; Inference; Cognitive psychology; Linguistics; Psychology; Artificial intelligence; Natural language processing; Social psychology; Data mining; Programming language; Economics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01234017,0.001000452,0.0009742958,0.001836309,0.0008063687,0.004598583,0.00204998,0.001810985,0.003365258],"category_scores_gemma":[0.1107141,0.001058743,0.001640067,0.0008283092,0.002641656,0.008982058,0.00305693,0.004333188,0.0008803084],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001659996,"about_ca_system_score_gemma":0.001724,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00369188,"about_ca_topic_score_gemma":0.004556925,"domain_scores_codex":[0.9913663,0.004909725,0.000472672,0.001572465,0.001307741,0.0003709138],"domain_scores_gemma":[0.882997,0.08448327,0.008193073,0.01960398,0.003578766,0.001143837],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00217898,0.0005015242,0.1001786,0.001450624,0.002301565,0.001204321,0.008052746,0.2187905,0.01231778,0.3019787,0.01482492,0.3362196],"study_design_scores_gemma":[0.00008546882,0.00008437445,0.004669781,0.0001390081,0.0001403721,0.0001998541,0.000352805,0.5156356,0.003717381,0.4717227,0.003177086,0.00007557285],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1741481,0.0009552895,0.811556,0.004634903,0.0001491892,0.0001564027,0.001031301,0.00310579,0.004263151],"genre_scores_gemma":[0.93166,0.0002545764,0.0639343,0.0008973294,0.0001587705,0.0001209672,0.001441245,0.0004971054,0.001035643],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01234017,"threshold_uncertainty_score":0.06526184,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06590286736689008,"score_gpt":0.1925870347735803,"score_spread":0.1266841674066902,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}