{"id":"W4408156536","doi":"10.1007/978-3-031-82481-4_10","title":"Robust Infidelity: When Faithfulness Measures on Masked Language Models Are Misleading","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Programming language; Speech recognition","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01538699,0.00134956,0.001782033,0.002392458,0.001319242,0.004009156,0.002629688,0.004462377,0.005611324],"category_scores_gemma":[0.1652961,0.001407761,0.0009792924,0.00218396,0.004895626,0.01314718,0.006737126,0.005253346,0.001203351],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00143357,"about_ca_system_score_gemma":0.001237472,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001463086,"about_ca_topic_score_gemma":0.0009991731,"domain_scores_codex":[0.9906268,0.004815026,0.0005387436,0.001606965,0.001888754,0.0005237017],"domain_scores_gemma":[0.8830209,0.09394559,0.004781796,0.0135833,0.003624753,0.001043662],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001484764,0.0001360866,0.01020129,0.000622595,0.0003205491,0.001476235,0.001969191,0.1058307,0.009877083,0.5910794,0.015183,0.2618191],"study_design_scores_gemma":[0.0000274188,0.00008704673,0.0008190137,0.00009893909,0.00004939107,0.0004275679,0.0001658678,0.306421,0.005112996,0.6845003,0.002244526,0.00004587547],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03987131,0.0008629985,0.9495091,0.002184493,0.0002289968,0.00006406307,0.0002619833,0.001397646,0.005619376],"genre_scores_gemma":[0.8111328,0.0005886642,0.1790122,0.001600974,0.0006554784,0.0001484422,0.0006092506,0.001288275,0.004963912],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01538699,"threshold_uncertainty_score":0.08137518,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04817885230213251,"score_gpt":0.2488063582291432,"score_spread":0.2006275059270106,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}