{"id":"W4407570660","doi":"10.1145/3706598.3714020","title":"Fostering Appropriate Reliance on Large Language Models: The Role of Explanations, Sources, and Inconsistencies","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":true,"ca_institutions":"Microsoft (Canada)","funders":"Princeton University; National Science Foundation","keywords":"Key (lock); Computer science; Scale (ratio); Raising (metalworking); Cognitive psychology; Psychology; Computer security; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02733528,0.0009112439,0.0006807227,0.00105083,0.0007434645,0.004714951,0.001531808,0.00171314,0.001650397],"category_scores_gemma":[0.2105626,0.0008434368,0.0004593596,0.0007121039,0.0015744,0.005655042,0.004122749,0.002363233,0.00047362],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006956666,"about_ca_system_score_gemma":0.001640656,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007896567,"about_ca_topic_score_gemma":0.001279861,"domain_scores_codex":[0.9732473,0.01995497,0.001436794,0.001533445,0.003361259,0.0004663138],"domain_scores_gemma":[0.619521,0.3310373,0.01648501,0.02427623,0.007118625,0.001561797],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001737213,0.001212658,0.1625419,0.003493648,0.0003856001,0.001375932,0.171275,0.007636584,0.1600612,0.01705097,0.003618516,0.4696108],"study_design_scores_gemma":[0.001003045,0.004061189,0.2290699,0.00442586,0.001747238,0.006107562,0.06731174,0.2593377,0.2058113,0.1354582,0.08433752,0.001328909],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8413632,0.0005400669,0.1482061,0.00180014,0.00003920755,0.000251899,0.00009732321,0.002723021,0.004978984],"genre_scores_gemma":[0.9333029,0.000154089,0.06535483,0.0002397301,0.00002331542,0.0001533781,0.00008965082,0.0002380928,0.0004438806],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02733528,"threshold_uncertainty_score":0.1445645,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03481069476889895,"score_gpt":0.2591592520451504,"score_spread":0.2243485572762514,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}